From 6ffd4efe0a6cb6b85a38b7f3b2f2fc4bd74654cd Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Sun, 9 Aug 2026 21:53:23 +1000 Subject: [PATCH 01/32] clusterer: controller-support API, gated behind a build-time flag Groundwork for clusterer_controller (added later in this series): an API the controller binds to, a way to declare which clusters it manages, and the hooks that let a controller-managed cluster take its identity at runtime instead of from the config. Declaring a managed cluster uses a per-cluster 'cluster_options' modparam - the same "key=value, key=value" idiom as my_node_info: modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") cluster_id is required; use_controller is a 0/1 flag defaulting to 0 (native), and only use_controller=1 pre-creates the controller-managed stub, which never touches the DB and is guarded against hijacking a native cluster of the same id. Native clusters need no cluster_options line at all, and every other clusterer setting (db_mode, ping_*, my_node_id, sharing_tag, ...) stays a global modparam. The interim 'use_controller' / 'cluster_id' int modparams (never released) are kept registered only to fail with a migration hint. The managed-id set is exported through the ctrl binds (managed_count / managed_ids) so the controller can cross-check it pre-fork: a managed id with no matching 'cluster' config has no BIN socket or crypto params, and a 'cluster' config for an unmanaged id has nothing legitimate to drive. Either mismatch aborts at startup naming the offending id, rather than half-forming a cluster. All of it is compiled only under CLUSTERER_CTRL_SUPPORT: - the clusterer_ctrl API (clusterer_ctrl.c) and the cluster_options modparam plus the load_clusterer_ctrl_binds export; - the controller stub pre-create loop, the child_init guard, the shm current_id mirror, the on-demand stub, and the shtag_managed / controller_managed logic; - the per-cluster identity and hybrid-db_mode accessors (cluster_self_id, cl_db_mode, GET_CURRENT_ID, use_controller), which get #else fallbacks to the stock globals (current_id / db_mode / 0) so their call sites compile to the exact upstream object code with no per-site #ifdef. add_node_info's internal self_id parameter is gated the same way. The top-level Makefile exports CLUSTERER_CTRL_SUPPORT=1 iff clusterer_controller is in the configured build - derived from the include/exclude lists rather than the current 'modules' subset, so it is stable across 'make all' and a single-module rebuild - and clusterer/Makefile turns that into the -D. The point of the gate: a build without clusterer_controller produces the stock clusterer module. No cluster_options parameter (rejected as unknown), no behavioural change, no added exports. Verified with 'unifdef -UCLUSTERER_CTRL_SUPPORT' against the base - no semantic difference from upstream. Enabling the controller rebuilds clusterer with the hooks; the two are a matched pair. --- Makefile | 13 + modules/clusterer/Makefile | 8 + modules/clusterer/clusterer.c | 161 ++++++++++-- modules/clusterer/clusterer_ctrl.c | 404 +++++++++++++++++++++++++++++ modules/clusterer/clusterer_ctrl.h | 184 +++++++++++++ modules/clusterer/clusterer_mod.c | 251 +++++++++++++++++- modules/clusterer/node_info.c | 59 +++-- modules/clusterer/node_info.h | 62 +++++ modules/clusterer/sharing_tags.c | 127 ++++++++- modules/clusterer/sharing_tags.h | 4 + modules/clusterer/sync.c | 38 +++ modules/clusterer/topology.c | 48 ++-- 12 files changed, 1293 insertions(+), 66 deletions(-) create mode 100644 modules/clusterer/clusterer_ctrl.c create mode 100644 modules/clusterer/clusterer_ctrl.h diff --git a/Makefile b/Makefile index aa87b754068..0f07348881b 100644 --- a/Makefile +++ b/Makefile @@ -70,6 +70,19 @@ include Makefile.defs # always exclude the SVN dir override exclude_modules+= .svn $(skip_modules) +# The clusterer module exposes its controller-support API (and the extra +# cluster_options modparam) only when the clusterer_controller module is part of +# this build; otherwise it is a stock module. Mirror the module-selection rule +# below (built if force-included, else unless excluded) and export the result so +# clusterer/Makefile can define -DCLUSTERER_CTRL_SUPPORT. Derived from the +# configured include/exclude lists (not the current 'modules' subset) so it stays +# consistent whether you 'make all' or rebuild just modules/clusterer. +ifneq ($(filter clusterer_controller,$(include_modules)),) +export CLUSTERER_CTRL_SUPPORT:=1 +else ifeq ($(filter clusterer_controller,$(exclude_modules)),) +export CLUSTERER_CTRL_SUPPORT:=1 +endif + #always include this modules #include_modules?= diff --git a/modules/clusterer/Makefile b/modules/clusterer/Makefile index 3ca15b036d3..d5b014e8ea8 100644 --- a/modules/clusterer/Makefile +++ b/modules/clusterer/Makefile @@ -6,4 +6,12 @@ NAME=clusterer.so #DEFS+= -DCLUSTERER_DBG #DEFS+= -DCLUSTERER_EXTRA_BIN_DBG +# Controller support (the clusterer_ctrl API, the cluster_options modparam and +# the controller hooks) is compiled only when the clusterer_controller module is +# part of this build - the top-level Makefile exports CLUSTERER_CTRL_SUPPORT=1 in +# that case. A stock build produces the vanilla clusterer module. +ifeq ($(CLUSTERER_CTRL_SUPPORT),1) +DEFS+= -DCLUSTERER_CTRL_SUPPORT +endif + include ../../Makefile.modules diff --git a/modules/clusterer/clusterer.c b/modules/clusterer/clusterer.c index ae6c16cbc70..de610588d19 100644 --- a/modules/clusterer/clusterer.c +++ b/modules/clusterer/clusterer.c @@ -87,6 +87,9 @@ void sync_check_timer(utime_t ticks, void *param) lock_start_read(cl_list_lock); for (cl = *cluster_list; cl; cl = cl->next) { +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cl->current_node) continue; +#endif lock_get(cl->current_node->lock); if (!(cl->current_node->flags & NODE_STATE_ENABLED)) { lock_release(cl->current_node->lock); @@ -109,11 +112,35 @@ void sync_check_timer(utime_t ticks, void *param) cap->flags &= ~(CAP_SYNC_PENDING|CAP_SYNC_STARTUP); sr_set_status(cl_srg, STR2CI(cap->reg.sr_id), CAP_SR_SYNCED, STR2CI(CAP_SR_STATUS_STR(CAP_SR_SYNCED)), 0); +#ifdef CLUSTERER_CTRL_SUPPORT + /* Seed-fallback for a still-PENDING sync (the transfer never + * started - no donor responded to our request within + * seed_fb_interval). This is exactly what the seed + * fallback is *for*: on a fresh or simultaneous cold start + * no peer has any data to donate, so the seed proceeds as + * authoritative. It is expected, not an error - the state + * is still recorded in the status report for visibility. A + * genuine partial-sync failure is a *different* branch (the + * SYNC_IN_PROGRESS timeout below) and is logged there. */ + if (cl->node_list == NULL) + sr_add_report_fmt(cl_srg, STR2CI(cap->reg.sr_id), 0, + "No peers present — self-synced as first node in cluster"); + else + sr_add_report_fmt(cl_srg, STR2CI(cap->reg.sr_id), 0, + "Seed fallback — no donor sent data in due time, " + "self-synced as seed"); + LM_DBG("Seed fallback in cluster %d: capability '%.*s' " + "self-marked as synced (%s)\n", cl->cluster_id, + cap->reg.name.len, cap->reg.name.s, + cl->node_list == NULL ? "first/lone node" + : "no donor sent data in due time"); +#else sr_add_report_fmt(cl_srg, STR2CI(cap->reg.sr_id), 0, "ERROR: Sync request aborted! (no donor found in due time)" " => fallback to synced state"); LM_ERR("Sync request aborted! (no donor found in due time)" ", falling back to synced state\n"); +#endif /* send update about the state of this capability */ send_single_cap_update(cl, cap, 1); @@ -155,7 +182,7 @@ int cl_set_state(int cluster_id, int node_id, enum cl_node_state state) return -1; } - if (node_id != current_id) { + if (node_id != cluster_self_id(cluster)) { node = get_node_by_id(cluster, node_id); if (!node) { lock_stop_read(cl_list_lock); @@ -190,13 +217,18 @@ int cl_set_state(int cluster_id, int node_id, enum cl_node_state state) LM_INFO("Set state: %s for node: %d in cluster: %d\n", state ? "enabled" : "disabled", node_id, cluster_id); - if (db_mode && update_db_state(cluster_id, node_id, state) < 0) + if (cl_db_mode(cluster) && update_db_state(cluster_id, node_id, state) < 0) LM_ERR("Failed to update state in clusterer DB for node [%d] cluster [%d]\n", node_id, cluster_id); return 0; } +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cluster->current_node) + return -1; +#endif + lock_get(cluster->current_node->lock); if (state == STATE_DISABLED && cluster->current_node->flags & NODE_STATE_ENABLED) @@ -226,7 +258,7 @@ int cl_set_state(int cluster_id, int node_id, enum cl_node_state state) LM_INFO("Set state: %s for local node in cluster: %d\n", state ? "enabled" : "disabled", cluster_id); - if (db_mode && update_db_state(cluster_id, current_id, state) < 0) + if (cl_db_mode(cluster) && update_db_state(cluster_id, current_id, state) < 0) LM_ERR("Failed to update state in clusterer DB for cluster [%d]\n", cluster->cluster_id); return 0; @@ -628,12 +660,35 @@ clusterer_bridges_bcast_msg(bin_packet_t *packet, int src_cid) } +#ifdef CLUSTERER_CTRL_SUPPORT +/* This node's id in @cluster_id, resolved without taking cl_list_lock: clusters + * persist for the module's lifetime, and callers may already hold cl_list_lock + * (read) or a different lock (shtags) whose order is cl_list_lock -> shtags_lock, + * so taking cl_list_lock here would risk a reverse-order deadlock. Matches the + * lock-free nature of the GET_CURRENT_ID this replaces. Returns -1 if the + * cluster is unknown or this node's identity in it is not yet established. */ +static int trailer_self_id(int cluster_id) +{ + cluster_info_t *cl; + + for (cl = *cluster_list; cl; cl = cl->next) + if (cl->cluster_id == cluster_id) + return cluster_self_id(cl); + return -1; +} +#endif + int msg_add_trailer(bin_packet_t *packet, int cluster_id, int dst_id) { if (bin_push_int(packet, cluster_id) < 0) return -1; +#ifdef CLUSTERER_CTRL_SUPPORT + if (bin_push_int(packet, trailer_self_id(cluster_id)) < 0) + return -1; +#else if (bin_push_int(packet, current_id) < 0) return -1; +#endif if (bin_push_int(packet, dst_id) < 0) return -1; @@ -947,7 +1002,7 @@ static void handle_cap_update(bin_packet_t *packet, node_info_t *source) for (i = 0; i < nr_nodes; i++) { bin_pop_int(packet, &node_id); - if (node_id == current_id) { + if (node_id == cluster_self_id(source->cluster)) { bin_pop_int(packet, &nr_cap); for (j = 0; j < nr_cap; j++) { bin_pop_str(packet, &cap); @@ -1128,12 +1183,17 @@ static void handle_remove_node(bin_packet_t *packet, cluster_info_t *cl) bin_pop_int(packet, &target_node); LM_DBG("Received remove node command for node id: [%d]\n", target_node); - if (db_mode) { + if (cl_db_mode(cl)) { LM_DBG("We are in DB mode, ignoring received remove node command\n"); return; } - if (target_node == current_id) { + if (target_node == cluster_self_id(cl)) { +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cl->current_node) + return; +#endif + lock_get(cl->current_node->lock); if (cl->current_node->flags & NODE_STATE_ENABLED) { @@ -1180,14 +1240,16 @@ void bin_rcv_cl_extra_packets(bin_packet_t *packet, int packet_type, LM_DBG("received clusterer message from: %s:%hu with source id: %d and" " cluster id: %d\n", ip, port, source_id, cluster_id); +#ifndef CLUSTERER_CTRL_SUPPORT if (source_id == current_id) { LM_ERR("Received message with bad source - same node id as this instance\n"); return; } +#endif gettimeofday(&now, NULL); - if (!db_mode && packet_type == CLUSTERER_REMOVE_NODE) + if ((!db_mode || use_controller) && packet_type == CLUSTERER_REMOVE_NODE) lock_start_write(cl_list_lock); else lock_start_read(cl_list_lock); @@ -1199,6 +1261,13 @@ void bin_rcv_cl_extra_packets(bin_packet_t *packet, int packet_type, goto exit; } +#ifdef CLUSTERER_CTRL_SUPPORT + if (source_id == cluster_self_id(cl)) { + LM_ERR("Received message with bad source - same node id as this instance\n"); + goto exit; + } +#endif + lock_get(cl->current_node->lock); if (!(cl->current_node->flags & NODE_STATE_ENABLED)) { lock_release(cl->current_node->lock); @@ -1236,7 +1305,7 @@ void bin_rcv_cl_extra_packets(bin_packet_t *packet, int packet_type, } else lock_release(node->lock); - if (dest_id != current_id) { + if (dest_id != cluster_self_id(cl)) { if (clusterer_enable_rerouting == 0) { LM_WARN("Received message for destination id [%d] but rerouting disabled\n", dest_id); goto exit; @@ -1292,7 +1361,7 @@ void bin_rcv_cl_extra_packets(bin_packet_t *packet, int packet_type, } exit: - if (!db_mode && packet_type == CLUSTERER_REMOVE_NODE) + if ((!db_mode || use_controller) && packet_type == CLUSTERER_REMOVE_NODE) lock_stop_write(cl_list_lock); else lock_stop_read(cl_list_lock); @@ -1318,12 +1387,14 @@ void bin_rcv_cl_packets(bin_packet_t *packet, int packet_type, LM_DBG("received clusterer message from: %s:%hu with source id: %d and " "cluster id: %d\n", ip, port, source_id, cl_id); +#ifndef CLUSTERER_CTRL_SUPPORT if (source_id == current_id) { LM_ERR("Received message with bad source - same node id as this instance\n"); return; } +#endif - if (!db_mode && (packet_type == CLUSTERER_NODE_DESCRIPTION || + if ((!db_mode || use_controller) && (packet_type == CLUSTERER_NODE_DESCRIPTION || packet_type == CLUSTERER_FULL_TOP_UPDATE)) lock_start_write(cl_list_lock); else @@ -1335,6 +1406,24 @@ void bin_rcv_cl_packets(bin_packet_t *packet, int packet_type, goto exit; } +#ifdef CLUSTERER_CTRL_SUPPORT + /* current_node is legitimately NULL while this node's identity is being + * (re)established for a dynamically constructed cluster (clusterer_ctrl + * update_identity: the cluster can already exist and receive BIN packets + * before current_node is assigned). Drop the packet instead of + * dereferencing NULL. */ + if (!cl->current_node) { + LM_INFO("Received message for cluster [%d] before local identity is " + "established, ignoring\n", cl_id); + goto exit; + } + + if (source_id == cluster_self_id(cl)) { + LM_ERR("Received message with bad source - same node id as this instance\n"); + goto exit; + } +#endif + lock_get(cl->current_node->lock); if (!(cl->current_node->flags & NODE_STATE_ENABLED)) { lock_release(cl->current_node->lock); @@ -1347,7 +1436,7 @@ void bin_rcv_cl_packets(bin_packet_t *packet, int packet_type, if (!node) { LM_INFO("Received message with unknown source id [%d]\n", source_id); - if (!db_mode) + if (!cl_db_mode(cl)) handle_internal_msg_unknown(packet, cl, packet_type, &ri->src_su, ri->proto, source_id); } else { @@ -1374,7 +1463,7 @@ void bin_rcv_cl_packets(bin_packet_t *packet, int packet_type, } exit: - if (!db_mode && (packet_type == CLUSTERER_NODE_DESCRIPTION || + if ((!db_mode || use_controller) && (packet_type == CLUSTERER_NODE_DESCRIPTION || packet_type == CLUSTERER_FULL_TOP_UPDATE)) lock_stop_write(cl_list_lock); else @@ -1472,10 +1561,12 @@ static void bin_rcv_mod_packets(bin_packet_t *packet, int packet_type, "cluster ids: %d->%d\n", ip, port, source_id, src_cluster_id, cluster_id); } +#ifndef CLUSTERER_CTRL_SUPPORT if (source_id == current_id) { LM_ERR("Received message with bad source - same node id as this instance\n"); return; } +#endif cap = (struct capability_reg *)ptr; if (!cap) { @@ -1493,6 +1584,13 @@ static void bin_rcv_mod_packets(bin_packet_t *packet, int packet_type, goto exit; } +#ifdef CLUSTERER_CTRL_SUPPORT + if (source_id == cluster_self_id(cl)) { + LM_ERR("Received message with bad source - same node id as this instance\n"); + goto exit; + } +#endif + lock_get(cl->current_node->lock); if (!(cl->current_node->flags & NODE_STATE_ENABLED)) { lock_release(cl->current_node->lock); @@ -1543,7 +1641,7 @@ static void bin_rcv_mod_packets(bin_packet_t *packet, int packet_type, } else lock_release(node->lock); - if (dest_id != current_id) { + if (dest_id != cluster_self_id(cl)) { /* route the message */ bin_push_int(packet, cluster_id); bin_push_int(packet, source_id); @@ -1619,6 +1717,9 @@ int send_single_cap_update(cluster_info_t *cluster, struct local_cap *cap, timestamp = (int)(unsigned long)time(NULL); +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cluster->current_node) return -1; +#endif lock_get(cluster->current_node->lock); for (neigh = cluster->current_node->neighbour_list; neigh; @@ -1637,7 +1738,7 @@ int send_single_cap_update(cluster_info_t *cluster, struct local_cap *cap, return -1; } bin_push_int(&packet, cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(cluster)); bin_push_int(&packet, ++cluster->current_node->cap_seq_no); bin_push_int(&packet, timestamp); @@ -1646,7 +1747,7 @@ int send_single_cap_update(cluster_info_t *cluster, struct local_cap *cap, /* only the current node */ bin_push_int(&packet, 1); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(cluster)); /* only a single capability */ bin_push_int(&packet, 1); @@ -1656,7 +1757,7 @@ int send_single_cap_update(cluster_info_t *cluster, struct local_cap *cap, bin_push_int(&packet, 0); /* don't require reply */ bin_push_int(&packet, 1); /* path length is 1, only current node at this point */ - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(cluster)); bin_get_buffer(&packet, &bin_buffer); for (i = 0; i < no_dests; i++) @@ -1704,7 +1805,7 @@ int send_cap_update(node_info_t *dest_node, int require_reply) return -1; } bin_push_int(&packet, dest_node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); lock_get(dest_node->cluster->current_node->lock); @@ -1719,7 +1820,7 @@ int send_cap_update(node_info_t *dest_node, int require_reply) for (cl_cap = dest_node->cluster->capabilities, nr_cap = 0; cl_cap; cl_cap = cl_cap->next, nr_cap++) ; if (nr_cap) { - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); bin_push_int(&packet, nr_cap); for (cl_cap=dest_node->cluster->capabilities;cl_cap;cl_cap=cl_cap->next) { bin_push_str(&packet, &cl_cap->reg.name); @@ -1750,7 +1851,7 @@ int send_cap_update(node_info_t *dest_node, int require_reply) bin_push_int(&packet, require_reply); bin_push_int(&packet, 1); /* path length is 1, only current node at this point */ - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); bin_get_buffer(&packet, &bin_buffer); if (msg_send(dest_node->cluster->send_sock, dest_node->proto, &dest_node->addr, @@ -1903,9 +2004,31 @@ int cl_register_cap(str *cap, cl_packet_cb_f packet_cb, cl_event_cb_f event_cb, cluster = get_cluster_by_id(cluster_id); if (!cluster) { +#ifdef CLUSTERER_CTRL_SUPPORT + if (use_controller) { + cluster = shm_malloc(sizeof *cluster); + if (!cluster) { LM_ERR("no shm\n"); return -1; } + memset(cluster, 0, sizeof *cluster); + cluster->cluster_id = cluster_id; + cluster->controller_managed = 1; /* on-demand controller stub */ + if ((cluster->lock = lock_alloc()) == NULL || !lock_init(cluster->lock)) { + shm_free(cluster); return -1; + } + if (cl_list_lock) lock_start_write(cl_list_lock); + cluster->next = *cluster_list; + *cluster_list = cluster; + if (cl_list_lock) lock_stop_write(cl_list_lock); + LM_INFO("clusterer: auto-created stub for cluster %d\n", cluster_id); + } else { + LM_ERR("cluster id %d is not defined in the %s\n", cluster_id, + db_mode ? "DB" : "script"); + return -1; + } +#else LM_ERR("cluster id %d is not defined in the %s\n", cluster_id, db_mode ? "DB" : "script"); return -1; +#endif } new_cl_cap = shm_malloc(sizeof *new_cl_cap + cap->len + CAP_SR_ID_PREFIX_LEN); diff --git a/modules/clusterer/clusterer_ctrl.c b/modules/clusterer/clusterer_ctrl.c new file mode 100644 index 00000000000..de9d5052107 --- /dev/null +++ b/modules/clusterer/clusterer_ctrl.c @@ -0,0 +1,404 @@ +/* + * clusterer_ctrl.c — Controller API implementation for clusterer + * + * Copyright (C) 2026 Yury Kirsanov + * VoIPLine Telecom + * + * This file is part of opensips, a free SIP server. + * + * opensips is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * (at your option) any later version. + */ + +#include "../../dprint.h" +#include "../../rw_locking.h" +#include "../../mem/shm_mem.h" +#include "../../locking.h" + +#include "node_info.h" /* add_node_info, remove_node_list, + get_cluster_by_id, get_node_by_id, + cluster_list, cl_list_lock, current_id */ +#include "clusterer.h" /* LS_DOWN, do_actions_node_ev, MAX_NO_CLUSTERS */ +#include "sharing_tags.h" +#include "topology.h" /* delete_neighbour */ /* shtag_event_handler */ +#include "clusterer_ctrl.h" + +/* This whole API is compiled only when the clusterer_controller module is part + * of the build (see modules/clusterer/Makefile). In a stock clusterer build the + * translation unit is empty. */ +#ifdef CLUSTERER_CTRL_SUPPORT + +/* declared in clusterer.c — raises E_CLUSTERER_NODE_STATE_CHANGED */ +int report_node_state(enum clusterer_event event, int cluster_id, int node_id); + +/* Free a current_node entry that is NOT in node_list. + * remove_node_list() walks node_list looking for the pointer — if current_node + * was never added there (our set_my_identity path) it crashes. */ +static void free_current_node(node_info_t *node) +{ + if (!node) return; + if (node->lock) { + lock_destroy(node->lock); + lock_dealloc(node->lock); + } + if (node->sp_info) shm_free(node->sp_info); + if (node->description.s) shm_free(node->description.s); + if (node->sip_addr.s) shm_free(node->sip_addr.s); + if (node->url.s) shm_free(node->url.s); + shm_free(node); +} + +/** + * clusterer_ctrl_set_identity() - register this node's own identity. + * + * CRITICAL: current_id MUST be set before calling add_node_info(). + * add_node_info() checks (node_id == current_id) to decide whether to + * place the entry in cluster->current_node (self, not pinged) or + * cluster->node_list (peer, pinged). Setting it after causes the local + * node to land in node_list and get pinged — "same node id" errors. + */ +int clusterer_ctrl_set_identity(int cluster_id, int node_id, str *bin_url) +{ + node_info_t *new_node = NULL; + cluster_info_t *cl; + int int_vals[NO_DB_INT_VALS]; + str str_vals[NO_DB_STR_VALS]; + static str desc = str_init("controller"); + static str seed = str_init("seed"); + + int_vals[INT_VALS_ID_COL] = 0; + int_vals[INT_VALS_CLUSTER_ID_COL] = cluster_id; + int_vals[INT_VALS_NODE_ID_COL] = node_id; + int_vals[INT_VALS_STATE_COL] = 1; + int_vals[INT_VALS_NO_PING_RETRIES_COL] = DEFAULT_NO_PING_RETRIES; + int_vals[INT_VALS_PRIORITY_COL] = DEFAULT_PRIORITY; + + memset(str_vals, 0, sizeof str_vals); + str_vals[STR_VALS_URL_COL] = *bin_url; + str_vals[STR_VALS_FLAGS_COL] = seed; + str_vals[STR_VALS_DESCRIPTION_COL] = desc; + + /* Set current_id BEFORE add_node_info */ + current_id = node_id; + if (_current_id_shm) *_current_id_shm = node_id; + + lock_start_write(cl_list_lock); + + if (add_node_info(&new_node, cluster_list, int_vals, str_vals, node_id) < 0) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: set_my_identity: add_node_info failed for " + "cluster %d node %d\n", cluster_id, node_id); + return -1; + } + + cl = get_cluster_by_id(cluster_id); + if (cl && new_node && !cl->current_node) + cl->current_node = new_node; + + lock_stop_write(cl_list_lock); + + LM_INFO("clusterer: [cluster %d] identity set: node_id=%d url=%.*s\n", + cluster_id, node_id, bin_url->len, bin_url->s); + return 0; +} + +/** + * clusterer_ctrl_add_node() - add a discovered peer at runtime. + */ +int clusterer_ctrl_add_node(int cluster_id, int node_id, str *bin_url) +{ + node_info_t *new_node = NULL; + cluster_info_t *cl; + int int_vals[NO_DB_INT_VALS]; + str str_vals[NO_DB_STR_VALS]; + static str desc = str_init("controller"); + static str seed = str_init("seed"); + + lock_start_write(cl_list_lock); + + cl = get_cluster_by_id(cluster_id); + if (cl && get_node_by_id(cl, node_id)) { + lock_stop_write(cl_list_lock); + LM_DBG("clusterer: [cluster %d] node %d already present\n", + cluster_id, node_id); + return 0; + } + + int_vals[INT_VALS_ID_COL] = 0; + int_vals[INT_VALS_CLUSTER_ID_COL] = cluster_id; + int_vals[INT_VALS_NODE_ID_COL] = node_id; + int_vals[INT_VALS_STATE_COL] = 1; + int_vals[INT_VALS_NO_PING_RETRIES_COL] = DEFAULT_NO_PING_RETRIES; + int_vals[INT_VALS_PRIORITY_COL] = DEFAULT_PRIORITY; + + memset(str_vals, 0, sizeof str_vals); + str_vals[STR_VALS_URL_COL] = *bin_url; + str_vals[STR_VALS_FLAGS_COL] = seed; + str_vals[STR_VALS_DESCRIPTION_COL] = desc; + + if (add_node_info(&new_node, cluster_list, int_vals, str_vals, cluster_self_id(cl)) < 0) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: add_node: add_node_info failed for " + "cluster %d node %d\n", cluster_id, node_id); + return -1; + } + + lock_stop_write(cl_list_lock); + + LM_INFO("clusterer: [cluster %d] added peer node_id=%d url=%.*s\n", + cluster_id, node_id, bin_url->len, bin_url->s); + return 0; +} + +/** + * clusterer_ctrl_remove_node() - remove a departed peer at runtime. + */ +int clusterer_ctrl_remove_node(int cluster_id, int node_id) +{ + cluster_info_t *cl; + node_info_t *node; + + lock_start_write(cl_list_lock); + + cl = get_cluster_by_id(cluster_id); + if (!cl) { + lock_stop_write(cl_list_lock); + LM_WARN("clusterer: remove_node: cluster %d not found\n", cluster_id); + return -1; + } + + node = get_node_by_id(cl, node_id); + if (!node) { + lock_stop_write(cl_list_lock); + LM_WARN("clusterer: remove_node: node %d not found in cluster %d\n", + node_id, cluster_id); + return -1; + } + + /* Purge all topology references to the departing node BEFORE + * freeing it: neighbour lists of current_node and every peer, + * plus next_hop pointers. Freed-node reuse (same node_id + * reassigned later) otherwise leaves dangling pointers that + * crash with bogus proto/node values. */ + { + node_info_t *it; + if (cl->current_node) + delete_neighbour(cl->current_node, node); + for (it = cl->node_list; it; it = it->next) { + if (it == node) continue; + lock_get(it->lock); + delete_neighbour(it, node); + if (it->next_hop && it->next_hop->node_id == node_id) + it->next_hop = NULL; + lock_release(it->lock); + } + } + + /* Remove node from list, then fire callbacks outside the lock. + * Callbacks (dialog rcv_cluster_event) call back into clusterer + * to send BIN packets and need cl_list_lock for read. */ + remove_node_list(cl, node); + + { + struct local_cap *cap_it; + struct local_cap *caps = cl->capabilities; + lock_stop_write(cl_list_lock); + for (cap_it = caps; cap_it; cap_it = cap_it->next) + if (cap_it->reg.event_cb) + cap_it->reg.event_cb(CLUSTER_NODE_DOWN, node_id); + report_node_state(CLUSTER_NODE_DOWN, cluster_id, node_id); + } + + LM_INFO("clusterer: [cluster %d] removed node_id=%d\n", + cluster_id, node_id); + return 0; +} + +/** + * clusterer_ctrl_update_identity() - correct this node's node_id. + * + * Replaces the optimistic node_id=1 with the real master-assigned id. + * No-op if id unchanged. + * + * CRITICAL: current_id must be set BEFORE free+add so add_node_info + * routes the new entry to current_node (self) not node_list (peer). + * current_node is NOT in node_list so we free it directly — calling + * remove_node_list() on it would crash walking the list for a pointer + * that isn't there. + */ +int clusterer_ctrl_update_identity(int cluster_id, int new_node_id, str *bin_url) +{ + cluster_info_t *cl; + node_info_t *new_node = NULL; + node_info_t *old_node; + int int_vals[NO_DB_INT_VALS]; + str str_vals[NO_DB_STR_VALS]; + static str desc = str_init("controller"); + static str seed = str_init("seed"); + + lock_start_write(cl_list_lock); + + cl = get_cluster_by_id(cluster_id); + if (!cl) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: update_identity: cluster %d not found\n", cluster_id); + return -1; + } + + if (cl->current_node && cl->current_node->node_id == new_node_id) { + lock_stop_write(cl_list_lock); + return 0; /* no-op */ + } + + /* Set current_id BEFORE free+add */ + current_id = new_node_id; + if (_current_id_shm) *_current_id_shm = new_node_id; + + old_node = cl->current_node; + cl->current_node = NULL; + + /* Purge all peer neighbour references to the old current_node before + * freeing it — same as clusterer_ctrl_remove_node does for peers. */ + if (old_node) { + node_info_t *it; + for (it = cl->node_list; it; it = it->next) { + lock_get(it->lock); + delete_neighbour(it, old_node); + if (it->next_hop && it->next_hop->node_id == old_node->node_id) + it->next_hop = NULL; + lock_release(it->lock); + } + } + + int_vals[INT_VALS_ID_COL] = 0; + int_vals[INT_VALS_CLUSTER_ID_COL] = cluster_id; + int_vals[INT_VALS_NODE_ID_COL] = new_node_id; + int_vals[INT_VALS_STATE_COL] = 1; + int_vals[INT_VALS_NO_PING_RETRIES_COL] = DEFAULT_NO_PING_RETRIES; + int_vals[INT_VALS_PRIORITY_COL] = DEFAULT_PRIORITY; + + memset(str_vals, 0, sizeof str_vals); + str_vals[STR_VALS_URL_COL] = *bin_url; + str_vals[STR_VALS_FLAGS_COL] = seed; + str_vals[STR_VALS_DESCRIPTION_COL] = desc; + + if (add_node_info(&new_node, cluster_list, int_vals, str_vals, new_node_id) < 0) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: update_identity: add_node_info failed for " + "cluster %d node %d\n", cluster_id, new_node_id); + free_current_node(old_node); + return -1; + } + + cl->current_node = new_node; + + lock_stop_write(cl_list_lock); + + /* Free old entry outside the lock */ + free_current_node(old_node); + + LM_INFO("clusterer: [cluster %d] identity updated to node_id=%d url=%.*s\n", + cluster_id, new_node_id, bin_url->len, bin_url->s); + return 0; +} + +/** + * load_clusterer_ctrl_binds() - fill the API struct for use by controller. + */ +int clusterer_ctrl_sync_current_id(void) +{ + cluster_info_t *cl; + + if (!cl_list_lock || !cluster_list || !*cluster_list) + return 0; + + lock_start_read(cl_list_lock); + for (cl = *cluster_list; cl; cl = cl->next) { + if (cl->current_node) { + current_id = cl->current_node->node_id; + break; + } + } + lock_stop_read(cl_list_lock); + return 0; +} + +int clusterer_ctrl_activate_backup_shtags(int cluster_id) +{ + return shtag_activate_all_backup(cluster_id); +} + +int clusterer_ctrl_force_backup_shtags(int cluster_id) +{ + return shtag_force_all_backup(cluster_id); +} + +int clusterer_ctrl_set_shtag_managed(int cluster_id) +{ + cluster_info_t *cl; + + lock_start_write(cl_list_lock); + cl = get_cluster_by_id(cluster_id); + if (!cl) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: set_shtag_managed: cluster %d not found\n", + cluster_id); + return -1; + } + cl->shtag_managed = 1; + lock_stop_write(cl_list_lock); + + LM_INFO("clusterer: [cluster %d] sharing tags are now " + "controller-managed (MI and script changes blocked)\n", + cluster_id); + return 0; +} + +int clusterer_ctrl_unset_shtag_managed(int cluster_id) +{ + cluster_info_t *cl; + + lock_start_write(cl_list_lock); + cl = get_cluster_by_id(cluster_id); + if (!cl) { + lock_stop_write(cl_list_lock); + LM_ERR("clusterer: unset_shtag_managed: cluster %d not found\n", + cluster_id); + return -1; + } + cl->shtag_managed = 0; + lock_stop_write(cl_list_lock); + + LM_INFO("clusterer: [cluster %d] sharing tags are no longer " + "controller-managed (MI and script changes allowed again)\n", + cluster_id); + return 0; +} + +/* 1 once a controller module has bound this API (see clusterer_ctrl.h). */ +int clusterer_ctrl_bound = 0; + +int load_clusterer_ctrl_binds(clusterer_ctrl_binds_t *binds) +{ + if (!binds) { + LM_ERR("clusterer: load_clusterer_ctrl_binds: NULL binds\n"); + return -1; + } + clusterer_ctrl_bound = 1; + binds->managed_count = cl_ctr_stub_count; + binds->managed_ids = cl_ctr_stub_ids; + binds->set_my_identity = clusterer_ctrl_set_identity; + binds->add_node = clusterer_ctrl_add_node; + binds->remove_node = clusterer_ctrl_remove_node; + binds->update_identity = clusterer_ctrl_update_identity; + binds->sync_current_id = clusterer_ctrl_sync_current_id; + binds->activate_backup_shtags = clusterer_ctrl_activate_backup_shtags; + binds->set_shtag_managed = clusterer_ctrl_set_shtag_managed; + binds->unset_shtag_managed = clusterer_ctrl_unset_shtag_managed; + binds->force_backup_shtags = clusterer_ctrl_force_backup_shtags; + return 0; +} + +#endif /* CLUSTERER_CTRL_SUPPORT */ diff --git a/modules/clusterer/clusterer_ctrl.h b/modules/clusterer/clusterer_ctrl.h new file mode 100644 index 00000000000..d5ba9f15289 --- /dev/null +++ b/modules/clusterer/clusterer_ctrl.h @@ -0,0 +1,184 @@ +/* + * clusterer_ctrl.h — Controller API for the clusterer module + * + * Allows an external module (clusterer_controller) to drive the clusterer + * topology at runtime without any DB or static configuration. + * + * Copyright (C) 2026 Yury Kirsanov + * VoIPLine Telecom + * + * This file is part of opensips, a free SIP server. + * + * opensips is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * (at your option) any later version. + */ + +#ifndef CLUSTERER_CTRL_H +#define CLUSTERER_CTRL_H + +#include "../../str.h" + +/** + * clusterer_ctrl_binds - API struct loaded by clusterer_controller. + * + * Usage in clusterer_controller mod_init(): + * + * #include "../clusterer/clusterer_ctrl.h" + * static clusterer_ctrl_binds_t clctl; + * if (load_clusterer_ctrl_binds(&clctl) < 0) { ... } + * + * Then from the worker process / callbacks: + * + * str url = str_init("bin:10.22.23.191:5566"); + * clctl.set_my_identity(1, my_node_id, &url); + * + * str peer_url = str_init("bin:10.22.23.192:5566"); + * clctl.add_node(1, peer_node_id, &peer_url); + * + * clctl.remove_node(1, departed_node_id); + */ +typedef struct clusterer_ctrl_binds { + /** + * set_my_identity() — register this node's identity in a cluster. + * + * Creates the cluster_info_t if it does not yet exist. + * Sets global current_id and marks cluster->current_node. + * Must be called before add_node() for the same cluster_id. + * + * Called by controller after node_id is allocated (either from + * existing master via NODE_ASSIGN or from join deadline expiry). + * + * @cluster_id integer cluster identifier (matches controller cluster) + * @node_id integer allocated by the controller master (>= 1) + * @bin_url str pointing to "bin:IP:PORT" + * @return 0 on success, -1 on error + */ + int (*set_my_identity)(int cluster_id, int node_id, str *bin_url); + + /** + * add_node() — add a peer node to a cluster at runtime. + * + * Creates the node_info_t, adds it to the cluster's node_list. + * The clusterer ping timer picks it up within one ping interval + * and establishes the BIN link automatically. + * + * Called on every CL_CTR_PKT_NODE_ASSIGN received for a peer. + * Safe to call if the node already exists — returns 0 (no-op). + * + * @cluster_id must match a cluster initialised by set_my_identity() + * @node_id peer's allocated node_id + * @bin_url peer's "bin:IP:PORT" string + * @return 0 on success, -1 on error + */ + int (*add_node)(int cluster_id, int node_id, str *bin_url); + + /** + * remove_node() — remove a peer node from a cluster at runtime. + * + * Removes from the node_list and cleans up routing state. + * The BIN connection is closed by the clusterer's own cleanup path. + * + * Called on CL_CTR_PKT_GOODBYE or when election-window expiry removes + * the peer from the controller's own peer table. + * + * @cluster_id cluster the node belongs to + * @node_id node_id to remove + * @return 0 on success, -1 if cluster or node not found + */ + int (*remove_node)(int cluster_id, int node_id); + /** + * update_identity() — correct this node's node_id after master assignment. + * + * Called when the real node_id arrives via NODE_ASSIGN and differs from + * the optimistic value set at startup. Removes the old current_node + * entry and adds a new one with the correct node_id, updating both + * global current_id and cluster->current_node atomically. + * + * Safe to call with the same node_id as already set — returns 0 (no-op). + * + * @cluster_id cluster to update + * @new_node_id the master-assigned node_id + * @bin_url this node's "bin:IP:PORT" (may be identical to current) + * @return 0 on success, -1 on error + */ + int (*update_identity)(int cluster_id, int new_node_id, str *bin_url); + + /** + * sync_current_id() - sync local current_id from shared memory. + * + * Must be called from child_init() in every child process after fork. + * current_id is a process-local global — after fork each child inherits + * the pre-fork value. This re-reads the correct id from the shared + * cluster->current_node so BIN packets carry the right source node_id. + * + * @return 0 always + */ + int (*sync_current_id)(void); + + /** + * activate_backup_shtags() - activate all BACKUP sharing tags + * for the given cluster. Called only by the controller master + * when a peer departs or when this node becomes new master. + */ + int (*activate_backup_shtags)(int cluster_id); + + /** + * force_backup_shtags() - force all local sharing tags to BACKUP + * regardless of =active config. Called by the controller at + * startup when it manages tags itself. + */ + int (*force_backup_shtags)(int cluster_id); + + /** + * set_shtag_managed() - mark a cluster's sharing tags as controller-managed. + * + * Once set, the MI command clusterer_set_tag_active and the $shtag() + * script variable setter are blocked for this cluster, returning an + * error to the caller. This prevents manual or event-route-driven + * shtag changes from conflicting with controller-managed failover. + * + * Called from clusterer_controller mod_init() for every cluster that + * has manage_shtags=1. + * + * @cluster_id cluster to lock + * @return 0 on success, -1 if cluster not found + */ + int (*set_shtag_managed)(int cluster_id); + + /** + * unset_shtag_managed() - stop treating a cluster's sharing tags as + * controller-managed, re-allowing MI/script changes. Used when a node + * adopts a running cluster's manage_shtags=0 setting at runtime + * (on_config_mismatch=adopt). + * + * @cluster_id cluster to unlock + * @return 0 on success, -1 if cluster not found + */ + int (*unset_shtag_managed)(int cluster_id); + + /* The set of cluster_ids the clusterer marked controller-managed + * (cluster_options use_controller=1): count, and a pointer to the clusterer's + * own array (valid for the process lifetime; read pre-fork in mod_init). The + * controller checks it has a matching 'cluster' config for every one of these - + * a clusterer-managed cluster with no controller config is a hard error. */ + int managed_count; + int *managed_ids; +} clusterer_ctrl_binds_t; + +/* Set to 1 by load_clusterer_ctrl_binds() when a controller module binds the + * API, so clusterer can warn if use_controller=1 yet no controller registered. */ +extern int clusterer_ctrl_bound; + +/** + * load_clusterer_ctrl_binds() — fill a clusterer_ctrl_binds_t struct. + * + * Called from clusterer_controller's mod_init(). Returns -1 if clusterer + * is not loaded or was not built with use_controller support. + */ +typedef int (*load_clusterer_ctrl_binds_f)(clusterer_ctrl_binds_t *binds); + +int load_clusterer_ctrl_binds(clusterer_ctrl_binds_t *binds); + +#endif /* CLUSTERER_CTRL_H */ diff --git a/modules/clusterer/clusterer_mod.c b/modules/clusterer/clusterer_mod.c index 92a6540b792..21181a31d41 100644 --- a/modules/clusterer/clusterer_mod.c +++ b/modules/clusterer/clusterer_mod.c @@ -39,6 +39,9 @@ #include "clusterer.h" #include "sync.h" #include "sharing_tags.h" +#ifdef CLUSTERER_CTRL_SUPPORT +#include "clusterer_ctrl.h" +#endif #include "clusterer_evi.h" int ping_interval = DEFAULT_PING_INTERVAL; @@ -50,6 +53,119 @@ int current_id = -1; int db_mode = 1; int clusterer_enable_rerouting = 1; +#ifdef CLUSTERER_CTRL_SUPPORT +/* shm-backed mirror of current_id so runtime identity changes are visible across + * forked workers; and the derived 'controller active' flag. Both exist only in a + * controller-enabled build - stock code reaches these through the GET_CURRENT_ID + * and use_controller fallbacks in node_info.h. */ +int *_current_id_shm = NULL; +int use_controller = 0; /* 1 if any cluster_options sets use_controller=1 */ + +/* cluster_ids marked controller-managed via 'cluster_options' (use_controller=1). + * Not static: the controller loads a pointer to these through the ctrl binds so it + * can verify it has a matching 'cluster' config for each. */ +int cl_ctr_stub_ids[64]; +int cl_ctr_stub_count = 0; + +/* extract "=" from a "key=value, key=value" cluster_options string. + * Returns 0 and sets *out on success, 1 if the key is absent, -1 if malformed. */ +static int cl_ctr_opt_int(str *descr, str *name, int *out) +{ + char *p, *pe; + str aux; + + p = str_strstr(descr, name); + if (!p) + return 1; + p += name->len; + p = q_memchr(p, '=', descr->s + descr->len - p); + if (!p) { + LM_ERR("clusterer: expected '=' after '%.*s' in cluster_options\n", + name->len, name->s); + return -1; + } + p++; + pe = q_memchr(p, ',', descr->s + descr->len - p); + aux.s = p; + aux.len = pe ? pe - p : descr->s + descr->len - p; + str_trim_spaces_lr(aux); + if (aux.len == 0 || str2int(&aux, (unsigned int *)out)) { + LM_ERR("clusterer: bad value for '%.*s' in cluster_options\n", + name->len, name->s); + return -1; + } + return 0; +} + +/* 'cluster_options' modparam: per-cluster settings, same key=value idiom as + * my_node_info. Format: "cluster_id=N, use_controller=0|1". cluster_id is + * required; use_controller is optional and defaults to 0 (native). Only + * use_controller=1 registers the cluster as controller-managed - its identity + * and peer list are then driven at runtime by clusterer_controller. Every other + * clusterer option stays a global modparam. */ +static int cl_ctr_add_cluster_options(modparam_t type, void *val) +{ + static str cid_prop = str_init("cluster_id"); + static str uc_prop = str_init("use_controller"); + str descr; + int cid = -1, uc = 0, rc, i; + + if (!val || !*(char *)val) { + LM_ERR("clusterer: empty 'cluster_options'\n"); + return -1; + } + descr.s = (char *)val; + descr.len = strlen(descr.s); + + rc = cl_ctr_opt_int(&descr, &cid_prop, &cid); + if (rc < 0) + return -1; + if (rc == 1 || cid < 1) { + LM_ERR("clusterer: 'cluster_options' requires cluster_id >= 1, " + "e.g. \"cluster_id=1, use_controller=1\"\n"); + return -1; + } + + rc = cl_ctr_opt_int(&descr, &uc_prop, &uc); /* absent (rc==1) -> uc stays 0 */ + if (rc < 0) + return -1; + if (uc != 0 && uc != 1) { + LM_ERR("clusterer: use_controller in 'cluster_options' must be 0 or 1 " + "(cluster_id %d)\n", cid); + return -1; + } + + if (uc == 0) + return 0; /* native (the default): nothing to register */ + + for (i = 0; i < cl_ctr_stub_count; i++) + if (cl_ctr_stub_ids[i] == cid) { + LM_ERR("clusterer: cluster_options declares cluster_id %d as " + "controller-managed more than once\n", cid); + return -1; + } + if (cl_ctr_stub_count >= (int)(sizeof cl_ctr_stub_ids / sizeof cl_ctr_stub_ids[0])) { + LM_ERR("clusterer: too many controller-managed clusters\n"); + return -1; + } + cl_ctr_stub_ids[cl_ctr_stub_count++] = cid; + use_controller = 1; /* at least one controller-managed cluster exists */ + return 0; +} + +/* Retired interim modparams (never released): controller-managed clusters are now + * declared via 'cluster_options'. Kept registered only to fail with a clear + * migration hint instead of a generic "unknown parameter". */ +static int cl_ctr_retired_param(modparam_t type, void *val) +{ + LM_ERR("clusterer: the 'use_controller'/'cluster_id' modparams have been " + "replaced; declare each controller-managed cluster with " + "modparam(\"clusterer\", \"cluster_options\", " + "\"cluster_id=N, use_controller=1\")\n"); + return -1; +} +#endif /* CLUSTERER_CTRL_SUPPORT */ + str clusterer_db_url = {NULL, 0}; extern db_con_t *db_hdl; @@ -106,6 +222,9 @@ int cmd_check_addr(struct sip_msg *msg, int *cluster_id, str *ip_str, */ static const cmd_export_t cmds[] = { +#ifdef CLUSTERER_CTRL_SUPPORT + {"load_clusterer_ctrl_binds", (cmd_function)load_clusterer_ctrl_binds, {{0,0,0}}, 0}, +#endif {"load_clusterer", (cmd_function)load_clusterer, {{0,0,0}}, 0}, {"cluster_broadcast_req", (cmd_function)cmd_broadcast_req, { {CMD_PARAM_INT,0,0}, @@ -158,6 +277,11 @@ static const param_export_t params[] = { {"flags_col", STR_PARAM, &flags_col.s }, {"description_col", STR_PARAM, &description_col.s }, {"db_mode", INT_PARAM, &db_mode }, +#ifdef CLUSTERER_CTRL_SUPPORT + {"cluster_options", STR_PARAM|USE_FUNC_PARAM, (void*)cl_ctr_add_cluster_options}, + {"use_controller", INT_PARAM|USE_FUNC_PARAM, (void*)cl_ctr_retired_param}, + {"cluster_id", INT_PARAM|USE_FUNC_PARAM, (void*)cl_ctr_retired_param}, +#endif {"neighbor_node_info", STR_PARAM|USE_FUNC_PARAM, (void*)&provision_neighbor}, {"my_node_info", STR_PARAM|USE_FUNC_PARAM, @@ -389,13 +513,32 @@ static int mod_init(void) flags_col.len = strlen(flags_col.s); description_col.len = strlen(description_col.s); - /* only allow the DB URL to be skipped in "P2P discovery" mode */ +#ifdef CLUSTERER_CTRL_SUPPORT + /* db_mode defaults to 1 (DB-backed). If no DB URL is configured there are + * no DB-backed native clusters, so drop to non-DB mode - this keeps + * controller-only and purely static (my_node_info) setups working without + * an explicit db_mode=0. A DB-native or hybrid setup provides a db_url and + * keeps db_mode. */ + if (!clusterer_db_url.s && !db_default_url) + db_mode = 0; +#endif + init_db_url(clusterer_db_url, db_mode == 0); +#ifdef CLUSTERER_CTRL_SUPPORT + /* my_node_id identifies this node within its *native* clusters (DB or + * static my_node_info). Controller clusters get their id assigned at + * runtime, so my_node_id is only required when native clusters exist. */ + if (current_id < 1 && (db_mode != 0 || (cluster_list && *cluster_list != NULL))) { + LM_CRIT("Invalid 'my_node_id' parameter (required for DB or static native clusters)\n"); + return -1; + } +#else if (current_id < 1) { LM_CRIT("Invalid 'my_node_id' parameter\n"); return -1; } +#endif if (ping_interval <= 0) { LM_WARN("Invalid ping_interval parameter, using default value\n"); ping_interval = DEFAULT_PING_INTERVAL; @@ -421,6 +564,23 @@ static int mod_init(void) LM_CRIT("Failed to init lock\n"); return -1; } +#ifdef CLUSTERER_CTRL_SUPPORT + /* Move GET_CURRENT_ID to shared memory so all forked processes + * see the same value when it is updated by the controller. */ + { + int *_shm_id = shm_malloc(sizeof(int)); + if (!_shm_id) { + LM_CRIT("No shm memory for GET_CURRENT_ID\n"); + return -1; + } + /* Seed with the static my_node_id so native (DB/static) clusters see + * the right identity via GET_CURRENT_ID at load time; -1 (a sentinel + * that matches no node) is only used when there is no static id, i.e. + * a controller-only node whose id is assigned at runtime. */ + *_shm_id = (current_id >= 1) ? current_id : -1; + _current_id_shm = _shm_id; + } +#endif /* if statistics are disabled, prevent their registration to core */ if (clusterer_enable_stats==0) @@ -443,16 +603,21 @@ static int mod_init(void) goto error; } - if (cl->current_node->node_id != current_id) { + if (cl->current_node && cl->current_node->node_id != current_id) { LM_ERR("Bad 'my_node_id' parameter, value: %d different than" - " the node_id property in the 'my_node_info' parameter\n", current_id); + " the node_id property in the 'my_node_info' parameter\n", GET_CURRENT_ID); goto error; } } } + /* Native clusters backed by the DB (db_mode != 0): bind and load them. In a + * hybrid this runs even when the controller is also active - DB-native and + * controller clusters are no longer mutually exclusive. Controller clusters + * are never stored in the DB, so load_db_info only ever brings in native + * ones. Static my_node_info/neighbor provisioning (db_mode==0) has already + * populated cluster_list at config-parse time. */ if (db_mode) { - /* bind to the mysql module */ if (db_bind_mod(&clusterer_db_url, &dr_dbf)) { LM_CRIT("Cannot bind to database module! " "Did you forget to load a database module ?\n"); @@ -462,7 +627,6 @@ static int mod_init(void) LM_CRIT("Given SQL DB does not provide query types needed by this module!\n"); goto error; } - /* init DB connection */ if ((db_hdl = dr_dbf.init(&clusterer_db_url)) == 0) { LM_ERR("cannot initialize database connection\n"); goto error; @@ -471,11 +635,59 @@ static int mod_init(void) LM_ERR("Failed to load info from DB\n"); goto error; } - dr_dbf.close(db_hdl); db_hdl = NULL; } + /* Controller-managed clusters: pre-create a stub for each cluster_id declared + * controller-managed (cluster_options use_controller=1), so cl_register_cap() + * succeeds for tm/dialog before clusterer_controller injects the real + * identity. Each stub is flagged controller_managed: it never touches the DB + * and behaves as db_mode=0, so it coexists with DB-native clusters. A + * cluster_id must be exclusively controller-managed OR native, never both - + * the native clusters (DB loaded just above, or static from parse time) are + * already in cluster_list, so an overlap is caught here. */ +#ifdef CLUSTERER_CTRL_SUPPORT + if (use_controller) { + int _ci; + for (_ci = 0; _ci < cl_ctr_stub_count; _ci++) { + cluster_info_t *_ex; + for (_ex = *cluster_list; _ex; _ex = _ex->next) + if (_ex->cluster_id == cl_ctr_stub_ids[_ci]) { + LM_ERR("clusterer: cluster_id %d is registered as " + "controller-managed (cluster_options use_controller=1) but is " + "already defined via native config " + "(my_node_info/neighbor_node_info/DB) or listed " + "twice - a cluster_id must be unique and either " + "controller-managed or native, not both\n", + cl_ctr_stub_ids[_ci]); + return -1; + } + + cluster_info_t *_cl = shm_malloc(sizeof *_cl); + if (!_cl) { + LM_ERR("clusterer: no shm for cluster %d stub\n", + cl_ctr_stub_ids[_ci]); + return -1; + } + memset(_cl, 0, sizeof *_cl); + _cl->cluster_id = cl_ctr_stub_ids[_ci]; + _cl->controller_managed = 1; + _cl->lock = lock_alloc(); + if (!_cl->lock || lock_init(_cl->lock) == NULL) { + LM_ERR("clusterer: lock_alloc failed for cluster %d\n", + cl_ctr_stub_ids[_ci]); + shm_free(_cl); + return -1; + } + _cl->next = *cluster_list; + *cluster_list = _cl; + LM_INFO("clusterer: pre-created controller-managed cluster %d stub\n", + cl_ctr_stub_ids[_ci]); + } + } +#endif /* CLUSTERER_CTRL_SUPPORT */ + /* register timer */ heartbeats_timer_interval = gcd(ping_interval*1000, ping_timeout); heartbeats_timer_interval = gcd(heartbeats_timer_interval, node_timeout*1000); @@ -536,7 +748,12 @@ static int mod_init(void) /* check if the cluster IDs in the the sharing tag list are valid */ shtag_init_list(); shtag_init_reporting(); +#ifdef CLUSTERER_CTRL_SUPPORT + if (!use_controller) + shtag_validate_list(); +#else shtag_validate_list(); +#endif return 0; error: @@ -559,6 +776,20 @@ static int child_init(int rank) return -1; } } + +#ifdef CLUSTERER_CTRL_SUPPORT + /* One-shot config sanity check (rank 1 fires once, and runs after every + * module's mod_init, so a clusterer_controller would already have bound the + * controller API): a controller-managed cluster is meaningless without that + * module - the pre-created stubs would never obtain an identity or form. */ + if (rank == 1 && use_controller && !clusterer_ctrl_bound) + LM_ERR("clusterer: cluster_options declares controller-managed cluster(s) " + "but no clusterer_controller module registered the controller API - " + "they will never obtain a node identity or form. Load the " + "clusterer_controller module, or drop use_controller from " + "cluster_options.\n"); +#endif /* CLUSTERER_CTRL_SUPPORT */ + return 0; } @@ -893,7 +1124,7 @@ static mi_response_t *clusterer_list_topology(const mi_params_t *params, if (!node_item) goto error; - if (add_mi_number(node_item, MI_SSTR("node_id"), current_id) < 0) + if (add_mi_number(node_item, MI_SSTR("node_id"), GET_CURRENT_ID) < 0) goto error; neigh_arr = add_mi_array(node_item, MI_SSTR("Neighbours")); @@ -925,7 +1156,7 @@ static mi_response_t *clusterer_list_topology(const mi_params_t *params, } if (n_info->link_state == LS_UP) - if (add_mi_number(neigh_arr, 0,0, current_id) < 0) { + if (add_mi_number(neigh_arr, 0,0, GET_CURRENT_ID) < 0) { lock_release(n_info->lock); goto error; } @@ -1054,7 +1285,7 @@ static mi_response_t *cluster_send_mi(const mi_params_t *params, return init_mi_param_error(); if (node_id < 1) return init_mi_error(400, MI_SSTR("Bad value for 'destination'")); - if (node_id == current_id) + if (node_id == GET_CURRENT_ID) return init_mi_error(400, MI_SSTR("Local node specified as destination")); if (get_mi_string_param(params, "cmd_name", &cmd_name.s, &cmd_name.len) < 0) @@ -1250,7 +1481,7 @@ static inline void generate_msg_tag(pv_value_t *tag_val, int cluster_id) memcpy(tag_val->rs.s, tmp, len); tag_val->rs.s[len] = '-'; tag_val->rs.len = len + 1; - tmp = int2str(current_id, &len); + tmp = int2str(GET_CURRENT_ID, &len); memcpy(tag_val->rs.s + tag_val->rs.len, tmp, len); tag_val->rs.s[tag_val->rs.len + len] = '-'; tag_val->rs.len += len + 1; diff --git a/modules/clusterer/node_info.c b/modules/clusterer/node_info.c index fd09726dd67..809c7299a5e 100644 --- a/modules/clusterer/node_info.c +++ b/modules/clusterer/node_info.c @@ -77,8 +77,16 @@ cluster_info_t **cluster_list; int load_cluster_bridges(cluster_info_t *cl_list); int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_vals, +#ifdef CLUSTERER_CTRL_SUPPORT + str *str_vals, int self_id) +{ +#else str *str_vals) { +/* Without the controller a node has a single global identity; the caller passes + * no per-cluster id, so self_id is just current_id here. */ +#define self_id current_id +#endif char *host; int hlen, port; int proto; @@ -140,7 +148,7 @@ int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_val else (*new_info)->flags &= ~NODE_STATE_ENABLED; - if (int_vals[INT_VALS_NODE_ID_COL] != current_id) + if (int_vals[INT_VALS_NODE_ID_COL] != self_id) (*new_info)->link_state = LS_RESTART_PINGING; else (*new_info)->link_state = LS_UP; @@ -213,7 +221,7 @@ int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_val (*new_info)->proto = proto; - if (int_vals[INT_VALS_NODE_ID_COL] != current_id) { + if (int_vals[INT_VALS_NODE_ID_COL] != self_id) { he = sip_resolvehost(&st, (unsigned short *) &port, (unsigned short *)&proto, 0, 0); if (!he) { @@ -259,7 +267,7 @@ int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_val } (*new_info)->sp_info->node = *new_info; - if (int_vals[INT_VALS_NODE_ID_COL] != current_id) { + if (int_vals[INT_VALS_NODE_ID_COL] != self_id) { (*new_info)->next = cluster->node_list; cluster->node_list = *new_info; cluster->no_nodes++; @@ -303,6 +311,9 @@ int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_val } return -1; } +#ifndef CLUSTERER_CTRL_SUPPORT +#undef self_id +#endif #define check_val( _col, _val, _type, _not_null, _is_empty_check) \ do { \ @@ -382,7 +393,7 @@ int load_db_info(db_func_t *dr_dbf, db_con_t* db_hdl, cluster_info_t **cl_list) LM_DBG("DB query - retrieve the list of clusters" " in which the local node runs\n"); - VAL_INT(&clusterer_node_id_value) = current_id; + VAL_INT(&clusterer_node_id_value) = GET_CURRENT_ID; /* first we see in which clusters the local node runs*/ if (dr_dbf->query(db_hdl, &clusterer_node_id_key, &op_eq, @@ -499,12 +510,16 @@ int load_db_info(db_func_t *dr_dbf, db_con_t* db_hdl, cluster_info_t **cl_list) strlen(str_vals[STR_VALS_DESCRIPTION_COL].s) : 0; /* add info to backing list */ - if ((rc = add_node_info(&_, cl_list, int_vals, str_vals)) != 0) { + if ((rc = add_node_info(&_, cl_list, int_vals, str_vals +#ifdef CLUSTERER_CTRL_SUPPORT + , GET_CURRENT_ID +#endif + )) != 0) { LM_ERR("Unable to add node info to backing list\n"); if (rc < 0) { /* serious error happened, better give up */ goto error; - } else if (int_vals[INT_VALS_NODE_ID_COL] == current_id) { + } else if (int_vals[INT_VALS_NODE_ID_COL] == GET_CURRENT_ID) { LM_ERR("Invalid info for local node\n"); /* the node info is bogus, but cannot be skipped * as it is the current node) */ @@ -770,7 +785,11 @@ int provision_neighbor(modparam_t type, void *val) *cluster_list = NULL; } - if (add_node_info(&new_info, cluster_list, int_vals, str_vals) < 0) { + if (add_node_info(&new_info, cluster_list, int_vals, str_vals +#ifdef CLUSTERER_CTRL_SUPPORT + , GET_CURRENT_ID +#endif + ) < 0) { LM_ERR("Unable to add node info to backing list\n"); return -1; } @@ -803,13 +822,13 @@ int provision_current(modparam_t type, void *val) return -1; } - if (int_vals[INT_VALS_NODE_ID_COL] == -1 && current_id == -1) { + if (int_vals[INT_VALS_NODE_ID_COL] == -1 && GET_CURRENT_ID == -1) { LM_ERR("Node ID not defined. Set either the value of the 'node_id' proprety" " of 'my_node_info' or set 'my_node_id' parameter before 'my_node_info'!\n"); return -1; } - if (current_id != -1 && int_vals[INT_VALS_NODE_ID_COL] != -1 && - int_vals[INT_VALS_NODE_ID_COL] != current_id) { + if (GET_CURRENT_ID != -1 && int_vals[INT_VALS_NODE_ID_COL] != -1 && + int_vals[INT_VALS_NODE_ID_COL] != GET_CURRENT_ID) { LM_ERR("Bad value in 'my_node_info' parameter, node_id: %d different" " than 'my_node_id' parameter\n", int_vals[INT_VALS_NODE_ID_COL]); return -1; @@ -817,7 +836,7 @@ int provision_current(modparam_t type, void *val) if (int_vals[INT_VALS_NODE_ID_COL] != -1) current_id = int_vals[INT_VALS_NODE_ID_COL]; else - int_vals[INT_VALS_NODE_ID_COL] = current_id; + int_vals[INT_VALS_NODE_ID_COL] = GET_CURRENT_ID; int_vals[INT_VALS_STATE_COL] = 1; if (int_vals[INT_VALS_NO_PING_RETRIES_COL] == -1) @@ -837,7 +856,11 @@ int provision_current(modparam_t type, void *val) *cluster_list = NULL; } - if (add_node_info(&new_info, cluster_list, int_vals, str_vals) != 0) { + if (add_node_info(&new_info, cluster_list, int_vals, str_vals +#ifdef CLUSTERER_CTRL_SUPPORT + , GET_CURRENT_ID +#endif + ) != 0) { LM_ERR("Unable to add node info to backing list\n"); return -1; } @@ -864,10 +887,10 @@ int update_db_state(int cluster_id, int node_id, int state) { VAL_NULL(&update_val) = 0; VAL_INT(&update_val) = state; - if (node_id == current_id) { + if (node_id == GET_CURRENT_ID) { VAL_TYPE(&node_id_val) = DB_INT; VAL_NULL(&node_id_val) = 0; - VAL_INT(&node_id_val) = current_id; + VAL_INT(&node_id_val) = GET_CURRENT_ID; if (dr_dbf.update(db_hdl, &node_id_key, 0, &node_id_val, &update_key, &update_val, 1, 1) < 0) @@ -1121,7 +1144,7 @@ void api_free_next_hop(clusterer_node_t *next_hop) int cl_get_my_id(void) { - return current_id; + return GET_CURRENT_ID; } int cl_get_my_sip_addr(int cluster_id, str *out_addr) @@ -1149,7 +1172,11 @@ int cl_get_my_sip_addr(int cluster_id, str *out_addr) memset(out_addr, 0, sizeof *out_addr); rc = 0; } else { +#ifdef CLUSTERER_CTRL_SUPPORT + if (cl->current_node && pkg_str_dup(out_addr, &cl->current_node->sip_addr) != 0) { +#else if (pkg_str_dup(out_addr, &cl->current_node->sip_addr) != 0) { +#endif LM_ERR("oom\n"); memset(out_addr, 0, sizeof *out_addr); rc = -1; @@ -1203,7 +1230,7 @@ int cl_get_my_index(int cluster_id, str *capability, int *nr_nodes) sorted[j+1] = tmp; } - for (i = 0; i < *nr_nodes && sorted[i] < current_id; i++) ; + for (i = 0; i < *nr_nodes && sorted[i] < cluster_self_id(cl); i++) ; (*nr_nodes)++; return i; diff --git a/modules/clusterer/node_info.h b/modules/clusterer/node_info.h index 649a7f7d829..953456bc790 100644 --- a/modules/clusterer/node_info.h +++ b/modules/clusterer/node_info.h @@ -137,6 +137,19 @@ struct cluster_info { cluster_bridge_t *bridges; /* replication links to other clusters */ +#ifdef CLUSTERER_CTRL_SUPPORT + /* Set by clusterer_controller when manage_shtags=1 for this cluster. + * Blocks MI and script-variable shtag activation to prevent conflicts + * with controller-managed failover. */ + int shtag_managed; + + /* 1 = this cluster's topology and identity are driven at runtime by + * clusterer_controller (registered via cluster_options use_controller=1); it + * never touches the DB and behaves as db_mode=0 regardless of global db_mode. + * 0 = a native cluster defined via DB or static my_node_info/neighbor. */ + int controller_managed; +#endif + struct cluster_info *next; }; @@ -167,16 +180,62 @@ extern str clnk_shtag_col; extern str clnk_dst_node_col; extern int current_id; + +/* Controller-support identity accessors. When the clusterer_controller module + * is NOT part of the build these all collapse to the stock global 'current_id', + * so every call site compiles to the exact upstream behaviour with no #ifdef of + * its own. With the controller, a node can hold a different node_id per cluster + * and its identity is (re)assigned at runtime, so the per-cluster shm value and a + * shm-backed global are authoritative instead. */ +#ifdef CLUSTERER_CTRL_SUPPORT +extern int *_current_id_shm; +/* Read current_id from shm if available (cross-process after fork) */ +#define GET_CURRENT_ID (_current_id_shm ? *_current_id_shm : current_id) + +/* This node's node_id *within a specific cluster*. Returns -1 (a node_id that + * matches nothing) when this cluster's identity is not yet established, so a + * not-yet-joined cluster can never accidentally match or stamp a real id. */ +static inline int cluster_self_id(const struct cluster_info *cl) +{ + return (cl && cl->current_node) ? cl->current_node->node_id : -1; +} +extern int use_controller; +/* cluster_ids declared controller-managed via cluster_options (use_controller=1) */ +extern int cl_ctr_stub_ids[]; +extern int cl_ctr_stub_count; +#else +#define GET_CURRENT_ID (current_id) +#define cluster_self_id(cl) (current_id) +#define use_controller 0 +#endif + extern int db_mode; extern rw_lock_t *cl_list_lock; extern cluster_info_t **cluster_list; +/* Effective db_mode *for one cluster*. Controller-managed clusters never use + * the DB (their topology is injected at runtime), so they always behave as + * db_mode=0 even in a hybrid where native clusters are DB-backed (db_mode!=0). + * Without the controller every cluster is native, so this is just db_mode. */ +#ifdef CLUSTERER_CTRL_SUPPORT +static inline int cl_db_mode(const struct cluster_info *cl) +{ + return (cl && cl->controller_managed) ? 0 : db_mode; +} +#else +#define cl_db_mode(cl) (db_mode) +#endif + int update_db_state(int cluster_id, int node_id, int state); int load_db_info(db_func_t *dr_dbf, db_con_t* db_hdl, cluster_info_t **cl_list); void free_info(cluster_info_t *cl_list); int add_node_info(node_info_t **new_info, cluster_info_t **cl_list, int *int_vals, +#ifdef CLUSTERER_CTRL_SUPPORT + str *str_vals, int self_id); +#else str *str_vals); +#endif void remove_node_list(cluster_info_t *cl, node_info_t *node); int provision_neighbor(modparam_t type, void* val); @@ -196,6 +255,9 @@ static inline cluster_info_t *get_cluster_by_id(int cluster_id) { cluster_info_t *cl; +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cluster_list || !*cluster_list) return NULL; +#endif for (cl = *cluster_list; cl; cl = cl->next) if (cl->cluster_id == cluster_id) return cl; diff --git a/modules/clusterer/sharing_tags.c b/modules/clusterer/sharing_tags.c index d486d813a62..21d507c6e3a 100644 --- a/modules/clusterer/sharing_tags.c +++ b/modules/clusterer/sharing_tags.c @@ -387,12 +387,38 @@ int shtag_modparam_func(modparam_t type, void *val_s) tag_name.len, tag_name.s); return -1; } +#ifdef CLUSTERER_CTRL_SUPPORT + /* Force the given state. Only a *controller-managed* cluster's active tag + * is forced to backup here (its controller master activates it when + * appropriate); a native cluster keeps its configured =active state - the + * old global use_controller check wrongly demoted native tags too in a + * hybrid. A controller cluster's stub is created later, in mod_init, so at + * this parse-time point it is normally not yet in cluster_list; the + * controller then forces its tags to backup at runtime via + * set_shtag_managed()/shtag_force_all_backup() (which also clears any pending + * active broadcast set below). */ + { + cluster_info_t *_tcl = get_cluster_by_id(c_id); + if (_tcl && _tcl->controller_managed && init_state == SHTAG_STATE_ACTIVE) { + tag->state = SHTAG_STATE_BACKUP; + LM_INFO("clusterer: [cluster %d] sharing tag [%.*s] " + "forced to backup (controller-managed)\n", + tag->cluster_id, tag->name.len, tag->name.s); + } else { + tag->state = init_state; + if (init_state == SHTAG_STATE_ACTIVE) + /* broadcast (later) in cluster that this tag is active */ + tag->send_active_msg = 1; + } + } +#else /* force the given state */ tag->state = init_state; if (init_state == SHTAG_STATE_ACTIVE) /* broadcast (later) in cluster that this tag is active */ tag->send_active_msg = 1; +#endif return 0; } @@ -876,6 +902,73 @@ int handle_shtag_active(bin_packet_t *packet, int cluster_id, int source_id) } +#ifdef CLUSTERER_CTRL_SUPPORT +/** + * shtag_force_all_backup() - force every sharing tag for the given + * cluster to BACKUP state, ignoring the =active config value. + * Called by clusterer_controller at startup when manage_shtags=1 so + * that tag activation is decided solely by the controller master. + * Runs pre-cluster-join: no BIN broadcast needed. + */ +int shtag_force_all_backup(int cluster_id) +{ + struct sharing_tag *tag; + int lock_old_flag; + + if (!shtags_list || !*shtags_list) + return 0; + + lock_start_sw_read(shtags_lock); + for (tag = *shtags_list; tag; tag = tag->next) { + if (tag->cluster_id != cluster_id || + tag->state != SHTAG_STATE_ACTIVE) + continue; + lock_switch_write(shtags_lock, lock_old_flag); + tag->state = SHTAG_STATE_BACKUP; + tag->send_active_msg = 0; /* suppress pending active broadcast */ + lock_switch_read(shtags_lock, lock_old_flag); + LM_INFO("clusterer: [cluster %d] sharing tag [%.*s] forced to " + "backup (controller-managed)\n", + cluster_id, tag->name.len, tag->name.s); + } + lock_stop_sw_read(shtags_lock); + return 0; +} + +/** + * shtag_activate_all_backup() - activate every BACKUP sharing tag for + * the given cluster. Called by clusterer_controller master on node departure. + */ +int shtag_activate_all_backup(int cluster_id) +{ + struct sharing_tag *tag; + /* collect names under lock to avoid O(n²) restart-from-head loop */ +#define SHTAG_MAX_ACTIVATE 64 + str to_activate[SHTAG_MAX_ACTIVATE]; + int n = 0, i; + + if (!shtags_list || !*shtags_list) + return 0; + + lock_start_read(shtags_lock); + for (tag = *shtags_list; tag && n < SHTAG_MAX_ACTIVATE; tag = tag->next) { + if (tag->cluster_id != cluster_id || + tag->state != SHTAG_STATE_BACKUP) + continue; + to_activate[n++] = tag->name; + } + lock_stop_read(shtags_lock); + + for (i = 0; i < n; i++) { + LM_INFO("clusterer: [cluster %d] promoting sharing tag " + "[%.*s] from backup to active (controller master)\n", + cluster_id, to_activate[i].len, to_activate[i].s); + shtag_activate(&to_activate[i], cluster_id, MI_SSTR("controller master")); + } + return 0; +} +#endif /* CLUSTERER_CTRL_SUPPORT */ + void shtag_event_handler(int cluster_id, enum clusterer_event ev, int node_id) { if (ev == CLUSTER_NODE_UP) @@ -960,10 +1053,27 @@ mi_response_t *shtag_mi_set_active(const mi_params_t *params, tag.len, tag.s, c_id); lock_start_read(cl_list_lock); +#ifdef CLUSTERER_CTRL_SUPPORT + { + cluster_info_t *_cl = get_cluster_by_id(c_id); + if (!_cl) { + lock_stop_read(cl_list_lock); + return init_mi_error(404, MI_SSTR("Cluster ID not found")); + } + if (_cl->shtag_managed) { + lock_stop_read(cl_list_lock); + LM_WARN("clusterer: MI shtag activation blocked for cluster %d " + "— sharing tags are controller-managed\n", c_id); + return init_mi_error(403, MI_SSTR("Sharing tag is " + "controller-managed; manual activation not allowed")); + } + } +#else if (!get_cluster_by_id(c_id)) { lock_stop_read(cl_list_lock); return init_mi_error(404, MI_SSTR("Cluster ID not found")); } +#endif lock_stop_read(cl_list_lock); if (shtag_activate( &tag, c_id, MI_SSTR("MI command"))<0) { @@ -1057,7 +1167,22 @@ int var_set_sh_tag(struct sip_msg* msg, pv_param_t *param, int op, return 0; } - if (shtag_activate( &v_name->shtag, v_name->cluster_id, +#ifdef CLUSTERER_CTRL_SUPPORT + lock_start_read(cl_list_lock); + { + cluster_info_t *_cl = get_cluster_by_id(v_name->cluster_id); + if (_cl && _cl->shtag_managed) { + lock_stop_read(cl_list_lock); + LM_WARN("clusterer: script shtag activation blocked for " + "tag <%.*s/%d> — sharing tags are controller-managed\n", + v_name->shtag.len, v_name->shtag.s, v_name->cluster_id); + return -1; + } + } + lock_stop_read(cl_list_lock); +#endif + + if (shtag_activate( &v_name->shtag, v_name->cluster_id, MI_SSTR("script variable"))==-1) { LM_ERR("failed to set sharing tag <%.*s/%d> to new state %d\n", v_name->shtag.len, v_name->shtag.s, v_name->cluster_id, state); diff --git a/modules/clusterer/sharing_tags.h b/modules/clusterer/sharing_tags.h index ae7a26e411c..0964ec73440 100644 --- a/modules/clusterer/sharing_tags.h +++ b/modules/clusterer/sharing_tags.h @@ -43,6 +43,10 @@ int send_shtag_active_info(int c_id, str *tag_name, int node_id); void shtag_flush_state(int c_id, int node_id); void shtag_event_handler(int cluster_id, enum clusterer_event ev, int node_id); +#ifdef CLUSTERER_CTRL_SUPPORT +int shtag_activate_all_backup(int cluster_id); +int shtag_force_all_backup(int cluster_id); +#endif mi_response_t *shtag_mi_list(const mi_params_t *params, struct mi_handler *async_hdl); diff --git a/modules/clusterer/sync.c b/modules/clusterer/sync.c index 187e825c0cf..89248ce73cf 100644 --- a/modules/clusterer/sync.c +++ b/modules/clusterer/sync.c @@ -70,8 +70,13 @@ static int get_sync_source(cluster_info_t *cluster, str *capability, if (get_next_hop(node) == 0) continue; +#ifdef CLUSTERER_CTRL_SUPPORT + if (!cluster->current_node || !match_node(cluster->current_node, node, match_cond)) + continue; +#else if (!match_node(cluster->current_node, node, match_cond)) continue; +#endif lock_get(node->lock); for (cap = node->capabilities; cap; cap = cap->next) @@ -94,14 +99,47 @@ static int get_sync_source(cluster_info_t *cluster, str *capability, int queue_sync_request(cluster_info_t *cluster, struct local_cap *lcap) { lock_get(cluster->lock); + +#ifdef CLUSTERER_CTRL_SUPPORT + /* If we are a seed node with no peers yet, skip the pending queue and + * self-mark as synced immediately. There is nobody to sync from, and + * the seed-fallback timer would just fire after seed_fb_interval and + * log a spurious ERROR. When peers join later the normal event-driven + * sync path (CLUSTER_NODE_UP callback) will re-trigger if needed. */ + if (cluster->current_node && + (cluster->current_node->flags & NODE_IS_SEED) && + cluster->node_list == NULL) { + lcap->flags |= CAP_STATE_OK; + lcap->flags &= ~(CAP_SYNC_PENDING | CAP_SYNC_STARTUP); + lock_release(cluster->lock); + LM_DBG("No peers in cluster %d — capability '%.*s' self-marked as " + "synced (first/lone seed node)\n", + cluster->cluster_id, lcap->reg.name.len, lcap->reg.name.s); + sr_set_status(cl_srg, STR2CI(lcap->reg.sr_id), CAP_SR_SYNCED, + STR2CI(CAP_SR_STATUS_STR(CAP_SR_SYNCED)), 0); + sr_add_report_fmt(cl_srg, STR2CI(lcap->reg.sr_id), 0, + "No peers present — self-synced as first node in cluster"); + send_single_cap_update(cluster, lcap, 1); + return 0; + } +#endif + lcap->flags |= CAP_SYNC_PENDING; if (sr_get_core_status() == STATE_INITIALIZING) lcap->flags |= CAP_SYNC_STARTUP; else lcap->flags &= ~CAP_SYNC_STARTUP; +#ifdef CLUSTERER_CTRL_SUPPORT + /* Always record when we started waiting — if current_node is not yet set + * (controller mode, identity assigned post-fork), sync_req_time would + * stay at zero (epoch) and TIME_DIFF would be huge, causing the seed + * fallback timer to fire immediately once identity is assigned. */ + gettimeofday(&lcap->sync_req_time, NULL); +#else if (cluster->current_node->flags & NODE_IS_SEED) gettimeofday(&lcap->sync_req_time, NULL); +#endif lock_release(cluster->lock); diff --git a/modules/clusterer/topology.c b/modules/clusterer/topology.c index 07603958637..b6baa9a413e 100644 --- a/modules/clusterer/topology.c +++ b/modules/clusterer/topology.c @@ -50,7 +50,7 @@ static int send_ping(node_info_t *node, int req_node_list) return -1; } bin_push_int(&packet, node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(node->cluster)); bin_push_int(&packet, req_node_list); /* request list of known nodes ? */ bin_get_buffer(&packet, &send_buffer); @@ -195,6 +195,10 @@ void heartbeats_timer(void) lock_start_read(cl_list_lock); for (clusters_it = *cluster_list; clusters_it; clusters_it = clusters_it->next) { +#ifdef CLUSTERER_CTRL_SUPPORT + if (!clusters_it->current_node) + continue; /* identity not yet set by controller */ +#endif lock_get(clusters_it->current_node->lock); if (!(clusters_it->current_node->flags & NODE_STATE_ENABLED)) { lock_release(clusters_it->current_node->lock); @@ -500,7 +504,7 @@ int flood_message(bin_packet_t *packet, cluster_info_t *cluster, bin_push_int(packet, path_len + 1); /* go to end of the buffer and include current node in path */ bin_skip_int_packet_end(packet, path_len); - bin_push_int(packet, current_id); + bin_push_int(packet, cluster_self_id(cluster)); bin_get_buffer(packet, &bin_buffer); msg_altered = 1; } @@ -562,7 +566,7 @@ static int send_full_top_update(node_info_t *dest_node, int nr_nodes, int *node_ return -1; } bin_push_int(&packet, dest_node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); bin_push_int(&packet, ++dest_node->cluster->current_node->top_seq_no); bin_push_int(&packet, timestamp); @@ -574,7 +578,7 @@ static int send_full_top_update(node_info_t *dest_node, int nr_nodes, int *node_ bin_push_int(&packet, dest_node->cluster->no_nodes); /* the first adjacency list in the message is for the current node */ - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); bin_push_int(&packet, 0); /* no description for current node */ bin_push_int(&packet, dest_node->cluster->current_node->ls_seq_no); bin_push_int(&packet, dest_node->cluster->current_node->ls_timestamp); @@ -622,7 +626,7 @@ static int send_full_top_update(node_info_t *dest_node, int nr_nodes, int *node_ } bin_push_int(&packet, 1); /* path length is 1, only current node at this point */ - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(dest_node->cluster)); bin_get_buffer(&packet, &bin_buffer); if (msg_send(dest_node->cluster->send_sock, dest_node->proto, &dest_node->addr, @@ -669,7 +673,7 @@ static int send_ls_update(node_info_t *node, clusterer_link_state new_ls) return -1; } bin_push_int(&packet, node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(node->cluster)); bin_push_int(&packet, ++node->cluster->current_node->ls_seq_no); bin_push_int(&packet, timestamp); @@ -680,7 +684,7 @@ static int send_ls_update(node_info_t *node, clusterer_link_state new_ls) /* path length is 1, only current node at this point */ bin_push_int(&packet, 1); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(node->cluster)); lock_release(node->cluster->current_node->lock); @@ -714,7 +718,11 @@ static node_info_t *add_node(bin_packet_t *received, cluster_info_t *cl, int_vals[INT_VALS_NODE_ID_COL] = src_node_id; int_vals[INT_VALS_STATE_COL] = 1; /* enabled */ - if (add_node_info(&new_node, &cl, int_vals, str_vals) != 0) { + if (add_node_info(&new_node, &cl, int_vals, str_vals +#ifdef CLUSTERER_CTRL_SUPPORT + , cluster_self_id(cl) +#endif + ) != 0) { LM_ERR("Unable to add node info to backing list\n"); return NULL; } @@ -1044,13 +1052,13 @@ void handle_full_top_update(bin_packet_t *packet, node_info_t *source, for (i = 0; i < no_nodes; i++) { skip = 0; - if (top_node_id[i] == current_id) + if (top_node_id[i] == cluster_self_id(source->cluster)) skip = 1; top_node = get_node_by_id(source->cluster, top_node_id[i]); if (!skip && !top_node) { - if (db_mode) { + if (cl_db_mode(source->cluster)) { skip = 1; } else if (!top_node_info[i][0]) { LM_WARN("Unknown node id [%d] in topology update with " @@ -1092,8 +1100,8 @@ void handle_full_top_update(bin_packet_t *packet, node_info_t *source, no_present_nodes = 0; for (j = 0; j < top_node_info[i][3]; j++) { top_neigh = get_node_by_id(source->cluster, top_node_info[i][j+4]); - if (!top_neigh && top_node_info[i][j+4] != current_id) { - if (db_mode) + if (!top_neigh && top_node_info[i][j+4] != cluster_self_id(source->cluster)) { + if (cl_db_mode(source->cluster)) continue; for (n_idx = 0; n_idx < no_nodes && top_node_info[i][j+4] != top_node_id[n_idx]; @@ -1117,7 +1125,7 @@ void handle_full_top_update(bin_packet_t *packet, node_info_t *source, } } - if (top_node_info[i][j+4] == current_id) { + if (top_node_info[i][j+4] == cluster_self_id(source->cluster)) { lock_get(top_node->lock); if (top_node->link_state == LS_DOWN && top_node->flags & NODE_STATE_ENABLED) { @@ -1180,7 +1188,7 @@ void handle_internal_msg_unknown(bin_packet_t *received, cluster_info_t *cl, return; } bin_push_int(&packet, cl->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(cl)); bin_get_buffer(&packet, &bin_buffer); if (msg_send(cl->send_sock, proto, src_su, 0, bin_buffer.s, @@ -1234,8 +1242,8 @@ void handle_ls_update(bin_packet_t *received, node_info_t *src_node, bin_pop_int(received, &neigh_id); bin_pop_int(received, &new_ls); ls_neigh = get_node_by_id(src_node->cluster, neigh_id); - if (!ls_neigh && neigh_id != current_id) { - if (!db_mode) + if (!ls_neigh && neigh_id != cluster_self_id(src_node->cluster)) { + if (!cl_db_mode(src_node->cluster)) LM_WARN("Received link state update about unknown node id [%d]\n", neigh_id); lock_release(src_node->lock); return; @@ -1244,7 +1252,7 @@ void handle_ls_update(bin_packet_t *received, node_info_t *src_node, LM_DBG("Received link state update with source [%d] about node [%d], new state=%s\n", src_node->node_id, neigh_id, new_ls ? "DOWN" : "UP"); - if (neigh_id == current_id) { + if (neigh_id == cluster_self_id(src_node->cluster)) { if ((new_ls == LS_UP && src_node->link_state == LS_DOWN) || (new_ls == LS_DOWN && src_node->link_state == LS_UP)) { lock_release(src_node->lock); @@ -1276,14 +1284,14 @@ void handle_unknown_id(node_info_t *src_node) return; } bin_push_int(&packet, src_node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(src_node->cluster)); /* include info about current node */ bin_push_node_info(&packet, src_node->cluster->current_node); /* path length is 1, only current node at this point */ bin_push_int(&packet, 1); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(src_node->cluster)); bin_get_buffer(&packet, &bin_buffer); if (msg_send(src_node->cluster->send_sock, src_node->proto, &src_node->addr, @@ -1316,7 +1324,7 @@ void handle_ping(bin_packet_t *received, node_info_t *src_node, return; } bin_push_int(&packet, src_node->cluster->cluster_id); - bin_push_int(&packet, current_id); + bin_push_int(&packet, cluster_self_id(src_node->cluster)); if (req_list) { /* include a list of known nodes */ From cb19612bfc35c1806a1dc95f82d868156217048e Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Sun, 9 Aug 2026 21:53:23 +1000 Subject: [PATCH 02/32] tm: render the anycast Via cid from the live node id, not a frozen one tm_init_cluster() read this node cluster id once, at mod_init, and baked it into the ";cid=" Via parameter (and into a cached tm_node_id used to decide whether an anycast reply is ours). That assumes the id is known and stable by mod_init, which holds for a statically configured clusterer but not for a controller-managed one: there the id is assigned at runtime, after the node joins, and can change on re-election. So mod_init froze the still-unassigned id (-1, printed as its unsigned form 18446744073709551615) into every outgoing Via, and every node compared incoming cids against -1 - t_anycast_replicate() could never route a reply to the owning node. Read the id live instead: pre-build only the fixed ";=" prefix at init, and have tm_via_cid() append the current get_my_id() per request (returning no parameter while the node still has no id); compare incoming cids against the live get_my_id() too. cl_get_my_id() already returns the runtime id, so this needs nothing from the caller and self-corrects across re-elections. --- modules/tm/cluster.c | 45 +++++++++++++++++++++++++++++++++----------- modules/tm/cluster.h | 7 +++++-- 2 files changed, 39 insertions(+), 13 deletions(-) diff --git a/modules/tm/cluster.c b/modules/tm/cluster.c index 3bdff5517c0..34b0b83a37a 100644 --- a/modules/tm/cluster.c +++ b/modules/tm/cluster.c @@ -32,7 +32,9 @@ str tm_cid; int tm_repl_cluster = 0; int tm_repl_auto_cancel = 1; str tm_cluster_param = str_init(TM_CLUSTER_DEFAULT_PARAM); -static int tm_node_id = 0; +/* length of the fixed ";=" prefix of tm_cid; the node id is appended + * live per-request by tm_via_cid() */ +static int tm_cid_prefix_len; static str tm_repl_cap = str_init("tm-repl"); struct clusterer_binds cluster_api; @@ -256,8 +258,6 @@ static void receive_tm_repl(bin_packet_t *packet) int tm_init_cluster(void) { - str cid; - if (tm_repl_cluster == 0) { LM_DBG("tm_replication_cluster not set - not engaging!\n"); return 0; @@ -282,11 +282,13 @@ int tm_init_cluster(void) /* overwrite structure to disable clusterer */ goto cluster_error; } - tm_node_id = cluster_api.get_my_id(); - - /* build the via param */ - cid.s = int2str(tm_node_id, &cid.len); - tm_cid.s = pkg_malloc(1/*;*/ + tm_cluster_param.len + 1/*=*/ + cid.len); + /* Pre-build only the fixed ";=" prefix. The node id is NOT read + * here: under a controller-managed cluster it is assigned at runtime + * (after this mod_init) and may change on re-election, so freezing it + * now would stamp a stale (typically -1) id into every Via. tm_via_cid() + * appends the current id per-request instead. Room is left for the + * largest id an int2str() can render. */ + tm_cid.s = pkg_malloc(1/*;*/ + tm_cluster_param.len + 1/*=*/ + INT2STR_MAX_LEN); if (!tm_cid.s) { LM_ERR("out of pkg memory!\n"); goto cluster_error; @@ -296,8 +298,7 @@ int tm_init_cluster(void) memcpy(tm_cid.s + tm_cid.len, tm_cluster_param.s, tm_cluster_param.len); tm_cid.len += tm_cluster_param.len; tm_cid.s[tm_cid.len++] = '='; - memcpy(tm_cid.s + tm_cid.len, cid.s, cid.len); - tm_cid.len += cid.len; + tm_cid_prefix_len = tm_cid.len; return 0; @@ -306,6 +307,28 @@ int tm_init_cluster(void) return -1; } +str *tm_via_cid(void) +{ + char *id_s; + int id, id_len; + + if (!tm_cluster_enabled()) + return NULL; + + /* read this node's id live - it is only known once the node has been + * assigned one in the cluster (immediately for a static clusterer, + * after the runtime join for a controller-managed one) and can change + * on re-election; advertise no cid until we actually have one */ + id = cluster_api.get_my_id(); + if (id < 0) + return NULL; + + id_s = int2str((unsigned long)id, &id_len); + memcpy(tm_cid.s + tm_cid_prefix_len, id_s, id_len); + tm_cid.len = tm_cid_prefix_len + id_len; + return &tm_cid; +} + #define TM_BIN_PUSH(_t, _f, _d) \ do { \ if (bin_push_##_t(&packet, _f) < 0) { \ @@ -522,7 +545,7 @@ int tm_reply_replicate(struct sip_msg *msg) /* if there was no parameter, or it was, but it was ours, handle it */ if (cid < 0) return 0; - if (cid == tm_node_id) { + if (cid == cluster_api.get_my_id()) { LM_DBG("reply should be processed by us (%d)\n", cid); return 0; } diff --git a/modules/tm/cluster.h b/modules/tm/cluster.h index 39783fe8810..9fad793ea57 100644 --- a/modules/tm/cluster.h +++ b/modules/tm/cluster.h @@ -51,7 +51,10 @@ int tm_anycast_cancel(struct sip_msg *msg); /* returns true if clusterer is enabled */ #define tm_cluster_enabled() (cluster_api.register_capability != 0) -/* returns the via parameter for the cluster */ -#define tm_via_cid() (tm_cluster_enabled()?&tm_cid:0) +/* Returns the ";=" Via parameter for the cluster, rendered + * with this node's CURRENT clusterer id (read live, not frozen at init - the + * id is assigned at runtime under a controller-managed cluster and may change + * on re-election). NULL when clustering is off or no id is assigned yet. */ +str *tm_via_cid(void); #endif /* _TM_CLUSTER_H_ */ From 1574da109ccc84ee560965f8e79475248b50585b Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Sun, 9 Aug 2026 21:53:23 +1000 Subject: [PATCH 03/32] clusterer_controller: zero-config HA clusterer control module A control plane for the clusterer module: nodes discover each other over encrypted UDP multicast, elect a master, and are assigned their cluster identity at runtime, so an HA cluster needs no per-node id in the config and no database behind it. Native, controller-managed and hybrid topologies coexist - a cluster is controller-managed only if the clusterer declares it so via cluster_options. Crypto is libsodium-only, no fallback: XChaCha20-Poly1305 for the payload AEAD, Argon2id as the bootstrap KDF, and Noise_NNpsk0 for the join handshake, with X25519 / HKDF-SHA256 / RNG underneath. libsodium is linked dynamically (distro static libs are usually non-PIC), so target hosts need the runtime package. The module is added to exclude_modules in Makefile.conf.template, like the other modules with external-lib dependencies - enable it via include_modules - and libsodium-dev joins the CI apt requirements. Membership and liveness: - master election, with the invariant "MASTER_ALIVE keepalive armed <=> I am the elected master", which is what stops a node demoted purely by election from continuing to broadcast and flapping between two masters; - master-mediated ALIVE, so liveness costs O(N) messages per round rather than every node pinging every other (O(N^2)); - a membership digest carried in MASTER_ALIVE, with a RESYNC repair path when a node's view diverges from the master's; - the join handshake is ACKed and retransmitted under a bounded budget, and both the KEY_GRANT and the join-time state snapshot are unicast to the joining node instead of broadcast to everyone. Several controller clusters can share one BIN socket - each has its own multicast group, shared-port unicast is routed by cluster_id, and multicast is used as the fallback when two clusters collide on a port. Hardening: the join path is flood-DoS rate-limited and its inputs are validated and bounds-checked (including the Noise msg2 decrypt output and the zero-length cipherstate passthrough). Decrypt failures are classified rather than logged uniformly - a bootstrap-key failure means a wrong password, a foreign cluster or tampering and warns, while a transient session-key mismatch during a rekey or a split-brain heal is expected and stays at DBG. The split-brain defer budget resets on a fresh higher-IP JOIN_REQ, so a lower-IP node waits for a live higher-IP peer instead of self-promoting, bounded by CL_CTR_JOIN_DEFER_HARDMAX. Multicast is sent and joined on this node's own interface rather than INADDR_ANY, which is what makes discovery work on a multi-homed host where the default route does not carry the cluster network. Validated end-to-end on a 3-node cluster: controller / native / hybrid modes, master failover and election stability (staggered and simultaneous starts both converge to a single stable master), multiple controller clusters on one BIN socket, the MI surface and its error paths, buffer overrun/underrun fuzzing, wrong-password rejection and config mismatch. Ships with the admin guide, the HA test appendix, the generated README and a join-rejection test. --- Makefile.conf.template | 2 +- modules/clusterer_controller/Makefile | 21 + modules/clusterer_controller/README | 1752 +++++ .../clusterer_controller.c | 6609 +++++++++++++++++ .../doc/clusterer_controller.xml | 22 + .../doc/clusterer_controller_admin.xml | 1931 +++++ .../doc/clusterer_controller_tests.xml | 265 + .../clusterer_controller/doc/contributors.xml | 25 + .../test/cl_ctr_join_reject_test.py | 135 + scripts/build/apt_requirements.txt | 1 + 10 files changed, 10762 insertions(+), 1 deletion(-) create mode 100644 modules/clusterer_controller/Makefile create mode 100644 modules/clusterer_controller/README create mode 100644 modules/clusterer_controller/clusterer_controller.c create mode 100644 modules/clusterer_controller/doc/clusterer_controller.xml create mode 100644 modules/clusterer_controller/doc/clusterer_controller_admin.xml create mode 100644 modules/clusterer_controller/doc/clusterer_controller_tests.xml create mode 100644 modules/clusterer_controller/doc/contributors.xml create mode 100755 modules/clusterer_controller/test/cl_ctr_join_reject_test.py diff --git a/Makefile.conf.template b/Makefile.conf.template index c6396cc5f49..971f35d277e 100644 --- a/Makefile.conf.template +++ b/Makefile.conf.template @@ -76,7 +76,7 @@ #uuid= UUID generator | uuid-dev # Modules omitted from the default build, generally due to external dependencies. -exclude_modules?= aaa_diameter aaa_radius auth_jwt auth_web3 b2b_logic_xml cachedb_cassandra cachedb_couchbase cachedb_dynamodb cachedb_memcached cachedb_mongodb cachedb_redis carrierroute cgrates compression cpl_c db_berkeley db_http db_mysql db_oracle db_perlvdb db_postgres db_sqlite db_unixodbc dialplan emergency event_rabbitmq event_kafka event_sqs h350 httpd http2d identity jabber json launch_darkly ldap lua mi_xmlrpc mmgeoip opentelemetry osp perl pi_http presence presence_dialoginfo presence_mwi presence_reginfo presence_xml presence_dfks proto_ipsec proto_sctp proto_tls proto_wss pua pua_bla pua_dialoginfo pua_mi pua_reginfo pua_usrloc pua_xmpp python regex rabbitmq_consumer rest_client rls rtp.io siprec sngtc snmpstats stir_shaken tls_mgm tls_openssl tls_wolfssl uuid xcap xcap_client xml xmpp +exclude_modules?= aaa_diameter aaa_radius auth_jwt auth_web3 b2b_logic_xml cachedb_cassandra cachedb_couchbase cachedb_dynamodb cachedb_memcached cachedb_mongodb cachedb_redis carrierroute cgrates clusterer_controller compression cpl_c db_berkeley db_http db_mysql db_oracle db_perlvdb db_postgres db_sqlite db_unixodbc dialplan emergency event_rabbitmq event_kafka event_sqs h350 httpd http2d identity jabber json launch_darkly ldap lua mi_xmlrpc mmgeoip opentelemetry osp perl pi_http presence presence_dialoginfo presence_mwi presence_reginfo presence_xml presence_dfks proto_ipsec proto_sctp proto_tls proto_wss pua pua_bla pua_dialoginfo pua_mi pua_reginfo pua_usrloc pua_xmpp python regex rabbitmq_consumer rest_client rls rtp.io siprec sngtc snmpstats stir_shaken tls_mgm tls_openssl tls_wolfssl uuid xcap xcap_client xml xmpp # Modules forced into the build, even when also listed in exclude_modules. include_modules?= diff --git a/modules/clusterer_controller/Makefile b/modules/clusterer_controller/Makefile new file mode 100644 index 00000000000..5a29721fabd --- /dev/null +++ b/modules/clusterer_controller/Makefile @@ -0,0 +1,21 @@ +# WARNING: do not run this directly, it should be run by the master Makefile + +include ../../Makefile.defs +auto_gen= +NAME= clusterer_controller.so + +# libsodium is a hard requirement: it provides the payload AEAD +# (XChaCha20-Poly1305), the bootstrap KDF (Argon2id), the Noise_NNpsk0 join +# handshake, and X25519 / HKDF-SHA256 / RNG. It is linked dynamically (distro +# static libs are usually non-PIC), so target hosts need the libsodium runtime +# package (e.g. `apt install libsodium23`). +SODIUM_EXISTS := $(shell pkg-config --exists libsodium 2>/dev/null && echo yes) +ifeq ($(SODIUM_EXISTS),yes) + DEFS += $(shell pkg-config --cflags libsodium) + LIBS += $(shell pkg-config --libs libsodium) + $(info clusterer_controller: libsodium found -> XChaCha20-Poly1305 + Argon2id + Noise_NNpsk0) +else + $(error clusterer_controller requires libsodium (install libsodium-dev / libsodium-devel)) +endif + +include ../../Makefile.modules diff --git a/modules/clusterer_controller/README b/modules/clusterer_controller/README new file mode 100644 index 00000000000..a0fe9651bdc --- /dev/null +++ b/modules/clusterer_controller/README @@ -0,0 +1,1752 @@ +CLUSTERER_CONTROLLER Module + __________________________________________________________ + + Table of Contents + + 1. Admin Guide + + 1.1. Overview + 1.2. Discovery Protocol + 1.3. Master Election + 1.4. Security Architecture + 1.5. Dependencies + + 1.5.1. OpenSIPS Modules + 1.5.2. External Libraries or Applications + + 1.6. Building the Module + 1.7. Exported Parameters + + 1.7.1. cluster (string) + 1.7.2. my_ip (string) + 1.7.3. interface (string) + 1.7.4. query_time (integer) + 1.7.5. password (string) + 1.7.6. manage_shtags (integer) + 1.7.7. master_stickiness (integer) + 1.7.8. on_config_mismatch (string) + + 1.8. Exported MI Functions + + 1.8.1. cl_ctr_list_members + 1.8.2. cl_ctr_node_info + 1.8.3. cl_ctr_list_config + 1.8.4. cl_ctr_shtag_force + 1.8.5. cl_ctr_shtag_auto + + 1.9. Exported Pseudo-Variables + 1.10. Exported Functions + 1.11. Multiple Clusters + 1.12. Hybrid Topologies (native + controller clusters) + 1.13. Configuration Example + 1.14. Limitations + 1.15. Planned Features + + A. HA Behaviour Tests + + A.1. Baseline + A.2. Test 1 — Stop the active node + A.3. Test 2 — Stop a backup node + A.4. Test 3 — Stop both backup nodes + A.5. Test 4 — Stop active node and one backup + A.6. Test 5 — Full cluster restart + A.7. Test 6 — Multiple clusters over one BIN socket + A.8. Summary + + 2. Contributors + + 2.1. Contributors + 2.2. Documentation Contributors + + List of Examples + + 1.1. Set cluster parameter — single cluster + 1.2. Set cluster parameter — multiple clusters on separate + networks (one BIN socket per network) + + 1.3. Set cluster parameter — multiple clusters sharing a single + BIN socket (recommended default) + + 1.4. Set my_ip parameter + 1.5. Set interface parameter + 1.6. Set query_time parameter + 1.7. Set password parameter + 1.8. Global manage_shtags — applies to all clusters + 1.9. Per-cluster override — opt one cluster out of automatic + failover + + 1.10. Global opt-out with one cluster opting in + 1.11. Typical full configuration with manage_shtags=1 (default) + 1.12. Set master_stickiness parameter + 1.13. Set on_config_mismatch parameter + 1.14. cl_ctr_list_members usage + 1.15. cl_ctr_node_info usage + 1.16. cl_ctr_list_config usage + 1.17. cl_ctr_shtag_force usage + 1.18. cl_ctr_shtag_auto usage + 1.19. Using $cl_ctr_* in the script + 1.20. Per-peer lookup functions + 1.21. Multiple clusters — dialog on cluster 1, usrloc on + cluster 2 + + 1.22. Multiple clusters — same multicast IP, different ports + 1.23. Hybrid — DB-native cluster 10 + controller cluster 1 + 1.24. Hybrid, no DB — static native cluster 7 + controller + cluster 1 + + 1.25. Minimal HA cluster configuration + +Chapter 1. Admin Guide + +1.1. Overview + + The clusterer_controller module provides automatic peer + discovery and topology management for the clusterer module via + authenticated, encrypted UDP multicast. It eliminates the need + for static node configuration or a database — nodes discover + each other automatically at startup and the cluster topology is + maintained dynamically at runtime. + + When a cluster is registered as controller-managed via + modparam("clusterer", "cluster_options", "cluster_id=N, + use_controller=1"), the clusterer_controller module takes over + all topology management: it allocates unique node IDs, + discovers peer BIN socket addresses, and calls the clusterer + internal API to add or remove nodes as they join or leave the + cluster. + + By default (manage_shtags=1), the module also provides fully + automatic sharing tag failover. The controller master node is + the single decision point for which node holds the active tag — + no event routes, MI commands, or seed_fallback_interval + configuration is needed. Sharing tags are forced to backup at + startup regardless of the =active config value, and the active + tag is claimed by the controller master automatically when the + cluster forms or when the active node departs. An operator can + override this automatic allocation and pin the active tag to a + chosen node with the cl_ctr_shtag_force MI command, reverting + to automatic allocation with cl_ctr_shtag_auto. + + The minimal configuration per node is a single multicast group + address. No IP addresses, node IDs, BIN URLs, or sharing tag + management scripts need to be hardcoded or maintained. Any + number of nodes can join or leave without any configuration + change on the remaining nodes. + +1.2. Discovery Protocol + + All traffic uses UDP multicast to the configured multicast + address and port. Clusters are kept apart in two independent + ways: by the multicast endpoint (two clusters may use different + IP addresses, or the same address with different UDP ports), + and by the id (cluster_id) carried in the cleartext of every + packet. A node silently ignores any packet whose cluster_id + differs from its own, so several clusters can safely share one + multicast group and port. Two clusters merge only if they share + all of the multicast address, the UDP port, the cluster_id and + the password — i.e. they are configured identically, which is + the operator's responsibility to avoid. + + Every packet is encrypted and authenticated with the + XChaCha20-Poly1305 AEAD (see the Security Architecture + section). Two distinct encryption keys are used depending on + the communication phase: + * Bootstrap key — derived from the configured password with + the Argon2id memory-hard KDF (per-cluster salt), so a + password captured from a bootstrap packet cannot be + brute-forced cheaply offline. Derived once at startup. It + is both the outer AEAD key for the admission handshake + (JOIN_REQ, KEY_GRANT, JOIN_REJECT) and the split-brain + MASTER_BEACON — traffic that must be readable before a + session key exists, or by masters holding different session + keys — and the pre-shared key for the Noise join handshake. + * Session key — derived via HKDF-SHA256 from the password and + a 32-byte master salt generated once when the cluster first + bootstraps. All normal cluster traffic uses this key. It is + preserved across master changes (a new master reuses the + key every member already holds), so failover needs no + re-keying. + + Each packet wire format begins with a 2-byte magic value + identifying the key type and a 2-byte cluster_id, both in + cleartext (the magic and cluster_id must be readable before + decryption to select the key and to filter foreign clusters), + followed by a random 24-byte AEAD nonce and the ciphertext. The + magic and cluster_id are additionally bound into the + authentication tag as AAD, so they cannot be altered + undetected. Only the payload is encrypted; the authenticated + plaintext begins with a 1-byte packet type and a 4-byte + monotonic sequence number used for replay protection. A 16-byte + authentication tag follows the ciphertext. Packets for a + different cluster_id are dropped before decryption; packets + encrypted with a different password fail authentication and are + silently discarded. + + The following packet types are defined: + * ALIVE — periodic heartbeat sent by every active node every + query_time seconds. Carries the sender IP, its X25519 + public key (so peers can prepare for key agreement) and a + small descriptor of the sender's consistency-critical + settings (manage_shtags, master_stickiness, query_time) + used for configuration-drift detection (see Security + Architecture). Encrypted with the session key. + * JOIN_REQ — sent at startup by a new node, carrying its IP, + BIN socket list, and Noise message 1 (the initiator's fresh + ephemeral public key). Encrypted with the bootstrap key so + it can be sent before a session key exists. + * MEMBER_LIST — sent by the master in response to a JOIN_REQ, + carrying the member count, the operator-forced sharing-tag + holder node_id (0 = automatic), and the full peer IP list + so the joining node can participate in elections. Only + accepted from the current master (except during initial + join when no master is yet known). + * NODE_ASSIGN — sent by the master to multicast, allocating a + node_id and BIN socket record for a joining node. All + cluster members receive and apply it. + * GOODBYE — sent on graceful shutdown so peers can remove the + node immediately without waiting for timeout. Uses the + sender's monotonic sequence counter to prevent forgery. + * MASTER_ALIVE — keepalive sent by the master every 1 second + (independent of query_time). Used by all peers to detect + master failure quickly (3-second timeout). Encrypted with + the session key. + * KEY_GRANT — the master's response to a JOIN_REQ, addressed + to the joining node. Carries Noise message 2 (the master's + ephemeral plus the AEAD-encrypted master salt), completing + the Noise_NNpsk0 handshake. Encrypted with the bootstrap + key. + * KEY_HANDOFF — sent by the outgoing master on graceful + shutdown to the next-highest-IP peer, delivering the master + salt (as an anonymous crypto_box sealed to that peer's + long-lived X25519 key, learned from its ALIVE) so it can + become the new master without a full re-join cycle. + Encrypted with the session key. + * JOIN_REJECT — sent by the master to a joining node whose + JOIN_REQ repeatedly fails authentication (wrong password). + After CL_CTR_JOIN_FAIL_LIMIT (3) consecutive bootstrap-key + decryption failures from the same source IP the master + sends a JOIN_REJECT to that IP. Encrypted with the + bootstrap key so it cannot be forged by a node that does + not know the cluster password. The joining node logs a + critical error and shuts down OpenSIPS on receipt. + * MASTER_BEACON — a master-only announcement multicast every + few MASTER_ALIVE ticks, carrying this partition's member + count. Unlike MASTER_ALIVE it is encrypted with the + bootstrap key, so it is readable even by a master that + holds a different session key. This is how a split brain + between two independently bootstrapped partitions is + detected and merged (see Master Election). + +1.3. Master Election + + Each cluster has three roles: master (the active coordinator), + backup (the standby promoted when the master fails, always the + highest-IP non-master) and member. The election uses a + quantized time window so that all nodes evaluate the same + eligible peer set and reach the same result deterministically. + No NTP synchronisation between nodes is required for correct + election results. + + The master_stickiness parameter (default 1) controls whether a + live master is kept when a higher-IP node joins. With + stickiness enabled, the master stays put and the higher-IP + joiner becomes the backup, minimising handovers; with + stickiness disabled the highest-IP node always becomes master. + In either mode two live masters are reconciled + deterministically (see Split-brain handling below). See the + master_stickiness parameter for details. + + Only the master handles JOIN_REQ packets, allocates node_ids, + and sends NODE_ASSIGN and MEMBER_LIST packets. Non-master nodes + are passive during join events. A joining node receives the + current session key from the master (via KEY_GRANT) and joins + as a member or backup; it never seizes mastership during the + join handshake. + + Preserved session key: the session key is generated once, when + the first node bootstraps the cluster, and is then preserved + across every master change. A new master does not re-key; + because every member already holds the key (obtained when it + joined), master transitions require no re-keying and no re-JOIN + cycle. + + Fast master failure detection: the master sends MASTER_ALIVE + packets every 1 second. All non-master peers maintain a + 3-second watchdog timer that fires if no MASTER_ALIVE is + received. On expiry the silent master is aged out of the + election window and each peer immediately re-elects, promoting + the backup (highest-IP survivor) — which already holds the + session key, so it starts serving within one keepalive + interval. + + Graceful master handoff: when the current master shuts down + cleanly, it sends a KEY_HANDOFF packet directly to the + next-highest-IP peer before sending GOODBYE to multicast. This + confirms the master salt to the incoming master so it can + assume control immediately. + + Split-brain handling. A split brain (more than one node + believing it is master) is prevented and, if it still occurs, + healed by three cooperating mechanisms: + * Prevention at join time. When several nodes start + simultaneously they all exchange (bootstrap-decryptable) + JOIN_REQs and thus learn about each other. At the join + deadline, a node that has seen a higher-IP node also still + joining defers its own self-promotion (for a few bounded + rounds) and joins that node instead, so only the highest-IP + starter becomes master and no independent-key lone masters + are created. + * Same-key yield. Two masters that share a session key (for + example after a network partition heals) can read each + other's MASTER_ALIVE; the lower-IP master immediately + yields to the higher-IP one. + * Divergent-key merge. Two masters that were bootstrapped + independently hold different session keys and so cannot + read each other's MASTER_ALIVE. Each therefore emits a + MASTER_BEACON encrypted with the shared bootstrap key. On + hearing a beacon from a superior partition — larger member + count, ties broken by higher IP — a node abandons its + partition, re-joins the superior master and adopts its + session key, converging the whole cluster onto a single + master and key. + +1.4. Security Architecture + + The module uses a two-phase key agreement to provide forward + secrecy and replay protection for all cluster traffic. + + Payload encryption and header binding: every packet's payload + is sealed with an AEAD. The 2-byte magic (a key selector that + must be readable before decryption) and the 2-byte cluster_id + that precede the nonce are cleartext framing, but they are + bound into the AEAD tag as additional authenticated data (AAD): + a captured packet cannot be re-stamped with a different + cluster_id and still authenticate, which matters when two + clusters share one multicast group and password. A node also + drops any packet whose cluster_id does not match its own before + attempting decryption, so foreign-cluster traffic on the group + never counts as an authentication failure. + + Crypto (all libsodium; a hard requirement): the payload AEAD is + XChaCha20-Poly1305 (24-byte nonce, whose 192-bit nonce space + removes any random-nonce collision concern); the bootstrap-key + KDF is Argon2id; and X25519 / HKDF-SHA256 / RNG also come from + libsodium. The active suite is reported in the startup log + (crypto=...). + + Phase 1 — join handshake (JOIN_REQ / KEY_GRANT): the join is a + Noise_NNpsk0_25519_ChaChaPoly_SHA256 handshake with the + pre-shared key set to the Argon2id bootstrap key: +-> psk, e JOIN_REQ carries Noise message 1 (a fresh ephemeral) +<- e, ee KEY_GRANT carries Noise message 2; its AEAD payload = mast +er_salt + + Authentication comes from the shared PSK, forward secrecy from + the ephemeral-ephemeral DH, and the Noise handshake hash binds + the whole transcript — so a stale KEY_GRANT for a superseded + JOIN_REQ simply fails to decrypt. Both handshake messages also + travel inside the bootstrap-key AEAD envelope, so a + wrong-password node is rejected at the envelope before the + handshake is even reached. This replaces the earlier + hand-rolled ECDH-and-XOR salt wrap. + + Phase 2 — session (all other packets): once the master salt is + known, all nodes derive the session key as: +session_key = HKDF-SHA256(IKM=password, salt=master_salt, info="cc-sessi +on-key") + + The session key is generated once, when the first node + bootstraps the cluster, and preserved across every master + change: a new master reuses the key that every member already + holds, so master transitions require no re-keying. All normal + cluster traffic (ALIVE, MEMBER_LIST, NODE_ASSIGN, GOODBYE, + MASTER_ALIVE, KEY_HANDOFF) is encrypted with this key. + + Replay protection: each sender maintains a monotonically + increasing 32-bit sequence number embedded in the authenticated + plaintext of every session-key packet. Each receiver tracks the + last accepted sequence number per source IP and rejects any + packet whose sequence is not strictly greater than the last + accepted value. Sequence counters are reset to zero whenever + the session key is (re)derived — at cluster bootstrap and when + a joiner adopts the key via KEY_GRANT/KEY_HANDOFF — and on peer + restart detection (JOIN_REQ received from a known IP resets + that peer's counter; MEMBER_LIST upsert resets all listed + peers). This protection does not depend on clock + synchronisation. + + Rate limiting: a per-source rate limiter (256 slots, 20 + packets/second limit) is applied before any decryption attempt. + This prevents CPU exhaustion from packet floods directed at the + multicast group. + + Join authentication and rejection: the master tracks + consecutive bootstrap-key decryption failures per source IP in + a small worker-local table (CL_CTR_JOIN_FAIL_TABLE_SZ = 8 slots). + When any source IP accumulates CL_CTR_JOIN_FAIL_LIMIT (3) + consecutive failures — indicating a node attempting to join + with the wrong password — the master sends an encrypted + JOIN_REJECT packet and stops responding to further JOIN_REQs + from that IP. + + On the joining side, a received JOIN_REJECT is only acted on + while the node is still in the initial join phase (CL_CTR_NODE_NEW + state) and is addressed to this node; an already-active cluster + member ignores any JOIN_REJECT unconditionally, so a node with + the correct password can never be evicted by a peer. + + A node joining with the wrong password cannot decrypt the + JOIN_REJECT (it is encrypted with the master's bootstrap key), + so it relies on a self-contained signal instead: while joining + it counts packets received from other peers that it cannot + decrypt. If, at the join deadline, the node is still unjoined + and has seen CL_CTR_JOIN_FAIL_LIMIT or more such undecryptable + packets, it concludes that a cluster it cannot authenticate to + exists on the group and shuts down OpenSIPS with a critical log + message — rather than promoting itself into a lone, split-brain + master (which, with managed sharing tags, would create a + duplicate active tag). This counter is reset the moment a + KEY_GRANT is successfully processed, so a legitimate joiner + that briefly saw an undecryptable packet before receiving its + key is never affected. + + Rogue traffic isolation: a node requests a re-key in response + to an undecryptable session-key packet only when that packet + came from its current master (a legitimate key rotation). + Undecryptable session packets from any other source — for + example a wrong-password or malicious node broadcasting on the + multicast group — are ignored, so such traffic cannot drive the + cluster into a re-JOIN churn. + + Peer table exhaustion defence: the peer table is bounded at + CL_CTR_MAX_PEERS (256) entries. When the table is full, the master + rejects JOIN_REQ packets from unknown IPs with a JOIN_REJECT + response. Known peers that are reconnecting after a restart + continue to be admitted regardless of the table count, since + they already own a slot. This prevents an attacker with the + cluster password from exhausting the peer table by flooding + JOIN_REQs from spoofed source addresses. + + Configuration-consistency enforcement: all nodes of a cluster + must use identical consistency-critical settings + (manage_shtags, master_stickiness and query_time); a per-node + mismatch would otherwise cause silent, inconsistent failover + and sharing-tag behaviour (for example, a master with + manage_shtags=0 would leave no node holding the active tag). + Each node advertises these effective settings in its ALIVE + heartbeat and in its JOIN_REQ, so mismatches are detected. What + happens then is controlled by the on_config_mismatch modparam: + * reject (default) - when a node tries to join an established + cluster (a master is alive) with different settings, the + master logs the attempt and returns a JOIN_REJECT; the + joining node logs the offending settings and shuts down, so + a misconfigured node never joins. + * warn - the node is allowed to join, but any peer that + observes a different value logs a single loud CONFIG + MISMATCH warning (repeated only if the peer's advertised + configuration changes, cleared once it matches). + * adopt - the joining node adopts the running cluster's + (master's) settings at runtime and continues; the adopted + values are what cl_ctr_list_config reports. + + This turns an easy-to-miss misconfiguration into an obvious log + line, a refused join, or a self-correction rather than a + hard-to-diagnose HA failure. + + Node identity and node_id allocation: node_id values are + allocated exclusively by the current master, serialised under + the peer-table lock, and a joining node never picks its own id. + The master hands out the lowest unused id by scanning the live + peer table, so a node that has failed but is not yet timed out + still occupies its slot and its id is never handed to a + different joiner — new nodes always receive a distinct id even + during the failure-detection window. A node that restarts and + rejoins from the same address reuses its previous id (and has + its replay counter reset), so ids stay stable across restarts. + Because peers are keyed by source IP address, every node in a + cluster must present a stable, unique source IP: two distinct + nodes that appear behind the same address (for example through + NAT) would share a single peer slot and node_id. Deploy the + cluster on a network where each member has its own routable + address on the BIN/multicast interface. + + Trust model — shared secret, not per-node identity: the cluster + is a single shared-secret trust domain. Authentication proves + only that a peer holds the cluster password; it does not bind a + cryptographic identity to an individual node, and there is no + per-node authorisation or revocation. Consequently any party in + possession of the password is a fully trusted member and can + legitimately win the highest-IP master election and assume the + master role — there is no distinction between "may be a member" + and "may be master". An attacker without the password cannot + affect the election at all: forged or replayed MASTER_ALIVE and + beacon packets fail AEAD authentication (or the strict + per-source sequence check) and are dropped before any election + logic runs. The residual exposure is therefore a malicious or + compromised insider that already holds the shared key. Protect + the password accordingly, and rotate it if a node is + decommissioned or suspected compromised. Removing this + limitation — per-node keypairs with enrolment and revocation, + so a single node can be distrusted without re-keying the whole + cluster — is planned future work. + +1.5. Dependencies + +1.5.1. OpenSIPS Modules + + The following modules are required by this module: + * proto_bin — required so that BIN listeners are registered + and available for discovery when clusterer_controller + initialises and scans the proto_bin listener list. + * clusterer — required, and it must register every + controller-managed cluster with modparam("clusterer", + "cluster_options", "cluster_id=N, use_controller=1"). That + per-cluster parameter is what pre-creates the + controller-managed cluster stubs, marks them so they never + touch the database, and arms the guard that stops the + controller from driving a native cluster of the same id. + The controller-managed ids declared here and the cluster + entries configured in clusterer_controller must match + exactly: if either side names a cluster the other does not, + the controller refuses to start with an error naming the + offending id — a managed id with no controller config has + no BIN socket or crypto parameters, and a controller config + for an unmanaged id has nothing to drive. + + Both dependencies are declared in the module's dep_export_t. + OpenSIPS will refuse to start if either dependency is not + satisfied. This does not imply that the modules must appear in + a particular order in the configuration file — OpenSIPS + resolves the dependency at runtime and will initialize the + required modules first regardless of loadmodule order. + + This is also checked from the other side: if a cluster_options + entry sets use_controller=1 but the clusterer_controller module + is not loaded at all, clusterer logs an error at startup, since + the controller-managed cluster stubs would never obtain a node + identity or form. clusterer itself keeps running, so any native + or DB-backed clusters are unaffected. + + Hybrid deployments are unaffected. use_controller is a + per-cluster option carried in cluster_options, defaulting to 0. + In a hybrid instance (native and controller-managed clusters + side by side), the controller-managed clusters each get a + cluster_options line with use_controller=1, while the native + ones are defined the usual way (DB rows or static + my_node_info/neighbor_node_info) and need no cluster_options + line. The exact-match check above compares only the + controller-managed ids against the controller's cluster + entries, so it never fires on a native cluster. + + All other modules that use the clusterer interface (tm, dialog, + dispatcher, usrloc etc.) may be loaded in any order relative to + clusterer_controller. The clusterer module automatically + creates a cluster stub when a cluster_options entry sets + use_controller=1 and a module attempts to register a capability + for that cluster. + + Because a controller-managed node receives its node_id at + runtime (after it joins) rather than from static configuration + at startup, any module that stamps this node's id into + on-the-wire data must read it live per message instead of + caching it once at initialisation — a value read at startup + would be the not-yet-assigned placeholder, and it may also + change on a re-election. In particular tm's anycast support + (tm_replication_cluster / t_anycast_replicate()) stamps this + node's id into the cid Via parameter so that a reply landing on + a different anycast member can be relayed to the node holding + the transaction; it renders that parameter from the current id + on each request, so anycast reply routing works unchanged under + a controller-managed cluster. + +1.5.2. External Libraries or Applications + + libsodium is required: + * libsodium (required) — provides the payload AEAD + (XChaCha20-Poly1305), the bootstrap-key KDF (Argon2id), the + Noise_NNpsk0 join handshake, and X25519 / HKDF-SHA256 / + RNG. The build fails with a clear error if libsodium + development files are not found. It is linked dynamically, + so each target host also needs the libsodium runtime + package (for example libsodium23). No other crypto library + is used or linked. + + The active suite is printed in the startup log (crypto=...). + +1.6. Building the Module + + clusterer_controller is excluded from the default build (it is + listed in exclude_modules in Makefile.conf.template), like the + other modules that depend on external libraries. A stock + OpenSIPS build therefore does not include it. To build it, add + it to include_modules in your Makefile.conf: +include_modules= clusterer_controller + + then rebuild (make all / make modules). libsodium development + files must be present on the build host or the build fails — + see External Libraries or Applications. + + The clusterer module is unmodified when clusterer_controller is + not built. The controller integration on the clusterer side + (the clusterer_ctrl API, the cluster_options modparam and all + controller hooks) is compiled only when clusterer_controller is + part of the build: the top-level Makefile detects this and + passes -DCLUSTERER_CTRL_SUPPORT to the clusterer module. A + build without clusterer_controller produces the stock clusterer + module, unchanged in behaviour and exported interface - in + particular the cluster_options parameter does not exist and is + rejected as unknown. Enabling clusterer_controller + automatically rebuilds clusterer with the support compiled in; + the two are always a matched pair. + +1.7. Exported Parameters + +1.7.1. cluster (string) + + Define a cluster to participate in. The value is a + comma-separated key=value string with the following fields: + * id (required) — positive integer cluster identifier, must + match the cluster_id used by clusterer consumers (dialog, + usrloc, dispatcher, etc.). + * multicast (required) — IPv4 multicast address and UDP port + in the form A.B.C.D:PORT. The address must be in the + 224.0.0.0/4 range. + * password (optional) — XChaCha20-Poly1305 encryption key + material. All nodes in the same cluster must use the same + password. Falls back to the global password modparam if not + set. + * bin_socket (optional) — BIN socket to advertise for this + cluster, in the form bin:IP:PORT. Required when multiple + clusters are defined (so the controller knows which + listener to advertise), but it need not be distinct — + several clusters may share the same BIN socket. When only + one cluster is defined and only one BIN socket exists, the + socket is auto-detected from the proto_bin listeners. + * manage_shtags (optional) — per-cluster override for the + global manage_shtags modparam. Set to 1 to enable automatic + sharing tag failover for this cluster, or 0 to disable it. + When omitted, the global manage_shtags value applies, + regardless of the order in which cluster and manage_shtags + modparams appear in the config file. + + This parameter may be set multiple times to participate in + multiple clusters simultaneously. Each cluster runs its own + independent worker process. + + No default value. At least one cluster must be defined. + + Example 1.1. Set cluster parameter — single cluster +... +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +... + + Example 1.2. Set cluster parameter — multiple clusters on + separate networks (one BIN socket per network) +... +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.1.10:5566") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,bin_socket=bin:10.0.2.10:5566") +... + + Example 1.3. Set cluster parameter — multiple clusters sharing + a single BIN socket (recommended default) +... +# one proto_bin listener serves both clusters; the cluster_id in each +# BIN packet keeps their replication traffic separate +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,bin_socket=bin:10.0.0.10:5566") +... + +Note + + One BIN socket can serve any number of clusters. Every + clusterer BIN packet carries its cluster_id, so a single + proto_bin listener demultiplexes traffic for all clusters. + Unless you specifically want different clusters on different + interfaces or networks, set the same bin_socket value on every + cluster entry (it is mandatory once more than one cluster is + defined, but need not be distinct). Using multiple BIN sockets + is for network/interface segregation, not throughput: the + replication data plane runs over BIN (TCP) and scales with the + shared TCP worker pool (tcp_children) and IPC dispatch, + independent of the number of BIN sockets. What does scale with + the number of clusters is the controller's own worker set — it + spawns one lightweight control worker (UDP multicast: + discovery, election, keepalives) per cluster — so it is the + cluster count, not the BIN-socket count, that adds processes. + +Note + + Each cluster needs a distinct multicast endpoint, but clusters + may share one BIN socket. The two sockets play opposite roles. + The controller's multicast endpoint is its control plane: every + cluster defined in one instance must use a different multicast + address:port — two cluster entries with the same multicast + value are rejected at startup (duplicate multicast). Use a + different port (e.g. :3333 and :3334) or a different group per + cluster. The bin_socket is the replication data plane and is + the opposite: it may be freely shared across clusters, since + every BIN packet carries its cluster_id. (The cluster_id filter + on the multicast wire header is for a different purpose: + letting separate deployments coexist on a shared multicast + group, not two clusters inside one instance.) + +1.7.2. my_ip (string) + + Explicitly set the local IPv4 address used by the controller + for its own node identity and master election. This is the IP + that the controller advertises to peers in JOIN_REQ and + NODE_ASSIGN packets and uses for the highest-IP master election + algorithm. + + Note: this parameter controls the controller's identity only. + The BIN socket address advertised to clusterer peers is + discovered separately from the proto_bin listener list and is + independent of this setting. Do not confuse my_ip with the BIN + socket IP defined by the socket=bin:IP:PORT core parameter. + + When set, the module walks the interface list to find which + local interface owns this address and uses that interface for + multicast traffic. Startup fails if no local interface owns the + given address. + + The module supports three identity resolution modes depending + on which modparams are provided: + + Mode 1 — my_ip set: The given IP is used directly. The owning + interface is resolved automatically from the system interface + list. Use this mode on multi-homed hosts where you want to pin + the controller identity to a specific IP. + + Mode 2 — interface set, my_ip not set: The first IPv4 address + on the named interface is used as the controller identity IP. A + warning is logged if the interface has multiple IPv4 addresses. + + Mode 3 — neither set (default): A throw-away UDP socket is + connected to the multicast group and getsockname() is called to + determine which source IP the kernel would select. The + interface name is resolved from the returned IP. Suitable for + single-homed hosts. + + Default: auto-detected (Mode 3). + + Example 1.4. Set my_ip parameter +... +modparam("clusterer_controller", "my_ip", "10.22.23.191") +... + +1.7.3. interface (string) + + Explicitly set the network interface name to use for multicast + traffic (e.g. eth0, enp6s18). The module takes the first IPv4 + address assigned to this interface as the controller's identity + IP. This corresponds to Mode 2 described in the my_ip parameter + documentation above. + + Like my_ip, this parameter affects the controller's own + identity only and has no effect on the BIN socket addresses + advertised to clusterer peers. + + If the interface has more than one IPv4 address, a warning is + logged and the first address (in the order returned by the + kernel) is used. Set my_ip explicitly to avoid ambiguity on + multi-address interfaces. + + Ignored if my_ip is also set — my_ip takes precedence. + + Default: auto-detected (Mode 3 — see my_ip). + + Example 1.5. Set interface parameter +... +modparam("clusterer_controller", "interface", "eth0") +... + +1.7.4. query_time (integer) + + How often (in seconds) each active node sends an ALIVE + heartbeat to the multicast group. This value also controls the + election window (3 × query_time) and the peer purge window (6 × + query_time). + + Smaller values mean faster failure detection but higher + multicast traffic. Valid range: 1–60. + + Default value is “5”. + + Example 1.6. Set query_time parameter +... +modparam("clusterer_controller", "query_time", 5) +... + +1.7.5. password (string) + + Global default encryption password for all clusters. All nodes + in a cluster must use the same password. The password serves + two purposes: + * Bootstrap key — the password is stretched with Argon2id + (memory-hard, per-cluster salt); it is the pre-shared key + for the Noise join handshake (JOIN_REQ / KEY_GRANT) and the + AEAD key for bootstrap traffic before a session key exists. + * Session key material — the password is fed into HKDF-SHA256 + together with the master salt to derive the session key + used for all normal cluster traffic. + + Can be overridden per cluster using the password= key in the + cluster parameter. + + Default value is “3eCrEt*5629”. Change this in production. Use + a long, high-entropy secret rather than a memorable phrase — + Argon2id raises the cost of an offline guess, but only a strong + secret removes the risk. A generated key is ideal, e.g. openssl + rand -base64 32. The module logs a startup warning if the + configured password is the default or has an estimated entropy + below 80 bits. + + Example 1.7. Set password parameter +... +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") +... + +1.7.6. manage_shtags (integer) + + When set to 1 (the default), the controller master node + automatically manages sharing tag failover for all clusters. + The controller becomes the single decision point for which node + holds the active tag, eliminating races between nodes and + requiring no script-level event routes or MI commands to handle + failover. While active, the clusterer_set_tag_active MI command + and the $shtag() script variable setter are blocked for + controller-managed clusters, returning an error to the caller. + + Behaviour when manage_shtags=1: + + Startup: all local sharing tags are forced to backup state + during module initialisation, regardless of the =active value + in the clusterer sharing_tag modparam. The deferred BIN + broadcast flag is also cleared so no SHTAG_ACTIVE packet is + ever sent at startup. This ensures that no node can steal the + active tag from an existing cluster member simply by + restarting. + + Bootstrap: when the first node starts alone and no existing + master responds within query_time seconds (join deadline), it + elects itself master and activates all local backup tags + exactly once. Nodes that join an existing cluster are never + eligible for this bootstrap path and never self-activate. + + Failover: when any node departs (graceful shutdown via GOODBYE + packet, or timeout-based removal), the controller master + activates its own backup tags for that cluster. This covers all + departure scenarios: last node standing, master still present, + and post re-election. + + Rejoin: a node rejoining an existing cluster always starts in + backup state and never reclaims the active tag from the current + holder, even if =active appears in its config. + + When set to 0, the controller does not touch sharing tag state + at all. The =active config value, seed_fallback_interval, and + external MI/event-route scripts behave exactly as in stock + clusterer without the controller. Use this when you have + existing tag management scripts and want to opt out of + automatic failover. + + Default value is “1”. + + Global vs per-cluster scope: This modparam sets a global + default that applies to every cluster defined via the cluster + modparam. Individual clusters can override it by including + manage_shtags=0 or manage_shtags=1 directly in the cluster + string. The global default is resolved at startup after all + modparams are processed, so the order of manage_shtags and + cluster lines in the config file does not matter. + + Example 1.8. Global manage_shtags — applies to all clusters +... +# clusters 1 and 2 registered as controller-managed on the clusterer sid +e +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +# Enable automatic failover for every cluster (this is also the default) +modparam("clusterer_controller", "manage_shtags", 1) +modparam("clusterer_controller", "cluster", "id=1,multicast=239.0.90.1:3 +333") +modparam("clusterer_controller", "cluster", "id=2,multicast=239.0.90.2:3 +333") +... + + Example 1.9. Per-cluster override — opt one cluster out of + automatic failover +... +# clusters 1 and 2 registered as controller-managed on the clusterer sid +e +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +# manage_shtags=1 globally, but cluster 2 uses its own MI/event-route sc +ripts +modparam("clusterer_controller", "manage_shtags", 1) +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,manage_shtags=0") +... + + Example 1.10. Global opt-out with one cluster opting in +... +# clusters 1 and 2 registered as controller-managed on the clusterer sid +e +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +# Disable automatic failover globally; enable it only for cluster 1 +modparam("clusterer_controller", "manage_shtags", 0) +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,manage_shtags=1") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333") +... + + Example 1.11. Typical full configuration with manage_shtags=1 + (default) +# All nodes use identical config — only the BIN socket IP differs per no +de. +# The =active tag value in sharing_tag is ignored by the controller; +# it is kept in the config only for compatibility with manage_shtags=0. + +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "sharing_tag", "vip1/1=active") +modparam("clusterer", "ping_interval", 4) +modparam("clusterer", "ping_timeout", 1500) + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", "id=1,multicast=239.0.90.1: +3333") +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") +# manage_shtags defaults to 1 — no need to set it explicitly + +1.7.7. master_stickiness (integer) + + Controls whether a live master keeps its role when a higher-IP + node joins. Default 1 (sticky). + + The module recognises three roles per cluster: master (the + active coordinator), backup (the standby that takes over when + the master fails), and member (all other nodes). The backup is + always the highest-IP node that is not the master. + * master_stickiness=1 (default): the master is sticky — a + live master keeps the role and is not displaced when a + higher-IP node joins. The newly joined node becomes the + backup (replacing the previous backup if it has a higher + IP); the master only changes when the current master + actually fails, at which point the backup is promoted. This + minimises the number of master handovers. + * master_stickiness=0: pure highest-IP election — a higher-IP + node takes over as master as soon as it appears. This + produces more handovers but always keeps the highest-IP + node as master. + + In both modes a split-brain (two nodes each believing they are + master, e.g. after a network partition heals) is resolved + deterministically: the lower-IP master yields to the higher-IP + one. + + Global vs per-cluster scope: like manage_shtags, this sets a + global default that individual clusters can override with + master_stickiness=0 or master_stickiness=1 in the cluster + string. Resolution happens at startup regardless of modparam + order. + + Example 1.12. Set master_stickiness parameter +... +# both clusters registered as controller-managed on the clusterer side +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +# Global default (sticky) — omit entirely for the same effect +modparam("clusterer_controller", "master_stickiness", 1) + +# Per-cluster override: cluster 2 always promotes the highest-IP node +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,master_stickiness=0") +... + +1.7.8. on_config_mismatch (string) + + Policy applied when a node's consistency-critical settings + (manage_shtags, master_stickiness, query_time) differ from + those of the running cluster. All nodes of a cluster are + expected to use identical values; this parameter decides what + happens when they do not. One of: + * reject (default) - the master refuses the join with a + JOIN_REJECT and the joining node shuts down after logging + which settings differ. + * warn - the node is allowed to join; a single CONFIG + MISMATCH warning is logged per mismatching peer. + * adopt - the joining node adopts the running cluster's + settings at runtime and continues. + + Global only (not per-cluster). Default: reject. + + Example 1.13. Set on_config_mismatch parameter +... +modparam("clusterer_controller", "on_config_mismatch", "reject") +... + +1.8. Exported MI Functions + +1.8.1. cl_ctr_list_members + + List all current cluster members with their node_id, status and + BIN socket addresses. Status is one of master, backup (the + standby that will be promoted if the master fails) or member. + Only peers within the current election window are shown. + + Parameters: none + + Example 1.14. cl_ctr_list_members usage +opensips-cli -x mi cl_ctr_list_members +[ + { + "cluster_id": 1, + "members": [ + { + "ip": "10.22.23.191", + "node_id": 1, + "status": "master", + "bin_sockets": [ "bin:10.22.23.191:3857" ] + }, + { + "ip": "10.22.23.193", + "node_id": 2, + "status": "backup", + "bin_sockets": [ "bin:10.22.23.193:3857" ] + }, + { + "ip": "10.22.23.192", + "node_id": 3, + "status": "member", + "bin_sockets": [ "bin:10.22.23.192:3857" ] + } + ] + } +] + +1.8.2. cl_ctr_node_info + + Return full information for a specific node identified by its + allocated node_id. + + Parameters: + * node_id — the integer node_id to look up. + + Example 1.15. cl_ctr_node_info usage +opensips-cli -x mi cl_ctr_node_info node_id=2 +{ + "node_id": 2, + "ip": "10.22.23.192", + "cluster_id": 1, + "status": "backup", + "bin_sockets": [ "bin:10.22.23.192:3857" ] +} + +1.8.3. cl_ctr_list_config + + List all configured clusters and their resolved settings — the + effective values actually in use after global defaults and + per-cluster overrides have been applied. Useful for confirming + that a per-cluster master_stickiness or manage_shtags override + took effect. The cluster password is never exposed. + + The shtag_mode field reports the current sharing-tag allocation + policy: auto when the active tag follows the master + automatically, or override: when an operator has + pinned a fixed holder with cl_ctr_shtag_force. + + Parameters: none + + Example 1.16. cl_ctr_list_config usage +opensips-cli -x mi cl_ctr_list_config +[ + { + "cluster_id": 1, + "multicast": "239.0.90.1:3333", + "my_ip": "10.22.23.191", + "bin_socket": "bin:10.22.23.191:3857", + "query_time": 5, + "master_stickiness": 1, + "manage_shtags": 1, + "shtag_mode": "auto", + "member_count": 3 + } +] + +1.8.4. cl_ctr_shtag_force + + Force a specific node to hold the active sharing tag, + overriding the normal master-driven allocation. This is useful + for planned maintenance or manual traffic steering: the chosen + node becomes the sole active shtag holder cluster-wide while + every other node — including the master — is put into backup + for that tag. + + The command must be issued on the current master (it returns an + error otherwise). The override is propagated to all members in + the MEMBER_LIST and survives master fail-over: a newly elected + master keeps honouring it rather than reclaiming the tag. + Automatic allocation stays suspended until cl_ctr_shtag_auto is + called. If the forced node leaves the cluster or times out, the + override is cleared automatically and automatic allocation + resumes. + + Parameters: + * cluster_id — the target cluster. + * node_id — the node that must hold the active tag; it must + be a current member of the cluster. + + Example 1.17. cl_ctr_shtag_force usage +opensips-cli -x mi cl_ctr_shtag_force cluster_id=1 node_id=3 + +1.8.5. cl_ctr_shtag_auto + + Clear any override set by cl_ctr_shtag_force and resume + automatic, master-driven sharing-tag allocation — the active + tag follows the master again. Must be issued on the current + master. + + Parameters: + * cluster_id — the target cluster. + + Example 1.18. cl_ctr_shtag_auto usage +opensips-cli -x mi cl_ctr_shtag_auto cluster_id=1 + +1.9. Exported Pseudo-Variables + + These read-only pseudo-variables expose live cluster state to + the routing script, so a decision such as “only the master runs + this job” can be made without an MI call. They read directly + from shared memory and are therefore available in every process + (SIP workers included). + + Each variable optionally takes a cluster id as its argument, + e.g. $cl_ctr_role(2). The bare form ($cl_ctr_role) resolves to + the only configured cluster; when several clusters are defined + the bare form returns NULL and logs a one-time warning, so the + cluster must be named explicitly. An unknown cluster id, or a + value that is not currently known (e.g. no master yet), returns + NULL. All of them are read-only - assigning to them fails. + * $cl_ctr_role — this node's role in the cluster: master, + backup, member, or joining (still authenticating / before + the first election). + * $cl_ctr_is_master — 1 if this node is the cluster master, 0 + otherwise. A fast path for the most common check. + * $cl_ctr_master_ip — IP of the current master (NULL if none + is elected yet). + * $cl_ctr_backup_ip — IP of the current backup / standby + master (NULL if none). + * $cl_ctr_node_id — this node's node_id within that cluster + (NULL until assigned). A node may hold different ids in + different clusters. + * $cl_ctr_my_ip — the controller identity IP of this node. + * $cl_ctr_members — number of live members currently in the + cluster. + * $cl_ctr_shtag_mode — auto (tags follow the elected master) + or forced (an operator pinned them with + cl_ctr_shtag_force). + * $cl_ctr_forced_node — the node_id the active sharing tag is + pinned to (NULL when in auto mode). + + Example 1.19. Using $cl_ctr_* in the script +# single cluster: run a periodic job only on the master +if ($cl_ctr_is_master) + route(do_master_only_work); + +# several clusters: name the one you mean +xlog("cluster 2 master is $cl_ctr_master_ip(2), I am $cl_ctr_role(2)\n") +; + + +1.10. Exported Functions + + Per-peer lookups take two arguments (cluster_id, node_id) and + are therefore script functions, not pseudo-variables (a comma + inside a variable's parentheses is ambiguous when the variable + is used as a function argument). Boolean checks return + true/false for use directly in an if; value lookups write into + an output variable. All are usable from any route. + * cl_ctr_node_is_master(cluster_id, node_id) — true if that + node is the cluster master. + * cl_ctr_node_present(cluster_id, node_id) — true if that + node_id is a live member. + * cl_ctr_get_node_role(cluster_id, node_id, out_var) — write + that node's role (master/backup/member) into out_var; + returns false if it is not a member. + * cl_ctr_get_node_ip(cluster_id, node_id, out_var) — write + that node's IP into out_var. + + Example 1.20. Per-peer lookup functions +if (cl_ctr_node_present(1, 3) && cl_ctr_node_is_master(1, 3)) { + cl_ctr_get_node_ip(1, 3, $var(ip)); + xlog("node 3 ($var(ip)) leads cluster 1\n"); +} + +1.11. Multiple Clusters + + A single OpenSIPS instance can participate in multiple clusters + simultaneously by repeating the cluster modparam. Each cluster + runs an independent worker process with its own multicast + socket, peer table, master election, and node_id space. + + Clusters are distinguished by the combination of their + multicast IP address and UDP port. Two useful topologies are + possible: + * Different ports, same multicast IP — convenient when all + clusters share the same L2 segment. The port number alone + separates traffic for each cluster. Each cluster's password + should also differ to provide an additional encryption + barrier. + * Different multicast IPs — useful when clusters span + different network segments or when multicast routing is + scoped differently per cluster. + + When multiple clusters are defined and the node has more than + one BIN socket, the bin_socket= key must be specified in each + cluster string to indicate which BIN socket to advertise for + that cluster. If only one BIN socket exists, it is used for all + clusters automatically. + + Each cluster has its own independent clusterer cluster_id, + allowing different OpenSIPS subsystems to replicate on + different clusters: + + Example 1.21. Multiple clusters — dialog on cluster 1, usrloc + on cluster 2 +# Two BIN sockets, one per cluster +socket=bin:10.0.1.10:5566 +socket=bin:10.0.2.10:5566 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +loadmodule "clusterer_controller.so" +# Cluster 1 — dialog replication group, LAN segment 10.0.1.0/24 +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.1.10:5566") +# Cluster 2 — usrloc replication group, LAN segment 10.0.2.0/24 +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,bin_socket=bin:10.0.2.10:5566") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) + +loadmodule "usrloc.so" +modparam("usrloc", "cluster_id", 2) + + Example 1.22. Multiple clusters — same multicast IP, different + ports +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,password=ClusterOneSecret") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,password=ClusterTwoSecret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 2) + +1.12. Hybrid Topologies (native + controller clusters) + + A single OpenSIPS instance can run native clusterer clusters + (defined in the database or statically via + my_node_info/neighbor_node_info) and controller-managed + clusters side by side. This allows, for example, a fixed, + DB-provisioned replication cluster to coexist with a + zero-config controller-driven HA cluster on the same node. + + Which kind a cluster is follows from how it is declared: + * Controller-managed — every cluster_id registered with the + clusterer module via modparam("clusterer", + "cluster_options", "cluster_id=N, use_controller=1") (and + matched by a cluster entry in this module). Its topology + and this node's node_id are driven at runtime by the + controller; it never touches the database and always + behaves as db_mode=0, regardless of the global db_mode. + * Native — every cluster loaded from the clusterer database + (db_mode!=0) or provisioned statically. It behaves exactly + as classic clusterer: fixed my_node_id, DB persistence + (when DB-backed), script/MI-managed sharing tags. + + The rules that keep the two kinds apart: + * A cluster_id is exclusively controller-managed or native — + declaring the same id both ways is rejected at startup. + * my_node_id identifies this node in its native clusters only + and is required only when native clusters exist; controller + clusters get their node id assigned at runtime, and this + node may well hold different node ids in different + clusters. + * db_url/db_mode apply to native clusters only. A + controller-only deployment needs neither. + * Sharing tags: only tags of controller-managed clusters are + forced to backup at startup and driven by the controller + master; tags of native clusters keep their configured state + (=active included) and stay script/MI-managed. + * Native and controller clusters may share the same BIN + socket - the cluster_id carried in every BIN packet keeps + their traffic apart. + + Example 1.23. Hybrid — DB-native cluster 10 + controller + cluster 1 +socket=bin:10.0.0.10:5566 + +loadmodule "db_mysql.so" +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +# native side: cluster 10 is defined entirely by rows in the clusterer D +B +# table (one row per node, including a row whose node_id = my_node_id be +low +# for THIS node) - there is no cluster_id modparam for native clusters. +modparam("clusterer", "db_mode", 1) +modparam("clusterer", "db_url", "mysql://opensips:pass@localhost/opensip +s") +modparam("clusterer", "my_node_id", 5) # this node's id in its nati +ve/DB clusters (global) +# controller side: cluster 1 is dynamic (no DB, no static rows) +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566,passwo +rd=S3cret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) # controller-manag +ed + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 10) # DB-native + + Example 1.24. Hybrid, no DB — static native cluster 7 + + controller cluster 1 +socket=bin:10.0.0.10:5566 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +# native side: cluster 7 is provisioned statically (no DB). +# db_mode=0 is REQUIRED, otherwise my_node_info/neighbor_node_info are i +gnored. +modparam("clusterer", "db_mode", 0) +modparam("clusterer", "my_node_id", 5) # this node's id in + native cluster 7 +modparam("clusterer", "my_node_info", "cluster_id=7, url=bin:10.0. +0.10:5566") +modparam("clusterer", "neighbor_node_info", "cluster_id=7, node_id=6, ur +l=bin:10.0.0.11:5566") +# controller side: cluster 1 is dynamic +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566,passwo +rd=S3cret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) # controller-manag +ed (cluster 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 7) # native (cluster 7 +) + + Here my_node_id=5 is this node's id in the native cluster 7; in + cluster 1 the controller assigns an id at runtime, which may + differ. + +1.13. Configuration Example + + The following example shows a minimal two-module configuration + for zero-config HA clustering with dialog replication. The + loadmodule order does not matter — the dependency system + enforces correct initialization order automatically. + + Example 1.25. Minimal HA cluster configuration +# Each node needs an explicit BIN socket (no wildcard) +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "sharing_tag", "vip1/1=active") +modparam("clusterer", "ping_interval", 4) +modparam("clusterer", "ping_timeout", 1500) + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") + +loadmodule "tm.so" +modparam("tm", "tm_replication_cluster", 1) + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) +modparam("dialog", "cluster_auto_sync", 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 1) +modparam("dispatcher", "cluster_probing_mode", "distributed") + + The same configuration file (with the node-specific + socket=bin:IP:PORT line changed per node) is used on every + node. No other per-node customization is required. + +1.14. Limitations + + * IPv4 only. IPv6 multicast is not currently supported. + * The module pre-allocates approximately 152 KB of shared + memory (-m) per cluster and 6 KB of private memory (-M) per + worker process at startup. If either allocation fails, + OpenSIPS will refuse to start with an error in the log. + * Wildcard BIN sockets (bin:*:PORT) are rejected at startup. + An explicit IP address must be used in the socket= line. + * Node IDs are not persistent across a full cluster restart + (all nodes down simultaneously). IDs are reallocated + starting from 1 when the cluster reforms. This has no + operational impact as long as at least one node remains up + during rolling restarts. + * The multicast network must support IP multicast routing + between all cluster nodes. Nodes on different L3 segments + require PIM or similar multicast routing. + * PIM-DM networks: PIM Dense Mode periodically re-floods + multicast traffic (prune state typically expires every 3 + minutes). On slow or congested networks this brief + re-flood/prune cycle could cause a gap in MASTER_ALIVE + delivery and trigger a spurious master re-election. If this + occurs, increase the effective timeout by raising + CL_CTR_MASTER_KA_MISSED in the source from 3 to 5 or higher. + PIM Sparse Mode (PIM-SM) or networks with IGMP snooping do + not have this issue. + * L2 overlay tunnels (Geneve, VXLAN, GRE): if the overlay + presents a flat L2 segment with multicast support, the + controller works transparently. However, some VXLAN + deployments disable multicast entirely (no underlay + multicast group, no BUM replication) — in that case the + controller will not function, as it has no unicast + fallback. Encapsulation overhead also adds latency and + jitter; on high-latency overlays consider raising + CL_CTR_MASTER_KA_MISSED to avoid spurious re-elections. + * IPsec-protected links: native IPsec multicast requires + GDOI/GET VPN (RFC 6407), which is rarely deployed. The + recommended approach is to run multicast inside an inner + tunnel (GRE-over-IPsec, Geneve-over-IPsec) that presents a + multicast-capable interface. Running the controller over + such a setup results in double encryption + (application-layer XChaCha20-Poly1305 plus IPsec ESP), + which is harmless but adds minor CPU overhead. IPsec ESP + tunnel mode also reduces the effective MTU by approximately + 50 bytes, which compounds the MEMBER_LIST fragmentation + issue described below. + * MEMBER_LIST fragmentation: the MEMBER_LIST packet grows + with cluster size and reaches approximately 4395 bytes at + the maximum of 256 nodes. This exceeds the 1472-byte UDP + payload budget of a standard 1500-byte MTU Ethernet link + and requires IP fragmentation: + + Standard Ethernet (1500 MTU): 3 fragments + + IPsec ESP tunnel (~1400 MTU): 4 fragments + + GRE-over-IPsec (~1350 MTU): 4–5 fragments + All other packet types (ALIVE, JOIN_REQ, KEY_GRANT, + GOODBYE, etc.) fit comfortably within a single datagram on + any of these links. The DF bit is not set, so IP + fragmentation occurs transparently where the network allows + it. However, firewalls or stateless middleboxes that + silently drop fragmented UDP will prevent new nodes from + joining, since MEMBER_LIST is required to complete the join + sequence. Verify that fragmented UDP is permitted on all + paths between cluster nodes, particularly over VPN tunnels + and across datacenter firewalls. + +1.15. Planned Features + + The following features are planned for future releases: + * Node maintenance mode — take a node out of duty for a + rolling upgrade while it stays in the cluster. A node in + maintenance keeps replicating and answering pings and stays + visible in cl_ctr_list_members, but is excluded from + election (never master or backup; if it is master it hands + over gracefully first) and sheds its sharing tags (a + cl_ctr_shtag_force pin on it auto-clears). Two levels are + planned: + + evicted — out of election and tags; the routing script + refuses new work while established dialogs finish. + + full — additionally marks the node down for clusterer + consumers so peers stop routing replication work to + it. + The state is cluster-wide (carried in the ALIVE/MEMBER_LIST + control plane, so every node agrees and it survives master + failover) and runtime-only — a restart brings the node back + in service. + Interfaces (following the read-only variables and MI + commands above): + + MI cl_ctr_maintenance (any node, the master propagates + it) sets a target node's state evicted/full/off; + cl_ctr_list_members gains a maintenance column. + + Script function cl_ctr_set_maintenance() (a verb - an + action) for a node to put itself in or out of + maintenance from the routing logic. + + Read-only variables $cl_ctr_maintenance (this node: + none/evicted/full) and $cl_ctr_node_maint(cluster_id, + node_id) (any peer); $cl_ctr_role gains a maintenance + value. + + Event E_CL_CTR_MAINTENANCE raised on every node when + any member's maintenance state changes, so an + event_route can react (e.g. shift dispatcher weights). + * Statistics — module statistics (current role, member count, + master changes, nodes joined/left, JOIN_REJECTs, decrypt + failures, config mismatches, split-brain merges) exposed + via get_statistics and monitoring exporters. + * Events — events raised on state transitions (became master, + demoted, node joined/left, split-brain merged, config + mismatch, authentication reject), named E_CL_CTR_* and + consumable from an event_route or any event subscriber + transport. + * IPv6 multicast — the control plane currently uses IPv4 + multicast (groups in 224.0.0.0/4). Add IPv6 multicast + support (ff00::/8 groups, AF_INET6 sockets and membership) + so the controller can run on IPv6-only or dual-stack + deployments. + + Read-only script variables for cluster state ($cl_ctr_role, + $cl_ctr_is_master, and the rest) are already available - see + Exported Pseudo-Variables. + +Appendix A. HA Behaviour Tests + + The following tests were performed on a three-node cluster + (nodes A=10.22.23.191, B=10.22.23.192, C=10.22.23.193) to + verify correct failover, tag stability, and no-steal-on-join + behaviour. The sharing tag under test is vip1 in cluster 1. All + nodes run with manage_shtags=1. + + Tag state was queried after each operation via: +opensips-cli -x mi clusterer_list_shtags + +A.1. Baseline + + All three nodes running. B holds the active tag; A and C are + backup. +A (10.22.23.191) svc=active tag=backup +B (10.22.23.192) svc=active tag=active +C (10.22.23.193) svc=active tag=backup + +A.2. Test 1 — Stop the active node + + B (active) is stopped. The remaining nodes must elect a new + active holder. B must rejoin as backup and must not steal the + tag from whichever node became active. +# Stop B +A svc=active tag=backup +B svc=inactive tag=(down) +C svc=active tag=active <-- C promoted + +# Start B +A svc=active tag=backup +B svc=active tag=backup <-- rejoined as backup +C svc=active tag=active <-- C retains active + + Result: PASS. Failover within the dead-node detection window; + rejoining node did not steal the active tag. + +A.3. Test 2 — Stop a backup node + + A (backup) is stopped. The active tag must remain on C without + any transition. A must rejoin as backup. +# Stop A +A svc=inactive tag=(down) +B svc=active tag=backup +C svc=active tag=active <-- unchanged + +# Start A +A svc=active tag=backup <-- rejoined as backup +B svc=active tag=backup +C svc=active tag=active <-- still active + + Result: PASS. Removing a backup node causes no tag movement; + rejoining node started in backup state. + +A.4. Test 3 — Stop both backup nodes + + A and B (both backup) are stopped simultaneously. The lone + remaining node C must retain the active tag. A and B must + rejoin as backup. +# Stop A and B +A svc=inactive tag=(down) +B svc=inactive tag=(down) +C svc=active tag=active <-- unchanged, lone node + +# Start A, then B +A svc=active tag=backup <-- rejoined as backup +B svc=active tag=backup <-- rejoined as backup +C svc=active tag=active <-- still active + + Result: PASS. Active node remained stable while running alone; + both rejoining nodes came up in backup state. + +A.5. Test 4 — Stop active node and one backup + + B (active) and C (backup) are stopped. The sole remaining node + A must become active. B and C must rejoin as backup. +# Stop B and C +A svc=active tag=active <-- A promoted, now lone node +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start B +A svc=active tag=active <-- retains active +B svc=active tag=backup <-- rejoined as backup +C svc=inactive tag=(down) + +# Start C +A svc=active tag=active <-- retains active +B svc=active tag=backup +C svc=active tag=backup <-- rejoined as backup + + Result: PASS. The surviving node correctly claimed the active + tag; each rejoining node started in backup state without + challenging the active holder. + +A.6. Test 5 — Full cluster restart + + All three nodes are stopped (full outage). Nodes are then + started one at a time. The first node up must self-elect as + active (no peers available to sync from). Subsequent nodes must + join as backup and must not steal the active tag. +# All stopped +A svc=inactive tag=(down) +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start A first +A svc=active tag=active <-- first/lone seed, self-synced +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start B +A svc=active tag=active <-- retains active +B svc=active tag=backup <-- joined as backup, did not steal +C svc=inactive tag=(down) + +# Start C +A svc=active tag=active <-- retains active +B svc=active tag=backup +C svc=active tag=backup <-- joined as backup, did not steal + + Result: PASS. The first node to start elected itself active and + no spurious sync errors were logged. Each subsequent node + joined as backup without challenging the active holder. + +A.7. Test 6 — Multiple clusters over one BIN socket + + Two controller clusters (id=1 and id=2) are configured on all + three nodes. Each cluster has its own multicast endpoint (a + distinct port is required - two clusters on the same multicast + address:port are rejected at startup with duplicate multicast), + but both advertise the same BIN socket (bin:IP:3857). Each + cluster must form independently and its replication must stay + isolated over the shared socket. +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1 +") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1 +") +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:IP:3857") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,bin_socket=bin:IP:3857") + +# both clusters converge, each with its own master/backup/member roles; +# clusterer_list shows cluster 1 and cluster 2 both using bin:IP:3857 + + Result: PASS. Both clusters formed independently (each + member_count=3 with consistent roles), the BIN links of both + were Up over the single shared socket, and there were zero + decrypt / cross-talk / foreign-cluster errors on any node - the + cluster_id in each BIN packet keeps the two clusters' + replication traffic separate. A control test confirmed that + configuring both clusters on the same multicast address:port is + refused at startup. + +A.8. Summary + + Test Scenario Result + 1 Stop active node; rejoin PASS + 2 Stop backup node; rejoin PASS + 3 Stop both backups simultaneously; rejoin PASS + 4 Stop active + one backup; rejoin PASS + 5 Full cluster restart; sequential startup PASS + 6 Two clusters sharing one BIN socket (distinct multicast) PASS + + In all five failover tests (1–5): exactly one node held the + active sharing tag at all times (including during the failure + window), and no rejoining node stole the active tag from the + current holder. Test 6 additionally confirmed that two clusters + can share a single BIN socket with fully isolated replication. + +Chapter 2. Contributors + +2.1. Contributors + + Definition, design and implementation of this module was made + by: + * Yury Kirsanov — VoIPLine Telecom + +2.2. Documentation Contributors + + Documentation was written by: + * Yury Kirsanov — VoIPLine Telecom + + Documentation Copyrights: + + Copyright © 2026 VoIPLine Telecom diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c new file mode 100644 index 00000000000..de93fa22948 --- /dev/null +++ b/modules/clusterer_controller/clusterer_controller.c @@ -0,0 +1,6609 @@ +/* + * clusterer_controller - multicast extension for the clusterer module + * + * Copyright (C) 2026 Yury Kirsanov + * VoIPLine Telecom + * + * This file is part of opensips, a free SIP server. + * + * opensips is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * (at your option) any later version. + * + * opensips is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program; if not, write to the Free Software + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + * + * ========================================================================= + * MODULE: clusterer_controller + * ========================================================================= + * + * PROTOCOL OVERVIEW + * ----------------- + * All traffic is UDP multicast to multicast_address:multicast_port. Every + * packet's payload is encrypted and authenticated with an AEAD. The cleartext + * framing that precedes it - a 2-byte magic (key selector, must be readable + * before decryption) and a 2-byte cluster_id - is bound into the AEAD tag as + * additional authenticated data (AAD), so it cannot be altered undetected (this + * blocks re-stamping a captured packet onto a different cluster_id when two + * clusters share a multicast group and password). + * + * CRYPTO SUITE (all libsodium; libsodium is a hard build requirement): + * - XChaCha20-Poly1305 payload AEAD, 24-byte nonce (its 192-bit nonce space + * removes any random-nonce collision worry). + * - Argon2id bootstrap-key KDF. + * - Noise_NNpsk0_25519_ChaChaPoly_SHA256 for the join handshake (PSK = the + * Argon2id bootstrap key); crypto_box for the KEY_HANDOFF salt delivery. + * The active suite is logged at startup ("crypto=..."). + * + * Two key types are used, selected by the 2-byte magic: + * + * Bootstrap key - KDF(password, salt="...v1:"+multicast) [scrypt or Argon2id] + * Used for the admission handshake (JOIN_REQ, KEY_GRANT, JOIN_REJECT) and + * the split-brain MASTER_BEACON - i.e. traffic that must be readable before + * a session key exists, or by masters holding different session keys. + * Memory-hard KDF (derived once at startup) so a password captured from a + * bootstrap packet cannot be brute-forced cheaply offline. + * + * Session key - HKDF-SHA256 over an X25519-ECDH-agreed random master_salt. + * Used for all normal traffic. Generated once when the first node + * bootstraps the cluster and then preserved across every master change: + * a new master reuses the key that every member already holds, so master + * transitions require no re-keying. + * + * Wire format (all packets): + * [2B magic] [2B cluster_id] [12B|24B nonce] [ciphertext] [16B tag] + * AAD = magic || cluster_id. A node drops packets whose cluster_id does not + * match its own BEFORE decryption, so foreign-cluster traffic sharing the + * group never counts as an authentication failure. + * + * Authenticated plaintext layout: + * [1B: packet type] [4B: seq BE] [payload] + * + * The 32-bit monotonic sequence number is per-sender and validated per source + * IP for session-key packets. Prevents replay without any dependency on clock + * synchronisation. + * + * PACKET TYPES + * ------------ + * ALIVE - session key, multicast + * Every active node every query_time seconds. + * Payload: IP(16B) + pubkey(32B). Peers learn X25519 pubkeys here. + * + * JOIN_REQ - bootstrap key, multicast + * Sent by a new node on startup. + * Payload: IP(16B) + bin_info + noise_msg1(48B) + config(4B) + * noise_msg1 is Noise_NNpsk0 message 1 (fresh ephemeral + AEAD tag). + * + * MEMBER_LIST - session key, multicast + * Master -> all: member count, the operator-forced sharing-tag holder + * node_id (0 = automatic), and the full peer IP list, so all nodes elect + * identically. Only accepted from the current master (CL_CTR_NODE_NEW aside). + * + * GOODBYE - session key, multicast + * Graceful shutdown. Peers remove sender immediately without timeout. + * + * NODE_ASSIGN - session key, multicast + * Master -> all: allocate node_id + BIN socket for a joining node. + * + * MASTER_ALIVE - session key, multicast + * Master-only keepalive every CL_CTR_MASTER_KA_INTERVAL seconds. Peers declare + * the master dead after CL_CTR_MASTER_KA_TIMEOUT seconds of silence and trigger + * re-election. Two masters that share a session key resolve split-brain + * here: the lower-IP one yields. + * + * KEY_GRANT - bootstrap key, multicast (addressed to joiner) + * Master reply to JOIN_REQ. + * Payload: IP(16B) + noise_msg2(80B) + * noise_msg2 is Noise_NNpsk0 message 2; its AEAD payload is the master_salt. + * + * KEY_HANDOFF - session key, multicast (addressed to next master) + * Outgoing master on graceful shutdown -> next-highest-IP peer. Transfers + * master_salt so the new master avoids a full re-join cycle. + * Payload: IP(16B) + crypto_box_seal(master_salt) sealed to the next + * master's long-lived X25519 key (learned from its ALIVE). + * + * JOIN_REJECT - bootstrap key, multicast + * Master -> a source whose bootstrap packets repeatedly fail to decrypt + * (wrong password). Encrypted, so only a correctly-configured node can + * read it; a wrong-password joiner also self-terminates at its deadline. + * + * MASTER_BEACON - bootstrap key, multicast + * Master-only, every CL_CTR_MASTER_BEACON_EVERY keepalive ticks. Because it + * uses the bootstrap key it is readable even by a master holding a + * DIFFERENT session key, which is how a split brain between independently + * bootstrapped partitions is detected and merged. Payload: member count. + * + * NODE STATE MACHINE + * ------------------ + * CL_CTR_NODE_NEW --> (MEMBER_LIST or KEY_GRANT received) --> CL_CTR_NODE_ACTIVE + * --> (join_deadline expired) --> CL_CTR_NODE_ACTIVE + * + * CL_CTR_NODE_NEW: receive only; do NOT send ALIVE or MASTER_ALIVE. + * CL_CTR_NODE_ACTIVE: send ALIVE every query_time seconds. + * Master also sends MASTER_ALIVE every CL_CTR_MASTER_KA_INTERVAL s. + * + * MASTER ELECTION + * --------------- + * Three roles per cluster: MASTER (active coordinator), BACKUP (standby, always + * the highest-IP non-master) and MEMBER. Election uses a quantized window so + * all nodes evaluate identical peer sets and reach the same result + * deterministically. No NTP synchronisation required. + * + * master_stickiness (modparam, default 1): a live master keeps the role - a + * higher-IP node that joins becomes the BACKUP instead of preempting the + * master, so handovers are minimised. With master_stickiness=0 the highest-IP + * node always becomes master. + * + * Fast failure detection: MASTER_ALIVE at 1 s; 3 s timeout. On master failure + * the silent master is aged out of the election window and the BACKUP + * (highest-IP survivor) is promoted immediately - it already holds the + * preserved session key, so there is no re-keying and no re-JOIN cycle. + * Graceful handoff: KEY_HANDOFF + GOODBYE before shutdown. + * + * SPLIT-BRAIN HANDLING (three layers) + * ----------------------------------- + * 1. Prevention at join time: simultaneously-starting nodes see each other's + * JOIN_REQs, so at the join deadline a node that has seen a higher-IP + * starter DEFERS self-promotion (bounded) and joins that node instead of + * everyone becoming an independent-key lone master. + * 2. Same-key yield: two masters that share a session key see each other's + * MASTER_ALIVE; the lower-IP one yields (see MASTER_ALIVE above). + * 3. Divergent-key merge: masters that DO NOT share a session key cannot read + * each other's MASTER_ALIVE, so each emits a bootstrap-key MASTER_BEACON. + * On hearing a superior beacon (larger member count, ties broken by higher + * IP) a node re-joins that master and adopts its key. + * + * SHARING TAGS (manage_shtags, default 1) + * --------------------------------------- + * The controller drives clusterer sharing tags: normally the MASTER is the sole + * active holder and every other node is backup. An operator can override this + * with the cl_ctr_shtag_force MI command (pin the active tag to a chosen node) and + * revert with cl_ctr_shtag_auto; the override is carried in MEMBER_LIST, survives + * master fail-over, and auto-clears if the forced node departs. cl_ctr_list_config + * reports the current mode (auto / override:). + * + * ========================================================================= + */ + +#include +#include +#include /* strcasecmp() */ +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include /* O_NONBLOCK, fcntl() */ +#include /* getifaddrs(), freeifaddrs() */ +#include /* IF_NAMESIZE, struct ifreq, SO_BINDTODEVICE */ + +#include "../../sr_module.h" /* module_exports, MODULE_VERSION, proc_export_t, + PROC_FLAG_*, dep_export_t, DEP_ABORT, + param_export_t, STR_PARAM, INT_PARAM */ +#include "../../dprint.h" /* LM_ERR / LM_WARN / LM_INFO / LM_DBG */ +#include "../../mem/shm_mem.h" /* shm_malloc / shm_free */ +#include "../../locking.h" /* gen_lock_t - base spinlock primitive */ +#include "../../rw_locking.h" /* rw_lock_t - reader-writer lock built on top */ +#include "../../mi/mi.h" /* mi_export_t, mi_response_t, MI helpers */ +#include "../../timer.h" /* get_uticks(), utime_t - us since start */ +#include "../../socket_info.h" /* struct socket_info, PROTO_BIN */ +#include "../../net/api_proto.h" /* protos[] array */ +#include "../../globals.h" /* process_no - this process's index */ +#include "../../ipc.h" /* ipc_send_rpc() - cross-process job dispatch */ +#include "../../pvar.h" /* pv_export_t - read-only $cl_ctr_* variables */ + +#include "../clusterer/clusterer_ctrl.h" /* set_my_identity, add_node, remove_node */ + +#include /* timerfd_create(), timerfd_settime() */ +#include "../../reactor_proc.h" /* reactor_proc_init/add_fd/loop */ + +/* All cryptography is provided by libsodium, which is a hard requirement (the + * module Makefile fails the build if it is not found). The payload AEAD is + * XChaCha20-Poly1305 (192-bit nonce -> no random-nonce collision worry, even for + * the static bootstrap key), the bootstrap KDF is Argon2id, the join handshake + * is Noise_NNpsk0_25519_ChaChaPoly_SHA256, and X25519 / HKDF-SHA256 / RNG also + * come from libsodium. */ +#include +#define CL_CTR_CRYPTO_SUITE "XChaCha20-Poly1305 + Argon2id + Noise_NNpsk0" + +/* ========================================================================= + * Wire-format constants + * ========================================================================= */ + +/* 2-byte wire magic - a cleartext key-selector at the start of every packet + * (it must be readable before decryption to choose bootstrap vs session key). + * Both share the 0xCC prefix; the second byte distinguishes the key tier. + * Only a sanity/routing tag on a dedicated multicast group:port - the real + * confidentiality and integrity come from the XChaCha20-Poly1305 payload. */ +#define CL_CTR_MAGIC_SZ 2 +static const unsigned char CL_CTR_PACKET_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x00 }; +static const unsigned char CL_CTR_BOOTSTRAP_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x01 }; + +/* Packet type bytes */ +#define CL_CTR_PKT_ALIVE 0x01 +#define CL_CTR_PKT_JOIN_REQ 0x02 +#define CL_CTR_PKT_MEMBER_LIST 0x03 /* master -> joining node: here is the cluster */ +#define CL_CTR_PKT_GOODBYE 0x04 /* graceful shutdown notification */ +#define CL_CTR_PKT_NODE_ASSIGN 0x05 /* master -> multicast: here is your node_id */ +#define CL_CTR_PKT_MASTER_ALIVE 0x06 /* master-only keepalive; ~1 s interval */ +#define CL_CTR_PKT_KEY_GRANT 0x07 /* master -> joiner: ECDH-wrapped master_salt */ +#define CL_CTR_PKT_KEY_HANDOFF 0x08 /* outgoing master -> next master: salt handoff */ +#define CL_CTR_PKT_JOIN_REJECT 0x09 /* master -> joiner: authentication rejected */ +#define CL_CTR_PKT_ACK 0x0B /* receiver -> sender: ack of a 1:1 handshake pkt */ +#define CL_CTR_PKT_RESYNC 0x0C /* member -> master: my view differs, resend state */ +#define CL_CTR_PKT_MASTER_BEACON 0x0A /* master-only announce (BOOTSTRAP key) so + * masters with divergent session keys can + * still discover each other and merge a + * split brain; payload = member count 2B BE */ + +/* Number of consecutive bootstrap-decrypt failures before a JOIN_REJECT is sent */ +#define CL_CTR_JOIN_FAIL_LIMIT 3 +#define CL_CTR_JOIN_FAIL_TABLE_SZ 8 /* max simultaneous rejected IPs tracked by master */ + +#define CL_CTR_MAX_IP_LEN 15 /* "255.255.255.255" without NUL */ +#define CL_CTR_PUBKEY_SZ 32 /* X25519 public key */ +#define CL_CTR_MASTER_SALT_SZ 32 /* random salt generated by each new master */ +/* MEMBER_LIST entry: IP (16B null-padded) + is_master (1B) = 17B. + * Pubkeys are NOT carried here - nodes learn them from ALIVE packets, + * keeping MEMBER_LIST small enough to avoid excessive IP fragmentation. */ +#define CL_CTR_IP_ENTRY_SZ 17 +#define CL_CTR_LIST_COUNT_SZ 2 /* MEMBER_LIST count field: uint16_t BE */ +#define CL_CTR_NODE_ID_SZ 2 /* uint16_t node_id, big-endian */ +#define CL_CTR_MAX_BIN_SOCKETS 8 /* max BIN listeners per node */ +#define CL_CTR_MAX_BIN_SOCK_LEN 64 /* "bin:255.255.255.255:65535" = 26 chars */ +/* BIN info block: [bin_count 1B][sock1 NUL-term]...[sockN NUL-term] */ +#define CL_CTR_BIN_INFO_MAX_SZ (1 + CL_CTR_MAX_BIN_SOCKETS * CL_CTR_MAX_BIN_SOCK_LEN) + +/* AEAD encryption constants + * wire: [magic 2B][cluster_id 2B BE][nonce 24B][ciphertext][tag 16B] + * plaintext: [type 1B][seq 4B][payload] + * The cluster_id is cleartext (like the magic) so a node can drop packets that + * belong to a different cluster sharing the same multicast group WITHOUT its + * key - before decryption, so foreign traffic never counts as an auth failure. */ +#define CL_CTR_NONCE_SZ 24 /* XChaCha20-Poly1305 nonce (192-bit) */ +#define CL_CTR_TAG_SZ 16 /* Poly1305 AEAD tag */ +#define CL_CTR_SEQ_SZ 4 /* uint32_t monotonic sequence in plaintext */ +#define CL_CTR_CLUSTER_ID_SZ 2 /* cleartext uint16 cluster_id (BE) selector */ +#define CL_CTR_NONCE_OFF (CL_CTR_MAGIC_SZ + CL_CTR_CLUSTER_ID_SZ) /* nonce starts here */ +#define CL_CTR_WIRE_HDR_SZ (CL_CTR_MAGIC_SZ + CL_CTR_CLUSTER_ID_SZ + CL_CTR_NONCE_SZ) /* 28 */ +#define CL_CTR_PLAIN_HDR_SZ (1 + CL_CTR_SEQ_SZ) /* type + seq = 5 */ + +/* Bootstrap-key hardening: the join/admission key (also the Noise PSK) is + * derived from the shared password with Argon2id (memory-hard) instead of a + * single SHA-256, so a password captured from a JOIN_REQ cannot be brute-forced + * cheaply offline. Derived ONCE in mod_init (main process, before fork), so the + * ~64 MiB working set is a transient startup cost; workers inherit the 32-byte + * key. Fixed parameters so every node derives the same key. */ +#define CL_CTR_ARGON2_OPSLIMIT 3UL +#define CL_CTR_ARGON2_MEMLIMIT (64UL * 1024 * 1024) +/* Minimum estimated password entropy (bits) before a startup warning fires. */ +#define CL_CTR_MIN_PASSWORD_BITS 80 +#define CL_CTR_DEFAULT_PASSWORD "3eCrEt*5629" /* insecure placeholder; warn if used */ + +/* Master keepalive: master sends CL_CTR_PKT_MASTER_ALIVE every CL_CTR_MASTER_KA_INTERVAL + * seconds. Peers declare master dead after CL_CTR_MASTER_KA_MISSED missed packets. */ +#define CL_CTR_MASTER_KA_INTERVAL 1 /* seconds between MASTER_ALIVE sends */ +#define CL_CTR_MASTER_KA_MISSED 3 /* missed keepalives before re-election */ +#define CL_CTR_MASTER_KA_TIMEOUT (CL_CTR_MASTER_KA_INTERVAL * CL_CTR_MASTER_KA_MISSED) + +/* Split-brain merge: a master emits a CL_CTR_PKT_MASTER_BEACON (bootstrap key) once + * every CL_CTR_MASTER_BEACON_EVERY MASTER_ALIVE ticks. This is the only traffic two + * masters with divergent session keys can both read, so it bounds split-brain + * convergence to ~CL_CTR_MASTER_BEACON_EVERY seconds while keeping bootstrap-key use + * (and thus exposure) rare compared with the 1 s session keepalive. */ +#define CL_CTR_MASTER_BEACON_EVERY 5 /* MASTER_ALIVE ticks between beacons (~5 s) */ + +/* Split-brain PREVENTION at join time. When several nodes cold-start together + * they all exchange (bootstrap-decryptable) JOIN_REQs, so each learns the other + * starters. At the join deadline a node that has seen a higher-IP starter does + * NOT self-promote; it defers (re-sending JOIN_REQ) so the highest-IP starter + * becomes the single master and everyone joins it - no divergent keys ever form. + * Bounded so a higher-IP node that heard-then-died cannot stall us forever. */ +#define CL_CTR_JOIN_DEFER_SECS 1 /* seconds per deferral round */ +#define CL_CTR_JOIN_DEFER_MAX 4 /* consecutive deferrals for a *silent* */ + /* higher-IP peer before we promote anyway */ +#define CL_CTR_JOIN_DEFER_HARDMAX 20 /* absolute cap on deferrals incl. resets - */ + /* a peer stuck joining can't stall forever */ +#define CL_CTR_JOIN_REQ_MIN_US 500000 /* min microseconds between JOIN_REQ sends */ + +/* + * Reliable delivery (ACK + bounded retransmit) for the strictly 1:1 packets of + * the join handshake - KEY_GRANT today, the joiner's NODE_ASSIGN burst next. + * The receiver ACKs each one (CL_CTR_PKT_ACK, echoing the packet's seq); the + * sender retransmits the cached bytes until ACKed or the budget runs out. + * + * Multicast/group packets are deliberately NOT ACKed - ACKing from every member + * is ACK implosion; their loss is healed by the periodic membership digest. + * + * The budget is pinned to the join timers so master-side ARQ nests inside one + * JOIN_REQ retry cycle: CL_CTR_RETX_MAX_RETRIES x CL_CTR_RETX_INTERVAL_US of + * retransmit (750 ms), then give up and let the joiner's own JOIN_REQ retry + * (~1 s) restart the handshake with fresh packets. ACKs are ONLY retransmit + * accounting - a never-ACKed packet is dropped, never gating a join or any + * cluster state, so a lost ACK or a dead joiner can never stall anything. + */ +#define CL_CTR_RETX_INTERVAL_US (CL_CTR_JOIN_REQ_MIN_US / 2) /* 250 ms */ +#define CL_CTR_RETX_MAX_RETRIES 3 /* then give up */ +#define CL_CTR_RETX_QUEUE_SZ 128 /* max outstanding unacked 1:1 packets */ +#if (CL_CTR_RETX_MAX_RETRIES * CL_CTR_RETX_INTERVAL_US) >= (CL_CTR_JOIN_DEFER_SECS * 1000000) +#error "retransmit budget must stay shorter than the JOIN_REQ retry interval" +#endif + +/* + * Membership digest, carried in every MASTER_ALIVE so a member can notice it is + * out of sync and pull a resend. A peer's node_id/BIN mapping travels only in + * NODE_ASSIGN, which is multicast (best-effort) - a member that dropped one + * knows the peer exists (from its ALIVE) but not its node_id, so its clusterer + * BIN mesh to that peer is incomplete with no other repair. The digest closes + * that: [member_count 2B BE][set_hash 8B BE] over the active peers; a mismatch + * triggers a rate-limited RESYNC, to which the master re-broadcasts the full + * NODE_ASSIGN set + MEMBER_LIST. Multicast/group packets stay unacked - this + * is their anti-entropy repair, the counterpart to the 1:1 ACK path. + */ +#define CL_CTR_DIGEST_SZ (2 + 8) +#define CL_CTR_RESYNC_MIN_US 1000000 /* member: min between RESYNCs; master: */ + /* min between full-state re-broadcasts (1 s) */ +/* + * Liveness bitmap relayed by the master in every MASTER_ALIVE: one bit per + * node_id. A settled non-master unicasts its ALIVE to the master and no longer + * hears peers' ALIVEs directly, so the master reports which node_ids it has seen + * within the election window and non-masters refresh those peers' last_seen from + * it - turning the old all-to-all O(N^2) liveness gossip into O(N). Built from + * the ELECTION-window cutoff (not the purge window) so a departed node drops out + * of the bitmap on the same schedule it drops out of an election. Sized to + * cover node_id 1..CL_CTR_MAX_PEERS. + */ +#define CL_CTR_ALIVE_BITMAP_SZ ((CL_CTR_MAX_PEERS / 8) + 1) + +/* Per-source-IP rate limiter: checked before decryption to shed floods cheaply. + * Tracks up to CL_CTR_RATE_TBL_SZ source IPs with a 1-second sliding window. */ +#define CL_CTR_RATE_TBL_SZ 256 /* one slot per peer; matches max cluster size */ +#define CL_CTR_RATE_LIMIT 20 /* max packets per second per source IP */ + +typedef struct { + uint32_t ip; /* network byte order; 0 = empty slot */ + time_t window_start; + int count; +} cl_ctr_rate_entry_t; + +/* Max packet sizes: wire(20) = magic(8) + nonce(12); plain(5) = type(1) + seq(4) + * MASTER_ALIVE : wire(20) + plain(5) + tag(16) = 41 bytes + * ALIVE : wire(20) + plain(5) + IP(16) + pubkey(32) + tag(16) = 89 bytes + * GOODBYE : wire(20) + plain(5) + IP(16) + tag(16) = 57 bytes + * KEY_GRANT : wire(20) + plain(5) + IP(16) + master_pubkey(32) + * + join_nonce(16) + wrapped_salt(32) + tag(16) = 137 bytes + * KEY_HANDOFF : wire(20) + plain(5) + IP(16) + sender_pubkey(32) + * + wrapped_salt(32) + tag(16) = 121 bytes + * JOIN_REQ max : wire(20) + plain(5) + IP(16) + bin_info(513) + pubkey(32) + * + join_nonce(16) + tag(16) = 618 bytes + * NODE_ASSIGN max : wire(20) + plain(5) + node_id(2) + IP(16) + bin_info(513) + * + tag(16) = 572 bytes + * MEMBER_LIST max : wire(20) + plain(5) + count(2) + 256x17 + tag(16) = 4395 bytes + * + * Pubkeys are distributed via ALIVE (89 bytes) rather than MEMBER_LIST so that + * MEMBER_LIST payload stays bounded to 256x17=4352 bytes max. + * + * Fragmentation note: all packets except MEMBER_LIST fit in a single datagram + * on any standard link (Ethernet 1472B payload budget). MEMBER_LIST at 4395B + * requires IP fragmentation: + * - Standard Ethernet (1500 MTU): 3 fragments + * - IPsec ESP tunnel (~1400 MTU): 4 fragments + * - GRE-over-IPsec (~1350 MTU): 4-5 fragments + * Firewalls that block fragmented UDP packets will silently drop MEMBER_LIST, + * preventing new nodes from joining. The DF bit is not set so fragmentation + * occurs transparently where the network allows it. */ +#define CL_CTR_SMALL_PKT_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 + CL_CTR_TAG_SZ) +/* Consistency-critical settings advertised in ALIVE so peers can detect + * accidental per-node config drift for the same cluster: + * manage_shtags(1B) + master_stickiness(1B) + query_time(2B BE). */ +#define CL_CTR_CONFIG_SZ 4 +/* Noise handshake message sizes (NNpsk0, X25519, ChaChaPoly, SHA-256): + * msg 1 = e(32) + tag over empty payload(16) = 48 + * msg 2 = e(32) + AEAD(master_salt 32 + tag 16) = 80 */ +#define CL_CTR_NOISE_MSG1_SZ (32 + CL_CTR_TAG_SZ) +#define CL_CTR_NOISE_MSG2_SZ (32 + CL_CTR_MASTER_SALT_SZ + CL_CTR_TAG_SZ) +/* JOIN_REQ: [ip NUL][bin_count 1B][sockets...][noise_msg1 48B][config 4B] */ +#define CL_CTR_JOIN_PKT_MAX_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 \ + + CL_CTR_BIN_INFO_MAX_SZ + CL_CTR_NOISE_MSG1_SZ \ + + CL_CTR_CONFIG_SZ + CL_CTR_TAG_SZ) +/* KEY_GRANT: [target_ip NUL][noise_msg2 80B] */ +#define CL_CTR_KEY_GRANT_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 \ + + CL_CTR_NOISE_MSG2_SZ + CL_CTR_TAG_SZ) +/* KEY_HANDOFF: [target_ip NUL][crypto_box_seal(master_salt)] */ +#define CL_CTR_KEY_HANDOFF_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 \ + + crypto_box_SEALBYTES + CL_CTR_MASTER_SALT_SZ + CL_CTR_TAG_SZ) +/* NODE_ASSIGN: [node_id 2B][ip NUL][bin_count 1B][sockets...] */ +#define CL_CTR_NODE_ASSIGN_MAX_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_NODE_ID_SZ \ + + CL_CTR_MAX_IP_LEN + 1 + CL_CTR_BIN_INFO_MAX_SZ + CL_CTR_TAG_SZ) +#define CL_CTR_LIST_PKT_MAX_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_LIST_COUNT_SZ \ + + CL_CTR_NODE_ID_SZ /* forced-shtag node_id */ \ + + CL_CTR_MAX_PEERS * CL_CTR_IP_ENTRY_SZ + CL_CTR_TAG_SZ) +/* Large enough to receive a fully reassembled UDP datagram (max 65507 bytes) */ +#define CL_CTR_RECV_BUF_SZ 65536 + +/* ========================================================================= + * Peer-table constants + * ========================================================================= */ + +#define CL_CTR_MAX_PEERS 256 + +/* + * CL_CTR_ELECT_FACTOR - election window = query_time x CL_CTR_ELECT_FACTOR. + * QUANTIZED: all nodes evaluate the same cutoff simultaneously. + * + * CL_CTR_PURGE_FACTOR - memory-cleanup window = query_time x CL_CTR_PURGE_FACTOR. + * Not quantized; only affects when entries are freed, not who is elected. + */ +#define CL_CTR_ELECT_FACTOR 3 +#define CL_CTR_PURGE_FACTOR 6 + +/* ========================================================================= + * Node-join state machine + * ========================================================================= */ + +typedef enum { + CL_CTR_NODE_NEW = 0, /* sent JOIN_REQ, awaiting MEMBER_LIST or timeout */ + CL_CTR_NODE_ACTIVE = 1 /* fully participating, sends ALIVE */ +} cl_ctr_node_state_t; + +/* Planned: node maintenance mode (see the module documentation roadmap). + * + * A CL_CTR_NODE_MAINTENANCE state would let a node be drained without leaving the + * cluster: never elected master, excluded from cl_ctr_elect_master (advertised in + * the ALIVE payload so peers drop it from the election window without waiting + * for a MEMBER_LIST refresh), and never the active sharing-tag holder even + * with manage_shtags=1. It is local policy, so it would live in cl_ctr_cluster_t + * (not the shared peer table) and survive MEMBER_LIST resets, toggled by a + * local MI command. + * + * It would build on what is already here: the sharing-tag override + * (cl_ctr_shtag_force / cl_ctr_shtag_auto) would auto-clear when the forced + * node enters maintenance, and cl_ctr_list_members / cl_ctr_list_config would + * gain a per-node maintenance indicator alongside the existing shtag_mode. + */ + +/* ========================================================================= + * Cluster table + * ========================================================================= */ + +#define CL_CTR_MAX_CLUSTERS 16 /* max cluster= entries */ + +/* Forward-declared so cl_ctr_cluster_t can embed a pointer */ +typedef struct cl_ctr_peers_ cl_ctr_peers_t; + +/* Noise handshake state (definitions with the Noise core further below). + * HASHLEN = 32 (SHA-256). cl_ctr_cluster_t embeds a cl_ctr_symstate_t for the + * initiator half of the join handshake. */ +#define CL_CTR_NOISE_HASHLEN 32 +typedef struct { unsigned char k[32]; uint64_t n; int has_key; } cl_ctr_cipherstate_t; +typedef struct { + unsigned char ck[CL_CTR_NOISE_HASHLEN]; + unsigned char h[CL_CTR_NOISE_HASHLEN]; + cl_ctr_cipherstate_t cs; +} cl_ctr_symstate_t; + +/* This node's own role in a cluster. Tracked across elections so + * cl_ctr_elect_master() can log an explicit transition (member->backup, + * backup->master, master->member, ...) whenever it changes. */ +enum cl_ctr_role { CL_CTR_ROLE_MEMBER = 0, CL_CTR_ROLE_BACKUP, CL_CTR_ROLE_MASTER }; + +static inline const char *cl_ctr_role_name(int r) +{ + switch (r) { + case CL_CTR_ROLE_MASTER: return "master"; + case CL_CTR_ROLE_BACKUP: return "backup"; + default: return "member"; + } +} + +/* + * One outstanding 1:1 handshake packet awaiting an ACK. The full sealed bytes + * are cached so a retransmit is a single sendto() with no re-encryption: the + * same seq/nonce is resent, which a peer that already got it (and ACKed) will + * not see again, while one that lost it still has an older last_seq and accepts + * it. Worker-local (only the controller worker touches the queue), so no lock. + */ +typedef struct { + int used; + uint32_t seq; /* packet seq, echoed by the ACK */ + unsigned char type; /* packet type, for logging */ + int retries_left; + utime_t next_due_us; /* get_uticks() deadline for next send */ + struct sockaddr_storage dest; /* unicast destination */ + socklen_t destlen; + int pkt_len; /* sealed length */ + unsigned char pkt[CL_CTR_NODE_ASSIGN_MAX_SZ]; /* cached sealed bytes */ +} cl_ctr_retx_entry_t; + +/** + * cl_ctr_cluster_t - per-cluster runtime state. + * One instance per "cluster" modparam; one worker process per instance. + */ +typedef struct cl_ctr_cluster_ { + int cluster_id; + char multicast_address[INET_ADDRSTRLEN]; + int multicast_port; + struct sockaddr_in mcast_dest; /* resolved once in cl_ctr_setup_socket */ + char password[1025]; + unsigned char key[32]; /* bootstrap key = SHA256(password); JOIN only */ + unsigned char session_key[32]; /* group key = HKDF(password, master_salt) */ + int manage_shtags; /* per-cluster override; defaults to global manage_shtags */ + int master_stickiness; /* per-cluster override; -1 = inherit global */ + cl_ctr_peers_t *peers; /* per-cluster peer table in shm */ + /* BIN socket resolved at mod_init - advertised in JOIN_REQ/NODE_ASSIGN */ + char bin_socket[CL_CTR_MAX_BIN_SOCK_LEN]; /* "bin:IP:PORT" */ + /* Worker-process fds and state - valid only inside cl_ctr_worker after fork */ + int sock; /* multicast UDP socket */ + int alive_tfd; /* periodic ALIVE timer */ + int join_tfd; /* one-shot join deadline */ + int rejoin_tfd; /* 1-second JOIN_REQ retry */ + int master_alive_tfd; /* master sends MASTER_ALIVE 1/s */ + int master_dead_tfd; /* non-master: fires on ka miss */ + int identity_registered; /* 1 once update_identity called */ + int shtag_bootstrapped; /* -1 = eligible, 1 = done */ + /* Long-lived X25519 keypair - generated in cl_ctr_worker after fork, never + * leaves the process. Advertised in ALIVE and used by crypto_box KEY_HANDOFF + * (NOT the Noise join handshake, which uses a fresh ephemeral per attempt). */ + unsigned char my_privkey[CL_CTR_PUBKEY_SZ]; + unsigned char my_pubkey[CL_CTR_PUBKEY_SZ]; + /* Noise (initiator) handshake state carried between our JOIN_REQ (msg 1) and + * the master's KEY_GRANT (msg 2): the SymmetricState and the fresh ephemeral + * private key. A new JOIN_REQ overwrites both, so a KEY_GRANT for a + * superseded attempt simply fails to decrypt and is dropped. Worker-local. */ + cl_ctr_symstate_t noise_hs; + unsigned char noise_e_priv[32]; + int noise_hs_valid; /* 1 after a JOIN_REQ, until KEY_GRANT/reset */ + /* Set while a re-key JOIN_REQ is in flight; cleared on KEY_GRANT success + * or master transition to prevent nonce stomping under packet flood. */ + int join_pending; + /* 1 once a valid session_key has been established - either generated at + * cluster bootstrap (cl_ctr_on_became_master) or adopted from the current master + * master via KEY_GRANT / KEY_HANDOFF. A node must NOT act as master + * (broadcast MASTER_ALIVE) while this is 0, or it would encrypt with an + * underived key that no member can decrypt. */ + int have_session_key; + /* Reliable-delivery queue for 1:1 handshake packets (worker-local, no lock). + * retx_tfd sweeps it every CL_CTR_RETX_INTERVAL_US while retx_count > 0. */ + int retx_tfd; + int retx_count; + cl_ctr_retx_entry_t retx_q[CL_CTR_RETX_QUEUE_SZ]; + /* Membership-digest resync throttles (worker-local): a member sends at most + * one RESYNC, and a master re-broadcasts full state at most once, per + * CL_CTR_RESYNC_MIN_US. */ + utime_t last_resync_us; + utime_t last_full_state_us; + /* Master-side per-IP table tracking bootstrap-decrypt failures. + * Worker-local (no shm, no lock needed). After CL_CTR_JOIN_FAIL_LIMIT + * failures from the same source IP the master sends JOIN_REJECT. */ + struct { + uint32_t ip_num; + char ip[CL_CTR_MAX_IP_LEN + 1]; + int count; + int rejected; /* 1 = JOIN_REJECT already sent; suppress repeats */ + } join_fail_tbl[CL_CTR_JOIN_FAIL_TABLE_SZ]; + /* Joiner-side auth-failure detection - no lock needed (worker-local). */ + int bootstrap_auth_fails; /* consecutive bootstrap decrypt failures + during CL_CTR_NODE_NEW; reset on KEY_GRANT */ + int join_attempt_count; /* rejoin_tfd fires since last KEY_GRANT */ + /* Count of packets from OTHER peers during CL_CTR_NODE_NEW that we could not + * decrypt (any magic). Non-zero means a cluster (or rogue) whose key we do + * not share exists on the group - evidence we may have the wrong password. + * This is only *evidence*: cl_ctr_on_join_tfd never self-terminates on it + * immediately (that would let start-up noise or a flood kill a healthy + * node); it defers and re-joins, and a KEY_GRANT resets this counter. Only + * a correct-password joiner ever receives a KEY_GRANT, so persistence of + * this counter across the whole defer budget is what marks a real + * wrong-password / foreign-cluster condition. */ + int auth_fail_pkts; + /* Number of join rounds we have already deferred because we still could not + * authenticate. We only give up (shut down) after CL_CTR_JOIN_DEFER_MAX such + * rounds, so a KEY_GRANT that is merely slow, or a brief burst of start-up + * noise or crafted garbage, never self-terminates a correctly-configured + * node. Reset on successful authentication. */ + int auth_defer_count; + /* master_salt lives in cl->peers->master_salt (shm) so mod_destroy can + * read it. session_key is the worker-local derived key cache. */ + /* Per-source-IP rate limiter table - pkg_malloc'd in cl_ctr_worker after fork */ + cl_ctr_rate_entry_t *rate_tbl; + /* Last shtag decision this worker applied, so cl_ctr_apply_shtags_decision() + * logs the *reason* only when the decision (or its cause) actually + * changes - not on every idempotent re-apply. Worker-local. + * shtag_last_active: -1 unknown, 0 backup, 1 active. + * shtag_last_forced: the forced node_id in effect at that time. */ + int shtag_last_active; + uint16_t shtag_last_forced; + /* Counts MASTER_ALIVE ticks so a beacon is emitted every + * CL_CTR_MASTER_BEACON_EVERY of them. Worker-local (master path only). */ + unsigned int beacon_tick; + /* How many times we have deferred self-promotion at the join deadline + * because a higher-IP node was also still joining (split-brain + * prevention). Reset to 0 whenever a fresh JOIN_REQ from a higher-IP peer + * arrives (it is demonstrably still alive and joining, so we keep waiting + * for it rather than self-promoting into a divergent-key split brain). + * Worker-local; reset once we leave the NEW state. */ + int join_defer_count; + /* Total deferrals across resets - an absolute cap (CL_CTR_JOIN_DEFER_HARDMAX) + * so a peer that keeps sending JOIN_REQ yet never becomes master cannot + * defer us forever. Worker-local; reset once we leave the NEW state. */ + int join_defer_total; + /* utime (us since start) of the last JOIN_REQ we transmitted, for a + * minimum-interval throttle so a key-mismatch/split-brain burst cannot + * flood the group with JOIN_REQs. 0 = never sent. Worker-local. */ + utime_t last_join_req_utime; + /* 1 while our MASTER_ALIVE keepalive timer is armed (i.e. we are acting as + * master and broadcasting). Set by cl_ctr_arm_master_timers(). Lets + * cl_ctr_elect_master() enforce the invariant "keepalive armed <=> I am the + * elected master": an election that demotes us (clears is_master) without + * going through a yield/member-list path must still stop the keepalive, + * otherwise a demoted node keeps broadcasting MASTER_ALIVE and lower-IP + * peers oscillate between two masters. Worker-local. */ + int master_ka_armed; + /* This node's last-known role in the cluster (enum cl_ctr_role). Compared in + * cl_ctr_elect_master() to emit a one-line transition log on every change. + * Zero-initialised to CL_CTR_ROLE_MEMBER, which matches a not-yet-joined node. + * Worker-local. */ + int my_role; +} cl_ctr_cluster_t; + +static cl_ctr_cluster_t cl_ctr_clusters[CL_CTR_MAX_CLUSTERS]; +static int cl_ctr_cluster_count = 0; + +/* Raw "cluster" strings collected during modparam parsing */ +static char *cl_ctr_cluster_strs[CL_CTR_MAX_CLUSTERS]; +static int cl_ctr_cluster_str_count = 0; + +/* ========================================================================= + * Module parameters + * ========================================================================= */ + +/* Global modparams - apply to all clusters unless overridden per-cluster */ +static char *my_ip = NULL; /* explicit IP, or NULL for auto-detect */ +static char *my_interface = NULL; /* explicit interface name, or NULL */ +static int query_time = 5; +static char *password = CL_CTR_DEFAULT_PASSWORD; /* default; falls back per cluster */ + +/* Policy when a node's consistency-critical settings (manage_shtags/ + * master_stickiness/query_time) differ from the running cluster (a master is + * alive). Set via the on_config_mismatch modparam string: + * "warn" - admit/keep the node but log a CONFIG MISMATCH warning; + * "reject" - the master refuses the join (JOIN_REJECT) and the node shuts + * down with a clear message (default); + * "adopt" - the node adopts the master's (authoritative) settings at + * runtime and continues. */ +#define CL_CTR_CFGMISMATCH_WARN 0 +#define CL_CTR_CFGMISMATCH_REJECT 1 +#define CL_CTR_CFGMISMATCH_ADOPT 2 +/* JOIN_REJECT reason codes (1 byte after the target IP in the payload). */ +#define CL_CTR_REJECT_GENERIC 0 /* wrong password / unauthorized / table full */ +#define CL_CTR_REJECT_CONFIG 1 /* different cluster settings (reject policy) */ +static char *on_config_mismatch_s = NULL; /* raw modparam string */ +static int on_config_mismatch = CL_CTR_CFGMISMATCH_REJECT; /* resolved; default reject */ + +/* Resolved at mod_init time - always valid after cl_ctr_resolve_local_identity() */ +static char my_ip_buf[INET_ADDRSTRLEN]; +static char my_interface_buf[IF_NAMESIZE]; + + +/* Local node identity - populated at mod_init by scanning the config file */ +static uint16_t my_node_id = 0; + +/* clusterer integration - loaded at mod_init if clusterer use_controller=1 */ +static clusterer_ctrl_binds_t clctl; +static int clctl_loaded = 0; +static int manage_shtags = 1; +/* master_stickiness (global default; per-cluster override via "cluster" string): + * 1 (default) = the master is "sticky": a live master keeps the role and is + * NOT displaced when a higher-IP node joins. The highest-IP + * non-master is designated BACKUP and takes over only when the + * master fails. A higher-IP joiner just replaces the backup. + * Result: fewer master handovers. + * 0 = not sticky - pure highest-IP election, so a higher-IP node + * takes over as master as soon as it appears (more handovers). */ +static int master_stickiness = 1; +static char my_bin_sockets[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; +static int my_bin_count = 0; + + +/** + * cl_ctr_add_cluster_param() - collect "cluster" modparam strings. + * Actual parsing happens in mod_init() after all params are set. + */ +static int cl_ctr_add_cluster_param(modparam_t type, void *val) +{ + if (cl_ctr_cluster_str_count >= CL_CTR_MAX_CLUSTERS) { + LM_ERR("clusterer_controller: too many clusters (max %d)\n", + CL_CTR_MAX_CLUSTERS); + return -1; + } + { + size_t _len = strlen((char *)val) + 1; + cl_ctr_cluster_strs[cl_ctr_cluster_str_count] = pkg_malloc(_len); + if (!cl_ctr_cluster_strs[cl_ctr_cluster_str_count]) { + LM_ERR("clusterer_controller: pkg_malloc failed\n"); + return -1; + } + memcpy(cl_ctr_cluster_strs[cl_ctr_cluster_str_count], (char *)val, _len); + } + cl_ctr_cluster_str_count++; + return 0; +} + +static const param_export_t params[] = { + {"cluster", STR_PARAM | USE_FUNC_PARAM, (void *)cl_ctr_add_cluster_param}, + {"my_ip", STR_PARAM, &my_ip}, + {"interface", STR_PARAM, &my_interface}, + {"query_time", INT_PARAM, &query_time}, + {"password", STR_PARAM, &password}, + {"manage_shtags", INT_PARAM, &manage_shtags}, + {"master_stickiness", INT_PARAM, &master_stickiness}, + {"on_config_mismatch", STR_PARAM, &on_config_mismatch_s}, + {0, 0, 0} +}; + +/* ========================================================================= + * Peer table (shared memory) + * ========================================================================= */ + +typedef struct cl_ctr_peer_ { + char ip[CL_CTR_MAX_IP_LEN + 1]; + unsigned int ip_num; + time_t last_seen; + int is_master; + int is_backup; /* 1 = standby master (highest-IP non-master) */ + int in_election; /* 1 = currently inside the election window */ + uint16_t node_id; /* allocated by master; 0 = not yet assigned */ + uint8_t bin_count; /* number of BIN listeners reported */ + char bin_sockets[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; + unsigned char pubkey[CL_CTR_PUBKEY_SZ]; /* long-lived X25519 pubkey (from ALIVE); + zero if unknown; used for KEY_HANDOFF */ + uint32_t last_seq; /* highest seq accepted from this peer */ + /* Peer's advertised consistency-critical config (from ALIVE), used to warn + * on accidental per-node config drift. cfg_known=0 until first advertised; + * cfg_warned deduplicates the mismatch warning. */ + int cfg_known; + int cfg_manage_shtags; + int cfg_master_stickiness; + int cfg_query_time; + int cfg_warned; +} cl_ctr_peer_t; + +struct cl_ctr_peers_ { + cl_ctr_peer_t entries[CL_CTR_MAX_PEERS]; + int count; + /* rw_lock_t allows concurrent readers (MI, future script functions) + * while still serialising the single writer (cl_ctr_worker). */ + rw_lock_t *lock; + cl_ctr_node_state_t node_state; + time_t join_deadline; + /* last elected master IP - used to detect and log master changes */ + char last_master[CL_CTR_MAX_IP_LEN + 1]; + /* master_salt: generated by each new master, shared here so mod_destroy + * (running in main process) can derive session_key for GOODBYE. */ + unsigned char master_salt[CL_CTR_MASTER_SALT_SZ]; + /* my_seq: monotonic send counter; in shm so mod_destroy can use it for + * GOODBYE without needing the worker's private state. Reset to 0 on + * every session key rotation so last_seq counters reset cleanly. */ + uint32_t my_seq; + /* Sharing-tag override: 0 = automatic (master-driven) allocation; nonzero = + * an operator has forced this node_id to be the active shtag holder for the + * cluster (cl_ctr_shtag_force MI), suspending automatic allocation until + * cl_ctr_shtag_auto clears it. Propagated to all nodes in the MEMBER_LIST. */ + uint16_t shtag_forced_node_id; + /* worker_proc_no: OpenSIPS process index of this cluster's cl_ctr_worker, + * published here (shm) after fork so MI handlers running in a different + * process can target the worker with ipc_send_rpc(). -1 until set. */ + int worker_proc_no; + /* Effective (possibly adopted) consistency-critical settings, mirrored in + * shm so MI handlers in another process (cl_ctr_list_config) report the value + * actually in force after an on_config_mismatch=adopt. The worker keeps + * these in sync with its own cl->manage_shtags / master_stickiness / + * query_time. Initialised from the resolved config at mod_init. */ + int eff_manage_shtags; + int eff_master_stickiness; + int eff_query_time; +}; + + +/* ========================================================================= + * timerfd helpers + * ========================================================================= */ + +/* Drain the expiration counter so the fd stops being readable. */ +static void cl_ctr_drain_tfd(int tfd) +{ + uint64_t exp; + if (read(tfd, &exp, sizeof(exp)) < 0 && errno != EAGAIN) + LM_WARN("clusterer_controller: timerfd read: %s\n", strerror(errno)); +} + +/* Arm a timerfd. Pass sec_value=0 to disarm. */ +static void cl_ctr_arm_tfd(int tfd, time_t sec_value, time_t sec_interval) +{ + struct itimerspec its; + memset(&its, 0, sizeof(its)); + its.it_value.tv_sec = sec_value; + its.it_interval.tv_sec = sec_interval; + if (timerfd_settime(tfd, 0, &its, NULL) < 0) + LM_WARN("clusterer_controller: timerfd_settime: %s\n", strerror(errno)); +} + +/* ========================================================================= + * Forward declarations + * ========================================================================= */ + +static int mod_init(void); +static int cl_ctr_child_init(int rank); +static void mod_destroy(void); +static void cl_ctr_worker(int rank); +static int cl_ctr_on_sock(int fd, void *param, int was_timeout); +static void cl_ctr_retx_flush(cl_ctr_cluster_t *cl); +static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout); +static void cl_ctr_membership_digest(cl_ctr_cluster_t *cl, uint16_t *count, uint64_t *hash); +static void cl_ctr_send_resync(int sock, cl_ctr_cluster_t *cl, + const struct sockaddr *dest, socklen_t destlen); +static void cl_ctr_broadcast_full_state(cl_ctr_cluster_t *cl); +static int cl_ctr_on_alive_tfd(int fd, void *param, int was_timeout); +static void cl_ctr_arm_master_timers(cl_ctr_cluster_t *cl, int i_am_master); +static int cl_ctr_on_join_tfd(int fd, void *param, int was_timeout); +static int cl_ctr_on_rejoin_tfd(int fd, void *param, int was_timeout); +static int cl_ctr_on_master_alive_tfd(int fd, void *param, int was_timeout); +static int cl_ctr_on_master_dead_tfd(int fd, void *param, int was_timeout); +static mi_response_t *mi_cl_ctr_members(const mi_params_t *params, + struct mi_handler *hdl); +static void cl_ctr_handle_member_list(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_len, + cl_ctr_cluster_t *cl, + const struct sockaddr *src, socklen_t src_len); +static void cl_ctr_handle_node_assign(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_handle_goodbye(int sock, const char *src_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_handle_master_alive(const char *sender_ip, cl_ctr_cluster_t *cl, + const char *payload, int payload_len, + const struct sockaddr *src, socklen_t src_len); +static void cl_ctr_handle_key_grant(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl, + uint32_t in_seq, const struct sockaddr *src, + socklen_t src_len); +static void cl_ctr_handle_key_handoff(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_handle_join_reject(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_cluster_t *cl, + int reason); +static mi_response_t *mi_cl_ctr_node_info(const mi_params_t *params, + struct mi_handler *hdl); +static mi_response_t *mi_cl_ctr_config(const mi_params_t *params, + struct mi_handler *hdl); +static mi_response_t *mi_cl_ctr_shtag_force(const mi_params_t *params, + struct mi_handler *hdl); +static mi_response_t *mi_cl_ctr_shtag_auto(const mi_params_t *params, + struct mi_handler *hdl); + +/* ========================================================================= + * Extra-process export (layout from mi_fifo.c) + * ========================================================================= */ + +static proc_export_t procs[] = { + {"clusterer_controller worker", 0, 0, cl_ctr_worker, 1, + PROC_FLAG_INITCHILD | PROC_FLAG_HAS_IPC}, + {0, 0, 0, 0, 0, 0} +}; + +/* ========================================================================= + * Module dependency + * ========================================================================= */ + +static const dep_export_t deps = { + { + /* proto_bin must load before us so its listeners are registered */ + { MOD_TYPE_DEFAULT, "proto_bin", DEP_ABORT }, + { MOD_TYPE_DEFAULT, "clusterer", DEP_ABORT }, + { MOD_TYPE_NULL, NULL, 0 }, + }, + { { NULL, NULL } }, +}; + +/* ========================================================================= + * MI command export table + * ========================================================================= */ + +static const mi_export_t mi_cmds[] = { + { + "cl_ctr_list_members", + "List all current cluster members with node_id, status and BIN sockets", + 0, 0, + { + {mi_cl_ctr_members, {0}}, + {EMPTY_MI_RECIPE} + }, {0} + }, + { + "cl_ctr_node_info", + "Return full info for a node_id across all clusters", + 0, 0, + { + {mi_cl_ctr_node_info, {"node_id", 0}}, + {EMPTY_MI_RECIPE} + }, {0} + }, + { + "cl_ctr_list_config", + "List all configured clusters and their resolved settings", + 0, 0, + { + {mi_cl_ctr_config, {0}}, + {EMPTY_MI_RECIPE} + }, {0} + }, + { + "cl_ctr_shtag_force", + "Force a node to hold the active sharing tag (master only); " + "suspends automatic allocation until cl_ctr_shtag_auto", + 0, 0, + { + {mi_cl_ctr_shtag_force, {"cluster_id", "node_id", 0}}, + {EMPTY_MI_RECIPE} + }, {0} + }, + { + "cl_ctr_shtag_auto", + "Resume automatic master-driven sharing-tag allocation (master only)", + 0, 0, + { + {mi_cl_ctr_shtag_auto, {"cluster_id", 0}}, + {EMPTY_MI_RECIPE} + }, {0} + }, + {EMPTY_MI_EXPORT} +}; + +/* ========================================================================= + * Read-only script variables ($cl_ctr_*) + * + * Nouns, never verbs (actions are script functions / MI). Each variable + * optionally takes a cluster id: $cl_ctr_role(2). The bare form + * ($cl_ctr_role) resolves to the sole configured cluster; with several + * clusters it returns NULL and warns once, so nobody silently reads the + * wrong cluster. All values are read from the shm peer table under a read + * lock, so every process (SIP workers included) sees live state. Read-only: + * no setter is exported. + * ========================================================================= */ + +enum cl_ctr_pv_field { + CL_CTR_PV_ROLE, /* master | backup | member | joining */ + CL_CTR_PV_IS_MASTER, /* 1 / 0 */ + CL_CTR_PV_MASTER_IP, /* current master's IP, NULL if none */ + CL_CTR_PV_BACKUP_IP, /* current backup's IP, NULL if none */ + CL_CTR_PV_NODE_ID, /* this node's id in the cluster, NULL if unassigned */ + CL_CTR_PV_MY_IP, /* controller identity IP */ + CL_CTR_PV_MEMBERS, /* live member count */ + CL_CTR_PV_SHTAG_MODE, /* auto | forced */ + CL_CTR_PV_FORCED_NODE, /* node pinned by cl_ctr_shtag_force, NULL if auto */ +}; + +/* Optional (cluster_id) argument; bare form leaves the spec zeroed (cid 0). */ +static int cl_ctr_pv_parse_cluster(pv_spec_p sp, const str *in) +{ + unsigned int cid; + str s; + + if (!sp) + return -1; + if (!in || !in->s || in->len == 0) { + sp->pvp.pvn.u.isname.name.n = 0; /* bare form -> sole cluster */ + return 0; + } + s = *in; + trim(&s); + if (s.len == 0) { + sp->pvp.pvn.u.isname.name.n = 0; + return 0; + } + /* str2int rejects any non-digit (so '-5', 'abc', '1.2' all fail); the + * length cap keeps a huge value from silently overflowing/wrapping into a + * valid id (999999999 < INT_MAX). */ + if (s.len > 9 || str2int(&s, &cid) < 0 || cid == 0) { + LM_ERR("clusterer_controller: invalid cluster id '%.*s' in $cl_ctr_* " + "variable (expected a positive integer 1..999999999)\n", + in->len, in->s); + return -1; + } + sp->pvp.pvn.u.isname.name.n = (int)cid; + return 0; +} + +static int cl_ctr_pv_get(struct sip_msg *msg, pv_param_t *param, pv_value_t *res, + enum cl_ctr_pv_field field) +{ + static char cl_ctr_pv_ipbuf[INET_ADDRSTRLEN]; + static int cl_ctr_pv_warned_ambiguous; + cl_ctr_cluster_t *cl = NULL; + cl_ctr_peer_t *me = NULL, *master = NULL, *backup = NULL; + const char *sval = NULL; + int cid, i, have_int = 0, ival = 0, joining; + + cid = param->pvn.u.isname.name.n; + if (cid == 0) { + if (cl_ctr_cluster_count == 1) { + cl = &cl_ctr_clusters[0]; + } else { + if (!cl_ctr_pv_warned_ambiguous) { + LM_WARN("clusterer_controller: bare $cl_ctr_* used with %d " + "clusters configured - specify the cluster id, e.g. " + "$cl_ctr_role(%d)\n", cl_ctr_cluster_count, + cl_ctr_cluster_count ? cl_ctr_clusters[0].cluster_id : 1); + cl_ctr_pv_warned_ambiguous = 1; + } + return pv_get_null(msg, param, res); + } + } else { + for (i = 0; i < cl_ctr_cluster_count; i++) + if (cl_ctr_clusters[i].cluster_id == cid) { + cl = &cl_ctr_clusters[i]; + break; + } + } + if (!cl || !cl->peers) + return pv_get_null(msg, param, res); + + lock_start_read(cl->peers->lock); + + joining = (cl->peers->node_state == CL_CTR_NODE_NEW); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + if (e->is_master) + master = e; + if (e->is_backup) + backup = e; + if (my_ip && strcmp(e->ip, my_ip) == 0) + me = e; + } + + switch (field) { + case CL_CTR_PV_ROLE: + if (joining) + sval = "joining"; + else if (me && me->is_master) + sval = "master"; + else if (me && me->is_backup) + sval = "backup"; + else + sval = "member"; + break; + case CL_CTR_PV_IS_MASTER: + have_int = 1; + ival = (!joining && me && me->is_master) ? 1 : 0; + break; + case CL_CTR_PV_MASTER_IP: + if (master) { + strncpy(cl_ctr_pv_ipbuf, master->ip, sizeof(cl_ctr_pv_ipbuf) - 1); + cl_ctr_pv_ipbuf[sizeof(cl_ctr_pv_ipbuf) - 1] = '\0'; + sval = cl_ctr_pv_ipbuf; + } + break; + case CL_CTR_PV_BACKUP_IP: + if (backup) { + strncpy(cl_ctr_pv_ipbuf, backup->ip, sizeof(cl_ctr_pv_ipbuf) - 1); + cl_ctr_pv_ipbuf[sizeof(cl_ctr_pv_ipbuf) - 1] = '\0'; + sval = cl_ctr_pv_ipbuf; + } + break; + case CL_CTR_PV_NODE_ID: + if (me && me->node_id > 0) { + have_int = 1; + ival = me->node_id; + } + break; + case CL_CTR_PV_MY_IP: + sval = my_ip; /* resolved in mod_init, constant afterwards */ + break; + case CL_CTR_PV_MEMBERS: + have_int = 1; + ival = cl->peers->count; + break; + case CL_CTR_PV_SHTAG_MODE: + sval = cl->peers->shtag_forced_node_id ? "forced" : "auto"; + break; + case CL_CTR_PV_FORCED_NODE: + if (cl->peers->shtag_forced_node_id) { + have_int = 1; + ival = cl->peers->shtag_forced_node_id; + } + break; + } + + lock_stop_read(cl->peers->lock); + + if (have_int) + return pv_get_uintval(msg, param, res, (unsigned int)ival); + if (sval) { + str s = {(char *)sval, (int)strlen(sval)}; + return pv_get_strval(msg, param, res, &s); + } + return pv_get_null(msg, param, res); +} + +#define CL_CTR_PV_WRAP(_fn, _field) \ +static int _fn(struct sip_msg *msg, pv_param_t *param, pv_value_t *res) \ +{ return cl_ctr_pv_get(msg, param, res, _field); } + +/* ------------------------------------------------------------------------- + * Per-peer lookups: query a *specific* node in a cluster. These are script + * FUNCTIONS, not pseudo-variables, because they take two arguments + * (cluster_id, node_id) - a comma inside a pvar's parentheses is ambiguous to + * the config parser when the pvar is used as a function argument. Boolean + * checks return true/false for use in if(); value lookups write an output var. + * ------------------------------------------------------------------------- */ + +/* Find peer (cluster_id, node_id); cid<=0 => the sole configured cluster. + * Returns 0 and fills the requested out params on success, -1 if the cluster + * or node is unknown. Caller must not hold the peers lock. */ +static int cl_ctr_find_peer(int cid, int nid, int *is_master, int *is_backup, + char *ipbuf, int ipbuf_sz) +{ + cl_ctr_cluster_t *cl = NULL; + int i, rc = -1; + + if (cid <= 0) { + if (cl_ctr_cluster_count == 1) + cl = &cl_ctr_clusters[0]; + } else { + for (i = 0; i < cl_ctr_cluster_count; i++) + if (cl_ctr_clusters[i].cluster_id == cid) { + cl = &cl_ctr_clusters[i]; + break; + } + } + if (!cl || !cl->peers) + return -1; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + if (e->node_id != nid) + continue; + if (is_master) *is_master = e->is_master; + if (is_backup) *is_backup = e->is_backup; + if (ipbuf) { + strncpy(ipbuf, e->ip, ipbuf_sz - 1); + ipbuf[ipbuf_sz - 1] = '\0'; + } + rc = 0; + break; + } + lock_stop_read(cl->peers->lock); + return rc; +} + +static int cl_ctr_out_str(struct sip_msg *msg, pv_spec_t *out, const char *str_s) +{ + pv_value_t val; + memset(&val, 0, sizeof val); + val.flags = PV_VAL_STR; + val.rs.s = (char *)str_s; + val.rs.len = strlen(str_s); + return pv_set_value(msg, out, 0, &val); +} + +/* cl_ctr_node_is_master(cluster_id, node_id) -> true if that node is master */ +static int w_cl_ctr_node_is_master(struct sip_msg *msg, int *cid, int *nid) +{ + int im = 0; + if (cl_ctr_find_peer(cid ? *cid : 0, nid ? *nid : 0, &im, NULL, NULL, 0) < 0) + return -1; + return im ? 1 : -1; +} + +/* cl_ctr_node_present(cluster_id, node_id) -> true if node_id is a live member */ +static int w_cl_ctr_node_present(struct sip_msg *msg, int *cid, int *nid) +{ + return cl_ctr_find_peer(cid ? *cid : 0, nid ? *nid : 0, NULL, NULL, NULL, 0) == 0 + ? 1 : -1; +} + +/* cl_ctr_get_node_role(cluster_id, node_id, out) -> out=master|backup|member */ +static int w_cl_ctr_get_node_role(struct sip_msg *msg, int *cid, int *nid, + pv_spec_t *out) +{ + int im = 0, ib = 0; + if (cl_ctr_find_peer(cid ? *cid : 0, nid ? *nid : 0, &im, &ib, NULL, 0) < 0) + return -1; + return cl_ctr_out_str(msg, out, im ? "master" : (ib ? "backup" : "member")) == 0 + ? 1 : -1; +} + +/* cl_ctr_get_node_ip(cluster_id, node_id, out) -> out=that node's IP */ +static int w_cl_ctr_get_node_ip(struct sip_msg *msg, int *cid, int *nid, + pv_spec_t *out) +{ + char ipbuf[INET_ADDRSTRLEN]; + if (cl_ctr_find_peer(cid ? *cid : 0, nid ? *nid : 0, NULL, NULL, + ipbuf, sizeof ipbuf) < 0) + return -1; + return cl_ctr_out_str(msg, out, ipbuf) == 0 ? 1 : -1; +} + +CL_CTR_PV_WRAP(cl_ctr_pv_role, CL_CTR_PV_ROLE) +CL_CTR_PV_WRAP(cl_ctr_pv_is_master, CL_CTR_PV_IS_MASTER) +CL_CTR_PV_WRAP(cl_ctr_pv_master_ip, CL_CTR_PV_MASTER_IP) +CL_CTR_PV_WRAP(cl_ctr_pv_backup_ip, CL_CTR_PV_BACKUP_IP) +CL_CTR_PV_WRAP(cl_ctr_pv_node_id, CL_CTR_PV_NODE_ID) +CL_CTR_PV_WRAP(cl_ctr_pv_my_ip, CL_CTR_PV_MY_IP) +CL_CTR_PV_WRAP(cl_ctr_pv_members, CL_CTR_PV_MEMBERS) +CL_CTR_PV_WRAP(cl_ctr_pv_shtag_mode, CL_CTR_PV_SHTAG_MODE) +CL_CTR_PV_WRAP(cl_ctr_pv_forced_node, CL_CTR_PV_FORCED_NODE) + +static const pv_export_t cl_ctr_mod_vars[] = { + { {"cl_ctr_role", sizeof("cl_ctr_role")-1}, 1100, + cl_ctr_pv_role, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_is_master", sizeof("cl_ctr_is_master")-1}, 1101, + cl_ctr_pv_is_master, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_master_ip", sizeof("cl_ctr_master_ip")-1}, 1102, + cl_ctr_pv_master_ip, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_backup_ip", sizeof("cl_ctr_backup_ip")-1}, 1103, + cl_ctr_pv_backup_ip, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_node_id", sizeof("cl_ctr_node_id")-1}, 1104, + cl_ctr_pv_node_id, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_my_ip", sizeof("cl_ctr_my_ip")-1}, 1105, + cl_ctr_pv_my_ip, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_members", sizeof("cl_ctr_members")-1}, 1106, + cl_ctr_pv_members, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_shtag_mode", sizeof("cl_ctr_shtag_mode")-1}, 1107, + cl_ctr_pv_shtag_mode, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {"cl_ctr_forced_node", sizeof("cl_ctr_forced_node")-1}, 1108, + cl_ctr_pv_forced_node, 0, cl_ctr_pv_parse_cluster, 0, 0, 0 }, + { {0, 0}, 0, 0, 0, 0, 0, 0, 0 } +}; + +/* Script functions: per-peer lookups (two args -> functions, not variables). */ +static const cmd_export_t cl_ctr_cmds[] = { + {"cl_ctr_node_is_master", (cmd_function)w_cl_ctr_node_is_master, { + {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {"cl_ctr_node_present", (cmd_function)w_cl_ctr_node_present, { + {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {"cl_ctr_get_node_role", (cmd_function)w_cl_ctr_get_node_role, { + {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_VAR, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {"cl_ctr_get_node_ip", (cmd_function)w_cl_ctr_get_node_ip, { + {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_VAR, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {0, 0, {{0, 0, 0}}, 0} +}; + +/* ========================================================================= + * module_exports + * ========================================================================= */ + +struct module_exports exports = { + "clusterer_controller", + MOD_TYPE_DEFAULT, + MODULE_VERSION, + DEFAULT_DLFLAGS, + 0, + &deps, + cl_ctr_cmds, /* cmds - per-peer lookup functions */ + 0, /* acmds */ + params, + 0, /* stats */ + mi_cmds, + cl_ctr_mod_vars, /* pvs - read-only $cl_ctr_* variables */ + 0, /* transforms */ + procs, + 0, /* pre_init_f */ + mod_init, + 0, /* response_f */ + mod_destroy, + cl_ctr_child_init, /* child_init_f */ + 0 /* reload_confirm_f */ +}; + +/* ========================================================================= + * Internal helpers + * ========================================================================= */ + +static unsigned int ip_to_num(const char *ip) +{ + struct in_addr addr; + if (inet_aton(ip, &addr) == 0) + return 0; + return ntohl(addr.s_addr); +} + +/** + * cl_ctr_election_cutoff() - quantized stale cutoff for master election. + * + * All NTP-synchronized nodes compute the same value at the same second, + * so they always evaluate the identical eligible-peer set and elect the + * same master. + */ +static time_t cl_ctr_election_cutoff(void) +{ + time_t now = time(NULL); + return (now / (time_t)query_time) * (time_t)query_time + - (time_t)(query_time * CL_CTR_ELECT_FACTOR); +} + +/** + * cl_ctr_elect_master(cl) - mark the peer with the highest IP as master. + * + * Uses the quantized election window so all NTP-synchronized nodes evaluate + * the same eligible set and elect the same master. + * + * Also tracks two state transitions and logs them at INFO: + * + * in_election 1->0 A peer's last_seen fell outside the election window - + * the node is considered down. Logged immediately so the + * operator sees the event without waiting for cl_ctr_prune_stale(cl) + * (which only fires at CL_CTR_PURGE_FACTOR x query_time). + * + * last_master The elected master IP changed - either because the + * previous master went down, or a higher-IP node joined. + * + * Must be called with cl->peers->lock held. + */ +static void cl_ctr_elect_master(cl_ctr_cluster_t *cl) +{ + time_t cutoff = cl_ctr_election_cutoff(); + unsigned int top_num = 0; + int i, n_in = 0, top_idx = -1, cur_master_idx = -1, master_idx = -1; + int i_am_elected = 0; + char prev_master[CL_CTR_MAX_IP_LEN + 1]; + char prev_backup[CL_CTR_MAX_IP_LEN + 1]; + + prev_backup[0] = '\0'; + /* Snapshot the master we had before this election, for change reporting. */ + { + size_t _l = strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN); + memcpy(prev_master, cl->peers->last_master, _l); + prev_master[_l] = '\0'; + } + + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + int now_in = (e->last_seen >= cutoff); + + /* Detect peer dropping out of the election window */ + if (e->in_election && !now_in) + LM_INFO("clusterer_controller: peer %s went down " + "(last seen %lds ago)\n", + e->ip, (long)(time(NULL) - e->last_seen)); + + e->in_election = now_in; + + /* Remember the live current master and the previous backup before + * clearing the flags - used for sticky election and change logging. */ + if (e->is_master && now_in) + cur_master_idx = i; + if (e->is_backup) { + size_t _l = strnlen(e->ip, CL_CTR_MAX_IP_LEN); + memcpy(prev_backup, e->ip, _l); + prev_backup[_l] = '\0'; + } + + e->is_master = 0; + e->is_backup = 0; + + if (now_in) { + n_in++; + if (e->ip_num > top_num) { + top_num = e->ip_num; + top_idx = i; + } + } + } + + /* Choose the master: + * master_stickiness == 1 (default, sticky): a live current master keeps the + * role - a higher-IP peer does NOT preempt it (fewer handovers). Only + * when there is no live master do we elect the highest-IP peer. + * master_stickiness == 0: pure highest-IP election - the highest-IP peer + * always wins, preempting any lower-IP current master. + * Split-brain (two live masters, e.g. after a partition heal) is resolved + * separately in cl_ctr_handle_master_alive by yielding to the highest IP. */ + if (cl->master_stickiness == 1 && cur_master_idx >= 0) + master_idx = cur_master_idx; + else + master_idx = top_idx; + + if (master_idx >= 0) { + const char *m_ip, *b_ip, *why; + int b_idx = -1, master_changed, backup_changed; + unsigned int b_num = 0; + int j; + + cl->peers->entries[master_idx].is_master = 1; + m_ip = cl->peers->entries[master_idx].ip; + i_am_elected = (strcmp(m_ip, my_ip) == 0); + + /* Designate the BACKUP: the highest-IP in-window peer that is not the + * master. Deterministic across all nodes, so everyone agrees who takes + * over next; on master failure cl_ctr_elect_master (no live current master) + * promotes exactly this node. */ + for (j = 0; j < cl->peers->count; j++) { + cl_ctr_peer_t *e = &cl->peers->entries[j]; + if (j == master_idx || !e->in_election) + continue; + if (e->ip_num > b_num) { + b_num = e->ip_num; + b_idx = j; + } + } + if (b_idx >= 0) + cl->peers->entries[b_idx].is_backup = 1; + b_ip = (b_idx >= 0) ? cl->peers->entries[b_idx].ip : NULL; + + master_changed = (strcmp(prev_master, m_ip) != 0); + backup_changed = (strcmp(prev_backup, b_ip ? b_ip : "") != 0); + + /* Persist the elected master for the next round / other handlers. */ + { + size_t _l = strnlen(m_ip, CL_CTR_MAX_IP_LEN); + memcpy(cl->peers->last_master, m_ip, _l); + cl->peers->last_master[_l] = '\0'; + } + + /* One clear line whenever the master or backup role changes, stating + * who holds each role, which is this node, and WHY the master was + * chosen (highest IP, sticky current master kept over a higher-IP peer, or + * sole surviving node). */ + if (master_changed || backup_changed) { + if (n_in <= 1) + why = "sole node in window"; + else if (cl->master_stickiness == 1 && master_idx == cur_master_idx && + top_idx >= 0 && top_idx != master_idx) + why = "sticky: current master kept over higher-IP node"; + else + why = "highest IP in window"; + + LM_INFO("clusterer_controller: [cluster %d] roles: MASTER=%s%s (%s); " + "BACKUP=%s%s (highest-IP non-master); %d node(s) in window\n", + cl->cluster_id, + m_ip, + strcmp(m_ip, my_ip) == 0 ? " [me]" : "", + why, + b_ip ? b_ip : "(none)", + (b_ip && strcmp(b_ip, my_ip) == 0) ? " [me]" : "", + n_in); + } + + /* Log this node's own role transition (member/backup/master) whenever + * it changes, so a failover or a peer (re)joining that demotes/promotes + * us gets its own line rather than being implied by the roles list. */ + { + int my_new_role = i_am_elected ? CL_CTR_ROLE_MASTER : + ((b_ip && strcmp(b_ip, my_ip) == 0) ? CL_CTR_ROLE_BACKUP : + CL_CTR_ROLE_MEMBER); + if (my_new_role != cl->my_role) { + LM_INFO("clusterer_controller: [cluster %d] my role changed: " + "%s -> %s\n", cl->cluster_id, + cl_ctr_role_name(cl->my_role), cl_ctr_role_name(my_new_role)); + cl->my_role = my_new_role; + } + } + } else { + /* No eligible peer - cluster has no master; we are a plain member. */ + if (cl->my_role != CL_CTR_ROLE_MEMBER) { + LM_INFO("clusterer_controller: [cluster %d] my role changed: %s -> " + "member\n", cl->cluster_id, cl_ctr_role_name(cl->my_role)); + cl->my_role = CL_CTR_ROLE_MEMBER; + } + if (cl->peers->last_master[0] != '\0') { + LM_INFO("clusterer_controller: [cluster %d] master lost (%s), " + "no eligible peers in election window\n", + cl->cluster_id, cl->peers->last_master); + cl->peers->last_master[0] = '\0'; + } + } + + /* Invariant: MASTER_ALIVE keepalive armed <=> I am the elected master. + * If this election demoted us (someone else won, or no master at all) but + * our keepalive is still running, stop it now - otherwise we keep + * broadcasting MASTER_ALIVE as a phantom master and lower-IP peers + * oscillate between two masters. Only disarm here: promoting a new master + * (arming) is done by the became-master paths, which also establish the + * session key. cl_ctr_arm_master_timers() only issues timerfd syscalls, so it + * is safe under cl->peers->lock. */ + if (!i_am_elected && cl->master_ka_armed) + cl_ctr_arm_master_timers(cl, 0); +} + +/** + * cl_ctr_i_am_master_locked(cl) - return 1 if my_ip is currently elected master. + * Must be called with cl->peers->lock held. + */ +static int cl_ctr_i_am_master_locked(cl_ctr_cluster_t *cl) +{ + time_t cutoff = cl_ctr_election_cutoff(); + int i; + + for (i = 0; i < cl->peers->count; i++) { + if (cl->peers->entries[i].is_master && + cl->peers->entries[i].last_seen >= cutoff && + strcmp(cl->peers->entries[i].ip, my_ip) == 0) + return 1; + } + return 0; +} + +/** + * cl_ctr_ip_beats_master_locked() - check whether a candidate IP would displace + * the current master in an election. + * + * Returns 1 (re-election is worth running) when: + * - ip_num is strictly greater than the current master's ip_num, OR + * - there is no current master in the election window (no-one to defend). + * + * Returns 0 when the current master has a higher or equal IP - it would + * win the election anyway, so running one is pointless. + * + * Must be called with cl->peers->lock held. + */ +static int cl_ctr_ip_beats_master_locked(unsigned int ip_num, cl_ctr_cluster_t *cl) +{ + time_t cutoff = cl_ctr_election_cutoff(); + int i; + + for (i = 0; i < cl->peers->count; i++) { + if (cl->peers->entries[i].is_master && + cl->peers->entries[i].last_seen >= cutoff) + return (ip_num > cl->peers->entries[i].ip_num); + } + return 1; /* no master in the election window - election is needed */ +} + +/** + * cl_ctr_apply_shtags_decision() - (de)activate this node's sharing tags per policy. + * + * Decides whether THIS node should be the active sharing-tag holder for the + * cluster and calls clusterer accordingly. Idempotent (activate/force-backup + * are no-ops when already in that state), so it is safe to call on any relevant + * event. Takes NO lock - the caller passes the state it already read, so this + * is safe both inside and outside cl->peers->lock (clctl uses its own locks). + * + * forced != 0 : the operator pinned node_id 'forced' as the active holder + * (cl_ctr_shtag_force). Only that node activates; all others go + * to backup. Automatic allocation is suspended. + * forced == 0 : automatic mode - the current master is the active holder. + */ +static void cl_ctr_apply_shtags_decision(cl_ctr_cluster_t *cl, int i_am_master, + uint16_t forced) +{ + int activate; + + if (!cl->manage_shtags || !clctl_loaded || !clctl.activate_backup_shtags) + return; + + if (forced != 0) + activate = (my_node_id != 0 && (uint16_t)my_node_id == forced); + else + activate = i_am_master; + + /* Log the reason on this node, but only when the decision or its cause + * changes - cl_ctr_apply_shtags_decision() is called on every relevant event + * and clusterer's own activate/force calls are idempotent, so logging + * unconditionally would flood. This makes the "why" visible on EVERY + * node (master, forced holder, and passive backups alike). */ + if (activate != cl->shtag_last_active || forced != cl->shtag_last_forced) { + if (activate && forced != 0) + LM_INFO("clusterer_controller: [cluster %d] activating sharing tags " + "- operator forced this node (node_id %u) as active holder\n", + cl->cluster_id, forced); + else if (activate) + LM_INFO("clusterer_controller: [cluster %d] activating sharing tags " + "- this node is the cluster master\n", cl->cluster_id); + else if (forced != 0) + LM_INFO("clusterer_controller: [cluster %d] keeping sharing tags in " + "backup - operator forced node_id %u as active holder\n", + cl->cluster_id, forced); + else + LM_INFO("clusterer_controller: [cluster %d] keeping sharing tags in " + "backup - active holder is the cluster master\n", + cl->cluster_id); + cl->shtag_last_active = activate; + cl->shtag_last_forced = forced; + } + + if (activate) + clctl.activate_backup_shtags(cl->cluster_id); + else if (clctl.force_backup_shtags) + clctl.force_backup_shtags(cl->cluster_id); +} + +/** + * cl_ctr_apply_shtags() - convenience wrapper that reads the current state under a + * read lock and applies the shtag decision. Call WITHOUT cl->peers->lock held. + */ +static void cl_ctr_apply_shtags(cl_ctr_cluster_t *cl) +{ + int i_am_master; + uint16_t forced; + + lock_start_read(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + forced = cl->peers->shtag_forced_node_id; + lock_stop_read(cl->peers->lock); + + cl_ctr_apply_shtags_decision(cl, i_am_master, forced); +} + +/** + * cl_ctr_prune_stale(cl) - free entries far outside the election window. + * Memory management only - does not affect election outcomes. + * Must be called with cl->peers->lock held. + */ +static void cl_ctr_prune_stale(cl_ctr_cluster_t *cl) +{ + time_t cutoff = time(NULL) - (time_t)(query_time * CL_CTR_PURGE_FACTOR); + int i; + + for (i = 0; i < cl->peers->count; i++) { + if (cl->peers->entries[i].last_seen < cutoff) { + uint16_t pruned_id = cl->peers->entries[i].node_id; + LM_INFO("clusterer_controller: purging timed-out peer %s\n", + cl->peers->entries[i].ip); + cl->peers->count--; + if (i < cl->peers->count) + cl->peers->entries[i] = cl->peers->entries[cl->peers->count]; + memset(&cl->peers->entries[cl->peers->count], 0, sizeof(cl_ctr_peer_t)); + i--; + /* cl_list_lock and cl->peers->lock are independent - no deadlock */ + if (clctl_loaded && pruned_id > 0) + clctl.remove_node(cl->cluster_id, pruned_id); + /* If the operator-forced shtag holder timed out, drop the override + * so automatic allocation resumes instead of leaving no active tag. */ + if (pruned_id != 0 && cl->peers->shtag_forced_node_id == pruned_id) { + LM_WARN("clusterer_controller: [cluster %d] forced shtag node " + "%u timed out - resuming automatic allocation\n", + cl->cluster_id, pruned_id); + cl->peers->shtag_forced_node_id = 0; + } + /* Re-apply shtag policy (override-aware); inside the lock, so pass + * the state directly rather than calling the locking wrapper. */ + cl_ctr_apply_shtags_decision(cl, cl_ctr_i_am_master_locked(cl), + cl->peers->shtag_forced_node_id); + } + } +} + +/** + * cl_ctr_apply_master_from_list_locked() - apply master designation from a + * received MEMBER_LIST/MEMBER_LIST packet. + * + * Zeros all is_master flags in the peer table, then sets is_master=1 for + * the entry matching master_ip. Updates last_master accordingly. + * Must be called with cl->peers->lock held. + */ +static void cl_ctr_apply_master_from_list_locked(const char *master_ip, cl_ctr_cluster_t *cl) +{ + int i; + + for (i = 0; i < cl->peers->count; i++) { + if (strcmp(cl->peers->entries[i].ip, master_ip) == 0) { + cl->peers->entries[i].is_master = 1; + } else { + cl->peers->entries[i].is_master = 0; + } + } + + memcpy(cl->peers->last_master, master_ip, + strnlen(master_ip, CL_CTR_MAX_IP_LEN)); + cl->peers->last_master[strnlen(master_ip, CL_CTR_MAX_IP_LEN)] = '\0'; +} + +/** + * cl_ctr_peer_by_ip_locked() - find a peer entry by IP string. + * Returns the entry pointer, or NULL if not present. + * Must be called with cl->peers->lock held (read or write). + */ +static cl_ctr_peer_t *cl_ctr_peer_by_ip_locked(cl_ctr_cluster_t *cl, const char *ip) +{ + int i; + for (i = 0; i < cl->peers->count; i++) + if (strcmp(cl->peers->entries[i].ip, ip) == 0) + return &cl->peers->entries[i]; + return NULL; +} + +/** + * cl_ctr_upsert_peer_locked() - insert or refresh a peer entry. + * Does NOT call cl_ctr_elect_master(cl); callers do that explicitly. + * Must be called with cl->peers->lock held. + */ +static void cl_ctr_upsert_peer_locked(const char *src_ip, cl_ctr_cluster_t *cl) +{ + unsigned int src_num = ip_to_num(src_ip); + time_t now = time(NULL); + cl_ctr_peer_t *found; + + if (src_num == 0) { + LM_WARN("clusterer_controller: ignoring invalid IP '%s'\n", src_ip); + return; + } + + found = cl_ctr_peer_by_ip_locked(cl, src_ip); + if (found) { + found->last_seen = now; + return; /* updated */ + } + + /* New entry */ + if (cl->peers->count >= CL_CTR_MAX_PEERS) { + LM_WARN("clusterer_controller: peer table full, ignoring %s\n", + src_ip); + return; + } + { + cl_ctr_peer_t *e = &cl->peers->entries[cl->peers->count]; + memcpy(e->ip, src_ip, strnlen(src_ip, CL_CTR_MAX_IP_LEN)); + e->ip[strnlen(src_ip, CL_CTR_MAX_IP_LEN)] = '\0'; + e->ip_num = src_num; + e->last_seen = now; + e->is_master = 0; + cl->peers->count++; + LM_INFO("clusterer_controller: new peer %s (total=%d)\n", + src_ip, cl->peers->count); + } +} + +/** + * cl_ctr_alloc_node_id_locked() - find the lowest unused node_id >= 1. + * Must be called with cl->peers->lock held for write. + */ +static uint16_t cl_ctr_alloc_node_id_locked(cl_ctr_cluster_t *cl) +{ + uint16_t id; + int i, used; + + for (id = 1; id < 65535; id++) { + used = 0; + for (i = 0; i < cl->peers->count; i++) { + if (cl->peers->entries[i].node_id == id) { + used = 1; + break; + } + } + if (!used) + return id; + } + return 0; /* table full (shouldn't happen with CL_CTR_MAX_PEERS=256) */ +} + +/** + * cl_ctr_update_peer_bin_locked() - store node_id and BIN sockets for a peer. + * Must be called with cl->peers->lock held. + */ +static void cl_ctr_update_peer_bin_locked(const char *ip, uint16_t node_id, + uint8_t bin_count, + const char (*bin_sockets)[CL_CTR_MAX_BIN_SOCK_LEN], + cl_ctr_cluster_t *cl) +{ + cl_ctr_peer_t *e = cl_ctr_peer_by_ip_locked(cl, ip); + if (e) { + e->node_id = node_id; + e->bin_count = bin_count; + if (bin_count > 0) + memcpy(e->bin_sockets, bin_sockets, bin_count * CL_CTR_MAX_BIN_SOCK_LEN); + } +} + +/* ========================================================================= + * Packet encryption / decryption (AEAD) + * + * Every packet is fully encrypted and authenticated with XChaCha20-Poly1305 + * (libsodium). A fresh random nonce per packet ensures that even identical + * payloads produce different ciphertext. + * + * A 4-byte monotonic sequence number is included inside the plaintext. + * Receivers track the highest sequence seen from each peer and reject any + * packet whose sequence is not strictly greater. This stops replays + * without requiring NTP-synchronised clocks or a finite nonce cache. + * The counter resets to 0 on every session key rotation; old packets + * encrypted with the previous key fail AEAD authentication anyway. + * Bootstrap-key packets (CL_CTR_BOOTSTRAP_MAGIC) skip the sequence check - + * their per-exchange join_nonce provides equivalent replay protection. + * + * Wire layout: + * [magic 2B][cluster_id 2B][nonce 24B][ciphertext][Poly1305 tag 16B] + * Plaintext: + * [type 1B][seq 4B][payload] + * ========================================================================= */ + +/* cl_ctr_random_bytes() - fill buf with cryptographically secure random bytes + * (libsodium's randombytes draws from the kernel CSPRNG). */ +static int cl_ctr_random_bytes(unsigned char *buf, size_t len) +{ + randombytes_buf(buf, len); + return 0; +} + +/* RFC 5869 HKDF-SHA256 built on libsodium's HMAC-SHA256 (libsodium 1.0.18 has + * no native HKDF; crypto_auth_hmacsha256_init accepts any key length). Every + * use in this module needs exactly 32 output bytes (= HashLen), so the expand + * phase is a single iteration: OKM = HMAC(PRK, info || 0x01). */ +static int cl_ctr_hkdf_sha256(const unsigned char *ikm, size_t ikm_len, + const unsigned char *salt, size_t salt_len, + const char *info, + unsigned char out[32]) +{ + crypto_auth_hmacsha256_state st; + unsigned char prk[crypto_auth_hmacsha256_BYTES]; + static const unsigned char zeros[crypto_auth_hmacsha256_BYTES]; + const unsigned char ctr = 0x01; + size_t info_len = info ? strlen(info) : 0; + + /* extract: PRK = HMAC(salt, IKM); an absent salt is HashLen zero bytes */ + if (crypto_auth_hmacsha256_init(&st, salt_len ? salt : zeros, + salt_len ? salt_len : sizeof(zeros)) != 0) + return -1; + crypto_auth_hmacsha256_update(&st, ikm, (unsigned long long)ikm_len); + crypto_auth_hmacsha256_final(&st, prk); + + /* expand: T(1) = HMAC(PRK, info || 0x01) */ + if (crypto_auth_hmacsha256_init(&st, prk, sizeof(prk)) != 0) + return -1; + if (info_len) + crypto_auth_hmacsha256_update(&st, (const unsigned char *)info, + (unsigned long long)info_len); + crypto_auth_hmacsha256_update(&st, &ctr, 1); + crypto_auth_hmacsha256_final(&st, out); + return 0; +} + +static int cl_ctr_gen_ecdh_keypair(unsigned char *privkey, unsigned char *pubkey) +{ + randombytes_buf(privkey, CL_CTR_PUBKEY_SZ); + if (crypto_scalarmult_base(pubkey, privkey) != 0) { + LM_ERR("clusterer_controller: X25519 keygen failed\n"); + return -1; + } + return 0; +} + +/* ========================================================================= + * Noise_NNpsk0_25519_ChaChaPoly_SHA256 - the join handshake. + * + * NNpsk0: ephemeral-only, mutual authentication via a pre-shared key (the + * Argon2id bootstrap key), forward secrecy via the ephemeral-ephemeral DH. + * -> psk, e (JOIN_REQ carries msg 1) + * <- e, ee (KEY_GRANT carries msg 2, whose AEAD payload = master_salt) + * Both endpoints run this identical code, so byte-for-byte interop with other + * Noise libraries is not required; the module verifies self-consistency and the + * security properties in its own test harness. See RFC "The Noise Protocol + * Framework" rev 34. HASHLEN = 32 (SHA-256), DHLEN = 32 (X25519). + * ========================================================================= */ +#define CL_CTR_NOISE_PROTO "Noise_NNpsk0_25519_ChaChaPoly_SHA256" + +static void cl_ctr_hmac256(const unsigned char *key, size_t keylen, + const unsigned char *data, size_t datalen, + unsigned char out[CL_CTR_NOISE_HASHLEN]) +{ + crypto_auth_hmacsha256_state st; + crypto_auth_hmacsha256_init(&st, key, keylen); + crypto_auth_hmacsha256_update(&st, data, datalen); + crypto_auth_hmacsha256_final(&st, out); +} + +/* Noise HKDF: chained HMAC outputs (num_outputs of 1..3, each HASHLEN bytes). */ +static void cl_ctr_noise_hkdf(const unsigned char ck[CL_CTR_NOISE_HASHLEN], + const unsigned char *ikm, size_t ikm_len, int num_outputs, + unsigned char o1[CL_CTR_NOISE_HASHLEN], + unsigned char o2[CL_CTR_NOISE_HASHLEN], + unsigned char o3[CL_CTR_NOISE_HASHLEN]) +{ + unsigned char tempkey[CL_CTR_NOISE_HASHLEN], buf[CL_CTR_NOISE_HASHLEN + 1]; + cl_ctr_hmac256(ck, CL_CTR_NOISE_HASHLEN, ikm, ikm_len, tempkey); /* extract */ + buf[0] = 0x01; + cl_ctr_hmac256(tempkey, CL_CTR_NOISE_HASHLEN, buf, 1, o1); + if (num_outputs >= 2) { + memcpy(buf, o1, CL_CTR_NOISE_HASHLEN); buf[CL_CTR_NOISE_HASHLEN] = 0x02; + cl_ctr_hmac256(tempkey, CL_CTR_NOISE_HASHLEN, buf, CL_CTR_NOISE_HASHLEN + 1, o2); + } + if (num_outputs >= 3) { + memcpy(buf, o2, CL_CTR_NOISE_HASHLEN); buf[CL_CTR_NOISE_HASHLEN] = 0x03; + cl_ctr_hmac256(tempkey, CL_CTR_NOISE_HASHLEN, buf, CL_CTR_NOISE_HASHLEN + 1, o3); + } + sodium_memzero(tempkey, sizeof tempkey); +} + +static void cl_ctr_cs_nonce(uint64_t n, unsigned char out[12]) +{ + int i; + memset(out, 0, 4); /* 32 bits of zeros */ + for (i = 0; i < 8; i++) out[4 + i] = (unsigned char)(n >> (8 * i)); /* LE64 */ +} +static int cl_ctr_cs_encrypt(cl_ctr_cipherstate_t *cs, const unsigned char *ad, size_t ad_len, + const unsigned char *pt, size_t pt_len, unsigned char *ct) +{ + unsigned char nonce[12]; unsigned long long clen = 0; + /* pt may legitimately be NULL for an empty payload (Noise msg 1 sends + * one); memcpy() is nonnull even for a zero length, so guard it. */ + if (!cs->has_key) { if (pt_len) memcpy(ct, pt, pt_len); return (int)pt_len; } + cl_ctr_cs_nonce(cs->n, nonce); + crypto_aead_chacha20poly1305_ietf_encrypt(ct, &clen, pt, pt_len, + ad, ad_len, NULL, nonce, cs->k); + cs->n++; + return (int)clen; +} +static int cl_ctr_cs_decrypt(cl_ctr_cipherstate_t *cs, const unsigned char *ad, size_t ad_len, + const unsigned char *ct, size_t ct_len, unsigned char *pt) +{ + unsigned char nonce[12]; unsigned long long plen = 0; + if (!cs->has_key) { if (ct_len) memcpy(pt, ct, ct_len); return (int)ct_len; } + cl_ctr_cs_nonce(cs->n, nonce); + if (crypto_aead_chacha20poly1305_ietf_decrypt(pt, &plen, NULL, ct, ct_len, + ad, ad_len, nonce, cs->k) != 0) + return -1; + cs->n++; + return (int)plen; +} + +static void cl_ctr_ss_mixhash(cl_ctr_symstate_t *s, const unsigned char *data, size_t len) +{ + crypto_hash_sha256_state st; + crypto_hash_sha256_init(&st); + crypto_hash_sha256_update(&st, s->h, CL_CTR_NOISE_HASHLEN); + crypto_hash_sha256_update(&st, data, len); + crypto_hash_sha256_final(&st, s->h); +} +static void cl_ctr_ss_init(cl_ctr_symstate_t *s) +{ + size_t nl = strlen(CL_CTR_NOISE_PROTO); /* <= HASHLEN, so pad it */ + memset(s->h, 0, CL_CTR_NOISE_HASHLEN); + memcpy(s->h, CL_CTR_NOISE_PROTO, nl); + memcpy(s->ck, s->h, CL_CTR_NOISE_HASHLEN); + s->cs.has_key = 0; s->cs.n = 0; +} +static void cl_ctr_ss_mixkey(cl_ctr_symstate_t *s, const unsigned char *ikm, size_t ikm_len) +{ + unsigned char o1[CL_CTR_NOISE_HASHLEN], o2[CL_CTR_NOISE_HASHLEN], o3[CL_CTR_NOISE_HASHLEN]; + cl_ctr_noise_hkdf(s->ck, ikm, ikm_len, 2, o1, o2, o3); + memcpy(s->ck, o1, CL_CTR_NOISE_HASHLEN); + memcpy(s->cs.k, o2, 32); s->cs.n = 0; s->cs.has_key = 1; +} +static void cl_ctr_ss_mixkeyhash(cl_ctr_symstate_t *s, const unsigned char *ikm, size_t ikm_len) +{ + unsigned char o1[CL_CTR_NOISE_HASHLEN], o2[CL_CTR_NOISE_HASHLEN], o3[CL_CTR_NOISE_HASHLEN]; + cl_ctr_noise_hkdf(s->ck, ikm, ikm_len, 3, o1, o2, o3); + memcpy(s->ck, o1, CL_CTR_NOISE_HASHLEN); + cl_ctr_ss_mixhash(s, o2, CL_CTR_NOISE_HASHLEN); + memcpy(s->cs.k, o3, 32); s->cs.n = 0; s->cs.has_key = 1; +} +static int cl_ctr_ss_encrypt_hash(cl_ctr_symstate_t *s, const unsigned char *pt, size_t pt_len, + unsigned char *ct) +{ + int cl = cl_ctr_cs_encrypt(&s->cs, s->h, CL_CTR_NOISE_HASHLEN, pt, pt_len, ct); + cl_ctr_ss_mixhash(s, ct, cl); + return cl; +} +static int cl_ctr_ss_decrypt_hash(cl_ctr_symstate_t *s, const unsigned char *ct, size_t ct_len, + unsigned char *pt) +{ + int pl = cl_ctr_cs_decrypt(&s->cs, s->h, CL_CTR_NOISE_HASHLEN, ct, ct_len, pt); + if (pl < 0) return -1; + cl_ctr_ss_mixhash(s, ct, ct_len); + return pl; +} + +/* Initiator: write msg 1 (-> psk, e). Stores e keypair in @e_priv for msg 2. */ +static int cl_ctr_noise_write1(cl_ctr_symstate_t *s, unsigned char e_priv[32], + const unsigned char psk[32], + const unsigned char *payload, size_t pl_len, + unsigned char *out) +{ + unsigned char e_pub[32]; + cl_ctr_ss_init(s); + cl_ctr_ss_mixkeyhash(s, psk, 32); /* psk token (psk0) */ + randombytes_buf(e_priv, 32); crypto_scalarmult_base(e_pub, e_priv); /* e */ + memcpy(out, e_pub, 32); cl_ctr_ss_mixhash(s, e_pub, 32); cl_ctr_ss_mixkey(s, e_pub, 32); + return 32 + cl_ctr_ss_encrypt_hash(s, payload, pl_len, out + 32); +} +/* Responder: read msg 1, then write msg 2 (<- e, ee) with @payload. One-shot, + * so no responder state is retained afterwards. @out gets msg 2. */ +static int cl_ctr_noise_respond(const unsigned char psk[32], + const unsigned char *msg1, size_t msg1_len, + const unsigned char *payload, size_t pl_len, + unsigned char *out) +{ + cl_ctr_symstate_t s; + unsigned char re[32], e_priv[32], e_pub[32], dh[32], scratch[64]; + if (msg1_len < 32) return -1; + cl_ctr_ss_init(&s); + cl_ctr_ss_mixkeyhash(&s, psk, 32); + memcpy(re, msg1, 32); cl_ctr_ss_mixhash(&s, re, 32); cl_ctr_ss_mixkey(&s, re, 32); + if (cl_ctr_ss_decrypt_hash(&s, msg1 + 32, msg1_len - 32, scratch) < 0) + return -1; /* bad msg 1 (wrong psk/tamper) */ + randombytes_buf(e_priv, 32); crypto_scalarmult_base(e_pub, e_priv); + memcpy(out, e_pub, 32); cl_ctr_ss_mixhash(&s, e_pub, 32); cl_ctr_ss_mixkey(&s, e_pub, 32); + if (crypto_scalarmult(dh, e_priv, re) != 0) return -1; /* ee */ + cl_ctr_ss_mixkey(&s, dh, 32); + return 32 + cl_ctr_ss_encrypt_hash(&s, payload, pl_len, out + 32); +} +/* Initiator: read msg 2 (-> e, ee), recover @payload (master_salt). */ +static int cl_ctr_noise_read2(cl_ctr_symstate_t *s, const unsigned char e_priv[32], + const unsigned char *msg2, size_t msg2_len, + unsigned char *payload_out) +{ + unsigned char re2[32], dh[32], scratch[64]; + int pl; + /* decrypt into a scratch large enough for the max ciphertext (mirrors + * the scratch[64] on the write2 side), then hand back exactly the + * master_salt. Bounding the decrypt output here keeps a malformed or + * oversized KEY_GRANT from ever writing past payload_out. */ + if (msg2_len < 32 || msg2_len - 32 > sizeof scratch) return -1; + memcpy(re2, msg2, 32); cl_ctr_ss_mixhash(s, re2, 32); cl_ctr_ss_mixkey(s, re2, 32); + if (crypto_scalarmult(dh, e_priv, re2) != 0) return -1; + cl_ctr_ss_mixkey(s, dh, 32); + pl = cl_ctr_ss_decrypt_hash(s, msg2 + 32, msg2_len - 32, scratch); + if (pl != CL_CTR_MASTER_SALT_SZ) return -1; + memcpy(payload_out, scratch, CL_CTR_MASTER_SALT_SZ); + return pl; +} + +/** + * cl_ctr_derive_session_key() - derive group key from password + master_salt. + * Reads master_salt from cl->peers->master_salt (shm, caller holds write lock). + * Stores result in cl->session_key (worker-local cache). + * Must be called with cl->peers->lock held for WRITE. + */ +static int cl_ctr_derive_session_key(cl_ctr_cluster_t *cl) +{ + int i; + size_t pass_len = strlen(cl->password); + if (cl_ctr_hkdf_sha256((unsigned char *)cl->password, pass_len, + cl->peers->master_salt, CL_CTR_MASTER_SALT_SZ, + "cl_ctr_session", cl->session_key) < 0) { + LM_ERR("clusterer_controller: [cluster %d] session key derivation failed\n", + cl->cluster_id); + return -1; + } + /* Reset sequence counters: old packets encrypted with the previous key + * fail AEAD authentication, so starting from 0 is safe. */ + cl->peers->my_seq = 0; + for (i = 0; i < cl->peers->count; i++) + cl->peers->entries[i].last_seq = 0; + cl->have_session_key = 1; /* a valid group key now exists */ + /* The salt (and my_seq) just changed, so any queued retransmit is now stale. */ + cl_ctr_retx_flush(cl); + return 0; +} + + +/** + * cl_ctr_password_entropy_bits() - conservative estimate of a password's entropy. + * bits ~= length * floor(log2(charset)), where charset is the union of the + * character classes present. floor() makes it a slight under-estimate, so the + * weak-password warning errs toward firing. No libm dependency. + */ +static int cl_ctr_password_entropy_bits(const char *p) +{ + int have_lower = 0, have_upper = 0, have_digit = 0, have_sym = 0; + int charset, bits_per_char = 0, tmp; + size_t i, len = strlen(p); + + for (i = 0; i < len; i++) { + unsigned char c = (unsigned char)p[i]; + if (c >= 'a' && c <= 'z') have_lower = 1; + else if (c >= 'A' && c <= 'Z') have_upper = 1; + else if (c >= '0' && c <= '9') have_digit = 1; + else have_sym = 1; + } + charset = have_lower * 26 + have_upper * 26 + have_digit * 10 + have_sym * 33; + if (charset < 2) + charset = 2; + for (tmp = charset; tmp > 1; tmp >>= 1) + bits_per_char++; + return (int)(len * (size_t)bits_per_char); +} + +/** + * cl_ctr_derive_key() - derive the 32-byte bootstrap (admission) key from the + * shared password. Used only for the join handshake (JOIN_REQ / KEY_GRANT / + * JOIN_REJECT); normal traffic uses the ECDH-agreed session key. + * + * Uses scrypt (memory-hard) rather than a single SHA-256 so that a password + * captured from a JOIN packet cannot be brute-forced cheaply offline. Called + * once per cluster from mod_init() in the main process before fork, so the + * ~64 MiB scrypt working set is a transient startup cost only; workers inherit + * the derived key and never run scrypt. + */ +static int cl_ctr_derive_key(cl_ctr_cluster_t *cl) +{ + char salt[64]; + int saltlen, bits; + + /* Warn on a weak or default admission password. scrypt raises the cost of + * each offline guess, but only a high-entropy secret removes the risk. */ + if (strcmp(cl->password, CL_CTR_DEFAULT_PASSWORD) == 0) { + LM_WARN("clusterer_controller: [cluster %d] using the built-in default " + "password - set a strong 'password' (e.g. `openssl rand -base64 32`)\n", + cl->cluster_id); + } else { + bits = cl_ctr_password_entropy_bits(cl->password); + if (bits < CL_CTR_MIN_PASSWORD_BITS) + LM_WARN("clusterer_controller: [cluster %d] weak password (~%d bits " + "of entropy) - an attacker who captures a JOIN packet can " + "brute-force it offline; use a long random string, e.g. " + "`openssl rand -base64 32`\n", cl->cluster_id, bits); + } + + /* Per-cluster salt: a fixed domain-separation label plus the multicast + * address, so the same password on different clusters yields different + * bootstrap keys. The salt is public by design - Argon2id's work factor, + * not salt secrecy, is what defeats brute force. */ + saltlen = snprintf(salt, sizeof(salt), "opensips-cl-ctr-bootstrap-v1:%s", + cl->multicast_address); + if (saltlen < 0 || saltlen >= (int)sizeof(salt)) + saltlen = (int)strlen(salt); + + /* Argon2id. crypto_pwhash needs a fixed-length 16-byte salt, so fold our + * variable-length domain-separation salt into one with BLAKE2b. */ + { + unsigned char salt16[crypto_pwhash_SALTBYTES]; + crypto_generichash(salt16, sizeof(salt16), + (const unsigned char *)salt, (size_t)saltlen, NULL, 0); + if (crypto_pwhash(cl->key, 32, + cl->password, strlen(cl->password), + salt16, CL_CTR_ARGON2_OPSLIMIT, CL_CTR_ARGON2_MEMLIMIT, + crypto_pwhash_ALG_ARGON2ID13) != 0) { + LM_ERR("clusterer_controller: key derivation (Argon2id) failed for " + "cluster %d (out of memory?)\n", cl->cluster_id); + return -1; + } + } + return 0; +} + +/** + * cl_ctr_encrypt_pkt() - encrypt plaintext in-place and append the Poly1305 tag. + * + * On entry: buf[0..CL_CTR_MAGIC_SZ-1] = magic (set by caller) + * buf[plain_off..] = plaintext to encrypt + * On return: buf[CL_CTR_MAGIC_SZ..] = cleartext cluster_id (BE) + * buf[CL_CTR_NONCE_OFF..] = random nonce + * buf[plain_off..] = ciphertext (same length) + * buf[plain_off+plain_len..+CL_CTR_TAG_SZ-1] = Poly1305 tag + * + * @return total packet length, or -1 on error + */ +static int cl_ctr_encrypt_pkt(char *buf, int plain_off, int plain_len, + const unsigned char *key, int cluster_id) +{ + uint16_t cid_be = htons((uint16_t)cluster_id); + + memcpy(buf + CL_CTR_MAGIC_SZ, &cid_be, CL_CTR_CLUSTER_ID_SZ); /* cleartext selector */ + + /* AAD = cleartext header (magic + cluster_id): binding it to the tag stops + * a captured packet being re-stamped with another cluster_id on a shared + * multicast+password group (cross-cluster injection). The nonce is the + * AEAD IV, already bound. XChaCha20's 192-bit nonce makes random nonces + * collision-safe outright. */ + { + unsigned char nonce[CL_CTR_NONCE_SZ]; + unsigned long long clen = 0; + randombytes_buf(nonce, CL_CTR_NONCE_SZ); + memcpy(buf + CL_CTR_NONCE_OFF, nonce, CL_CTR_NONCE_SZ); + if (crypto_aead_xchacha20poly1305_ietf_encrypt( + (unsigned char *)buf + plain_off, &clen, + (const unsigned char *)buf + plain_off, (unsigned long long)plain_len, + (const unsigned char *)buf, CL_CTR_MAGIC_SZ + CL_CTR_CLUSTER_ID_SZ, + NULL, nonce, key) != 0) { + LM_ERR("clusterer_controller: XChaCha20-Poly1305 encrypt failed\n"); + return -1; + } + return plain_off + (int)clen; /* clen = plain_len + CL_CTR_TAG_SZ */ + } +} + +/** + * cl_ctr_decrypt_pkt() - authenticate and decrypt a received packet in-place. + * + * On entry: buf = [magic 2B][cluster_id 2B][nonce][ciphertext][tag 16B] + * On return: buf[CL_CTR_WIRE_HDR_SZ..] = plaintext (type + seq + payload) + * + * AAD must match cl_ctr_encrypt_pkt(): the cleartext header (magic+cluster_id), so + * a packet re-stamped with another cluster_id fails authentication. + * + * @return 0 on success, -1 to drop (wrong key, tampered, or too short) + */ +static int cl_ctr_decrypt_pkt(char *buf, ssize_t n, const char *sender_ip, + const unsigned char *key, int is_bootstrap) +{ + ssize_t cipher_len = n - CL_CTR_WIRE_HDR_SZ - CL_CTR_TAG_SZ; /* plaintext length */ + + if (cipher_len <= 0) { + LM_INFO("clusterer_controller: packet from %s too short to decrypt " + "(%zd bytes)\n", sender_ip, n); + return -1; + } + + { + unsigned char *nonce = (unsigned char *)buf + CL_CTR_NONCE_OFF; + unsigned long long mlen = 0; + if (crypto_aead_xchacha20poly1305_ietf_decrypt( + (unsigned char *)buf + CL_CTR_WIRE_HDR_SZ, &mlen, NULL, + (const unsigned char *)buf + CL_CTR_WIRE_HDR_SZ, + (unsigned long long)(cipher_len + CL_CTR_TAG_SZ), + (const unsigned char *)buf, CL_CTR_MAGIC_SZ + CL_CTR_CLUSTER_ID_SZ, + nonce, key) != 0) { + if (is_bootstrap) + LM_WARN("clusterer_controller: bootstrap decryption failed " + "from %s - wrong password, foreign cluster, or " + "tampered packet\n", sender_ip); + else + LM_DBG("clusterer_controller: session packet from %s did not " + "decrypt - transient key mismatch during (re)key or " + "split-brain heal\n", sender_ip); + return -1; + } + } + return 0; +} + +/** + * cl_ctr_check_and_update_seq() - reject replayed or reordered packets. + * Looks up sender_ip in the peer table; requires pkt_seq > last_seq. + * Updates last_seq on accept. Unknown senders (new nodes not yet in + * the peer table) are accepted so their first packet (ALIVE/JOIN_REQ) + * can populate the table. + * Only called for CL_CTR_PACKET_MAGIC packets; bootstrap packets use join_nonce. + * Single-threaded caller (cl_ctr_worker reactor); no lock needed for the check. + * @return 0 to accept, -1 to drop. + */ +static int cl_ctr_check_and_update_seq(const char *sender_ip, uint32_t pkt_seq, + cl_ctr_cluster_t *cl) +{ + int i; + for (i = 0; i < cl->peers->count; i++) { + if (strcmp(cl->peers->entries[i].ip, sender_ip) == 0) { + if (pkt_seq <= cl->peers->entries[i].last_seq) { + LM_WARN("clusterer_controller: replay from %s seq=%u last=%u, dropping\n", + sender_ip, pkt_seq, cl->peers->entries[i].last_seq); + return -1; + } + cl->peers->entries[i].last_seq = pkt_seq; + return 0; + } + } + return 0; /* unknown sender: accept, handler will upsert into peer table */ +} + +/* ========================================================================= + * Socket setup + * ========================================================================= */ + +static int cl_ctr_setup_socket(cl_ctr_cluster_t *cl) +{ + int sock; + int yes = 1; + unsigned char loop = 1, ttl = 32; + struct sockaddr_in local; + struct ip_mreq mreq; + + sock = socket(AF_INET, SOCK_DGRAM, 0); + if (sock < 0) { + LM_ERR("clusterer_controller: socket(): %s\n", strerror(errno)); + return -1; + } + + if (setsockopt(sock, SOL_SOCKET, SO_REUSEADDR, &yes, sizeof(yes)) < 0) { + LM_ERR("clusterer_controller: SO_REUSEADDR: %s\n", strerror(errno)); + close(sock); + return -1; + } + + /* Expand the kernel receive buffer so it can hold a fully reassembled + * MEMBER_LIST datagram (up to ~4 KB with 256 peers) even when IP + * fragmentation is in play on a low-MTU link such as PPP at 128 bytes. */ + { + int rcvbuf = 1 << 20; /* request 1 MB; kernel may cap lower */ + if (setsockopt(sock, SOL_SOCKET, SO_RCVBUF, + &rcvbuf, sizeof(rcvbuf)) < 0) + LM_WARN("clusterer_controller: SO_RCVBUF: %s\n", strerror(errno)); + } + + memset(&local, 0, sizeof(local)); + local.sin_family = AF_INET; + local.sin_port = htons((uint16_t)cl->multicast_port); + local.sin_addr.s_addr = htonl(INADDR_ANY); + + if (bind(sock, (struct sockaddr *)&local, sizeof(local)) < 0) { + LM_ERR("clusterer_controller: bind() port %d: %s\n", + cl->multicast_port, strerror(errno)); + close(sock); + return -1; + } + + memset(&mreq, 0, sizeof(mreq)); + mreq.imr_multiaddr.s_addr = inet_addr(cl->multicast_address); + /* Bind the join to our OWN resolved interface (my_ip, set by + * cl_ctr_resolve_local_identity() before this runs), not + * INADDR_ANY. INADDR_ANY leaves the actual interface choice to + * the kernel's default-route selection, which is NOT necessarily + * the interface this cluster was configured for - on a host whose + * default route goes out a different (e.g. external-facing) + * interface than the one named by the "interface" modparam, the + * join silently lands on the wrong interface: setsockopt still + * succeeds (no error logged), but peers on the intended interface + * never see it, so every node times out waiting for a master that + * was never reachable and self-elects instead. Found live: two + * production nodes each joined on their external interface + * (matching their default route) instead of the internal one the + * cluster was actually meant to run on, and neither ever saw the + * other. The send side already pins to my_ip a few lines below + * (local_if.s_addr = inet_addr(my_ip)) - this makes the join side + * consistent with it. */ + mreq.imr_interface.s_addr = inet_addr(my_ip); + + if (setsockopt(sock, IPPROTO_IP, IP_ADD_MEMBERSHIP, + &mreq, sizeof(mreq)) < 0) { + LM_ERR("clusterer_controller: IP_ADD_MEMBERSHIP (%s): %s\n", + cl->multicast_address, strerror(errno)); + close(sock); + return -1; + } + + if (setsockopt(sock, IPPROTO_IP, IP_MULTICAST_LOOP, + &loop, sizeof(loop)) < 0) + LM_WARN("clusterer_controller: IP_MULTICAST_LOOP: %s\n", + strerror(errno)); + + if (setsockopt(sock, IPPROTO_IP, IP_MULTICAST_TTL, + &ttl, sizeof(ttl)) < 0) + LM_WARN("clusterer_controller: IP_MULTICAST_TTL: %s\n", + strerror(errno)); + + /* Pin the sending interface to my_ip so that loopback packets carry + * my_ip as source address - this makes self-loopback detection in + * cl_ctr_handle_member_list() reliable on multi-homed hosts. */ + { + struct in_addr local_if; + local_if.s_addr = inet_addr(my_ip); + if (setsockopt(sock, IPPROTO_IP, IP_MULTICAST_IF, + &local_if, sizeof(local_if)) < 0) + LM_WARN("clusterer_controller: [cluster %d] IP_MULTICAST_IF: %s\n", + cl->cluster_id, + strerror(errno)); + } + + /* Bind socket to the resolved interface by name for stricter routing. + * This is more reliable than IP_MULTICAST_IF alone on multi-homed hosts + * because it works at the socket level regardless of routing tables. */ + if (my_interface_buf[0] != '\0') { + struct ifreq ifr; + memset(&ifr, 0, sizeof(ifr)); + memcpy(ifr.ifr_name, my_interface_buf, + strnlen(my_interface_buf, IF_NAMESIZE - 1)); + if (setsockopt(sock, SOL_SOCKET, SO_BINDTODEVICE, + &ifr, sizeof(ifr)) < 0) + /* Requires CAP_NET_RAW - not available after privilege drop. + * IP_MULTICAST_IF (set above) already pins the interface by + * IP, so this is belt-and-suspenders only; failure is safe. */ + LM_DBG("clusterer_controller: SO_BINDTODEVICE (%s): %s " + "(non-fatal, IP_MULTICAST_IF covers this)\n", + my_interface_buf, strerror(errno)); + } + + LM_INFO("clusterer_controller: [cluster %d] socket ready, joined %s:%d\n", + cl->cluster_id, cl->multicast_address, cl->multicast_port); + + /* cl->mcast_dest was resolved in mod_init (main process) and inherited + * across fork; no need to rebuild it here. */ + + /* Set non-blocking so sendto() never hangs the worker if the kernel + * UDP send buffer fills up. recvfrom() already relies on select() + * for readiness, so O_NONBLOCK is safe and consistent for both. */ + if (fcntl(sock, F_SETFL, fcntl(sock, F_GETFL, 0) | O_NONBLOCK) < 0) { + LM_WARN("clusterer_controller: fcntl O_NONBLOCK: %s\n", + strerror(errno)); + /* non-fatal - we continue; send paths handle EAGAIN explicitly */ + } + + return sock; +} + +/* ========================================================================= + * Packet senders + * ========================================================================= */ + +/** + * cl_ctr_seal_and_send_to() - encrypt a prepared packet in place and send it to + * an explicit destination. + * + * On entry @pkt holds the cleartext framing (magic already stamped) plus the + * @plain_len-byte plaintext starting at CL_CTR_WIRE_HDR_SZ; cl_ctr_encrypt_pkt() fills + * the cluster_id + nonce and appends the AEAD tag. @dest/@destlen is the target + * sockaddr: the cluster's multicast group for group packets (see the + * cl_ctr_seal_and_send() wrapper), or a single peer for the strictly 1:1 packets + * of the join handshake (e.g. KEY_GRANT), which every other member would + * otherwise decrypt only to discard. @dest is passed straight to sendto(), so + * its address family is whatever the caller supplies - v4 today, v6-ready. + * @type is only used for logging. + * + * @return 0 on success, -1 on encrypt or send failure. + */ +static int cl_ctr_seal_and_send_to(int sock, cl_ctr_cluster_t *cl, char *pkt, + int plain_len, const unsigned char *key, + unsigned char type, + const struct sockaddr *dest, socklen_t destlen) +{ + int total_len; + + /* Always unicast to @dest as the caller asked. When several local clusters + * share a multicast port a unicast reply may be delivered by the kernel to + * the wrong sibling's socket (unicast is demuxed by port alone); the + * receiver recovers it by cluster_id in cl_ctr_maybe_forward(), so we no + * longer fall back to the multicast group here. */ + total_len = cl_ctr_encrypt_pkt(pkt, CL_CTR_WIRE_HDR_SZ, plain_len, key, + cl->cluster_id); + if (total_len < 0) + return -1; + + if (sendto(sock, pkt, total_len, 0, dest, destlen) < 0) { + if (errno == EAGAIN || errno == EWOULDBLOCK) + LM_DBG("clusterer_controller: [cluster %d] sendto (type=0x%02x) " + "would block\n", cl->cluster_id, type); + else + LM_ERR("clusterer_controller: [cluster %d] sendto (type=0x%02x): %s\n", + cl->cluster_id, type, strerror(errno)); + return -1; + } + LM_DBG("clusterer_controller: [cluster %d] sent 0x%02x\n", cl->cluster_id, type); + return 0; +} + +/** + * cl_ctr_seal_and_send() - multicast wrapper over cl_ctr_seal_and_send_to(). + * + * Sends to the cluster's multicast group (cached in cl->mcast_dest); used by + * every group packet (ALIVE, MASTER_ALIVE, MEMBER_LIST, NODE_ASSIGN, ...). + */ +static int cl_ctr_seal_and_send(int sock, cl_ctr_cluster_t *cl, char *pkt, int plain_len, + const unsigned char *key, unsigned char type) +{ + return cl_ctr_seal_and_send_to(sock, cl, pkt, plain_len, key, type, + (const struct sockaddr *)&cl->mcast_dest, + sizeof(cl->mcast_dest)); +} + +/* ========================================================================= + * Reliable delivery: ACK + bounded retransmit for 1:1 handshake packets. + * All of this runs only in the controller worker, so the queue needs no lock. + * ========================================================================= */ + +/* Arm a timerfd with microsecond resolution (one-shot; 0/0 disarms). */ +static void cl_ctr_arm_tfd_us(int tfd, uint64_t usec_value, uint64_t usec_interval) +{ + struct itimerspec its; + memset(&its, 0, sizeof(its)); + its.it_value.tv_sec = (time_t)(usec_value / 1000000); + its.it_value.tv_nsec = (long)((usec_value % 1000000) * 1000); + its.it_interval.tv_sec = (time_t)(usec_interval / 1000000); + its.it_interval.tv_nsec = (long)((usec_interval % 1000000) * 1000); + if (timerfd_settime(tfd, 0, &its, NULL) < 0) + LM_WARN("clusterer_controller: timerfd_settime(us): %s\n", strerror(errno)); +} + +/* + * Drop every outstanding retransmit. Called on loss of mastership and on + * session-key rotation, so a demoted or re-keyed node never keeps delivering a + * stale master-keyed KEY_GRANT/NODE_ASSIGN that would push a joiner onto a dead + * key - the split-brain guard for the ARQ layer. ACKs never influence election + * or membership, so dropping these entries can only stop redundant sends. + */ +static void cl_ctr_retx_flush(cl_ctr_cluster_t *cl) +{ + if (cl->retx_count == 0) + return; + LM_DBG("clusterer_controller: [cluster %d] flushing %d pending retransmit(s)\n", + cl->cluster_id, cl->retx_count); + memset(cl->retx_q, 0, sizeof(cl->retx_q)); + cl->retx_count = 0; + cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); /* disarm */ +} + +/* + * Queue a just-sent 1:1 packet (already sealed in @pkt, @pkt_len bytes) for + * retransmit until ACKed. Best-effort: if the queue is full the packet still + * went out once and the joiner's JOIN_REQ retry remains the backstop. + */ +static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned char type, + const unsigned char *pkt, int pkt_len, + const struct sockaddr *dest, socklen_t destlen) +{ + cl_ctr_retx_entry_t *e; + int i, slot = -1, was_empty; + + if (pkt_len <= 0 || pkt_len > (int)sizeof(cl->retx_q[0].pkt) || + destlen == 0 || destlen > (socklen_t)sizeof(cl->retx_q[0].dest)) + return; + + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) + if (!cl->retx_q[i].used) { slot = i; break; } + if (slot < 0) { + LM_DBG("clusterer_controller: [cluster %d] retransmit queue full, " + "0x%02x sent best-effort\n", cl->cluster_id, type); + return; + } + + was_empty = (cl->retx_count == 0); + e = &cl->retx_q[slot]; + memset(e, 0, sizeof(*e)); + e->used = 1; + e->seq = seq; + e->type = type; + e->retries_left = CL_CTR_RETX_MAX_RETRIES; + e->next_due_us = get_uticks() + CL_CTR_RETX_INTERVAL_US; + memcpy(&e->dest, dest, destlen); + e->destlen = destlen; + memcpy(e->pkt, pkt, pkt_len); + e->pkt_len = pkt_len; + cl->retx_count++; + if (was_empty) + cl_ctr_arm_tfd_us(cl->retx_tfd, CL_CTR_RETX_INTERVAL_US, 0); +} + +/* An ACK arrived: drop the queued packet whose seq it echoes. */ +static void cl_ctr_handle_ack(const char *payload, int payload_len, cl_ctr_cluster_t *cl) +{ + uint32_t acked_be, acked; + int i; + + if (payload_len < (int)sizeof(uint32_t)) + return; + memcpy(&acked_be, payload, sizeof(acked_be)); + acked = ntohl(acked_be); + + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) { + cl_ctr_retx_entry_t *e = &cl->retx_q[i]; + if (!e->used || e->seq != acked) + continue; + /* seq is unique within a key epoch (a rekey flushes the queue), so this + * is the acknowledged packet; drop it. */ + LM_DBG("clusterer_controller: [cluster %d] ACK for 0x%02x seq %u\n", + cl->cluster_id, e->type, acked); + e->used = 0; + cl->retx_count--; + break; + } + if (cl->retx_count == 0) + cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); /* nothing pending - disarm */ +} + +/* + * Retransmit sweep: resend every entry whose deadline has passed, one step of + * its budget at a time; drop an entry once the budget is spent (the joiner's + * JOIN_REQ retry is the outer backstop). Re-arms only while work remains. + */ +static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + utime_t now = get_uticks(); + int i; + (void)was_timeout; + + cl_ctr_drain_tfd(fd); + + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) { + cl_ctr_retx_entry_t *e = &cl->retx_q[i]; + if (!e->used || now < e->next_due_us) + continue; + + if (sendto(cl->sock, e->pkt, e->pkt_len, 0, + (struct sockaddr *)&e->dest, e->destlen) < 0 && + errno != EAGAIN && errno != EWOULDBLOCK) + LM_DBG("clusterer_controller: [cluster %d] retransmit 0x%02x: %s\n", + cl->cluster_id, e->type, strerror(errno)); + + if (--e->retries_left <= 0) { + LM_DBG("clusterer_controller: [cluster %d] 0x%02x (seq %u) unacked " + "after %d retransmits, giving up - joiner will re-JOIN_REQ\n", + cl->cluster_id, e->type, e->seq, CL_CTR_RETX_MAX_RETRIES); + e->used = 0; + cl->retx_count--; + } else { + e->next_due_us = now + CL_CTR_RETX_INTERVAL_US; + } + } + + if (cl->retx_count > 0) + cl_ctr_arm_tfd_us(cl->retx_tfd, CL_CTR_RETX_INTERVAL_US, 0); + return 0; +} + +/* + * Send an ACK for @acked_seq back to @dest. Sealed in the same key class as the + * packet being acknowledged (@use_bootstrap: KEY_GRANT is bootstrap-keyed), so + * the sender can read it without assuming the receiver already holds a key. + */ +static void cl_ctr_send_ack(int sock, cl_ctr_cluster_t *cl, uint32_t acked_seq, + int use_bootstrap, const struct sockaddr *dest, socklen_t destlen) +{ + char pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + sizeof(uint32_t) + + CL_CTR_TAG_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint32_t acked_be = htonl(acked_seq); + const unsigned char *key = use_bootstrap ? cl->key : cl->session_key; + + memcpy(pkt, use_bootstrap ? CL_CTR_BOOTSTRAP_MAGIC : CL_CTR_PACKET_MAGIC, + CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_ACK; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ, &acked_be, sizeof(acked_be)); + + cl_ctr_seal_and_send_to(sock, cl, pkt, CL_CTR_PLAIN_HDR_SZ + (int)sizeof(acked_be), + key, CL_CTR_PKT_ACK, dest, destlen); +} + +/** + * cl_ctr_send_pkt_with_ip() - build and multicast a small (ALIVE/GOODBYE) packet. + * JOIN_REQ is handled by cl_ctr_send_join_req_pkt() which carries BIN socket info. + * + * ALIVE: [type 1B][seq 4B][ip NUL][pubkey 32B] - peers learn our pubkey here + * GOODBYE: [type 1B][seq 4B][ip NUL] - no pubkey needed + */ +static void cl_ctr_send_pkt_with_ip(int sock, unsigned char type, cl_ctr_cluster_t *cl, + const struct sockaddr *dest, socklen_t destlen) +{ + /* Sized for ALIVE which carries an extra pubkey + config descriptor */ + char pkt[CL_CTR_SMALL_PKT_SZ + CL_CTR_PUBKEY_SZ + CL_CTR_CONFIG_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + int ip_len = (int)strlen(my_ip); + int plain_len; + + if (ip_len > CL_CTR_MAX_IP_LEN) + ip_len = CL_CTR_MAX_IP_LEN; + + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + + pkt[CL_CTR_WIRE_HDR_SZ] = (char)type; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1 + CL_CTR_SEQ_SZ, my_ip, ip_len); + pkt[CL_CTR_WIRE_HDR_SZ + 1 + CL_CTR_SEQ_SZ + ip_len] = '\0'; + plain_len = 1 + CL_CTR_SEQ_SZ + ip_len + 1; + + /* ALIVE appends our X25519 public key so peers accumulate pubkeys + * without bloating MEMBER_LIST (avoids excessive IP fragmentation). */ + if (type == CL_CTR_PKT_ALIVE) { + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + plain_len, cl->my_pubkey, CL_CTR_PUBKEY_SZ); + plain_len += CL_CTR_PUBKEY_SZ; + /* Advertise our effective consistency-critical config so peers can + * detect accidental per-node config drift for the same cluster. */ + { + char *c = pkt + CL_CTR_WIRE_HDR_SZ + plain_len; + uint16_t qt = htons((uint16_t)(query_time & 0xFFFF)); + c[0] = (char)(cl->manage_shtags ? 1 : 0); + c[1] = (char)(cl->master_stickiness ? 1 : 0); + memcpy(c + 2, &qt, 2); + plain_len += CL_CTR_CONFIG_SZ; + } + } + + if (dest) + cl_ctr_seal_and_send_to(sock, cl, pkt, plain_len, cl->session_key, type, + dest, destlen); + else + cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->session_key, type); +} + +#define cl_ctr_send_alive(sock, cl) cl_ctr_send_pkt_with_ip((sock), CL_CTR_PKT_ALIVE, (cl), NULL, 0) +#define cl_ctr_send_alive_to(sock, cl, dest, dlen) \ + cl_ctr_send_pkt_with_ip((sock), CL_CTR_PKT_ALIVE, (cl), (dest), (dlen)) +#define cl_ctr_send_join_req(sock, cl) cl_ctr_send_join_req_pkt((sock), (cl)) + +/** + * cl_ctr_send_list_pkt() - encrypt and multicast the active peer table. + * + * Wire: [magic 2B][cluster_id 2B][nonce 24B][XChaCha20-Poly1305([type 1B][seq 4B][count 2B][entries...])][tag 16B] + */ +static void cl_ctr_send_list_pkt(int sock, unsigned char type, cl_ctr_cluster_t *cl, + const struct sockaddr *dest, socklen_t destlen) +{ + char pkt[CL_CTR_LIST_PKT_MAX_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint16_t count = 0; + uint16_t count_be; + char *p; + time_t cutoff; + int i, plain_len; + + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + /* nonce at [8..19] written by cl_ctr_encrypt_pkt */ + + /* Plaintext: [type][seq][count BE][forced_shtag_node_id BE][entries...] */ + pkt[CL_CTR_WIRE_HDR_SZ] = (char)type; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + /* count filled after iteration; forced_shtag_node_id filled below */ + p = pkt + CL_CTR_WIRE_HDR_SZ + 1 + CL_CTR_SEQ_SZ + CL_CTR_LIST_COUNT_SZ + CL_CTR_NODE_ID_SZ; + + cutoff = time(NULL) - (time_t)(query_time * CL_CTR_ELECT_FACTOR); + + lock_start_read(cl->peers->lock); + { + uint16_t forced_be = htons(cl->peers->shtag_forced_node_id); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1 + CL_CTR_SEQ_SZ + CL_CTR_LIST_COUNT_SZ, + &forced_be, CL_CTR_NODE_ID_SZ); + } + + for (i = 0; i < cl->peers->count && count < CL_CTR_MAX_PEERS; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + if (e->last_seen < cutoff) + continue; + /* Entry layout: [ip 16B null-padded][is_master 1B] = 17B */ + memset(p, 0, CL_CTR_IP_ENTRY_SZ); + memcpy(p, e->ip, strnlen(e->ip, CL_CTR_MAX_IP_LEN)); + p[CL_CTR_IP_ENTRY_SZ - 1] = (char)(e->is_master ? 1 : 0); + p += CL_CTR_IP_ENTRY_SZ; + count++; + } + + lock_stop_read(cl->peers->lock); + + count_be = htons(count); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1 + CL_CTR_SEQ_SZ, &count_be, CL_CTR_LIST_COUNT_SZ); + + plain_len = 1 + CL_CTR_SEQ_SZ + CL_CTR_LIST_COUNT_SZ + CL_CTR_NODE_ID_SZ + + count * CL_CTR_IP_ENTRY_SZ; + if ((dest ? cl_ctr_seal_and_send_to(sock, cl, pkt, plain_len, cl->session_key, + type, dest, destlen) + : cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->session_key, type)) == 0) + LM_INFO("clusterer_controller: [cluster %d] sent MEMBER_LIST (%d members)%s\n", + cl->cluster_id, count, dest ? " (unicast)" : ""); +} + +#define cl_ctr_send_member_list(sock, cl) cl_ctr_send_list_pkt((sock), CL_CTR_PKT_MEMBER_LIST, (cl), NULL, 0) + +/** + * cl_ctr_send_join_req_pkt() - send CL_CTR_PKT_JOIN_REQ with BIN socket info. + * + * Payload: [ip NUL][bin_count 1B][sock1 NUL]...[sockN NUL] + */ +static void cl_ctr_send_join_req_pkt(int sock, cl_ctr_cluster_t *cl) +{ + char pkt[CL_CTR_JOIN_PKT_MAX_SZ]; + uint32_t seq; + + /* Rate-limit JOIN_REQ transmissions. During a key-mismatch / split-brain + * heal several code paths (defer timer, session-mismatch re-key, rejoin + * timer) can each ask to (re)send a JOIN_REQ within the same second. A + * JOIN_REQ is idempotent - the master answers whichever one arrives - so + * drop any that lands within CL_CTR_JOIN_REQ_MIN_US of the previous send; the + * next timer tick resends if the join is still pending. The very first + * send (last_join_req_utime == 0) is never throttled. */ + { + utime_t now_us = get_uticks(); + if (cl->last_join_req_utime != 0 && + (utime_t)(now_us - cl->last_join_req_utime) < CL_CTR_JOIN_REQ_MIN_US) + return; + cl->last_join_req_utime = now_us; + } + + seq = htonl(++cl->peers->my_seq); + int ip_len = (int)strlen(my_ip); + char *p; + int plain_len; + + if (ip_len > CL_CTR_MAX_IP_LEN) + ip_len = CL_CTR_MAX_IP_LEN; + + memcpy(pkt, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ); /* JOIN_REQ uses bootstrap key */ + + /* Plaintext: [type][seq][ip NUL][bin_count][sock1 NUL]...[sockN NUL][pubkey 32B] */ + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_JOIN_REQ; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + + memcpy(p, my_ip, ip_len); + p[ip_len] = '\0'; + p += ip_len + 1; + + /* Advertise only the BIN socket resolved for this specific cluster */ + { + int slen = (int)strnlen(cl->bin_socket, CL_CTR_MAX_BIN_SOCK_LEN - 1); + *p++ = 1; /* bin_count */ + memcpy(p, cl->bin_socket, slen); + p[slen] = '\0'; + p += slen + 1; + } + + /* Noise msg 1 (-> psk, e): a fresh ephemeral public key plus an empty + * authenticated payload, PSK = the Argon2id bootstrap key (cl->key). Keep + * the initiator handshake state so we can read the master's KEY_GRANT + * (msg 2). A later JOIN_REQ overwrites it, so a stale KEY_GRANT just fails + * to decrypt and is dropped. */ + { + int m1 = cl_ctr_noise_write1(&cl->noise_hs, cl->noise_e_priv, cl->key, + NULL, 0, (unsigned char *)p); + cl->noise_hs_valid = 1; + p += m1; /* 48 bytes: 32 ephemeral + 16 tag */ + } + + /* Advertise our consistency-critical config so the master can reject (or + * warn about) a join with settings that differ from the running cluster. */ + { + uint16_t qt = htons((uint16_t)(query_time & 0xFFFF)); + *p++ = (char)(cl->manage_shtags ? 1 : 0); + *p++ = (char)(cl->master_stickiness ? 1 : 0); + memcpy(p, &qt, 2); + p += 2; + } + + plain_len = (int)(p - (pkt + CL_CTR_WIRE_HDR_SZ)); + if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->key, CL_CTR_PKT_JOIN_REQ) == 0) + LM_DBG("clusterer_controller: [cluster %d] sent JOIN_REQ bin=%s\n", + cl->cluster_id, cl->bin_socket); +} + +/** + * cl_ctr_send_node_assign() - send CL_CTR_PKT_NODE_ASSIGN to multicast. + * + * Payload: [node_id 2B BE][ip NUL][bin_count 1B][sock1 NUL]...[sockN NUL] + * + * Sent by master after allocating a node_id. All cluster members receive + * it and update their peer tables accordingly. + */ +static void cl_ctr_send_node_assign(int sock, const char *ip, uint16_t node_id, + uint8_t bin_count, + const char (*bin_sockets)[CL_CTR_MAX_BIN_SOCK_LEN], + cl_ctr_cluster_t *cl, + const struct sockaddr *dest, socklen_t destlen) +{ + char pkt[CL_CTR_NODE_ASSIGN_MAX_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint16_t nid_be = htons(node_id); + int ip_len = (int)strnlen(ip, CL_CTR_MAX_IP_LEN); + char *p; + int i, plain_len; + + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_NODE_ASSIGN; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + + /* node_id (2B BE) */ + memcpy(p, &nid_be, CL_CTR_NODE_ID_SZ); + p += CL_CTR_NODE_ID_SZ; + + /* IP NUL */ + memcpy(p, ip, ip_len); + p[ip_len] = '\0'; + p += ip_len + 1; + + /* BIN sockets */ + *p++ = (char)bin_count; + for (i = 0; i < bin_count; i++) { + int slen = (int)strnlen(bin_sockets[i], CL_CTR_MAX_BIN_SOCK_LEN - 1); + memcpy(p, bin_sockets[i], slen); + p[slen] = '\0'; + p += slen + 1; + } + + plain_len = (int)(p - (pkt + CL_CTR_WIRE_HDR_SZ)); + /* NODE_ASSIGN carries the CL_CTR_PACKET_MAGIC session magic, so the receiver + * decrypts it with session_key. It MUST therefore be encrypted with + * session_key, not the bootstrap key - otherwise every NODE_ASSIGN fails + * AEAD auth on receipt ("session key mismatch"), driving a JOIN_REQ storm. + * KEY_GRANT is sent before NODE_ASSIGN so the joiner already holds the + * session key by the time this arrives. */ + if ((dest ? cl_ctr_seal_and_send_to(sock, cl, pkt, plain_len, cl->session_key, + CL_CTR_PKT_NODE_ASSIGN, dest, destlen) + : cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->session_key, + CL_CTR_PKT_NODE_ASSIGN)) == 0) + LM_INFO("clusterer_controller: [cluster %d] NODE_ASSIGN node_id=%u ip=%s%s\n", + cl->cluster_id, node_id, ip, dest ? " (unicast)" : ""); +} + + +/* + * Order-independent digest of the active peer set: the XOR of a per-peer FNV-1a + * over (node_id, ip_num). A member that missed a peer's NODE_ASSIGN carries + * node_id 0 for it and so computes a different hash from the master's - which is + * how the gap is detected. Roles are deliberately excluded (they reconcile via + * MASTER_ALIVE itself) to avoid spurious resyncs on transient role churn. + * Takes the read lock itself - the caller must NOT hold cl->peers->lock. + */ +static void cl_ctr_membership_digest(cl_ctr_cluster_t *cl, uint16_t *count, uint64_t *hash) +{ + uint64_t h = 0; + uint16_t c = 0; + time_t cutoff = time(NULL) - (time_t)(query_time * CL_CTR_ELECT_FACTOR); + int i, k; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + uint64_t ph = 1469598103934665603ULL; /* FNV-1a offset basis */ + unsigned char b[6]; + if (e->last_seen < cutoff) + continue; + b[0] = (unsigned char)(e->node_id >> 8); b[1] = (unsigned char)e->node_id; + b[2] = (unsigned char)(e->ip_num >> 24); b[3] = (unsigned char)(e->ip_num >> 16); + b[4] = (unsigned char)(e->ip_num >> 8); b[5] = (unsigned char)e->ip_num; + for (k = 0; k < 6; k++) { ph ^= b[k]; ph *= 1099511628211ULL; } + h ^= ph; + c++; + } + lock_stop_read(cl->peers->lock); + *count = c; + *hash = h; +} + +/* Build a sockaddr_in for an IPv4 dotted string + port (for unicast sends). */ +static void cl_ctr_sockaddr_in(const char *ip, int port, struct sockaddr_in *out) +{ + memset(out, 0, sizeof(*out)); + out->sin_family = AF_INET; + out->sin_port = htons((uint16_t)port); + out->sin_addr.s_addr = inet_addr(ip); +} + +/* Fill @bm (CL_CTR_ALIVE_BITMAP_SZ bytes) with a bit set for every peer the + * master has seen inside the election window - the liveness it relays. */ +static void cl_ctr_build_alive_bitmap(cl_ctr_cluster_t *cl, unsigned char *bm) +{ + time_t cutoff = time(NULL) - (time_t)(query_time * CL_CTR_ELECT_FACTOR); + int i; + + memset(bm, 0, CL_CTR_ALIVE_BITMAP_SZ); + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + uint16_t nid = e->node_id; + if (e->last_seen < cutoff || nid == 0 || nid > CL_CTR_MAX_PEERS) + continue; + bm[nid >> 3] |= (unsigned char)(1u << (nid & 7)); + } + lock_stop_read(cl->peers->lock); +} + +/* Refresh last_seen for every peer whose node_id the master reports alive in @bm, + * so a settled non-master keeps its election window populated without hearing the + * peers' ALIVEs directly. Only a non-master applies this. */ +static void cl_ctr_apply_alive_bitmap(cl_ctr_cluster_t *cl, const unsigned char *bm) +{ + time_t now = time(NULL); + int i; + + lock_start_write(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + uint16_t nid = e->node_id; + if (nid == 0 || nid > CL_CTR_MAX_PEERS) + continue; + if (bm[nid >> 3] & (unsigned char)(1u << (nid & 7))) + e->last_seen = now; + } + lock_stop_write(cl->peers->lock); +} + +/** + * cl_ctr_send_master_alive() - master-only keepalive. Encrypted with session_key + * (CL_CTR_PACKET_MAGIC). Carries the membership digest ([count 2B][hash 8B]) so + * members can detect a missed NODE_ASSIGN and pull a RESYNC. + */ +static void cl_ctr_send_master_alive(int sock, cl_ctr_cluster_t *cl) +{ + char pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + + CL_CTR_DIGEST_SZ + CL_CTR_ALIVE_BITMAP_SZ + CL_CTR_TAG_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint16_t cnt; + uint64_t hash; + char *p; + int k; + + cl_ctr_membership_digest(cl, &cnt, &hash); + + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_MASTER_ALIVE; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + *p++ = (char)(cnt >> 8); *p++ = (char)cnt; + for (k = 0; k < 8; k++) *p++ = (char)(hash >> (56 - 8 * k)); + /* relay the per-node liveness bitmap so settled non-masters can refresh + * their election window without hearing peers' ALIVEs directly */ + cl_ctr_build_alive_bitmap(cl, (unsigned char *)p); + p += CL_CTR_ALIVE_BITMAP_SZ; + + cl_ctr_seal_and_send(sock, cl, pkt, + CL_CTR_PLAIN_HDR_SZ + CL_CTR_DIGEST_SZ + CL_CTR_ALIVE_BITMAP_SZ, + cl->session_key, CL_CTR_PKT_MASTER_ALIVE); +} + +/* + * cl_ctr_send_resync() - member -> master (unicast, session-keyed): "my membership + * view differs from your digest, please resend". Carries our own digest for + * the master's logs. Best-effort and rate-limited by the caller. + */ +static void cl_ctr_send_resync(int sock, cl_ctr_cluster_t *cl, + const struct sockaddr *dest, socklen_t destlen) +{ + char pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_DIGEST_SZ + + CL_CTR_TAG_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint16_t cnt; + uint64_t hash; + char *p; + int k; + + cl_ctr_membership_digest(cl, &cnt, &hash); + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_RESYNC; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + *p++ = (char)(cnt >> 8); *p++ = (char)cnt; + for (k = 0; k < 8; k++) *p++ = (char)(hash >> (56 - 8 * k)); + + cl_ctr_seal_and_send_to(sock, cl, pkt, CL_CTR_PLAIN_HDR_SZ + CL_CTR_DIGEST_SZ, + cl->session_key, CL_CTR_PKT_RESYNC, dest, destlen); +} + +/* + * cl_ctr_broadcast_full_state() - master re-multicasts every peer's NODE_ASSIGN + * (node_id + BIN mapping) followed by the MEMBER_LIST. This is the same state a + * join emits; here it repairs members that missed a NODE_ASSIGN. Master path. + */ +static void cl_ctr_broadcast_full_state(cl_ctr_cluster_t *cl) +{ + int i; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + if (e->node_id == 0) + continue; + cl_ctr_send_node_assign(cl->sock, e->ip, e->node_id, e->bin_count, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])e->bin_sockets, + cl, NULL, 0); /* re-broadcast to the group */ + } + lock_stop_read(cl->peers->lock); + cl_ctr_send_member_list(cl->sock, cl); +} + +/* + * cl_ctr_handle_resync() - a member reported a membership mismatch. Only the + * master answers, and coalesces a burst of RESYNCs into one full-state + * re-broadcast per CL_CTR_RESYNC_MIN_US. + */ +static void cl_ctr_handle_resync(const char *sender_ip, cl_ctr_cluster_t *cl) +{ + int i_am_master; + utime_t now; + + lock_start_read(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + lock_stop_read(cl->peers->lock); + if (!i_am_master) + return; + + now = get_uticks(); + if ((utime_t)(now - cl->last_full_state_us) < CL_CTR_RESYNC_MIN_US) { + LM_DBG("clusterer_controller: [cluster %d] RESYNC from %s coalesced\n", + cl->cluster_id, sender_ip); + return; + } + cl->last_full_state_us = now; + LM_INFO("clusterer_controller: [cluster %d] RESYNC from %s - re-broadcasting " + "full state\n", cl->cluster_id, sender_ip); + cl_ctr_broadcast_full_state(cl); +} + +/** + * cl_ctr_send_master_beacon() - master-only split-brain merge announcement. + * + * Encrypted with the BOOTSTRAP key (CL_CTR_BOOTSTRAP_MAGIC) - unlike MASTER_ALIVE, + * which uses the per-cluster session key. Every correctly-configured node + * shares the bootstrap key, so two masters that hold *different* session keys + * (e.g. after an all-node simultaneous cold start) can still read each other's + * beacon and reconcile. Payload is this partition's member count (2B BE); the + * sender IP comes from the datagram source. See cl_ctr_handle_master_beacon(). + */ +static void cl_ctr_send_master_beacon(int sock, cl_ctr_cluster_t *cl) +{ + char pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + 2 + CL_CTR_TAG_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + uint16_t cnt_be; + + lock_start_read(cl->peers->lock); + cnt_be = htons((uint16_t)cl->peers->count); + lock_stop_read(cl->peers->lock); + + memcpy(pkt, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_MASTER_BEACON; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ, &cnt_be, 2); + + cl_ctr_seal_and_send(sock, cl, pkt, CL_CTR_PLAIN_HDR_SZ + 2, cl->key, + CL_CTR_PKT_MASTER_BEACON); +} + +/** + * cl_ctr_send_key_grant() - send ECDH-wrapped master_salt to a joining node. + * Encrypted with bootstrap key (CL_CTR_BOOTSTRAP_MAGIC) so the joining node can + * read it before having the session key. Unicast to the joiner (@dest, its + * JOIN_REQ datagram source): it is strictly 1:1, so multicasting it only made + * every other member spend an AEAD decrypt to discard a KEY_GRANT not addressed + * to it. The wire format is unchanged (target_ip still carried), so this + * interoperates with peers that still multicast it. + * + * Runs the Noise responder over the joiner's msg 1 and answers with msg 2, + * whose AEAD payload is the master_salt. The Noise handshake authenticates the + * exchange (PSK = bootstrap key) and binds it to the joiner's ephemeral, so a + * stale KEY_GRANT for a superseded JOIN_REQ fails on the joiner's side. + * + * Payload: [target_ip NUL][noise_msg2 64B] + */ +static void cl_ctr_send_key_grant(int sock, const char *target_ip, cl_ctr_cluster_t *cl, + const unsigned char *msg1, int msg1_len, + const struct sockaddr *dest, socklen_t destlen) +{ + char pkt[CL_CTR_KEY_GRANT_SZ]; + uint32_t seq_h = ++cl->peers->my_seq; + uint32_t seq = htonl(seq_h); + unsigned char msg2[32 + CL_CTR_MASTER_SALT_SZ + CL_CTR_TAG_SZ]; + char *p; + int ip_len, plain_len, m2; + + m2 = cl_ctr_noise_respond(cl->key, msg1, (size_t)msg1_len, + cl->peers->master_salt, CL_CTR_MASTER_SALT_SZ, msg2); + if (m2 < 0) { + LM_DBG("clusterer_controller: [cluster %d] Noise msg1 from %s did not " + "verify - no KEY_GRANT sent\n", cl->cluster_id, target_ip); + return; + } + + ip_len = (int)strnlen(target_ip, CL_CTR_MAX_IP_LEN); + + memcpy(pkt, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_KEY_GRANT; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + + memcpy(p, target_ip, ip_len); p[ip_len] = '\0'; p += ip_len + 1; + memcpy(p, msg2, m2); p += m2; + + plain_len = (int)(p - (pkt + CL_CTR_WIRE_HDR_SZ)); + if (cl_ctr_seal_and_send_to(sock, cl, pkt, plain_len, cl->key, + CL_CTR_PKT_KEY_GRANT, dest, destlen) == 0) { + LM_INFO("clusterer_controller: [cluster %d] sent KEY_GRANT to %s (unicast)\n", + cl->cluster_id, target_ip); + /* Critical 1:1 packet - retransmit until the joiner ACKs it. */ + cl_ctr_retx_enqueue(cl, seq_h, CL_CTR_PKT_KEY_GRANT, (const unsigned char *)pkt, + CL_CTR_WIRE_HDR_SZ + plain_len + CL_CTR_TAG_SZ, dest, destlen); + } +} + +/** + * cl_ctr_send_key_handoff() - send master_salt to the next master before departing. + * Outer envelope is the session key (CL_CTR_PACKET_MAGIC); the salt itself is sealed + * to the next master's long-lived X25519 key (learned via ALIVE) with an + * anonymous crypto_box - only that node can open it. Sender authenticity comes + * from the session-key envelope (only cluster members can send it). + * + * Payload: [next_master_ip NUL][sealed_box 64B] + */ +static void cl_ctr_send_key_handoff(int sock, const char *next_master_ip, + const unsigned char *next_master_pubkey, + cl_ctr_cluster_t *cl) +{ + char pkt[CL_CTR_KEY_HANDOFF_SZ]; + uint32_t seq = htonl(++cl->peers->my_seq); + unsigned char sealed[crypto_box_SEALBYTES + CL_CTR_MASTER_SALT_SZ]; + char *p; + int ip_len, plain_len; + + if (crypto_box_seal(sealed, cl->peers->master_salt, CL_CTR_MASTER_SALT_SZ, + next_master_pubkey) != 0) { + LM_ERR("clusterer_controller: [cluster %d] crypto_box_seal failed\n", + cl->cluster_id); + return; + } + + ip_len = (int)strnlen(next_master_ip, CL_CTR_MAX_IP_LEN); + + memcpy(pkt, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_KEY_HANDOFF; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + p = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + + memcpy(p, next_master_ip, ip_len); p[ip_len] = '\0'; p += ip_len + 1; + memcpy(p, sealed, sizeof(sealed)); p += sizeof(sealed); + + plain_len = (int)(p - (pkt + CL_CTR_WIRE_HDR_SZ)); + if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->session_key, + CL_CTR_PKT_KEY_HANDOFF) == 0) + LM_INFO("clusterer_controller: [cluster %d] sent KEY_HANDOFF to %s\n", + cl->cluster_id, next_master_ip); +} + +/* ========================================================================= + * State transition helper + * ========================================================================= */ + +/** + * cl_ctr_on_became_master() - side effects when this node wins an election. + * Generates fresh master_salt, derives session_key, arms MASTER_ALIVE timer, + * disarms master_dead watchdog. Must be called with cl->peers->lock held WRITE. + */ +static void cl_ctr_on_became_master(cl_ctr_cluster_t *cl) +{ + cl->join_pending = 0; /* any pending re-key join is now moot */ + if (cl_ctr_random_bytes(cl->peers->master_salt, CL_CTR_MASTER_SALT_SZ) != 0) { + LM_ERR("clusterer_controller: [cluster %d] RNG for master_salt failed\n", + cl->cluster_id); + return; + } + cl_ctr_derive_session_key(cl); /* reads cl->peers->master_salt */ + LM_INFO("clusterer_controller: [cluster %d] became master - " + "new master_salt generated, session key rotated\n", cl->cluster_id); + /* Timer ops outside lock (timerfd is worker-local, no shm concern) */ +} + +/** + * cl_ctr_arm_master_timers() - arm/disarm keepalive timers after election. + * Call WITHOUT cl->peers->lock held. + * i_am_master: 1 = we won, arm MASTER_ALIVE, disarm dead-watchdog. + * 0 = we lost, arm dead-watchdog, disarm MASTER_ALIVE. + */ +static void cl_ctr_arm_master_timers(cl_ctr_cluster_t *cl, int i_am_master) +{ + cl->master_ka_armed = i_am_master ? 1 : 0; + if (i_am_master) { + cl_ctr_arm_tfd(cl->master_alive_tfd, CL_CTR_MASTER_KA_INTERVAL, CL_CTR_MASTER_KA_INTERVAL); + cl_ctr_arm_tfd(cl->master_dead_tfd, 0, 0); /* disarm watchdog */ + } else { + /* No longer master: drop any queued master->joiner retransmits so a + * demoted node can't keep pushing a joiner onto a now-stale key. */ + cl_ctr_retx_flush(cl); + cl_ctr_arm_tfd(cl->master_alive_tfd, 0, 0); /* disarm sender */ + cl_ctr_arm_tfd(cl->master_dead_tfd, CL_CTR_MASTER_KA_TIMEOUT, 0); + } +} + +/** + * cl_ctr_request_rekey() - ask the current key-holder for the session key. + * Sends a single JOIN_REQ (bootstrap key), guarded by join_pending so a fresh + * nonce is not stomped while one exchange is in flight. Used when a node is + * elected master but has not yet adopted the cluster key (KEY_GRANT lost). + * Call WITHOUT cl->peers->lock held. + */ +static void cl_ctr_request_rekey(cl_ctr_cluster_t *cl) +{ + if (cl->join_pending) + return; + cl->join_pending = 1; + cl_ctr_send_join_req(cl->sock, cl); +} + +/** + * cl_ctr_transition_to_active() - switch from CL_CTR_NODE_NEW to CL_CTR_NODE_ACTIVE. + * + * Disarms the join-phase timers, sends the first ALIVE immediately, then + * arms the periodic ALIVE timer. Called from both: + * - the join_tfd handler (deadline expired, fresh cluster), and + * - cl_ctr_handle_member_list (master responded before deadline). + * Must be called with cl->peers->lock NOT held. + */ +static void cl_ctr_transition_to_active(cl_ctr_cluster_t *cl) +{ + cl_ctr_arm_tfd(cl->join_tfd, 0, 0); /* disarm one-shot deadline */ + cl_ctr_arm_tfd(cl->rejoin_tfd, 0, 0); /* disarm JOIN_REQ retry */ + cl_ctr_send_alive(cl->sock, cl); /* first heartbeat now */ + cl_ctr_arm_tfd(cl->alive_tfd, query_time, query_time); /* periodic from here */ +} + +/* ========================================================================= + * Packet handlers + * ========================================================================= */ + +/** + * cl_ctr_fmt_cfg_diff() - render only the consistency-critical settings that + * actually DIFFER into 'out' (e.g. "manage_shtags cluster=1 node=0"), so a + * mismatch log names just the offending setting(s) rather than all three. + * 'la'/'lb' are the labels for the local/peer sides ("cluster"/"node" or + * "local"/"peer"). + */ +static void cl_ctr_fmt_cfg_diff(char *out, int outsz, const char *la, const char *lb, + int a_manage, int b_manage, int a_stick, int b_stick, + int a_qt, int b_qt) +{ + int n = 0; + out[0] = '\0'; +#define CL_CTR_DIFF_APPEND(cond, name, av, bv) \ + do { \ + if (cond) { \ + int _w = snprintf(out + n, (n < outsz) ? (outsz - n) : 0, \ + "%s" name " %s=%d %s=%d", \ + n ? ", " : "", la, (av), lb, (bv)); \ + if (_w > 0) { n += _w; if (n > outsz) n = outsz; } \ + } \ + } while (0) + CL_CTR_DIFF_APPEND(a_manage != b_manage, "manage_shtags", a_manage, b_manage); + CL_CTR_DIFF_APPEND(a_stick != b_stick, "master_stickiness", a_stick, b_stick); + CL_CTR_DIFF_APPEND(a_qt != b_qt, "query_time", a_qt, b_qt); +#undef CL_CTR_DIFF_APPEND +} + +/** + * cl_ctr_adopt_config() - adopt the running cluster's consistency-critical settings + * (on_config_mismatch=adopt). Called on a non-master node when the master's + * advertised config differs from ours. Call WITHOUT cl->peers->lock held. + */ +static void cl_ctr_adopt_config(cl_ctr_cluster_t *cl, int new_manage, int new_stick, + int new_qt, int is_active) +{ + int old_manage = cl->manage_shtags ? 1 : 0; + + LM_INFO("clusterer_controller: [cluster %d] adopting cluster settings from " + "master (manage_shtags %d->%d, master_stickiness %d->%d, " + "query_time %d->%d)\n", + cl->cluster_id, old_manage, new_manage ? 1 : 0, + cl->master_stickiness ? 1 : 0, new_stick ? 1 : 0, query_time, new_qt); + + cl->master_stickiness = new_stick ? 1 : 0; + + /* query_time is declared global but is a per-process copy after fork, and + * there is one worker process per cluster, so updating it here affects only + * THIS cluster's worker - never another cluster on a multi-cluster node. + * Re-arm the periodic ALIVE timer if we are already active; the election + * window and purge derive from query_time dynamically and need no re-arm. */ + if (new_qt >= 1 && new_qt != query_time) { + query_time = new_qt; + if (is_active) + cl_ctr_arm_tfd(cl->alive_tfd, query_time, query_time); + } + + /* Mirror the effective values into shm so cl_ctr_list_config (MI process) + * reports what is actually in force after adoption. */ + cl->peers->eff_manage_shtags = new_manage ? 1 : 0; + cl->peers->eff_master_stickiness = cl->master_stickiness; + cl->peers->eff_query_time = query_time; + + if ((new_manage ? 1 : 0) != old_manage) { + cl->manage_shtags = new_manage ? 1 : 0; + if (clctl_loaded) { + if (cl->manage_shtags) { + if (clctl.set_shtag_managed) + clctl.set_shtag_managed(cl->cluster_id); + } else if (clctl.unset_shtag_managed) { + clctl.unset_shtag_managed(cl->cluster_id); + } + } + /* Reconcile tag state under the new policy (no-op when now unmanaged). */ + cl_ctr_apply_shtags(cl); + } +} + +/** + * cl_ctr_handle_alive() - process a CL_CTR_PKT_ALIVE packet. + * + * Regular heartbeat path: upsert the sender, re-elect. + * Only called in CL_CTR_NODE_ACTIVE state; ignored while joining (cl_ctr_recv_one + * still dispatches them so the peer table builds up before the timeout). + */ +static void cl_ctr_handle_alive(const char *src_ip, + const unsigned char *pubkey, /* may be NULL */ + int cfg_present, int peer_manage, + int peer_stick, int peer_qt, + cl_ctr_cluster_t *cl) +{ + int prev_master, now_master; + int warn = 0, adopt = 0, is_active = 0, ent = -1; + char warn_ip[CL_CTR_MAX_IP_LEN + 1] = ""; + int loc_manage = cl->manage_shtags ? 1 : 0; + int loc_stick = cl->master_stickiness ? 1 : 0; + int loc_qt = query_time; + int mism = 0, changed = 0; + + lock_start_write(cl->peers->lock); + prev_master = cl_ctr_i_am_master_locked(cl); + cl_ctr_upsert_peer_locked(src_ip, cl); + { + int _i; + for (_i = 0; _i < cl->peers->count; _i++) { + cl_ctr_peer_t *e = &cl->peers->entries[_i]; + if (strcmp(e->ip, src_ip) != 0) + continue; + ent = _i; + /* Store pubkey so master can use it for KEY_HANDOFF / KEY_GRANT */ + if (pubkey) + memcpy(e->pubkey, pubkey, CL_CTR_PUBKEY_SZ); + if (cfg_present) { + changed = (!e->cfg_known + || e->cfg_manage_shtags != peer_manage + || e->cfg_master_stickiness != peer_stick + || e->cfg_query_time != peer_qt); + mism = (peer_manage != loc_manage + || peer_stick != loc_stick + || peer_qt != loc_qt); + e->cfg_known = 1; + e->cfg_manage_shtags = peer_manage; + e->cfg_master_stickiness = peer_stick; + e->cfg_query_time = peer_qt; + } + break; + } + } + cl_ctr_elect_master(cl); + now_master = cl_ctr_i_am_master_locked(cl); + is_active = (cl->peers->node_state == CL_CTR_NODE_ACTIVE); + /* Config-consistency handling, decided once the master is known. In + * 'adopt' mode a non-master node takes the master's settings; otherwise a + * mismatch is logged once per peer (re-logged only if the peer's advertised + * config changes, cleared when it matches). cl_ctr_elect_master does not + * reorder entries, so the captured index stays valid. */ + if (cfg_present && ent >= 0) { + cl_ctr_peer_t *e = &cl->peers->entries[ent]; + int sender_is_master = (cl->peers->last_master[0] != '\0' + && strcmp(src_ip, cl->peers->last_master) == 0); + if (!mism) { + e->cfg_warned = 0; + } else if (on_config_mismatch == CL_CTR_CFGMISMATCH_ADOPT + && sender_is_master && !now_master) { + adopt = 1; + e->cfg_warned = 0; + } else if (changed || !e->cfg_warned) { + warn = 1; + e->cfg_warned = 1; + memcpy(warn_ip, e->ip, sizeof(warn_ip)); + } + } + lock_stop_write(cl->peers->lock); + + if (warn) { + char diff[160]; + cl_ctr_fmt_cfg_diff(diff, sizeof(diff), "local", "peer", + loc_manage, peer_manage, loc_stick, peer_stick, + loc_qt, peer_qt); + LM_WARN("clusterer_controller: [cluster %d] CONFIG MISMATCH with peer %s " + "- all nodes of a cluster MUST use identical settings; mismatched " + "values cause inconsistent failover/sharing-tag behaviour (%s)\n", + cl->cluster_id, warn_ip, diff); + } + if (adopt) + cl_ctr_adopt_config(cl, peer_manage, peer_stick, peer_qt, is_active); + /* Defer acting as master until we hold the cluster key. In normal + * operation a higher-IP joiner has already adopted the key via KEY_GRANT + * before winning here, so this only guards the pathological case where the + * KEY_GRANT was lost during the join. Request a re-key instead of + * broadcasting an undecryptable MASTER_ALIVE. */ + if (now_master && !cl->have_session_key) { + cl_ctr_request_rekey(cl); + return; + } + if (prev_master != now_master) + cl_ctr_arm_master_timers(cl, now_master); +} + +/** + * cl_ctr_handle_join_req() - process a CL_CTR_PKT_JOIN_REQ packet. + * + * Payload: [ip NUL][bin_count 1B][sock1 NUL]...[sockN NUL] + * + * Non-masters ignore JOIN_REQ - the master handles discovery exclusively. + * + * Master behaviour: + * 1. Parse joining node's IP and BIN sockets from payload. + * 2. Upsert peer; store BIN info and allocate a node_id. + * 3. Send NODE_ASSIGN (joining node + all existing peers) so every + * node in the cluster learns the full updated picture. + * 4. If joining IP > own IP: run election, send MEMBER_LIST. + * 5. If joining IP <= own IP: send MEMBER_LIST (self still master). + */ +static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_len, + cl_ctr_cluster_t *cl, + const struct sockaddr *src, socklen_t src_len) +{ + const char *p = payload; + const char *end = payload + payload_len; + char src_ip[CL_CTR_MAX_IP_LEN + 1]; + char bin_socks[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; + const unsigned char *noise_msg1 = NULL; /* points into payload */ + int noise_msg1_len = 0; + uint8_t bin_cnt = 0; + int ip_len, was_master, i; + int j_cfg_present = 0, j_manage = 0, j_stick = 0, j_qt = 0; + uint16_t new_id; + + /* --- Parse IP --- */ + ip_len = (int)strnlen(p, CL_CTR_MAX_IP_LEN); + if (p + ip_len >= end) { + LM_WARN("clusterer_controller: JOIN_REQ payload truncated\n"); + return; + } + memcpy(src_ip, p, ip_len); + src_ip[ip_len] = '\0'; + p += ip_len + 1; + + /* Ignore our own JOIN_REQ via loopback */ + if (strcmp(src_ip, my_ip) == 0) + return; + + /* --- Parse BIN sockets --- */ + memset(bin_socks, 0, sizeof(bin_socks)); + if (p < end) { + bin_cnt = (uint8_t)*p++; + if (bin_cnt > CL_CTR_MAX_BIN_SOCKETS) + bin_cnt = CL_CTR_MAX_BIN_SOCKETS; + for (i = 0; i < (int)bin_cnt && p < end; i++) { + int slen = (int)strnlen(p, CL_CTR_MAX_BIN_SOCK_LEN - 1); + memcpy(bin_socks[i], p, slen); + bin_socks[i][slen] = '\0'; + p += slen + 1; + } + } + + /* --- Parse the Noise msg 1 (appended after BIN info) --- */ + if (p + CL_CTR_NOISE_MSG1_SZ <= end) { + noise_msg1 = (const unsigned char *)p; + noise_msg1_len = CL_CTR_NOISE_MSG1_SZ; + p += CL_CTR_NOISE_MSG1_SZ; + } + + /* --- Parse the joiner's consistency-critical config (after join_nonce) --- */ + if (p + CL_CTR_CONFIG_SZ <= end) { + uint16_t qt_be; + j_manage = (unsigned char)p[0]; + j_stick = (unsigned char)p[1]; + memcpy(&qt_be, p + 2, 2); + j_qt = ntohs(qt_be); + j_cfg_present = 1; + p += CL_CTR_CONFIG_SZ; + } + + LM_INFO("clusterer_controller: [cluster %d] JOIN_REQ from %s " + "(%d BIN socket(s))\n", cl->cluster_id, src_ip, bin_cnt); + + lock_start_write(cl->peers->lock); + + was_master = (cl->peers->node_state == CL_CTR_NODE_ACTIVE) && + cl_ctr_i_am_master_locked(cl); + + if (!was_master) { + /* Split-brain prevention: while we are ourselves still joining, record + * the other joiner so the join-deadline election can defer to the + * highest-IP starter instead of every node self-promoting into a + * divergent-key lone master. (An active non-master simply ignores it; + * the master drives discovery.) */ + if (cl->peers->node_state == CL_CTR_NODE_NEW) { + cl_ctr_upsert_peer_locked(src_ip, cl); + /* A fresh JOIN_REQ from a *higher-IP* peer proves it is still alive + * and joining, so reset our split-brain defer budget: keep waiting + * for it to become master instead of prematurely self-promoting into + * a divergent-key split brain. join_defer_total still grows (hard + * cap) so a peer that is stuck forever cannot stall us indefinitely. */ + if (ip_to_num(src_ip) > ip_to_num(my_ip)) + cl->join_defer_count = 0; + } + lock_stop_write(cl->peers->lock); + LM_DBG("clusterer_controller: non-master ignoring JOIN_REQ from %s\n", + src_ip); + return; + } + + /* Config-consistency gate (master side): if the joiner advertises + * consistency-critical settings that differ from the running cluster and + * the policy is 'reject', refuse the join so the node shuts down rather + * than joining and behaving inconsistently. (warn/adopt admit the node; + * the joiner then warns or adopts on the master's ALIVE.) */ + if (j_cfg_present && on_config_mismatch == CL_CTR_CFGMISMATCH_REJECT) { + int loc_manage = cl->manage_shtags ? 1 : 0; + int loc_stick = cl->master_stickiness ? 1 : 0; + if (j_manage != loc_manage || j_stick != loc_stick || j_qt != query_time) { + char diff[160]; + lock_stop_write(cl->peers->lock); + cl_ctr_fmt_cfg_diff(diff, sizeof(diff), "cluster", "node", + loc_manage, j_manage, loc_stick, j_stick, + query_time, j_qt); + LM_WARN("clusterer_controller: [cluster %d] rejecting JOIN_REQ from %s: " + "different settings than the running cluster (%s)\n", + cl->cluster_id, src_ip, diff); + cl_ctr_send_join_reject(sock, src_ip, cl, CL_CTR_REJECT_CONFIG); + return; + } + } + + /* Reject JOIN_REQ from an unknown IP when the peer table is full. + * Known peers (reconnecting after restart) are still allowed through + * since they already occupy a slot. */ + if (cl->peers->count >= CL_CTR_MAX_PEERS) { + int _found = 0, _fi; + for (_fi = 0; _fi < cl->peers->count; _fi++) { + if (strcmp(cl->peers->entries[_fi].ip, src_ip) == 0) { + _found = 1; + break; + } + } + if (!_found) { + lock_stop_write(cl->peers->lock); + LM_WARN("clusterer_controller: [cluster %d] peer table full " + "(%d/%d), rejecting JOIN_REQ from %s\n", + cl->cluster_id, cl->peers->count, CL_CTR_MAX_PEERS, src_ip); + cl_ctr_send_join_reject(sock, src_ip, cl, CL_CTR_REJECT_GENERIC); + return; + } + } + + /* Upsert peer and store BIN info + node_id. + * If this IP already has a node_id (rejoining after crash/restart), + * reuse it so the id stays stable and clusterer stays in sync. */ + cl_ctr_upsert_peer_locked(src_ip, cl); + { + int _i; + new_id = 0; + for (_i = 0; _i < cl->peers->count; _i++) { + if (strcmp(cl->peers->entries[_i].ip, src_ip) == 0) { + new_id = cl->peers->entries[_i].node_id; + break; + } + } + if (new_id == 0) + new_id = cl_ctr_alloc_node_id_locked(cl); + } + cl_ctr_update_peer_bin_locked(src_ip, new_id, + bin_cnt, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, + cl); + + /* Reset last_seq in the peer table. Essential: a restarted node begins its + * seq counter from 0, and without this reset peers would permanently reject + * its new packets (old last_seq > new seq) until the next key rotation. The + * joiner's long-lived pubkey (for a future KEY_HANDOFF) is learned from its + * ALIVE, not here - the JOIN_REQ now carries only an ephemeral Noise key. */ + { + cl_ctr_peer_t *e = cl_ctr_peer_by_ip_locked(cl, src_ip); + if (e) + e->last_seq = 0; + } + + lock_stop_write(cl->peers->lock); + + /* JOIN_REQ decrypted successfully - clear any failure record for this IP + * so a node that fixes its password isn't immediately rejected again. */ + { + uint32_t _ip_num = ip_to_num(src_ip); + int _fi; + for (_fi = 0; _fi < CL_CTR_JOIN_FAIL_TABLE_SZ; _fi++) { + if (cl->join_fail_tbl[_fi].ip_num == _ip_num) { + memset(&cl->join_fail_tbl[_fi], 0, sizeof(cl->join_fail_tbl[_fi])); + break; + } + } + } + + /* Send KEY_GRANT first (bootstrap key) so joiner can derive session_key, + * then NODE_ASSIGN + MEMBER_LIST (session key). */ + if (noise_msg1) /* Noise msg 1 present -> run responder, answer msg 2 */ + cl_ctr_send_key_grant(sock, src_ip, cl, noise_msg1, noise_msg1_len, + src, src_len); + + lock_start_write(cl->peers->lock); + + /* Send NODE_ASSIGN for joining node */ + /* the newcomer's assignment must reach every member - multicast (a member + * that misses it self-heals via the membership digest). */ + cl_ctr_send_node_assign(sock, src_ip, new_id, bin_cnt, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, cl, NULL, 0); + + /* Send NODE_ASSIGN for each existing peer so the joining node learns all + * current node_ids and BIN sockets. Only the joiner needs these, so unicast + * them to it rather than re-multicasting to the whole cluster; a member that + * would once have re-learned a peer from this burst now self-heals via the + * membership digest instead. */ + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + if (strcmp(e->ip, src_ip) == 0 || e->node_id == 0) + continue; + cl_ctr_send_node_assign(sock, e->ip, e->node_id, e->bin_count, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])e->bin_sockets, + cl, src, src_len); + } + + /* A current master NEVER hands mastership to a joining node during the + * join handshake - doing so forced the joiner to broadcast MASTER_ALIVE + * before it had adopted the cluster key, producing a key-mismatch loop. + * + * Instead we stay master and designate ourselves in the MEMBER_LIST. The + * joiner adopts our session key via the KEY_GRANT sent above and joins as + * a member. If it has a higher IP it will win the very next ALIVE-driven + * election (cl_ctr_handle_alive) - but by then it holds the key, so when it + * takes over it broadcasts with a key every member already has. This + * keeps the deterministic "highest IP is master" outcome while deferring + * the actual takeover until after the key has been transferred. */ + lock_stop_write(cl->peers->lock); + LM_INFO("clusterer_controller: [cluster %d] I am master, " + "new node %s assigned node_id=%u\n", + cl->cluster_id, src_ip, new_id); + /* the joiner's authoritative snapshot - unicast to it (a member that misses + * it self-heals via the membership digest; the joiner, via its JOIN_REQ retry). */ + cl_ctr_send_list_pkt(sock, CL_CTR_PKT_MEMBER_LIST, cl, src, src_len); +} + +/** + * cl_ctr_handle_member_list() - process a CL_CTR_PKT_MEMBER_LIST from the master. + * + * This is THE authoritative packet for cluster state. All nodes - both + * the joining node and existing active members - update their peer tables + * and master designation directly from this list. No independent election + * is run; the master's word is final. + * + * Joining node (CL_CTR_NODE_NEW): + * - Replaces its empty peer table with the master's list. + * - Transitions to CL_CTR_NODE_ACTIVE. + * - If the list designates us as master (our IP has is_master=1): we + * accept mastership immediately and log accordingly. + * + * Active member / old master (CL_CTR_NODE_ACTIVE): + * - Upserts any new peers from the list. + * - Applies master designation from the list (may demote old master). + * - Logs who is now master. + */ +static void cl_ctr_handle_member_list(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl) +{ + uint16_t count; + uint16_t forced_node_id; + int i; + const char *p; + char designated_master[CL_CTR_MAX_IP_LEN + 1]; + + /* MEMBER_LIST is authoritative cluster state and must only come from the + * current master. Any cluster member holding the session key could forge + * one; accepting it would let an insider demote the real master, inject + * fake peers, or trigger spurious re-elections. + * Own loopback (sender == my_ip) is allowed when we are the master. */ + if (strcmp(sender_ip, my_ip) == 0) { + LM_DBG("clusterer_controller: ignoring own MEMBER_LIST loopback\n"); + return; + } + { + int _from_master; + lock_start_read(cl->peers->lock); + _from_master = (cl->peers->last_master[0] != '\0' && + strcmp(sender_ip, cl->peers->last_master) == 0); + lock_stop_read(cl->peers->lock); + if (!_from_master) { + /* Also allow during CL_CTR_NODE_NEW: we have no master yet, so any + * MEMBER_LIST is our first authoritative view of the cluster. */ + int _is_new; + lock_start_read(cl->peers->lock); + _is_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + lock_stop_read(cl->peers->lock); + if (!_is_new) { + LM_WARN("clusterer_controller: MEMBER_LIST from non-master %s " + "(master is %s), dropping\n", + sender_ip, + cl->peers->last_master[0] ? cl->peers->last_master : "(none)"); + return; + } + } + } + + designated_master[0] = '\0'; + + if (payload_len < CL_CTR_LIST_COUNT_SZ + CL_CTR_NODE_ID_SZ) { + LM_WARN("clusterer_controller: MEMBER_LIST too short\n"); + return; + } + + { + uint16_t count_be, forced_be; + memcpy(&count_be, payload, CL_CTR_LIST_COUNT_SZ); + count = ntohs(count_be); + /* forced-shtag node_id follows the count (0 = automatic allocation) */ + memcpy(&forced_be, payload + CL_CTR_LIST_COUNT_SZ, CL_CTR_NODE_ID_SZ); + forced_node_id = ntohs(forced_be); + } + + if (count > CL_CTR_MAX_PEERS) { + LM_WARN("clusterer_controller: MEMBER_LIST count %u exceeds " + "max peers %d, dropping\n", count, CL_CTR_MAX_PEERS); + return; + } + + if (payload_len < CL_CTR_LIST_COUNT_SZ + CL_CTR_NODE_ID_SZ + + (int)count * CL_CTR_IP_ENTRY_SZ) { + LM_WARN("clusterer_controller: MEMBER_LIST truncated " + "(count=%u, got %d bytes)\n", count, payload_len); + return; + } + + p = payload + CL_CTR_LIST_COUNT_SZ + CL_CTR_NODE_ID_SZ; + + /* First pass: collect designated master IP */ + for (i = 0; i < (int)count; i++) { + const char *entry = p + i * CL_CTR_IP_ENTRY_SZ; + unsigned char is_master = (unsigned char)entry[CL_CTR_IP_ENTRY_SZ - 1]; + if (is_master) { + memcpy(designated_master, entry, CL_CTR_MAX_IP_LEN); + designated_master[CL_CTR_MAX_IP_LEN] = '\0'; + break; + } + } + + lock_start_write(cl->peers->lock); + + /* Second pass: upsert all peers and reset their last_seq. + * Resetting last_seq here covers the case where a peer restarted and + * sent JOIN_REQ: the MEMBER_LIST is the broadcast announcement that a + * join event occurred. Without the reset, non-master peers would reject + * the restarted node's new packets (old last_seq > new low seq). */ + for (i = 0; i < (int)count; i++, p += CL_CTR_IP_ENTRY_SZ) { + char ip_buf[CL_CTR_MAX_IP_LEN + 1]; + int _j; + memcpy(ip_buf, p, CL_CTR_MAX_IP_LEN); + ip_buf[CL_CTR_MAX_IP_LEN] = '\0'; + if (ip_buf[0] == '\0') + continue; + cl_ctr_upsert_peer_locked(ip_buf, cl); + for (_j = 0; _j < cl->peers->count; _j++) { + if (strcmp(cl->peers->entries[_j].ip, ip_buf) == 0) { + cl->peers->entries[_j].last_seq = 0; + break; + } + } + } + + /* Snapshot master status BEFORE cl_ctr_apply_master_from_list_locked clears it */ + int was_master_before = cl_ctr_i_am_master_locked(cl); + + /* Apply the master designation from the list - no local election */ + if (designated_master[0] != '\0') + cl_ctr_apply_master_from_list_locked(designated_master, cl); + + /* The master is authoritative for the shtag override too. */ + cl->peers->shtag_forced_node_id = forced_node_id; + + int was_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + if (was_new) { + /* Only act as master if the list designates us AND we already hold the + * cluster key. Incumbent masters no longer hand over during a join, so + * in normal operation a MEMBER_LIST never designates a NEW node - this + * guard is defensive: without a key we would broadcast MASTER_ALIVE that + * no member can decrypt. Without the key we join as a plain member and + * let a later election promote us once KEY_GRANT has landed. */ + int _taking_over = (designated_master[0] != '\0' + && strcmp(designated_master, my_ip) == 0 + && cl->have_session_key); + cl->peers->node_state = CL_CTR_NODE_ACTIVE; + if (_taking_over) { + lock_stop_write(cl->peers->lock); + LM_INFO("clusterer_controller: received MEMBER_LIST (%u members) " + "from existing master %s - taking over mastership " + "(my IP %s is higher)\n", + count, sender_ip, my_ip); + } else { + lock_stop_write(cl->peers->lock); + LM_INFO("clusterer_controller: received MEMBER_LIST (%u members) " + "from %s - joined cluster as member, master is %s\n", + count, sender_ip, + designated_master[0] ? designated_master : "(none)"); + } + cl_ctr_transition_to_active(cl); + if (_taking_over) + cl_ctr_arm_master_timers(cl, 1); /* start MASTER_ALIVE, disarm dead watchdog */ + } else { + /* Active node - log the master update (may be self-demotion) */ + int i_am_master = (designated_master[0] != '\0' && + strcmp(designated_master, my_ip) == 0); + lock_stop_write(cl->peers->lock); + if (i_am_master) { + LM_INFO("clusterer_controller: MEMBER_LIST received - " + "I am master (%d members)\n", count); + } else if (was_master_before) { + /* A genuine demotion: we held mastership until this MEMBER_LIST. */ + LM_INFO("clusterer_controller: demoted to member - new master is %s " + "(%d members in cluster)\n", + designated_master[0] ? designated_master : "(none)", + count); + /* Fix up keepalive timers: stop sending MASTER_ALIVE, arm watchdog. */ + cl_ctr_arm_master_timers(cl, 0); + } else { + /* Already a member - this is just a routine MEMBER_LIST refresh (e.g. + * a periodic re-broadcast or a shtag-override update). No role change, + * so don't cry "demoted"; log quietly at debug level. */ + LM_DBG("clusterer_controller: MEMBER_LIST refreshed - master is %s " + "(%d members)\n", + designated_master[0] ? designated_master : "(none)", count); + } + } + + /* (Re)apply the shtag decision now that the forced-node override and the + * master designation from this MEMBER_LIST have both been stored. */ + cl_ctr_apply_shtags(cl); +} + +/** + * cl_ctr_handle_goodbye() - process a CL_CTR_PKT_GOODBYE packet. + * + * Remove the departing node from the peer table immediately. + * + * Re-election is triggered ONLY when: + * 1. Only one node remains - we are alone and must assume mastership. + * 2. Our IP is higher than the current master's IP, or the master entry + * no longer exists because the departing node was the master. + * cl_ctr_ip_beats_master_locked() covers both cases: it returns 1 when + * no is_master entry is present (departed master) or when our IP + * numerically exceeds the current master's. + * + * All other departures (a member leaves while a higher-IP master is alive) + * require no immediate action - the next periodic ALIVE cycle runs + * cl_ctr_elect_master(cl) within query_time seconds and self-corrects if needed. + */ +static void cl_ctr_handle_goodbye(int sock, const char *src_ip, cl_ctr_cluster_t *cl) +{ + int i, i_am_master, master_unchanged, remaining; + char prev_master[CL_CTR_MAX_IP_LEN + 1]; + char new_master[CL_CTR_MAX_IP_LEN + 1]; + uint16_t departed_node_id = 0; + + LM_INFO("clusterer_controller: GOODBYE from %s\n", src_ip); + + lock_start_write(cl->peers->lock); + + for (i = 0; i < cl->peers->count; i++) { + if (strcmp(cl->peers->entries[i].ip, src_ip) == 0) { + departed_node_id = cl->peers->entries[i].node_id; + cl->peers->count--; + if (i < cl->peers->count) + cl->peers->entries[i] = cl->peers->entries[cl->peers->count]; + memset(&cl->peers->entries[cl->peers->count], 0, sizeof(cl_ctr_peer_t)); + break; + } + } + + /* If the operator-forced shtag holder departed, drop the override so + * automatic allocation resumes rather than leaving no active holder. */ + if (departed_node_id != 0 && + cl->peers->shtag_forced_node_id == departed_node_id) { + LM_WARN("clusterer_controller: [cluster %d] forced shtag node %u " + "departed - resuming automatic allocation\n", + cl->cluster_id, departed_node_id); + cl->peers->shtag_forced_node_id = 0; + } + + remaining = cl->peers->count; + + /* --- Decide whether re-election is warranted --- */ + if (remaining <= 1) { + /* We are the only node left - no election needed, promote directly. */ + int was_master = cl_ctr_i_am_master_locked(cl); + cl_ctr_apply_master_from_list_locked(my_ip, cl); + lock_stop_write(cl->peers->lock); + if (was_master) { + LM_INFO("clusterer_controller: %s departed - I am the only " + "node remaining, I remain master\n", src_ip); + } else { + LM_INFO("clusterer_controller: %s departed - I am the only " + "node remaining, promoted myself to master\n", src_ip); + } + if (clctl_loaded && departed_node_id > 0) + clctl.remove_node(cl->cluster_id, departed_node_id); + cl_ctr_apply_shtags(cl); + return; + } + + if (cl_ctr_ip_beats_master_locked(ip_to_num(my_ip), cl)) { + /* Two sub-cases both return 1 from cl_ctr_ip_beats_master_locked: + * a) departing node was the master (no master entry remains) + * b) our IP is genuinely higher than the current master (anomaly) */ + if (strcmp(src_ip, cl->peers->last_master) == 0) { + LM_INFO("clusterer_controller: %s departed - it was the master, " + "triggering re-election\n", src_ip); + } else { + LM_INFO("clusterer_controller: %s departed - our IP %s is higher " + "than current master %s, triggering re-election\n", + src_ip, my_ip, + cl->peers->last_master[0] ? cl->peers->last_master : "(none)"); + } + } else { + /* Snapshot before releasing - must not read last_master after unlock */ + { + size_t _l = strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN); + memcpy(prev_master, cl->peers->last_master, _l); + prev_master[_l] = '\0'; + } + lock_stop_write(cl->peers->lock); + LM_INFO("clusterer_controller: %s departed - master %s still " + "active, no re-election needed (%d node(s) remaining)\n", + src_ip, + prev_master[0] ? prev_master : "(none)", + remaining); + if (clctl_loaded && departed_node_id > 0) + clctl.remove_node(cl->cluster_id, departed_node_id); + /* Re-apply shtag policy: the still-active master takes over the + * departed node's tags, unless an operator override is in effect. */ + cl_ctr_apply_shtags(cl); + return; + } + + memcpy(prev_master, cl->peers->last_master, + strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN)); + prev_master[strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN)] = '\0'; + + { + cl_ctr_elect_master(cl); + i_am_master = cl_ctr_i_am_master_locked(cl); + } + + master_unchanged = (strcmp(prev_master, cl->peers->last_master) == 0); + i_am_master = cl_ctr_i_am_master_locked(cl); + /* Snapshot post-election master before releasing - used in member log */ + { + size_t _l = strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN); + memcpy(new_master, cl->peers->last_master, _l); + new_master[_l] = '\0'; + } + + lock_stop_write(cl->peers->lock); + + if (i_am_master) { + if (master_unchanged) { + LM_INFO("clusterer_controller: re-election complete - " + "I remain master (%d node(s) in cluster)\n", remaining); + } else { + LM_INFO("clusterer_controller: re-election complete - " + "I reclaimed mastership after %s departed " + "(%d node(s) remaining)\n", + src_ip, remaining); + cl_ctr_arm_master_timers(cl, 1); + /* No full MEMBER_LIST is broadcast: every member re-elected the same + * deterministic master locally on this same GOODBYE, MASTER_ALIVE + * re-asserts it within a keepalive, and the membership digest + * reconciles anyone who missed the departure - so the O(N) re-broadcast + * (which fragments at large N) is redundant. */ + } + } else { + /* This node's own role change (if any) is logged separately by + * cl_ctr_elect_master() as a "my role changed" transition line. */ + LM_INFO("clusterer_controller: re-election complete - " + "master is %s (%d node(s) remaining)\n", + new_master[0] ? new_master : "(none)", + remaining); + } + if (clctl_loaded && departed_node_id > 0) + clctl.remove_node(cl->cluster_id, departed_node_id); + cl_ctr_apply_shtags(cl); +} + +/** + * cl_ctr_handle_node_assign() - process a CL_CTR_PKT_NODE_ASSIGN from the master. + * + * Payload: [node_id 2B BE][ip NUL][bin_count 1B][sock1 NUL]...[sockN NUL] + * + * All nodes (including master via loopback) apply the assignment: + * - Upsert the peer entry if not already present. + * - Store node_id and BIN sockets. + * - If ip == my_ip: record my_node_id. + */ +static void cl_ctr_handle_node_assign(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl) +{ + const char *p = payload; + const char *end = payload + payload_len; + uint16_t node_id; + char ip[CL_CTR_MAX_IP_LEN + 1]; + char bin_socks[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; + uint8_t bin_cnt = 0; + int ip_len, i; + + if (payload_len < (int)(CL_CTR_NODE_ID_SZ + 2)) { + LM_WARN("clusterer_controller: NODE_ASSIGN payload too short\n"); + return; + } + + /* node_id (2B BE) */ + memcpy(&node_id, p, CL_CTR_NODE_ID_SZ); + node_id = ntohs(node_id); + p += CL_CTR_NODE_ID_SZ; + + /* IP */ + ip_len = (int)strnlen(p, CL_CTR_MAX_IP_LEN); + memcpy(ip, p, ip_len); + ip[ip_len] = '\0'; + p += ip_len + 1; + + /* BIN sockets */ + memset(bin_socks, 0, sizeof(bin_socks)); + if (p < end) { + bin_cnt = (uint8_t)*p++; + if (bin_cnt > CL_CTR_MAX_BIN_SOCKETS) + bin_cnt = CL_CTR_MAX_BIN_SOCKETS; + for (i = 0; i < (int)bin_cnt && p < end; i++) { + int slen = (int)strnlen(p, CL_CTR_MAX_BIN_SOCK_LEN - 1); + memcpy(bin_socks[i], p, slen); + bin_socks[i][slen] = '\0'; + p += slen + 1; + } + } + + lock_start_write(cl->peers->lock); + cl_ctr_upsert_peer_locked(ip, cl); + cl_ctr_update_peer_bin_locked(ip, node_id, bin_cnt, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, + cl); + lock_stop_write(cl->peers->lock); + + /* Record our own node_id - also the first signal that a master exists */ + if (strcmp(ip, my_ip) == 0) { + /* Always our own entry - update identity regardless of my_node_id */ + if (my_node_id == 0) { + int is_joining; + lock_start_read(cl->peers->lock); + is_joining = (cl->peers->node_state == CL_CTR_NODE_NEW); + lock_stop_read(cl->peers->lock); + if (is_joining) + LM_INFO("clusterer_controller: [cluster %d] found existing master " + "at %s - receiving cluster state\n", + cl->cluster_id, sender_ip); + } + my_node_id = node_id; + LM_INFO("clusterer_controller: [cluster %d] master %s assigned us " + "node_id=%u\n", cl->cluster_id, sender_ip, node_id); + /* Correct the optimistic node_id=1 set at startup if needed */ + if (clctl_loaded) { + str url = {cl->bin_socket, (int)strlen(cl->bin_socket)}; + clctl.update_identity(cl->cluster_id, node_id, &url); + } + } else { + LM_INFO("clusterer_controller: [cluster %d] master %s assigned " + "node_id=%u to %s\n", cl->cluster_id, sender_ip, node_id, ip); + /* Add peer to clusterer */ + if (clctl_loaded && bin_cnt > 0) { + str url = {bin_socks[0], (int)strlen(bin_socks[0])}; + clctl.add_node(cl->cluster_id, node_id, &url); + } + } +} + +/** + * cl_ctr_handle_master_alive() - process a master keepalive. + * + * Beyond rearming the master-dead watchdog, this keeps every node agreeing on + * the master's identity (the 1s keepalive is more frequent than MEMBER_LIST, + * so it is the authoritative "who is master" signal) and resolves split-brain: + * if this node also believes it is master and the announcing node has a higher + * IP, it yields. Highest-IP-wins is the deterministic tiebreak in BOTH + * preemption modes, so two masters (e.g. after a network partition heals) can + * never both stick. + */ +static void cl_ctr_handle_master_alive(const char *sender_ip, cl_ctr_cluster_t *cl, + const char *payload, int payload_len, + const struct sockaddr *src, socklen_t src_len) +{ + int i_am_master, yielded = 0; + int from_self = (strcmp(sender_ip, my_ip) == 0); + + lock_start_write(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + + if (i_am_master) { + if (!from_self && ip_to_num(sender_ip) > ip_to_num(my_ip)) { + /* Split-brain: a higher-IP node also claims mastership - yield. */ + cl_ctr_upsert_peer_locked(sender_ip, cl); + cl_ctr_apply_master_from_list_locked(sender_ip, cl); + yielded = 1; + } + /* else: my own loopback, or a lower-IP claimant that will yield to us */ + } else if (!from_self) { + /* Track the announcing node as the current master so all nodes agree + * even if a MEMBER_LIST was missed, and so sticky cl_ctr_elect_master + * preserves the correct current master. */ + cl_ctr_upsert_peer_locked(sender_ip, cl); + cl_ctr_apply_master_from_list_locked(sender_ip, cl); + } + lock_stop_write(cl->peers->lock); + + if (yielded) { + LM_INFO("clusterer_controller: [cluster %d] yielding mastership to " + "higher-IP master %s (split-brain resolution)\n", + cl->cluster_id, sender_ip); + cl_ctr_arm_master_timers(cl, 0); /* stop MASTER_ALIVE, arm dead watchdog */ + } + + /* Non-masters (including a node that just yielded) watch the keepalive. */ + if (!i_am_master || yielded) + cl_ctr_arm_tfd(cl->master_dead_tfd, CL_CTR_MASTER_KA_TIMEOUT, 0); + + /* Liveness relay: the master reports which node_ids it has seen in-window; + * refresh those peers' last_seen so our election window stays populated even + * though a settled non-master no longer hears the peers' ALIVEs directly. */ + if (!i_am_master && !from_self && + payload_len >= CL_CTR_DIGEST_SZ + CL_CTR_ALIVE_BITMAP_SZ) + cl_ctr_apply_alive_bitmap(cl, + (const unsigned char *)payload + CL_CTR_DIGEST_SZ); + + /* Membership digest: if the master advertises a set we do not match, we may + * have missed a NODE_ASSIGN (a peer known from its ALIVE but with node_id/BIN + * unknown). Pull a rate-limited RESYNC. Only as a member holding the key, + * and never off our own loopback. */ + if (!i_am_master && !from_self && cl->have_session_key && + payload_len >= CL_CTR_DIGEST_SZ) { + const unsigned char *d = (const unsigned char *)payload; + uint16_t adv_cnt = (uint16_t)((d[0] << 8) | d[1]); + uint16_t my_cnt; + uint64_t adv_hash = 0, my_hash; + int k; + for (k = 0; k < 8; k++) adv_hash = (adv_hash << 8) | d[2 + k]; + cl_ctr_membership_digest(cl, &my_cnt, &my_hash); + if (adv_cnt != my_cnt || adv_hash != my_hash) { + utime_t now = get_uticks(); + if ((utime_t)(now - cl->last_resync_us) >= CL_CTR_RESYNC_MIN_US) { + cl->last_resync_us = now; + LM_DBG("clusterer_controller: [cluster %d] membership digest " + "mismatch with master %s (theirs %u, ours %u) - RESYNC\n", + cl->cluster_id, sender_ip, adv_cnt, my_cnt); + cl_ctr_send_resync(cl->sock, cl, src, src_len); + } + } + } +} + +/** + * cl_ctr_rejoin_superior_master() - abandon our current allegiance and merge into + * the partition led by 'superior_ip', adopting its session key. + * + * Used by the split-brain merge (cl_ctr_handle_master_beacon): our node - whether a + * lone/independent master or a member of a smaller partition - records + * superior_ip as master, stops asserting mastership, and issues a JOIN_REQ so + * the superior master answers with NODE_ASSIGN + KEY_GRANT. Once the KEY_GRANT + * lands we can decrypt the superior partition's session traffic and are fully + * merged. Call WITHOUT cl->peers->lock held. + */ +static void cl_ctr_rejoin_superior_master(cl_ctr_cluster_t *cl, const char *superior_ip) +{ + int was_master; + + lock_start_write(cl->peers->lock); + was_master = cl_ctr_i_am_master_locked(cl); + cl_ctr_upsert_peer_locked(superior_ip, cl); + cl_ctr_apply_master_from_list_locked(superior_ip, cl); + cl->peers->node_state = CL_CTR_NODE_ACTIVE; + lock_stop_write(cl->peers->lock); + + if (was_master) { + LM_INFO("clusterer_controller: [cluster %d] superior master %s found via " + "beacon - demoting and merging (split-brain resolution)\n", + cl->cluster_id, superior_ip); + cl_ctr_arm_master_timers(cl, 0); /* stop MASTER_ALIVE, arm dead watchdog */ + } else { + LM_INFO("clusterer_controller: [cluster %d] moving to superior master %s " + "via beacon (split-brain merge)\n", cl->cluster_id, superior_ip); + } + + /* Drop any locally-held active shtag now that we are no longer master. */ + cl_ctr_apply_shtags(cl); + + /* Fetch the superior master's session key. join_pending guards the nonce + * so a beacon storm cannot stomp an exchange already in flight. */ + if (!cl->join_pending) { + cl->join_pending = 1; + cl_ctr_send_join_req(cl->sock, cl); + } + cl_ctr_arm_tfd(cl->master_dead_tfd, CL_CTR_MASTER_KA_TIMEOUT, 0); +} + +/** + * cl_ctr_handle_master_beacon() - process a CL_CTR_PKT_MASTER_BEACON (bootstrap key). + * + * A master announced itself on the bootstrap layer. If it outranks our own + * partition we merge into it; otherwise we ignore it (that master will yield to + * us when it hears our beacon). Ranking: larger member count wins, ties broken + * by higher IP - the same deterministic tiebreak used elsewhere, so exactly one + * master survives. A healthy single-master cluster sees only its own master's + * beacon (sender == our master) and never acts, so this adds no churn. + */ +static void cl_ctr_handle_master_beacon(const char *sender_ip, uint16_t sender_count, + cl_ctr_cluster_t *cl) +{ + int i_am_master, is_new, our_count, superior; + char our_master[CL_CTR_MAX_IP_LEN + 1]; + + if (strcmp(sender_ip, my_ip) == 0) + return; /* our own beacon looped back */ + + lock_start_read(cl->peers->lock); + is_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + i_am_master = cl_ctr_i_am_master_locked(cl); + our_count = cl->peers->count; + if (i_am_master) { + size_t l = strnlen(my_ip, CL_CTR_MAX_IP_LEN); + memcpy(our_master, my_ip, l); our_master[l] = '\0'; + } else { + size_t l = strnlen(cl->peers->last_master, CL_CTR_MAX_IP_LEN); + memcpy(our_master, cl->peers->last_master, l); our_master[l] = '\0'; + } + lock_stop_read(cl->peers->lock); + + /* Still running the join protocol - a JOIN_REQ is already outstanding. */ + if (is_new) + return; + + /* Already following the beacon's sender: normal steady state, nothing to do. */ + if (our_master[0] != '\0' && strcmp(sender_ip, our_master) == 0) + return; + + /* Rank the sender's partition against ours. */ + if (our_master[0] == '\0') + superior = 1; /* we have no master yet */ + else if (sender_count != (uint16_t)our_count) + superior = (sender_count > (uint16_t)our_count);/* larger partition wins */ + else + superior = (ip_to_num(sender_ip) > ip_to_num(our_master)); /* IP tiebreak */ + + if (!superior) + return; /* we outrank the sender; it will yield to us on our beacon */ + + cl_ctr_rejoin_superior_master(cl, sender_ip); +} + +/** + * cl_ctr_handle_key_grant() - process CL_CTR_PKT_KEY_GRANT from master. + * Payload: [target_ip NUL][master_pubkey 32B][join_nonce 16B][wrapped_salt 32B] + * + * Recover master_salt via ECDH unwrap, derive session_key, update shm. + * Only processed if target_ip == my_ip (multicast - all nodes receive it). + */ +static void cl_ctr_handle_key_grant(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl, + uint32_t in_seq, const struct sockaddr *src, + socklen_t src_len) +{ + const char *p = payload; + const char *end = payload + payload_len; + char target_ip[CL_CTR_MAX_IP_LEN + 1]; + unsigned char new_salt[CL_CTR_MASTER_SALT_SZ]; + int ip_len; + + /* Parse target_ip */ + ip_len = (int)strnlen(p, CL_CTR_MAX_IP_LEN); + if (p + ip_len >= end) return; + memcpy(target_ip, p, ip_len); target_ip[ip_len] = '\0'; + p += ip_len + 1; + + /* Only process if addressed to us */ + if (strcmp(target_ip, my_ip) != 0) return; + + /* Masters are the key source - they never accept KEY_GRANTs. + * Without this guard a stale KEY_GRANT (from the previous master, + * responding to a JOIN_REQ we sent while CL_CTR_NODE_NEW) can arrive + * after we self-promoted via cl_ctr_on_join_tfd and overwrite the + * session key we just generated, causing a key-mismatch loop. */ + { + int _im; + lock_start_read(cl->peers->lock); + _im = cl_ctr_i_am_master_locked(cl); + lock_stop_read(cl->peers->lock); + if (_im) { + LM_DBG("clusterer_controller: KEY_GRANT from %s ignored - " + "I am master\n", sender_ip); + return; + } + } + + if (p + CL_CTR_NOISE_MSG2_SZ > end) { + LM_WARN("clusterer_controller: KEY_GRANT too short\n"); + return; + } + + /* Complete the Noise handshake: read msg 2 with the initiator state stored + * when we sent the JOIN_REQ. Its AEAD payload is the master_salt. A stale + * KEY_GRANT (for a superseded JOIN_REQ) fails here and is dropped - this + * subsumes the old join_nonce echo check. */ + if (!cl->noise_hs_valid || + cl_ctr_noise_read2(&cl->noise_hs, cl->noise_e_priv, + (const unsigned char *)p, CL_CTR_NOISE_MSG2_SZ, new_salt) < 0) { + LM_DBG("clusterer_controller: KEY_GRANT from %s did not complete the " + "Noise handshake (stale or wrong key) - dropping\n", sender_ip); + /* Clear join_pending so the next decryption failure or rejoin_tfd can + * issue a fresh JOIN_REQ. */ + cl->join_pending = 0; + /* If we already hold a session key, this is a duplicate of a KEY_GRANT + * we handled whose ACK was lost - re-ACK so the master stops retrying. */ + if (cl->have_session_key) + cl_ctr_send_ack(cl->sock, cl, in_seq, 1/*bootstrap*/, src, src_len); + return; + } + cl->noise_hs_valid = 0; /* handshake consumed */ + + /* Apply: write to shm, derive session_key */ + lock_start_write(cl->peers->lock); + memcpy(cl->peers->master_salt, new_salt, CL_CTR_MASTER_SALT_SZ); + cl_ctr_derive_session_key(cl); + lock_stop_write(cl->peers->lock); + + cl->join_pending = 0; + cl->bootstrap_auth_fails = 0; + cl->join_attempt_count = 0; + cl->auth_fail_pkts = 0; /* authenticated: clear wrong-password evidence */ + cl->auth_defer_count = 0; + LM_INFO("clusterer_controller: [cluster %d] KEY_GRANT from %s - " + "session key updated\n", cl->cluster_id, sender_ip); + /* Acknowledge to the master (bootstrap key class, matching KEY_GRANT) so it + * stops retransmitting. Best-effort: a lost ACK just draws a retransmit, + * which we re-ACK via the duplicate path above; never gates our join. */ + cl_ctr_send_ack(cl->sock, cl, in_seq, 1/*bootstrap*/, src, src_len); +} + +/** + * cl_ctr_handle_key_handoff() - process CL_CTR_PKT_KEY_HANDOFF. + * Only acted on by the node that wins the next election (highest IP). + * Payload: [next_master_ip NUL][crypto_box_seal(master_salt)] + */ +static void cl_ctr_handle_key_handoff(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl) +{ + const char *p = payload; + const char *end = payload + payload_len; + char target_ip[CL_CTR_MAX_IP_LEN + 1]; + unsigned char new_salt[CL_CTR_MASTER_SALT_SZ]; + int ip_len; + + ip_len = (int)strnlen(p, CL_CTR_MAX_IP_LEN); + if (p + ip_len >= end) return; + memcpy(target_ip, p, ip_len); target_ip[ip_len] = '\0'; + p += ip_len + 1; + + /* Only the intended next master unwraps this */ + if (strcmp(target_ip, my_ip) != 0) return; + + if (p + (int)(crypto_box_SEALBYTES + CL_CTR_MASTER_SALT_SZ) > end) { + LM_WARN("clusterer_controller: KEY_HANDOFF too short\n"); + return; + } + /* Open the anonymous sealed box with our own long-lived keypair. */ + if (crypto_box_seal_open(new_salt, (const unsigned char *)p, + crypto_box_SEALBYTES + CL_CTR_MASTER_SALT_SZ, + cl->my_pubkey, cl->my_privkey) != 0) { + LM_WARN("clusterer_controller: [cluster %d] KEY_HANDOFF from %s did not " + "open - dropping\n", cl->cluster_id, sender_ip); + return; + } + + /* Store salt and re-derive so session_key is guaranteed consistent with + * the adopted salt (the incoming master's salt may differ from ours). */ + lock_start_write(cl->peers->lock); + memcpy(cl->peers->master_salt, new_salt, CL_CTR_MASTER_SALT_SZ); + cl_ctr_derive_session_key(cl); /* sets have_session_key = 1 */ + lock_stop_write(cl->peers->lock); + + LM_INFO("clusterer_controller: [cluster %d] KEY_HANDOFF from %s - " + "master_salt preserved for seamless transition\n", + cl->cluster_id, sender_ip); +} + +/* ========================================================================= + * Receive dispatcher + * ========================================================================= */ + +/** + * cl_ctr_rate_check() - per-source-IP rate limiter, called before decryption. + * Finds or creates a 1-second sliding-window counter for src_ip. + * @return 0 if within CL_CTR_RATE_LIMIT packets/s, -1 to drop. + */ +static int cl_ctr_rate_check(cl_ctr_cluster_t *cl, uint32_t src_ip) +{ + time_t now = time(NULL); + cl_ctr_rate_entry_t *oldest = NULL; + int i; + + for (i = 0; i < CL_CTR_RATE_TBL_SZ; i++) { + cl_ctr_rate_entry_t *e = &cl->rate_tbl[i]; + if (e->ip == 0) { + if (!oldest) oldest = e; /* prefer empty slot */ + continue; + } + if (e->ip == src_ip) { + if (now > e->window_start) { /* new second */ + e->window_start = now; + e->count = 1; + return 0; + } + if (++e->count > CL_CTR_RATE_LIMIT) + return -1; + return 0; + } + /* track oldest entry for eviction when table is full */ + if (!oldest || e->window_start < oldest->window_start) + oldest = e; + } + + /* new source IP - claim oldest/empty slot */ + oldest->ip = src_ip; + oldest->window_start = now; + oldest->count = 1; + return 0; +} + +/* -- Join-reject helpers ---------------------------------------------------- + * + * Security model: JOIN_REJECT is sent as a normal CL_CTR_BOOTSTRAP_MAGIC packet + * (XChaCha20-Poly1305 authenticated with the bootstrap key). An attacker + * without the cluster password cannot forge an authenticated reject, so they + * cannot kick nodes out or block joins. cl_ctr_handle_join_reject() also guards + * on CL_CTR_NODE_NEW state so that even a legitimate cluster member with the + * correct password cannot send a JOIN_REJECT to an already-active node. + * + * cl_ctr_join_fail_check() - master: track per-IP BOOTSTRAP_MAGIC decrypt failures. + * Returns 1 the first time a source IP reaches CL_CTR_JOIN_FAIL_LIMIT failures. + * + * cl_ctr_send_join_reject() - master: send encrypted JOIN_REJECT via BOOTSTRAP_MAGIC. + * Wire payload: [target_ip NUL] + * + * cl_ctr_handle_join_reject() - joiner: stop OpenSIPS if the reject is for us and + * we are still in CL_CTR_NODE_NEW (i.e., have not successfully joined yet). + * + * WRONG-PASSWORD fallback: if the joiner has the wrong password it cannot + * decrypt the encrypted JOIN_REJECT. Instead it detects the situation + * through bootstrap_auth_fails (incremented on any BOOTSTRAP_MAGIC decrypt + * failure during CL_CTR_NODE_NEW) combined with join_attempt_count. After + * CL_CTR_JOIN_FAIL_LIMIT rejoin retries with at least one bootstrap failure + * observed, the joiner concludes the master rejected it and exits. + */ + +static int cl_ctr_join_fail_check(const char *src_ip, cl_ctr_cluster_t *cl) +{ + uint32_t ip_num = ip_to_num(src_ip); + int evict = 0; /* index of lowest-count slot for eviction */ + int i; + + if (ip_num == 0) + return 0; + + for (i = 0; i < CL_CTR_JOIN_FAIL_TABLE_SZ; i++) { + if (cl->join_fail_tbl[i].ip_num != ip_num) + continue; + if (cl->join_fail_tbl[i].rejected) + return 0; /* reject already sent; don't repeat */ + cl->join_fail_tbl[i].count++; + if (cl->join_fail_tbl[i].count >= CL_CTR_JOIN_FAIL_LIMIT) { + cl->join_fail_tbl[i].rejected = 1; + return 1; + } + return 0; + } + + /* Not found - insert, evicting the slot with the smallest count */ + for (i = 1; i < CL_CTR_JOIN_FAIL_TABLE_SZ; i++) { + if (cl->join_fail_tbl[i].count < cl->join_fail_tbl[evict].count) + evict = i; + } + memset(&cl->join_fail_tbl[evict], 0, sizeof(cl->join_fail_tbl[evict])); + cl->join_fail_tbl[evict].ip_num = ip_num; + snprintf(cl->join_fail_tbl[evict].ip, sizeof(cl->join_fail_tbl[evict].ip), + "%s", src_ip); + cl->join_fail_tbl[evict].count = 1; + return 0; +} + +static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_cluster_t *cl, + int reason) +{ + char pkt[CL_CTR_SMALL_PKT_SZ + 1]; /* +1 for the reason byte */ + uint32_t seq = htonl(++cl->peers->my_seq); + int ip_len, plain_len; + + ip_len = (int)strnlen(target_ip, CL_CTR_MAX_IP_LEN); + + memcpy(pkt, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)CL_CTR_PKT_JOIN_REJECT; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ, target_ip, ip_len); + pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + ip_len] = '\0'; + /* reason byte follows the NUL-terminated target IP */ + pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + ip_len + 1] = (char)reason; + + plain_len = CL_CTR_PLAIN_HDR_SZ + ip_len + 1 + 1; + if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->key, CL_CTR_PKT_JOIN_REJECT) == 0) + LM_WARN("clusterer_controller: [cluster %d] sent JOIN_REJECT to %s (%s)\n", + cl->cluster_id, target_ip, + reason == CL_CTR_REJECT_CONFIG ? "different cluster settings" + : "repeated auth failure - wrong password?"); +} + +static void cl_ctr_handle_join_reject(const char *payload, int payload_len, + const char *sender_ip, cl_ctr_cluster_t *cl) +{ + char target_ip[CL_CTR_MAX_IP_LEN + 1]; + int l, still_new; + + /* Only act during the initial join phase. An active member receiving a + * JOIN_REJECT (e.g. from a cluster peer with the correct password who + * wanted to test the mechanism, or a stale in-flight packet) must ignore + * it - this prevents any cluster member from silently evicting another. */ + lock_start_read(cl->peers->lock); + still_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + lock_stop_read(cl->peers->lock); + if (!still_new) return; + + l = (int)strnlen(payload, CL_CTR_MAX_IP_LEN); + if (l >= payload_len) return; + memcpy(target_ip, payload, l); + target_ip[l] = '\0'; + + if (strcmp(target_ip, my_ip) != 0) + return; /* not addressed to this node */ + + /* reason byte follows the NUL-terminated target IP (older senders omit it) */ + { + int reason = CL_CTR_REJECT_GENERIC; + if (payload_len > l + 1) + reason = (unsigned char)payload[l + 1]; + if (reason == CL_CTR_REJECT_CONFIG) + LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " + "running cluster has different settings than this node; fix the " + "local config (manage_shtags/master_stickiness/query_time) to " + "match and restart; shutting down\n", + cl->cluster_id, sender_ip); + else + LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - " + "wrong password or unauthorized node; shutting down\n", + cl->cluster_id, sender_ip); + } + exit(-1); +} + +/** + * cl_ctr_recv_one() - read one datagram, validate header, dispatch by type. + */ +/* A datagram received on a shared ip:port for a sibling local cluster, handed + * to that cluster's worker (still encrypted) by its cleartext cluster_id. */ +struct cl_ctr_fwd_pkt { + cl_ctr_cluster_t *cl; /* target cluster (owns the reply socket) */ + struct sockaddr_in src; /* original datagram source */ + int n; /* length of buf */ + unsigned char buf[1]; /* the encrypted packet (flexible array) */ +}; +static void cl_ctr_rpc_forward(int sender, void *param); +static void cl_ctr_maybe_forward(const char *buf, int n, + const struct sockaddr_in *src, uint16_t pkt_cid, + cl_ctr_cluster_t *from); + +static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, + const char *fwd_buf, int fwd_n, + const struct sockaddr_in *fwd_src) +{ + /* Static buffer: avoids a 64 KB stack frame; safe because cl_ctr_recv_one + * is called only from the single-threaded cl_ctr_worker process. */ + static char buf[CL_CTR_RECV_BUF_SZ]; + struct sockaddr_in src_addr; + socklen_t src_len = sizeof(src_addr); + ssize_t n; + unsigned char pkt_type; + const char *payload; + int payload_len; + + if (fwd_buf) { + /* Forwarded from a sibling cluster's worker that received this on our + * shared ip:port (see cl_ctr_maybe_forward). Process it exactly as if + * it had arrived on our own socket. */ + n = (fwd_n > (int)sizeof(buf) - 1) ? (int)sizeof(buf) - 1 : fwd_n; + memcpy(buf, fwd_buf, (size_t)n); + src_addr = *fwd_src; + (void)src_len; + } else { + n = recvfrom(sock, buf, sizeof(buf) - 1, 0, + (struct sockaddr *)&src_addr, &src_len); + if (n < 0) { + if (errno != EAGAIN && errno != EWOULDBLOCK) + LM_ERR("clusterer_controller: recvfrom(): %s\n", strerror(errno)); + return; + } + } + + /* Minimum: magic(2)+cluster_id(2)+nonce(12)+type(1)+seq(4)+tag(16) = 37 */ + if (n < CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_TAG_SZ) { + LM_WARN("clusterer_controller: short packet (%zd bytes), dropping\n", + n); + return; + } + + if (memcmp(buf, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ) != 0 && + memcmp(buf, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ) != 0) { + LM_DBG("clusterer_controller: bad magic, dropping\n"); + return; + } + + /* Cleartext cluster_id filter: silently ignore packets for a different + * cluster sharing this multicast group. Done BEFORE decryption so foreign + * traffic never counts as an auth failure (auth_fail_pkts) - a wrong-cluster + * packet is not a wrong-password attempt. */ + { + uint16_t pkt_cid_be; + memcpy(&pkt_cid_be, buf + CL_CTR_MAGIC_SZ, CL_CTR_CLUSTER_ID_SZ); + if (ntohs(pkt_cid_be) != (uint16_t)cl->cluster_id) { + /* Not ours. On a shared ip:port the kernel routes unicast by port + * alone, so it may have delivered a sibling local cluster's packet + * to us - hand it to that cluster's worker by its cleartext + * cluster_id (still encrypted; the sibling decrypts with its own + * key). A packet we were already forwarded is never re-forwarded. */ + if (!fwd_buf) + cl_ctr_maybe_forward(buf, (int)n, &src_addr, ntohs(pkt_cid_be), cl); + else + LM_DBG("clusterer_controller: [cluster %d] forwarded packet for " + "cluster %u is not ours, dropping\n", cl->cluster_id, + ntohs(pkt_cid_be)); + return; + } + } + + /* Rate-limit before any crypto work to shed floods cheaply. */ + if (cl_ctr_rate_check(cl, src_addr.sin_addr.s_addr) < 0) + return; + + /* Resolve sender IP once - used for HMAC warning and MEMBER_LIST dispatch */ + { + char sender_ip_buf[INET_ADDRSTRLEN]; + const unsigned char *dec_key; + inet_ntop(AF_INET, &src_addr.sin_addr, + sender_ip_buf, sizeof(sender_ip_buf)); + + /* Select decryption key by magic: + * CL_CTR_BOOTSTRAP_MAGIC -> bootstrap key (JOIN_REQ, KEY_GRANT) + * CL_CTR_PACKET_MAGIC -> session key (all normal traffic) + * If session key decryption fails, schedule a re-JOIN to refresh it. */ + int is_bootstrap = (memcmp(buf, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ) == 0); + dec_key = is_bootstrap ? cl->key : cl->session_key; + + if (cl_ctr_decrypt_pkt(buf, n, sender_ip_buf, dec_key, is_bootstrap) < 0) { + int _im, _new; + char _lm[CL_CTR_MAX_IP_LEN + 1]; + lock_start_read(cl->peers->lock); + _im = cl_ctr_i_am_master_locked(cl); + _new = (cl->peers->node_state == CL_CTR_NODE_NEW); + memcpy(_lm, cl->peers->last_master, sizeof(_lm)); + lock_stop_read(cl->peers->lock); + + /* A packet from another peer we cannot decrypt, seen while still + * joining, means a cluster (or rogue) whose key we do not share is + * present on this group. We count it (any magic) as evidence that we + * may not belong here: session-key failures matter too, because a + * running cluster's steady traffic (MASTER_ALIVE/MEMBER_LIST) is + * session-encrypted, so a wrong-password joiner would otherwise see too + * little bootstrap traffic to notice the cluster and would self-promote + * into a split-brain lone master. This is only *evidence*, never an + * immediate death sentence: cl_ctr_on_join_tfd defers and keeps re-joining, + * and a KEY_GRANT arriving in the grace window resets this counter, so a + * healthy node whose key is merely slow is not affected. Our own + * loopback decrypts fine, so guard on my_ip. */ + if (_new && strcmp(sender_ip_buf, my_ip) != 0) + cl->auth_fail_pkts++; + + if (is_bootstrap) { + + /* Master: track per-IP bootstrap failures; send JOIN_REJECT on limit. + * JOIN_REJECT is encrypted (BOOTSTRAP_MAGIC/AEAD) so only nodes with + * the correct password can read it. Forgeries are impossible without + * the bootstrap key. */ + if (_im && cl_ctr_join_fail_check(sender_ip_buf, cl)) + cl_ctr_send_join_reject(sock, sender_ip_buf, cl, CL_CTR_REJECT_GENERIC); + + /* Joiner fallback: count bootstrap decrypt failures during CL_CTR_NODE_NEW + * only when the packet came from the known master. This filters out + * rogue nodes on the multicast group whose JOIN_REQs (encrypted with + * their own wrong key) would otherwise increment this counter and + * trigger a spurious exit on a legitimate joining node. + * If last_master is empty we have no master reference yet, so we + * conservatively skip counting - no master means no rejection. */ + if (_new && _lm[0] != '\0' && strcmp(sender_ip_buf, _lm) == 0) + cl->bootstrap_auth_fails++; + } + if (!is_bootstrap && !cl->join_pending) { + /* Session key mismatch: request a re-key ONLY when the packet we + * could not decrypt came from OUR current master (a legitimate key + * rotation). Undecryptable session packets from any other source + * are rogue or stale traffic (e.g. a wrong-password node broadcasting + * on the group); reacting to them would drive an endless re-JOIN + * churn across the whole cluster. Masters never re-JOIN. */ + if (!_im && _lm[0] != '\0' && strcmp(sender_ip_buf, _lm) == 0) { + LM_INFO("clusterer_controller: [cluster %d] session key mismatch " + "from master %s - sending JOIN_REQ to re-key\n", + cl->cluster_id, sender_ip_buf); + cl->join_pending = 1; + cl_ctr_send_join_req(cl->sock, cl); + } + } + return; + } + + /* Sequence check: reject replays for all session-key packets including + * GOODBYE. my_seq lives in shm so mod_destroy increments the same + * counter the worker uses - GOODBYE gets a valid monotonic seq number + * without any special-casing. + * Bootstrap packets (CL_CTR_BOOTSTRAP_MAGIC) use join_nonce instead. */ + if (!is_bootstrap) { + uint32_t pkt_seq; + memcpy(&pkt_seq, buf + CL_CTR_WIRE_HDR_SZ + 1, CL_CTR_SEQ_SZ); + pkt_seq = ntohl(pkt_seq); + if (cl_ctr_check_and_update_seq(sender_ip_buf, pkt_seq, cl) < 0) + return; + } + + pkt_type = (unsigned char)buf[CL_CTR_WIRE_HDR_SZ]; + payload = buf + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ; + /* Non-negative: the minimum-length gate above guarantees + * n >= CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_TAG_SZ. */ + payload_len = (int)(n - CL_CTR_WIRE_HDR_SZ - CL_CTR_PLAIN_HDR_SZ - CL_CTR_TAG_SZ); + + switch (pkt_type) { + + case CL_CTR_PKT_ALIVE: { + char ip_buf[CL_CTR_MAX_IP_LEN + 1]; + int ip_len = (int)strnlen(payload, CL_CTR_MAX_IP_LEN); + const unsigned char *pubkey = NULL; + int cfg_present = 0, p_manage = 0, p_stick = 0, p_qt = 0; + memcpy(ip_buf, payload, ip_len); + ip_buf[ip_len] = '\0'; + /* Pubkey appended after NUL-terminated IP */ + if (payload_len >= ip_len + 1 + (int)CL_CTR_PUBKEY_SZ) + pubkey = (const unsigned char *)payload + ip_len + 1; + /* Config descriptor appended after the pubkey (optional) */ + if (payload_len >= ip_len + 1 + (int)CL_CTR_PUBKEY_SZ + CL_CTR_CONFIG_SZ) { + const char *c = payload + ip_len + 1 + CL_CTR_PUBKEY_SZ; + uint16_t qt_be; + p_manage = (unsigned char)c[0]; + p_stick = (unsigned char)c[1]; + memcpy(&qt_be, c + 2, 2); + p_qt = ntohs(qt_be); + cfg_present = 1; + } + cl_ctr_handle_alive(ip_buf, pubkey, cfg_present, p_manage, p_stick, p_qt, cl); + break; + } + + case CL_CTR_PKT_JOIN_REQ: + cl_ctr_handle_join_req(sock, payload, payload_len, cl, + (const struct sockaddr *)&src_addr, src_len); + break; + + case CL_CTR_PKT_MEMBER_LIST: + cl_ctr_handle_member_list(payload, payload_len, sender_ip_buf, cl); + break; + + case CL_CTR_PKT_GOODBYE: { + char ip_buf[CL_CTR_MAX_IP_LEN + 1]; + int ip_len = payload_len > CL_CTR_MAX_IP_LEN ? CL_CTR_MAX_IP_LEN : payload_len; + memcpy(ip_buf, payload, ip_len); + ip_buf[ip_len] = '\0'; + cl_ctr_handle_goodbye(sock, ip_buf, cl); + break; + } + + case CL_CTR_PKT_NODE_ASSIGN: + cl_ctr_handle_node_assign(payload, payload_len, sender_ip_buf, cl); + break; + + case CL_CTR_PKT_MASTER_ALIVE: + cl_ctr_handle_master_alive(sender_ip_buf, cl, payload, payload_len, + (const struct sockaddr *)&src_addr, src_len); + break; + + case CL_CTR_PKT_RESYNC: + cl_ctr_handle_resync(sender_ip_buf, cl); + break; + + case CL_CTR_PKT_KEY_GRANT: { + uint32_t in_seq; + memcpy(&in_seq, buf + CL_CTR_WIRE_HDR_SZ + 1, CL_CTR_SEQ_SZ); + in_seq = ntohl(in_seq); + cl_ctr_handle_key_grant(payload, payload_len, sender_ip_buf, cl, + in_seq, (const struct sockaddr *)&src_addr, src_len); + break; + } + + case CL_CTR_PKT_ACK: + cl_ctr_handle_ack(payload, payload_len, cl); + break; + + case CL_CTR_PKT_KEY_HANDOFF: + cl_ctr_handle_key_handoff(payload, payload_len, sender_ip_buf, cl); + break; + + case CL_CTR_PKT_JOIN_REJECT: + cl_ctr_handle_join_reject(payload, payload_len, sender_ip_buf, cl); + break; + + case CL_CTR_PKT_MASTER_BEACON: { + uint16_t cnt_be, sender_count = 0; + if (payload_len >= 2) { + memcpy(&cnt_be, payload, 2); + sender_count = ntohs(cnt_be); + } + cl_ctr_handle_master_beacon(sender_ip_buf, sender_count, cl); + break; + } + + default: + LM_WARN("clusterer_controller: unknown packet type 0x%02x " + "from %s, dropping\n", pkt_type, sender_ip_buf); + break; + } + } +} + +/* + * cl_ctr_maybe_forward() - hand a datagram received on a shared ip:port to the + * sibling local cluster it actually belongs to. + * + * On a shared multicast_port every controller cluster on this node binds the + * same ip:port; the kernel demultiplexes multicast by group membership but + * unicast by port alone, so a 1:1 reply meant for a co-located cluster can be + * delivered to the wrong worker. We recover it here: find the local cluster + * whose cluster_id matches the packet's cleartext header and forward the still- + * encrypted bytes to its worker via IPC. The target decrypts with its own key. + * + * Called only from the original receiver (never re-entered for a forwarded + * packet), so no forwarding loop is possible. + */ +static void cl_ctr_maybe_forward(const char *buf, int n, + const struct sockaddr_in *src, uint16_t pkt_cid, + cl_ctr_cluster_t *from) +{ + cl_ctr_cluster_t *target = NULL; + struct cl_ctr_fwd_pkt *fp; + int i, proc_no; + + /* Bound the work: only a legitimately-sized unicast is worth forwarding. */ + if (n <= 0 || n > CL_CTR_LIST_PKT_MAX_SZ) + return; + + for (i = 0; i < cl_ctr_cluster_count; i++) { + if ((uint16_t)cl_ctr_clusters[i].cluster_id == pkt_cid && + &cl_ctr_clusters[i] != from) { + target = &cl_ctr_clusters[i]; + break; + } + } + if (!target) { + LM_DBG("clusterer_controller: [cluster %d] no local cluster %u for " + "packet on shared port, dropping\n", from->cluster_id, pkt_cid); + return; + } + + /* Shed floods before allocating: charge the source against our own limiter + * (the target re-checks after it receives the forward). */ + if (cl_ctr_rate_check(from, src->sin_addr.s_addr) < 0) + return; + + /* worker_proc_no is written once at worker fork and stable thereafter. */ + proc_no = target->peers ? target->peers->worker_proc_no : -1; + if (proc_no < 0) + return; /* target worker not up yet */ + + fp = shm_malloc(sizeof(*fp) + (size_t)n - 1); + if (!fp) { + LM_ERR("clusterer_controller: out of shm forwarding to cluster %u\n", + pkt_cid); + return; + } + fp->cl = target; + fp->src = *src; + fp->n = n; + memcpy(fp->buf, buf, (size_t)n); + + if (ipc_send_rpc(proc_no, cl_ctr_rpc_forward, fp) < 0) { + LM_ERR("clusterer_controller: ipc forward to cluster %u failed\n", + pkt_cid); + shm_free(fp); + } +} + +/* + * cl_ctr_rpc_forward() - runs in the target cluster's worker; process a datagram + * a sibling worker forwarded to us (see cl_ctr_maybe_forward). + */ +static void cl_ctr_rpc_forward(int sender, void *param) +{ + struct cl_ctr_fwd_pkt *fp = (struct cl_ctr_fwd_pkt *)param; + (void)sender; + + cl_ctr_recv_one(fp->cl->sock, fp->cl, (char *)fp->buf, fp->n, &fp->src); + shm_free(fp); +} + +/* ========================================================================= + * Background worker process + * ========================================================================= */ + +/* ========================================================================= + * Reactor callbacks - one per event source + * ========================================================================= */ + +static int cl_ctr_on_sock(int fd, void *param, int was_timeout) +{ + cl_ctr_recv_one(fd, (cl_ctr_cluster_t *)param, NULL, 0, NULL); + return 0; +} + +static int cl_ctr_on_alive_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + int prev_master, now_master; + cl_ctr_drain_tfd(fd); + /* ALIVE transport: a settled non-master unicasts its heartbeat to the master, + * which relays liveness to the whole group via the MASTER_ALIVE bitmap - so + * the old all-to-all O(N^2) becomes O(N). During formation (no settled master + * or key yet) fall back to multicast for discovery; the master itself keeps + * multicasting its own ALIVE so its liveness and config still reach everyone. */ + { + int _im, _active; + char _lm[CL_CTR_MAX_IP_LEN + 1]; + lock_start_read(cl->peers->lock); + _im = cl_ctr_i_am_master_locked(cl); + _active = (cl->peers->node_state == CL_CTR_NODE_ACTIVE); + memcpy(_lm, cl->peers->last_master, sizeof(_lm)); + lock_stop_read(cl->peers->lock); + if (!_im && _active && cl->have_session_key && _lm[0] != '\0') { + struct sockaddr_in d; + cl_ctr_sockaddr_in(_lm, cl->multicast_port, &d); + cl_ctr_send_alive_to(cl->sock, cl, (struct sockaddr *)&d, sizeof(d)); + } else { + cl_ctr_send_alive(cl->sock, cl); + } + } + lock_start_write(cl->peers->lock); + /* Refresh our own liveness first: a settled non-master unicasts its ALIVE to + * the master and no longer hears a multicast loopback, so without this it + * would age itself out of its own election window once the master is gone + * and briefly find "no eligible peers" during a failover. */ + cl_ctr_upsert_peer_locked(my_ip, cl); + prev_master = cl_ctr_i_am_master_locked(cl); + cl_ctr_prune_stale(cl); + cl_ctr_elect_master(cl); + now_master = cl_ctr_i_am_master_locked(cl); + lock_stop_write(cl->peers->lock); + /* Do not broadcast MASTER_ALIVE before we hold the cluster key - + * request a re-key from the current key-holder instead. */ + if (now_master && !cl->have_session_key) { + cl_ctr_request_rekey(cl); + return 0; + } + if (prev_master != now_master) + cl_ctr_arm_master_timers(cl, now_master); + + /* Belt-and-suspenders identity registration (covers paths where + * cl_ctr_handle_node_assign has not yet run). */ + if (!cl->identity_registered && clctl_loaded && my_node_id > 0) { + str url = {cl->bin_socket, (int)strlen(cl->bin_socket)}; + clctl.update_identity(cl->cluster_id, my_node_id, &url); + cl->identity_registered = 1; + } + + /* Bootstrap shtag activation: only for fresh clusters + * (shtag_bootstrapped == -1). Retries until we are master. */ + if (cl->manage_shtags && clctl_loaded && cl->shtag_bootstrapped == -1 + && clctl.activate_backup_shtags) { + int _im; + lock_start_read(cl->peers->lock); + _im = cl_ctr_i_am_master_locked(cl); + lock_stop_read(cl->peers->lock); + if (_im) { + cl_ctr_apply_shtags(cl); /* override-aware */ + cl->shtag_bootstrapped = 1; + } + } + return 0; +} + +static int cl_ctr_on_join_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + cl_ctr_drain_tfd(fd); + + int was_new, auth_fails; + lock_start_write(cl->peers->lock); + was_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + auth_fails = cl->auth_fail_pkts; + + /* Wrong-password / unauthorized guard: if we are still joining at the + * deadline AND we received packets from peers we could not decrypt, then a + * cluster whose key we do not share exists on this multicast group. + * Self-promoting would create a split-brain lone master (and, with managed + * shtags, a duplicate active tag). A wrong-password node also cannot read + * the master's JOIN_REJECT, so this is where it must shut down. */ + if (was_new && auth_fails >= CL_CTR_JOIN_FAIL_LIMIT) { + /* We could not decrypt traffic from peers on our cluster_id: a cluster + * using a different password (or we do) shares this group. Do NOT + * self-terminate on the first deadline - a KEY_GRANT that is merely slow, a + * brief burst of start-up noise, or a flood of crafted garbage would + * otherwise kill a healthy node. Defer and keep re-joining: a real + * KEY_GRANT arriving in the grace window resets auth_fail_pkts + * (cl_ctr_handle_key_grant) and we join normally. Only give up after + * CL_CTR_JOIN_DEFER_MAX rounds still short of authentication - by then the + * wrong-password / foreign-cluster condition is sustained, not transient, + * and self-promoting would create a split-brain lone master. */ + if (cl->auth_defer_count < CL_CTR_JOIN_DEFER_MAX) { + cl->auth_defer_count++; + lock_stop_write(cl->peers->lock); + LM_WARN("clusterer_controller: [cluster %d] %d undecryptable packet(s) " + "on cluster_id %d (%s:%d) - cannot authenticate yet, deferring " + "self-promotion (%d/%d); a slow KEY_GRANT would clear this\n", + cl->cluster_id, auth_fails, cl->cluster_id, + cl->multicast_address, cl->multicast_port, + cl->auth_defer_count, CL_CTR_JOIN_DEFER_MAX); + cl->join_pending = 0; + cl_ctr_send_join_req(cl->sock, cl); + cl_ctr_arm_tfd(cl->join_tfd, CL_CTR_JOIN_DEFER_SECS, 0); + return 0; + } + lock_stop_write(cl->peers->lock); + LM_CRIT("clusterer_controller: [cluster %d] cannot authenticate on %s:%d - " + "%d packet(s) stayed undecryptable and no KEY_GRANT arrived across %d " + "re-join round(s) (wrong password, or a foreign cluster on " + "cluster_id %d). Shutting down.\n", + cl->cluster_id, cl->multicast_address, cl->multicast_port, + auth_fails, cl->auth_defer_count, cl->cluster_id); + exit(-1); + } + + /* Split-brain PREVENTION: if a higher-IP node is also still joining (we + * learned it from its JOIN_REQ), defer self-promotion so it becomes the + * single master and we join it, rather than both self-promoting with + * independent session keys. Bounded by CL_CTR_JOIN_DEFER_MAX so a higher-IP + * node that was heard but never finished starting cannot stall us. */ + if (was_new && cl->join_defer_count < CL_CTR_JOIN_DEFER_MAX + && cl->join_defer_total < CL_CTR_JOIN_DEFER_HARDMAX) { + unsigned int my_ipn = ip_to_num(my_ip); + int higher_seen = 0, _i; + for (_i = 0; _i < cl->peers->count; _i++) { + if (strcmp(cl->peers->entries[_i].ip, my_ip) == 0) + continue; + if (cl->peers->entries[_i].ip_num > my_ipn) { + higher_seen = 1; + break; + } + } + if (higher_seen) { + cl->join_defer_count++; + cl->join_defer_total++; + lock_stop_write(cl->peers->lock); + LM_INFO("clusterer_controller: [cluster %d] join deadline: a higher-IP " + "node is still joining - deferring self-promotion (%d/%d) to " + "avoid split brain\n", + cl->cluster_id, cl->join_defer_count, CL_CTR_JOIN_DEFER_MAX); + /* Re-send a JOIN_REQ now (clear join_pending so it is not suppressed) + * so the higher-IP node answers as soon as it becomes master, then + * extend the join window for one more short round. */ + cl->join_pending = 0; + cl_ctr_send_join_req(cl->sock, cl); + cl_ctr_arm_tfd(cl->join_tfd, CL_CTR_JOIN_DEFER_SECS, 0); + return 0; + } + } + + if (was_new) { + LM_INFO("clusterer_controller: [cluster %d] join deadline expired, " + "no master found - transitioning to CL_CTR_NODE_ACTIVE\n", + cl->cluster_id); + cl->join_defer_count = 0; /* leaving NEW state */ + cl->join_defer_total = 0; + cl->shtag_bootstrapped = -1; /* fresh cluster: eligible to claim active */ + cl_ctr_upsert_peer_locked(my_ip, cl); + my_node_id = cl_ctr_alloc_node_id_locked(cl); + { + char self_sock[1][CL_CTR_MAX_BIN_SOCK_LEN]; + memcpy(self_sock[0], cl->bin_socket, CL_CTR_MAX_BIN_SOCK_LEN); + cl_ctr_update_peer_bin_locked(my_ip, my_node_id, 1, + (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN]) + self_sock, cl); + } + cl->peers->node_state = CL_CTR_NODE_ACTIVE; + /* Run election first so is_master and last_master are set before + * cl_ctr_on_became_master. Without this, cl_ctr_handle_alive would see + * is_master=0 on the loopback ALIVE and call cl_ctr_on_became_master + * a second time, regenerating the session key unnecessarily. */ + cl_ctr_elect_master(cl); + cl_ctr_on_became_master(cl); + } + lock_stop_write(cl->peers->lock); + if (!was_new) + return 0; /* already active via MEMBER_LIST - nothing to do */ + cl_ctr_transition_to_active(cl); + cl_ctr_arm_master_timers(cl, 1); + + if (!cl->identity_registered && clctl_loaded && my_node_id > 0) { + str url = {cl->bin_socket, (int)strlen(cl->bin_socket)}; + clctl.update_identity(cl->cluster_id, my_node_id, &url); + cl->identity_registered = 1; + } + return 0; +} + +static int cl_ctr_on_rejoin_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + int still_new; + + cl_ctr_drain_tfd(fd); + lock_start_read(cl->peers->lock); + still_new = (cl->peers->node_state == CL_CTR_NODE_NEW); + lock_stop_read(cl->peers->lock); + if (still_new && !cl->join_pending) { + cl->join_attempt_count++; + + /* Wrong-password fallback: if the master keeps sending bootstrap + * packets we cannot decrypt (bootstrap_auth_fails > 0) and we have + * already retried CL_CTR_JOIN_FAIL_LIMIT times, give up. This fires when + * the encrypted JOIN_REJECT from the master was lost in transit (if it + * arrived intact, cl_ctr_handle_join_reject would have already exited). */ + if (cl->join_attempt_count >= CL_CTR_JOIN_FAIL_LIMIT + && cl->bootstrap_auth_fails > 0) { + LM_CRIT("clusterer_controller: [cluster %d] join failed after %d " + "attempts with %d bootstrap auth error(s) - wrong password? " + "Shutting down.\n", + cl->cluster_id, cl->join_attempt_count, + cl->bootstrap_auth_fails); + exit(-1); + } + + cl_ctr_send_join_req(cl->sock, cl); + cl->join_pending = 1; + LM_DBG("clusterer_controller: [cluster %d] resending JOIN_REQ " + "(attempt %d)\n", cl->cluster_id, cl->join_attempt_count); + } + return 0; +} + +static int cl_ctr_on_master_alive_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + cl_ctr_drain_tfd(fd); + cl_ctr_send_master_alive(cl->sock, cl); + /* Emit a bootstrap-key beacon every CL_CTR_MASTER_BEACON_EVERY ticks so any + * peer master holding a different session key can find us and merge. */ + if (++cl->beacon_tick >= CL_CTR_MASTER_BEACON_EVERY) { + cl->beacon_tick = 0; + cl_ctr_send_master_beacon(cl->sock, cl); + } + return 0; +} + +static int cl_ctr_on_master_dead_tfd(int fd, void *param, int was_timeout) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + int now_master; + char dead_master[CL_CTR_MAX_IP_LEN + 1]; + + cl_ctr_drain_tfd(fd); + + dead_master[0] = '\0'; + + lock_start_write(cl->peers->lock); + /* The master has been silent for CL_CTR_MASTER_KA_TIMEOUT (3s). Age the silent + * master OUT of the election window before re-electing, otherwise + * cl_ctr_elect_master would just re-select it: the election window is + * query_time * CL_CTR_ELECT_FACTOR (~15s), far longer than the keepalive + * timeout, so a just-declared-dead master stays "eligible" and keeps + * winning - delaying real failover by ~12s. Zeroing its last_seen makes + * the immediate re-election pick the next-highest LIVE peer. If the + * master was only briefly unreachable, its next MASTER_ALIVE/ALIVE + * refreshes last_seen and the normal highest-IP election restores it. */ + { + int _i; + for (_i = 0; _i < cl->peers->count; _i++) { + if (cl->peers->entries[_i].is_master && + strcmp(cl->peers->entries[_i].ip, my_ip) != 0) { + size_t _l = strnlen(cl->peers->entries[_i].ip, CL_CTR_MAX_IP_LEN); + memcpy(dead_master, cl->peers->entries[_i].ip, _l); + dead_master[_l] = '\0'; + cl->peers->entries[_i].last_seen = 0; + cl->peers->entries[_i].in_election = 0; + } + cl->peers->entries[_i].is_master = 0; + } + cl->peers->last_master[0] = '\0'; + } + + LM_INFO("clusterer_controller: [cluster %d] master %s went silent " + "(no keepalive for %ds) - re-electing\n", + cl->cluster_id, + dead_master[0] ? dead_master : "(unknown)", + CL_CTR_MASTER_KA_TIMEOUT); + + /* cl_ctr_elect_master logs the resulting MASTER/BACKUP roles and why. */ + cl_ctr_elect_master(cl); + now_master = cl_ctr_i_am_master_locked(cl); + lock_stop_write(cl->peers->lock); + + /* Preserve-key recovery: every surviving member already holds the session + * key, so the winner has it too. Guard defensively against the impossible + * keyless-winner case rather than broadcasting an undecryptable MEMBER_LIST. */ + if (now_master && !cl->have_session_key) { + LM_WARN("clusterer_controller: [cluster %d] elected master after " + "keepalive timeout but hold no session key - deferring\n", + cl->cluster_id); + return 0; + } + + cl_ctr_arm_master_timers(cl, now_master); + + /* No full MEMBER_LIST is broadcast: every surviving member ran the same + * deterministic re-election on its own keepalive timeout, MASTER_ALIVE + * (which the new master now emits) re-asserts the winner, and the + * membership digest reconciles any divergence - so the O(N) re-broadcast + * (which fragments at large N) is redundant. A node still joining reaches + * the new master through its own JOIN_REQ retry. */ + return 0; +} + +/** + * cl_ctr_worker() - the single dedicated background process. + * + * JOIN PROTOCOL: + * 1. Open socket, join multicast group. + * 2. Send CL_CTR_PKT_JOIN_REQ and set state = CL_CTR_NODE_NEW with a deadline + * of (now + query_time). + * 3. Listen for incoming packets. If CL_CTR_PKT_MEMBER_LIST arrives: + * -> cl_ctr_handle_member_list() sets state = CL_CTR_NODE_ACTIVE. + * If deadline expires with no MEMBER_LIST: + * -> no master exists yet; transition to CL_CTR_NODE_ACTIVE and + * join the normal election cycle. + * + * ACTIVE LOOP: + * Fully event-driven via OpenSIPS reactor (epoll by default). + * Each event source is a registered fd with a dedicated callback: + * cl_ctr_on_sock - incoming UDP packet + * cl_ctr_on_alive_tfd - periodic ALIVE heartbeat (query_time seconds) + * cl_ctr_on_join_tfd - one-shot join deadline + * cl_ctr_on_rejoin_tfd - 1-second JOIN_REQ retry while in CL_CTR_NODE_NEW + * reactor_proc_init() also wires in IPC (shutdown, load stats, reload). + */ +static void cl_ctr_worker(int rank) +{ + cl_ctr_cluster_t *cl; + + if (rank >= cl_ctr_cluster_count) { + /* Extra process slot - no cluster assigned, exit cleanly */ + return; + } + cl = &cl_ctr_clusters[rank]; + + LM_INFO("clusterer_controller: [cluster %d] worker started (pid=%d)\n", + cl->cluster_id, getpid()); + + /* Publish our process index so MI handlers (other processes) can reach us + * via ipc_send_rpc() to apply operator-driven shtag overrides promptly. */ + cl->peers->worker_proc_no = process_no; + + cl->shtag_last_active = -1; /* unknown - first decision logs its reason */ + cl->shtag_last_forced = 0; + + cl->sock = cl_ctr_setup_socket(cl); + if (cl->sock < 0) { + LM_CRIT("clusterer_controller: [cluster %d] cannot open multicast socket, " + "worker exits\n", cl->cluster_id); + exit(-1); + } + + /* Generate ephemeral X25519 keypair - private key never leaves this process */ + if (cl_ctr_gen_ecdh_keypair(cl->my_privkey, cl->my_pubkey) < 0) { + LM_CRIT("clusterer_controller: [cluster %d] ECDH keypair generation failed\n", + cl->cluster_id); + exit(-1); + } + + /* Store our pubkey in peer table entry for ourselves so MEMBER_LIST broadcasts it */ + lock_start_write(cl->peers->lock); + { + int _i; + for (_i = 0; _i < cl->peers->count; _i++) { + if (strcmp(cl->peers->entries[_i].ip, my_ip) == 0) { + memcpy(cl->peers->entries[_i].pubkey, cl->my_pubkey, CL_CTR_PUBKEY_SZ); + break; + } + } + } + lock_stop_write(cl->peers->lock); + + cl->alive_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + cl->join_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + cl->rejoin_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + cl->master_alive_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + cl->master_dead_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + cl->retx_tfd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK); + if (cl->alive_tfd < 0 || cl->join_tfd < 0 || cl->rejoin_tfd < 0 || + cl->master_alive_tfd < 0 || cl->master_dead_tfd < 0 || cl->retx_tfd < 0) { + LM_CRIT("clusterer_controller: [cluster %d] timerfd_create: %s\n", + cl->cluster_id, strerror(errno)); + exit(-1); + } + + cl->rate_tbl = pkg_malloc(CL_CTR_RATE_TBL_SZ * sizeof(cl_ctr_rate_entry_t)); + if (!cl->rate_tbl) { + LM_CRIT("clusterer_controller: [cluster %d] no pkg memory for rate table\n", + cl->cluster_id); + exit(-1); + } + memset(cl->rate_tbl, 0, CL_CTR_RATE_TBL_SZ * sizeof(cl_ctr_rate_entry_t)); + + /* ---- Phase 1: join protocol ---- */ + lock_start_write(cl->peers->lock); + cl->peers->node_state = CL_CTR_NODE_NEW; + cl->peers->join_deadline = time(NULL) + (time_t)query_time; + lock_stop_write(cl->peers->lock); + + cl_ctr_send_join_req(cl->sock, cl); + cl_ctr_arm_tfd(cl->join_tfd, (time_t)query_time, 0); /* one-shot deadline */ + cl_ctr_arm_tfd(cl->rejoin_tfd, 1, 1); /* retry every 1 s */ + /* alive_tfd left disarmed - armed by cl_ctr_transition_to_active() */ + + LM_INFO("clusterer_controller: [cluster %d] sent JOIN_REQ, waiting up to %ds " + "for master response\n", cl->cluster_id, query_time); + + /* ---- Register with reactor and enter event loop ---- */ + if (reactor_proc_init("clusterer_controller worker") < 0) { + LM_CRIT("clusterer_controller: [cluster %d] reactor_proc_init failed\n", + cl->cluster_id); + exit(-1); + } + + if (reactor_proc_add_fd(cl->sock, cl_ctr_on_sock, cl) < 0 || + reactor_proc_add_fd(cl->alive_tfd, cl_ctr_on_alive_tfd, cl) < 0 || + reactor_proc_add_fd(cl->join_tfd, cl_ctr_on_join_tfd, cl) < 0 || + reactor_proc_add_fd(cl->rejoin_tfd, cl_ctr_on_rejoin_tfd, cl) < 0 || + reactor_proc_add_fd(cl->master_alive_tfd, cl_ctr_on_master_alive_tfd, cl) < 0 || + reactor_proc_add_fd(cl->master_dead_tfd, cl_ctr_on_master_dead_tfd, cl) < 0 || + reactor_proc_add_fd(cl->retx_tfd, cl_ctr_on_retx_tfd, cl) < 0) { + LM_CRIT("clusterer_controller: [cluster %d] reactor_proc_add_fd failed\n", + cl->cluster_id); + exit(-1); + } + + reactor_proc_loop(); + + /* ---- Graceful shutdown epilogue ---- */ + { + int i_am_master; + lock_start_read(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + lock_stop_read(cl->peers->lock); + + if (i_am_master && cl->peers->count > 1) { + /* Find the peer that will win the next election (highest IP, not us) */ + unsigned int best_ip = 0; + int best_idx = -1; + int _i; + lock_start_read(cl->peers->lock); + for (_i = 0; _i < cl->peers->count; _i++) { + cl_ctr_peer_t *e = &cl->peers->entries[_i]; + if (strcmp(e->ip, my_ip) == 0) continue; + if (e->ip_num > best_ip) { best_ip = e->ip_num; best_idx = _i; } + } + if (best_idx >= 0) { + char next_ip[CL_CTR_MAX_IP_LEN + 1]; + unsigned char next_pub[CL_CTR_PUBKEY_SZ]; + memcpy(next_ip, cl->peers->entries[best_idx].ip, CL_CTR_MAX_IP_LEN + 1); + memcpy(next_pub, cl->peers->entries[best_idx].pubkey, CL_CTR_PUBKEY_SZ); + lock_stop_read(cl->peers->lock); + cl_ctr_send_key_handoff(cl->sock, next_ip, next_pub, cl); + } else { + lock_stop_read(cl->peers->lock); + } + } + } + + close(cl->sock); + close(cl->alive_tfd); + close(cl->join_tfd); + close(cl->rejoin_tfd); + close(cl->master_alive_tfd); + close(cl->master_dead_tfd); +} + +/* ========================================================================= + * MI command handlers + * ========================================================================= */ + +/** + * mi_cl_ctr_members() - list active cluster members with their role. + * + * opensips-cli -x mi cl_ctr_list_members + * + * [ + * {"ip": "10.0.0.3", "status": "master"}, + * {"ip": "10.0.0.1", "status": "member"}, + * {"ip": "10.0.0.2", "status": "member"} + * ] + * + * Only peers within the current quantized election window are shown, + * consistent with what cl_ctr_elect_master(cl) considers. + */ +static mi_response_t *mi_cl_ctr_members(const mi_params_t *params, + struct mi_handler *hdl) +{ + mi_response_t *resp; + mi_item_t *arr, *cl_obj, *members_arr, *peer_obj, *bin_arr; + int i, j, ci; + cl_ctr_cluster_t *cl; + + resp = init_mi_result_array(&arr); + if (!resp) + return NULL; + + for (ci = 0; ci < cl_ctr_cluster_count; ci++) { + cl = &cl_ctr_clusters[ci]; + if (!cl->peers) + continue; + + cl_obj = add_mi_object(arr, NULL, 0); + if (!cl_obj) goto error; + + if (add_mi_number(cl_obj, MI_SSTR("cluster_id"), cl->cluster_id) < 0) + goto error; + + members_arr = add_mi_array(cl_obj, MI_SSTR("members")); + if (!members_arr) goto error; + + lock_start_read(cl->peers->lock); + + for (i = 0; i < cl->peers->count; i++) { + cl_ctr_peer_t *e = &cl->peers->entries[i]; + + peer_obj = add_mi_object(members_arr, NULL, 0); + if (!peer_obj) { + lock_stop_read(cl->peers->lock); + goto error; + } + if (add_mi_string(peer_obj, MI_SSTR("ip"), + e->ip, strlen(e->ip)) < 0 || + add_mi_number(peer_obj, MI_SSTR("node_id"), e->node_id) < 0 || + add_mi_string(peer_obj, MI_SSTR("status"), + e->is_master ? "master" : (e->is_backup ? "backup" : "member"), + 6) < 0) { + lock_stop_read(cl->peers->lock); + goto error; + } + + bin_arr = add_mi_array(peer_obj, MI_SSTR("bin_sockets")); + if (!bin_arr) { + lock_stop_read(cl->peers->lock); + goto error; + } + for (j = 0; j < (int)e->bin_count; j++) { + if (add_mi_string(bin_arr, NULL, 0, + e->bin_sockets[j], + strlen(e->bin_sockets[j])) < 0) { + lock_stop_read(cl->peers->lock); + goto error; + } + } + } + + lock_stop_read(cl->peers->lock); + } + + return resp; + +error: + LM_ERR("clusterer_controller: mi_cl_ctr_members: failed to build response\n"); + free_mi_response(resp); + return NULL; +} + +/** + * mi_cl_ctr_node_info() - return full info for a specific node_id. + * + * opensips-cli -x mi cl_ctr_node_info node_id=2 + */ +static mi_response_t *mi_cl_ctr_node_info(const mi_params_t *params, + struct mi_handler *hdl) +{ + mi_response_t *resp; + mi_item_t *root, *bin_arr; + int target_id; + int i, j, ci; + cl_ctr_cluster_t *cl; + cl_ctr_peer_t *e; + + if (get_mi_int_param(params, "node_id", &target_id) < 0) + return init_mi_param_error(); + + for (ci = 0; ci < cl_ctr_cluster_count; ci++) { + cl = &cl_ctr_clusters[ci]; + if (!cl->peers) + continue; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + e = &cl->peers->entries[i]; + if ((int)e->node_id != target_id) + continue; + + resp = init_mi_result_object(&root); + if (!resp) { + lock_stop_read(cl->peers->lock); + return NULL; + } + if (add_mi_number(root, MI_SSTR("node_id"), e->node_id) < 0 || + add_mi_string(root, MI_SSTR("ip"), e->ip, strlen(e->ip)) < 0 || + add_mi_number(root, MI_SSTR("cluster_id"), cl->cluster_id) < 0 || + add_mi_string(root, MI_SSTR("status"), + e->is_master ? "master" : (e->is_backup ? "backup" : "member"), + 6) < 0) { + lock_stop_read(cl->peers->lock); + goto error_node; + } + bin_arr = add_mi_array(root, MI_SSTR("bin_sockets")); + if (!bin_arr) { + lock_stop_read(cl->peers->lock); + goto error_node; + } + for (j = 0; j < (int)e->bin_count; j++) { + if (add_mi_string(bin_arr, NULL, 0, + e->bin_sockets[j], + strlen(e->bin_sockets[j])) < 0) { + lock_stop_read(cl->peers->lock); + goto error_node; + } + } + lock_stop_read(cl->peers->lock); + return resp; + } + lock_stop_read(cl->peers->lock); + } + + return init_mi_error(404, MI_SSTR("node_id not found")); + +error_node: + LM_ERR("clusterer_controller: mi_cl_ctr_node_info: failed to build response\n"); + free_mi_response(resp); + return NULL; +} + +/** + * mi_cl_ctr_config() - list all configured clusters and their resolved settings. + * + * opensips-cli -x mi cl_ctr_list_config + * + * Reports per cluster: id, multicast endpoint, master_stickiness, + * manage_shtags, query_time, this node's BIN socket and current member count. + * The password is intentionally NOT exposed. + */ +static mi_response_t *mi_cl_ctr_config(const mi_params_t *params, + struct mi_handler *hdl) +{ + mi_response_t *resp; + mi_item_t *arr, *cl_obj; + int ci, members; + char mcast[INET_ADDRSTRLEN + 8]; /* "IP:PORT" */ + char shtag_mode[24]; /* "auto" / "override:" */ + cl_ctr_cluster_t *cl; + + resp = init_mi_result_array(&arr); + if (!resp) + return NULL; + + for (ci = 0; ci < cl_ctr_cluster_count; ci++) { + cl = &cl_ctr_clusters[ci]; + + cl_obj = add_mi_object(arr, NULL, 0); + if (!cl_obj) goto error; + + snprintf(mcast, sizeof(mcast), "%s:%d", + cl->multicast_address, cl->multicast_port); + + members = 0; + /* Report the EFFECTIVE (possibly adopted) settings from shm, so the + * values reflect what is actually in force after on_config_mismatch= + * adopt (the worker mirrors them there). Fall back to the configured + * values if the peer table is not up yet. */ + int eff_manage = cl->manage_shtags, eff_stick = cl->master_stickiness, + eff_qt = query_time; + { + uint16_t forced = 0; + if (cl->peers) { + lock_start_read(cl->peers->lock); + members = cl->peers->count; + forced = cl->peers->shtag_forced_node_id; + eff_manage = cl->peers->eff_manage_shtags; + eff_stick = cl->peers->eff_master_stickiness; + eff_qt = cl->peers->eff_query_time; + lock_stop_read(cl->peers->lock); + } + /* Current shtag allocation mode: "auto" = master-driven automatic + * allocation; "override:" = operator forced a fixed holder + * via cl_ctr_shtag_force. ("manual" is reserved for the future + * maintenance mode.) */ + if (forced) + snprintf(shtag_mode, sizeof(shtag_mode), "override:%u", forced); + else + snprintf(shtag_mode, sizeof(shtag_mode), "auto"); + } + + if (add_mi_number(cl_obj, MI_SSTR("cluster_id"), cl->cluster_id) < 0 || + add_mi_string(cl_obj, MI_SSTR("multicast"), mcast, strlen(mcast)) < 0 || + add_mi_string(cl_obj, MI_SSTR("my_ip"), my_ip, strlen(my_ip)) < 0 || + add_mi_string(cl_obj, MI_SSTR("bin_socket"), + cl->bin_socket, strlen(cl->bin_socket)) < 0 || + add_mi_number(cl_obj, MI_SSTR("query_time"), eff_qt) < 0 || + add_mi_number(cl_obj, MI_SSTR("master_stickiness"), eff_stick) < 0 || + add_mi_number(cl_obj, MI_SSTR("manage_shtags"), eff_manage) < 0 || + add_mi_string(cl_obj, MI_SSTR("shtag_mode"), + shtag_mode, strlen(shtag_mode)) < 0 || + add_mi_number(cl_obj, MI_SSTR("member_count"), members) < 0) + goto error; + } + + return resp; + +error: + LM_ERR("clusterer_controller: mi_cl_ctr_config: failed to build response\n"); + free_mi_response(resp); + return NULL; +} + +/** + * cl_ctr_rpc_apply_shtags() - IPC job run inside the cl_ctr_worker process. + * + * An MI handler (running in a different process) has already updated + * cl->peers->shtag_forced_node_id in shm; this job makes the change take + * effect promptly: it re-applies the local shtag decision and, if we are the + * master, re-broadcasts the MEMBER_LIST so every member learns the new + * override without waiting for the next periodic announcement. + */ +static void cl_ctr_rpc_apply_shtags(int sender, void *param) +{ + cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; + int i_am_master; + + (void)sender; + if (!cl || !cl->peers) + return; + + cl_ctr_apply_shtags(cl); + + lock_start_read(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + lock_stop_read(cl->peers->lock); + + if (i_am_master && cl->sock >= 0) + cl_ctr_send_member_list(cl->sock, cl); +} + +/** + * cl_ctr_mi_find_cluster() - locate a configured cluster by its cluster_id. + * Returns NULL if no cluster matches. + */ +static cl_ctr_cluster_t *cl_ctr_mi_find_cluster(int cluster_id) +{ + int ci; + for (ci = 0; ci < cl_ctr_cluster_count; ci++) + if (cl_ctr_clusters[ci].cluster_id == cluster_id) + return &cl_ctr_clusters[ci]; + return NULL; +} + +/** + * mi_cl_ctr_shtag_force() - force a specific node to hold the active sharing tag. + * + * opensips-cli -x mi cl_ctr_shtag_force cluster_id=1 node_id=2 + * + * Must be issued on the current master. Suspends automatic (master-driven) + * shtag allocation until cl_ctr_shtag_auto is called; the chosen node becomes the + * sole active shtag holder cluster-wide. The override is propagated to every + * member via the MEMBER_LIST and survives master fail-over, but is cleared + * automatically if the forced node leaves or times out. + */ +static mi_response_t *mi_cl_ctr_shtag_force(const mi_params_t *params, + struct mi_handler *hdl) +{ + int cluster_id, node_id; + cl_ctr_cluster_t *cl; + int i_am_master, found = 0, proc_no; + + if (get_mi_int_param(params, "cluster_id", &cluster_id) < 0 || + get_mi_int_param(params, "node_id", &node_id) < 0) + return init_mi_param_error(); + + if (node_id <= 0 || node_id > 0xFFFF) + return init_mi_error(400, MI_SSTR("node_id out of range")); + + cl = cl_ctr_mi_find_cluster(cluster_id); + if (!cl || !cl->peers) + return init_mi_error(404, MI_SSTR("cluster_id not found")); + + if (!cl->manage_shtags) + return init_mi_error(409, + MI_SSTR("shtag management is disabled for this cluster")); + + lock_start_write(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + if (i_am_master) { + int i; + for (i = 0; i < cl->peers->count; i++) { + if ((int)cl->peers->entries[i].node_id == node_id) { + found = 1; + break; + } + } + /* TODO(maintenance-mode): once node maintenance mode lands, reject + * forcing the shtag onto a node that is currently in maintenance. */ + if (found) + cl->peers->shtag_forced_node_id = (uint16_t)node_id; + } + proc_no = cl->peers->worker_proc_no; + lock_stop_write(cl->peers->lock); + + if (!i_am_master) + return init_mi_error(409, + MI_SSTR("not the master - issue cl_ctr_shtag_force on the master node")); + if (!found) + return init_mi_error(404, MI_SSTR("node_id not a member of this cluster")); + + /* Apply locally and broadcast the new override from the worker process. */ + if (proc_no >= 0) + ipc_send_rpc(proc_no, cl_ctr_rpc_apply_shtags, cl); + + LM_INFO("clusterer_controller: [cluster %d] operator forced shtag onto " + "node %d\n", cluster_id, node_id); + return init_mi_result_ok(); +} + +/** + * mi_cl_ctr_shtag_auto() - resume automatic (master-driven) shtag allocation. + * + * opensips-cli -x mi cl_ctr_shtag_auto cluster_id=1 + * + * Clears any operator override set by cl_ctr_shtag_force so the active shtag + * follows the master again. Must be issued on the current master. + */ +static mi_response_t *mi_cl_ctr_shtag_auto(const mi_params_t *params, + struct mi_handler *hdl) +{ + int cluster_id; + cl_ctr_cluster_t *cl; + int i_am_master, proc_no; + + if (get_mi_int_param(params, "cluster_id", &cluster_id) < 0) + return init_mi_param_error(); + + cl = cl_ctr_mi_find_cluster(cluster_id); + if (!cl || !cl->peers) + return init_mi_error(404, MI_SSTR("cluster_id not found")); + + lock_start_write(cl->peers->lock); + i_am_master = cl_ctr_i_am_master_locked(cl); + if (i_am_master) + cl->peers->shtag_forced_node_id = 0; + proc_no = cl->peers->worker_proc_no; + lock_stop_write(cl->peers->lock); + + if (!i_am_master) + return init_mi_error(409, + MI_SSTR("not the master - issue cl_ctr_shtag_auto on the master node")); + + if (proc_no >= 0) + ipc_send_rpc(proc_no, cl_ctr_rpc_apply_shtags, cl); + + LM_INFO("clusterer_controller: [cluster %d] operator resumed automatic " + "shtag allocation\n", cluster_id); + return init_mi_result_ok(); +} +/* ========================================================================= + * Lifecycle + * ========================================================================= */ + +/** + * cl_ctr_resolve_local_identity() - determine my_ip and my_interface_buf. + * + * Three modes depending on which modparams were provided: + * + * Mode 1 - ip= only: + * Walk getifaddrs() to find the interface that owns the given IP. + * Fails if no interface owns it. + * + * Mode 2 - interface= only: + * Walk getifaddrs() to find the interface and take its first IPv4 address. + * Warns if the interface has multiple IPv4 addresses; uses the first one + * (the kernel's enumeration order matches `ip addr show`). + * + * Mode 3 - neither: + * Connect a throw-away UDP socket to the multicast group (no data sent). + * getsockname() returns the source IP the kernel would select. + * Reverse-look up the interface name via getifaddrs(). + * + * On success: my_ip points to a valid dotted-decimal IPv4 string and + * my_interface_buf holds the interface name (may be empty if the + * reverse lookup failed in mode 3 - non-fatal). + */ +/** + * cl_ctr_parse_cluster_str() - parse one "cluster" modparam string into cl. + * + * Format: "id=N,multicast=A.B.C.D:PORT[,password=STRING][,bin_socket=bin:IP:PORT]" + * - id= required, positive integer + * - multicast= required, IPv4:port + * - password= optional, falls back to global password modparam + * - bin_socket= optional, BIN socket for this cluster; falls back to + * first discovered socket (or only socket if one exists) + */ +static int cl_ctr_parse_cluster_str(const char *str, cl_ctr_cluster_t *cl) +{ + char buf[2048]; + char *p, *tok, *key, *val, *colon; + struct in_addr addr; + unsigned char first_octet; + int has_id = 0, has_mcast = 0; + + strncpy(buf, str, sizeof(buf) - 1); + buf[sizeof(buf) - 1] = '\0'; + + /* Defaults */ + cl->multicast_port = 3333; + strncpy(cl->password, password, sizeof(cl->password) - 1); + cl->password[sizeof(cl->password) - 1] = '\0'; + cl->manage_shtags = -1; /* sentinel: inherit global default in mod_init */ + cl->master_stickiness = -1; /* sentinel: inherit global default in mod_init */ + + for (tok = strtok_r(buf, ",", &p); tok; tok = strtok_r(NULL, ",", &p)) { + while (*tok == ' ' || *tok == '\t') tok++; + key = tok; + val = strchr(tok, '='); + if (!val) continue; + *val++ = '\0'; + + if (strcmp(key, "id") == 0) { + cl->cluster_id = atoi(val); + if (cl->cluster_id <= 0 || cl->cluster_id > 65535) { + LM_ERR("clusterer_controller: cluster id must be 1..65535 " + "(carried as a 2-byte field on the wire)\n"); + return -1; + } + has_id = 1; + + } else if (strcmp(key, "multicast") == 0) { + colon = strrchr(val, ':'); + if (colon) { + *colon = '\0'; + cl->multicast_port = atoi(colon + 1); + if (cl->multicast_port <= 0 || cl->multicast_port > 65535) { + LM_ERR("clusterer_controller: invalid port in cluster '%s'\n", str); + return -1; + } + } + strncpy(cl->multicast_address, val, INET_ADDRSTRLEN - 1); + cl->multicast_address[INET_ADDRSTRLEN - 1] = '\0'; + if (inet_aton(cl->multicast_address, &addr) == 0) { + LM_ERR("clusterer_controller: invalid multicast address '%s'\n", val); + return -1; + } + first_octet = (unsigned char)((ntohl(addr.s_addr) >> 24) & 0xFF); + if (first_octet < 224 || first_octet > 239) { + LM_ERR("clusterer_controller: '%s' is not a multicast address\n", val); + return -1; + } + has_mcast = 1; + + } else if (strcmp(key, "password") == 0) { + strncpy(cl->password, val, sizeof(cl->password) - 1); + cl->password[sizeof(cl->password) - 1] = '\0'; + + } else if (strcmp(key, "bin_socket") == 0) { + if (strlen(val) >= CL_CTR_MAX_BIN_SOCK_LEN) { + LM_ERR("clusterer_controller: bin_socket value too long\n"); + return -1; + } + strncpy(cl->bin_socket, val, CL_CTR_MAX_BIN_SOCK_LEN - 1); + cl->bin_socket[CL_CTR_MAX_BIN_SOCK_LEN - 1] = '\0'; + + } else if (strcmp(key, "manage_shtags") == 0) { + cl->manage_shtags = atoi(val) ? 1 : 0; + + } else if (strcmp(key, "master_stickiness") == 0) { + cl->master_stickiness = atoi(val) ? 1 : 0; + } + } + + if (!has_id) { + LM_ERR("clusterer_controller: cluster string missing id= in '%s'\n", str); + return -1; + } + if (!has_mcast) { + LM_ERR("clusterer_controller: cluster string missing multicast= in '%s'\n", str); + return -1; + } + return 0; +} + +static int cl_ctr_resolve_local_identity(void) +{ + struct ifaddrs *ifap = NULL, *ifa; + int found = 0; + + if (getifaddrs(&ifap) < 0) { + LM_ERR("clusterer_controller: getifaddrs() failed: %s\n", + strerror(errno)); + return -1; + } + + if (my_ip && *my_ip) { + /* ---- Mode 1: ip= explicitly provided - find owning interface ---- */ + struct in_addr target; + + if (inet_aton(my_ip, &target) == 0) { + LM_ERR("clusterer_controller: cannot parse 'my_ip' '%s'\n", my_ip); + freeifaddrs(ifap); + return -1; + } + for (ifa = ifap; ifa; ifa = ifa->ifa_next) { + if (!ifa->ifa_addr || ifa->ifa_addr->sa_family != AF_INET) + continue; + if (((struct sockaddr_in *)ifa->ifa_addr)->sin_addr.s_addr + == target.s_addr) { + strncpy(my_interface_buf, ifa->ifa_name, IF_NAMESIZE - 1); + my_interface_buf[IF_NAMESIZE - 1] = '\0'; + found = 1; + break; + } + } + if (!found) { + LM_ERR("clusterer_controller: no local interface owns IP '%s'\n", + my_ip); + freeifaddrs(ifap); + return -1; + } + LM_INFO("clusterer_controller: using IP %s on interface %s\n", + my_ip, my_interface_buf); + + } else if (my_interface && *my_interface) { + /* ---- Mode 2: interface= provided - derive IP from it ---- */ + int addr_count = 0; + + strncpy(my_interface_buf, my_interface, IF_NAMESIZE - 1); + my_interface_buf[IF_NAMESIZE - 1] = '\0'; + + for (ifa = ifap; ifa; ifa = ifa->ifa_next) { + if (!ifa->ifa_addr || ifa->ifa_addr->sa_family != AF_INET) + continue; + if (strcmp(ifa->ifa_name, my_interface) != 0) + continue; + addr_count++; + if (addr_count == 1) { + struct in_addr a = + ((struct sockaddr_in *)ifa->ifa_addr)->sin_addr; + inet_ntop(AF_INET, &a, my_ip_buf, sizeof(my_ip_buf)); + my_ip = my_ip_buf; + found = 1; + } + } + if (!found) { + LM_ERR("clusterer_controller: interface '%s' not found or " + "has no IPv4 address\n", my_interface); + freeifaddrs(ifap); + return -1; + } + if (addr_count > 1) + LM_WARN("clusterer_controller: interface '%s' has %d IPv4 " + "addresses, using %s (first returned by kernel) - " + "use 'my_ip' modparam to override\n", + my_interface, addr_count, my_ip); + else + LM_INFO("clusterer_controller: using IP %s on interface %s\n", + my_ip, my_interface_buf); + + } else { + /* ---- Mode 3: neither - auto-detect via kernel routing table ---- */ + struct sockaddr_in dest, local; + socklen_t local_len = sizeof(local); + int probe; + + memset(&dest, 0, sizeof(dest)); + dest.sin_family = AF_INET; + dest.sin_port = htons((uint16_t)cl_ctr_clusters[0].multicast_port); + dest.sin_addr.s_addr = inet_addr(cl_ctr_clusters[0].multicast_address); + + probe = socket(AF_INET, SOCK_DGRAM, 0); + if (probe < 0) { + LM_ERR("clusterer_controller: auto-detect socket: %s\n", + strerror(errno)); + freeifaddrs(ifap); + return -1; + } + if (connect(probe, (struct sockaddr *)&dest, sizeof(dest)) < 0) { + LM_ERR("clusterer_controller: auto-detect connect: %s\n", + strerror(errno)); + close(probe); + freeifaddrs(ifap); + return -1; + } + memset(&local, 0, sizeof(local)); + if (getsockname(probe, (struct sockaddr *)&local, &local_len) < 0) { + LM_ERR("clusterer_controller: auto-detect getsockname: %s\n", + strerror(errno)); + close(probe); + freeifaddrs(ifap); + return -1; + } + close(probe); + + inet_ntop(AF_INET, &local.sin_addr, my_ip_buf, sizeof(my_ip_buf)); + my_ip = my_ip_buf; + + /* Reverse-look up the interface name */ + for (ifa = ifap; ifa; ifa = ifa->ifa_next) { + if (!ifa->ifa_addr || ifa->ifa_addr->sa_family != AF_INET) + continue; + if (((struct sockaddr_in *)ifa->ifa_addr)->sin_addr.s_addr + == local.sin_addr.s_addr) { + strncpy(my_interface_buf, ifa->ifa_name, IF_NAMESIZE - 1); + my_interface_buf[IF_NAMESIZE - 1] = '\0'; + found = 1; + break; + } + } + if (found) + LM_INFO("clusterer_controller: auto-detected IP %s on " + "interface %s\n", my_ip, my_interface_buf); + else + LM_WARN("clusterer_controller: auto-detected IP %s but could " + "not determine interface name\n", my_ip); + } + + freeifaddrs(ifap); + return 0; +} + +/** + * cl_ctr_discover_bin_sockets() - enumerate BIN listeners via proto_bin. + * + * Walks the protos[PROTO_BIN].listeners list and collects every + * entry with proto == PROTO_BIN. proto_bin must be loaded before this + * module so the listeners are already registered when mod_init() runs. + * + * Populates my_bin_sockets[] and my_bin_count. + * Returns 0 on success, -1 if no BIN sockets are found. + */ +static int cl_ctr_discover_bin_sockets(void) +{ + struct socket_info_full *sif; + struct socket_info *si; + char buf[CL_CTR_MAX_BIN_SOCK_LEN]; + int len; + + /* On devel, protos[].listeners is a list of socket_info_full (next/prev), + * each embedding the socket_info as its first member. */ + for (sif = protos[PROTO_BIN].listeners; sif; sif = sif->next) { + si = &sif->socket_info; + /* all entries here are PROTO_BIN by construction */ + if (si->proto != PROTO_BIN) + continue; + /* Reject wildcard - clusterer needs an explicit IP to set send_sock. + * Use socket=bin:IP:PORT instead of socket=bin:*:PORT. */ + if (si->address_str.len == 0 + || (si->address_str.len == 1 && si->address_str.s[0] == '*') + || (si->address_str.len == 7 + && memcmp(si->address_str.s, "0.0.0.0", 7) == 0)) { + LM_ERR("clusterer_controller: wildcard BIN socket " + "(bin:*:%u) is not allowed - use an explicit IP " + "(e.g. socket=bin:%s:%u)\n", + si->port_no, my_ip ? my_ip : "YOUR_IP", si->port_no); + return -1; + } + if (my_bin_count >= CL_CTR_MAX_BIN_SOCKETS) { + LM_WARN("clusterer_controller: more than %d BIN sockets, " + "ignoring the rest\n", CL_CTR_MAX_BIN_SOCKETS); + break; + } + len = snprintf(buf, sizeof(buf), "bin:%.*s:%u", + si->address_str.len, si->address_str.s, + si->port_no); + if (len <= 0 || len >= CL_CTR_MAX_BIN_SOCK_LEN) { + LM_WARN("clusterer_controller: BIN socket name too long, " + "skipping\n"); + continue; + } + memcpy(my_bin_sockets[my_bin_count], buf, len + 1); + LM_INFO("clusterer_controller: found BIN socket: %s\n", buf); + my_bin_count++; + } + + if (my_bin_count == 0) { + LM_ERR("clusterer_controller: no BIN sockets found - " + "is proto_bin loaded and socket=bin: configured?\n"); + return -1; + } + + return 0; +} + +static int mod_init(void) +{ + struct in_addr addr; + int i, j; + + LM_INFO("clusterer_controller: initialising\n"); + + if (sodium_init() < 0) { + LM_ERR("clusterer_controller: sodium_init() failed\n"); + return -1; + } + + /* Resolve the on_config_mismatch policy string. */ + if (on_config_mismatch_s) { + if (strcasecmp(on_config_mismatch_s, "warn") == 0) + on_config_mismatch = CL_CTR_CFGMISMATCH_WARN; + else if (strcasecmp(on_config_mismatch_s, "reject") == 0) + on_config_mismatch = CL_CTR_CFGMISMATCH_REJECT; + else if (strcasecmp(on_config_mismatch_s, "adopt") == 0) + on_config_mismatch = CL_CTR_CFGMISMATCH_ADOPT; + else { + LM_ERR("clusterer_controller: invalid on_config_mismatch '%s' " + "(expected warn|reject|adopt)\n", on_config_mismatch_s); + return -1; + } + } + + /* ---- Require at least one cluster ---------------------------------- */ + + if (cl_ctr_cluster_str_count == 0) { + LM_ERR("clusterer_controller: no 'cluster' modparam defined\n"); + return -1; + } + + /* ---- Global param validation --------------------------------------- */ + + if (query_time < 1) { + LM_WARN("clusterer_controller: 'query_time' %d below min, clamping to 1s\n", + query_time); + query_time = 1; + } else if (query_time > 60) { + LM_WARN("clusterer_controller: 'query_time' %d exceeds max, clamping to 60s\n", + query_time); + query_time = 60; + } + + /* ---- Parse and validate all cluster strings ------------------------ */ + + for (i = 0; i < cl_ctr_cluster_str_count; i++) { + cl_ctr_cluster_t *cl = &cl_ctr_clusters[i]; + if (cl_ctr_parse_cluster_str(cl_ctr_cluster_strs[i], cl) < 0) + return -1; + /* Resolve the multicast destination once, in the main process, so both + * the forked workers (via fork) and mod_destroy's GOODBYE path (main + * process) can send without rebuilding it. */ + memset(&cl->mcast_dest, 0, sizeof(cl->mcast_dest)); + cl->mcast_dest.sin_family = AF_INET; + cl->mcast_dest.sin_port = htons((uint16_t)cl->multicast_port); + cl->mcast_dest.sin_addr.s_addr = inet_addr(cl->multicast_address); + cl_ctr_cluster_count++; + pkg_free(cl_ctr_cluster_strs[i]); + cl_ctr_cluster_strs[i] = NULL; + } + + /* Validate cluster_id / (multicast,port) uniqueness. Every local controller + * socket binds INADDR_ANY:multicast_port, so a unicast to my_ip:port is + * demultiplexed by port alone - if two local clusters share a port (distinct + * groups), a unicast can land on the wrong cluster's socket. That is now + * recovered at the receiver by the packet's cleartext cluster_id (see + * cl_ctr_maybe_forward()), so shared ports stay fully unicast; we only note + * it. */ + for (i = 0; i < cl_ctr_cluster_count; i++) { + for (j = i + 1; j < cl_ctr_cluster_count; j++) { + if (cl_ctr_clusters[i].cluster_id == cl_ctr_clusters[j].cluster_id) { + LM_ERR("clusterer_controller: duplicate cluster_id %d\n", + cl_ctr_clusters[i].cluster_id); + return -1; + } + if (cl_ctr_clusters[i].multicast_port != cl_ctr_clusters[j].multicast_port) + continue; + if (strcmp(cl_ctr_clusters[i].multicast_address, + cl_ctr_clusters[j].multicast_address) == 0) { + LM_ERR("clusterer_controller: duplicate multicast %s:%d\n", + cl_ctr_clusters[i].multicast_address, + cl_ctr_clusters[i].multicast_port); + return -1; + } + /* same port, different group: unicast is demuxed by port alone and + * routed to the right cluster by cluster_id at the receiver */ + LM_INFO("clusterer_controller: clusters %d and %d share multicast " + "port %d - unicast is routed by cluster_id\n", + cl_ctr_clusters[i].cluster_id, cl_ctr_clusters[j].cluster_id, + cl_ctr_clusters[i].multicast_port); + } + } + + /* ---- Discover BIN sockets from opensips config file --------------- */ + /* Called after cl_ctr_resolve_local_identity() so my_ip is available for */ + /* wildcard substitution (bin:*:PORT -> bin:my_ip:PORT). */ + + /* ---- Resolve local identity using first cluster for Mode 3 probe --- */ + + if (cl_ctr_resolve_local_identity() < 0) + return -1; + + if (cl_ctr_discover_bin_sockets() < 0) + return -1; + + if (strlen(my_ip) > CL_CTR_MAX_IP_LEN) { + LM_ERR("clusterer_controller: resolved my_ip too long\n"); + return -1; + } + if (inet_aton(my_ip, &addr) == 0) { + LM_ERR("clusterer_controller: cannot parse resolved my_ip '%s'\n", my_ip); + return -1; + } + + /* ---- Multi-cluster: each cluster must name its BIN socket ---------- */ + + if (cl_ctr_cluster_count > 1) { + for (i = 0; i < cl_ctr_cluster_count; i++) { + if (cl_ctr_clusters[i].bin_socket[0] == '\0') { + LM_ERR("clusterer_controller: cluster %d has no bin_socket= " + "defined - required when multiple clusters are configured " + "(e.g. id=%d,multicast=...,bin_socket=bin:IP:PORT)\n", + cl_ctr_clusters[i].cluster_id, cl_ctr_clusters[i].cluster_id); + return -1; + } + } + } + + /* ---- Per-cluster: resolve BIN socket, derive key, allocate peers --- */ + + for (i = 0; i < cl_ctr_cluster_count; i++) { + cl_ctr_cluster_t *cl = &cl_ctr_clusters[i]; + + /* Resolve sentinels to the global default when the cluster string did + * not set them explicitly. Done here (unconditionally, before workers + * fork and before any MI query) so every cl->* setting always holds a + * concrete 0/1 value regardless of whether clusterer is loaded - a mix + * of global and per-cluster overrides always reports correctly. */ + if (cl->master_stickiness == -1) + cl->master_stickiness = master_stickiness ? 1 : 0; + if (cl->manage_shtags == -1) + cl->manage_shtags = manage_shtags ? 1 : 0; + + /* Resolve which BIN socket to use for this cluster. + * Priority: explicit bin_socket= in cluster string > + * sole discovered socket > + * first discovered socket (warn if multiple) */ + if (cl->bin_socket[0] != '\0') { + /* Explicit override - validate it was actually discovered */ + int found_bs = 0, bi; + for (bi = 0; bi < my_bin_count; bi++) { + if (strcmp(my_bin_sockets[bi], cl->bin_socket) == 0) { + found_bs = 1; + break; + } + } + if (!found_bs) { + char _disc[CL_CTR_MAX_BIN_SOCKETS * (CL_CTR_MAX_BIN_SOCK_LEN + 2)]; + int _o = 0, _b; + _disc[0] = '\0'; + for (_b = 0; _b < my_bin_count; _b++) + _o += snprintf(_disc + _o, sizeof(_disc) - _o, "%s%s", + _b ? ", " : "", my_bin_sockets[_b]); + LM_ERR("clusterer_controller: cluster %d bin_socket='%s' does not " + "match any configured BIN listener (discovered: %s) - peers " + "cannot connect and clusterer replication would silently " + "fail; fix bin_socket= or the socket=bin: line\n", + cl->cluster_id, cl->bin_socket, _disc); + return -1; + } + } else if (my_bin_count == 1) { + /* Only one socket - unambiguous */ + { + size_t _l = strnlen(my_bin_sockets[0], CL_CTR_MAX_BIN_SOCK_LEN - 1); + memcpy(cl->bin_socket, my_bin_sockets[0], _l); + cl->bin_socket[_l] = '\0'; + } + } else { + /* Multiple sockets, no explicit override - use first, warn */ + { + size_t _l = strnlen(my_bin_sockets[0], CL_CTR_MAX_BIN_SOCK_LEN - 1); + memcpy(cl->bin_socket, my_bin_sockets[0], _l); + cl->bin_socket[_l] = '\0'; + } + LM_WARN("clusterer_controller: cluster %d has no bin_socket= override " + "and multiple BIN sockets exist - using %s; add bin_socket= " + "to the cluster string to be explicit\n", + cl->cluster_id, cl->bin_socket); + } + LM_INFO("clusterer_controller: cluster %d: bin_socket=%s\n", + cl->cluster_id, cl->bin_socket); + + if (cl_ctr_derive_key(cl) < 0) + return -1; + + cl->peers = shm_malloc(sizeof(cl_ctr_peers_t)); + if (!cl->peers) { + LM_ERR("clusterer_controller: no shm memory for cluster %d peer table\n", + cl->cluster_id); + return -1; + } + memset(cl->peers, 0, sizeof(cl_ctr_peers_t)); + cl->peers->node_state = CL_CTR_NODE_NEW; + cl->peers->worker_proc_no = -1; /* published by cl_ctr_worker after fork */ + cl->peers->eff_manage_shtags = cl->manage_shtags; + cl->peers->eff_master_stickiness = cl->master_stickiness; + cl->peers->eff_query_time = query_time; + + cl->peers->lock = lock_init_rw(); + if (!cl->peers->lock) { + LM_ERR("clusterer_controller: lock_init_rw() failed for cluster %d\n", + cl->cluster_id); + shm_free(cl->peers); + cl->peers = NULL; + return -1; + } + + LM_INFO("clusterer_controller: cluster %d: multicast=%s:%d bin=%s\n", + cl->cluster_id, cl->multicast_address, cl->multicast_port, + cl->bin_socket); + } + + LM_INFO("clusterer_controller: my_ip=%s interface=%s query_time=%ds " + "clusters=%d bin_sockets=%d crypto=%s\n", + my_ip, my_interface_buf[0] ? my_interface_buf : "(unknown)", + query_time, cl_ctr_cluster_count, my_bin_count, CL_CTR_CRYPTO_SUITE); + + /* Set worker process count dynamically - one per cluster */ + procs[0].no = cl_ctr_cluster_count; + + /* Load clusterer controller API if clusterer.so is present and + * use_controller=1 is set. Soft dependency - controller works + * standalone even without clusterer loaded. */ + { + load_clusterer_ctrl_binds_f load_fn; + load_fn = (load_clusterer_ctrl_binds_f) + find_export("load_clusterer_ctrl_binds", 0); + if (load_fn && load_fn(&clctl) == 0) { + clctl_loaded = 1; + LM_INFO("clusterer_controller: clusterer API loaded - " + "topology will be driven dynamically\n"); + + /* The clusterer marks a cluster controller-managed with + * cluster_options use_controller=1; that pre-creates its stub, sets the + * controller_managed flag (so it never touches the DB), and arms the + * guard against hijacking a native cluster of the same id. The set of + * clusterer-managed ids and the set of 'cluster' configs here must match + * exactly - either direction of mismatch is a hard error, since neither + * half is usable without the other (a managed id with no controller + * config has no bin socket or crypto params; a controller config for an + * unmanaged id has nothing legitimate to drive). Abort naming the id. */ + + /* (a) clusterer marks a cluster managed that we have no config for */ + { + int m, k, found; + for (m = 0; m < clctl.managed_count; m++) { + found = 0; + for (k = 0; k < cl_ctr_cluster_count; k++) + if (cl_ctr_clusters[k].cluster_id == clctl.managed_ids[m]) { + found = 1; + break; + } + if (!found) { + LM_ERR("clusterer_controller: the clusterer marks cluster %d " + "controller-managed (cluster_options use_controller=1) " + "but this module has no configuration for it - add " + "modparam(\"clusterer_controller\", \"cluster\", " + "\"id=%d, ...\"), or drop use_controller for that " + "cluster.\n", + clctl.managed_ids[m], clctl.managed_ids[m]); + return -1; + } + } + } + + /* (b) we have a config for a cluster the clusterer did not mark managed */ + { + int k, m, managed; + for (k = 0; k < cl_ctr_cluster_count; k++) { + managed = 0; + for (m = 0; m < clctl.managed_count; m++) + if (clctl.managed_ids[m] == cl_ctr_clusters[k].cluster_id) { + managed = 1; + break; + } + if (!managed) { + LM_ERR("clusterer_controller: cluster %d is configured here " + "but the clusterer did not mark it controller-managed " + "- add modparam(\"clusterer\", \"cluster_options\", " + "\"cluster_id=%d, use_controller=1\"), or remove this " + "module's 'cluster' config for it.\n", + cl_ctr_clusters[k].cluster_id, cl_ctr_clusters[k].cluster_id); + return -1; + } + } + } + + /* When we manage sharing tags, start every tag as BACKUP and + * lock out MI/script changes - the controller master decides + * who becomes active. (manage_shtags sentinels were already + * resolved to concrete 0/1 in the unconditional loop above.) */ + { + int _ci; + for (_ci = 0; _ci < cl_ctr_cluster_count; _ci++) { + if (!cl_ctr_clusters[_ci].manage_shtags) + continue; + if (clctl.force_backup_shtags) + clctl.force_backup_shtags(cl_ctr_clusters[_ci].cluster_id); + if (clctl.set_shtag_managed) + clctl.set_shtag_managed(cl_ctr_clusters[_ci].cluster_id); + } + } + + } else { + LM_DBG("clusterer_controller: clusterer not loaded or " + "use_controller not set - running standalone\n"); + } + } + + return 0; +} + +static int cl_ctr_child_init(int rank) +{ + /* Re-seed the CSPRNG after fork - each worker must have independent state. */ + randombytes_stir(); + + /* Sync current_id from shared memory in every child process. + * The global current_id diverges after fork - each process needs + * to re-read the correct value from cluster->current_node. */ + if (clctl_loaded && clctl.sync_current_id) + clctl.sync_current_id(); + return 0; +} + +static void mod_destroy(void) +{ + int i, sock; + unsigned char ttl = 32; + cl_ctr_cluster_t *cl; + + if (!my_ip) + goto cleanup; + + /* Send GOODBYE on each cluster's multicast group so peers re-elect */ + for (i = 0; i < cl_ctr_cluster_count; i++) { + cl = &cl_ctr_clusters[i]; + if (!cl->peers) + continue; + if (cl->peers->count <= 1) { + LM_INFO("clusterer_controller: [cluster %d] sole node, " + "skipping GOODBYE\n", cl->cluster_id); + continue; + } + sock = socket(AF_INET, SOCK_DGRAM, 0); + if (sock < 0) { + LM_ERR("clusterer_controller: [cluster %d] goodbye socket(): %s\n", + cl->cluster_id, strerror(errno)); + continue; + } + setsockopt(sock, IPPROTO_IP, IP_MULTICAST_TTL, &ttl, sizeof(ttl)); + /* Derive session_key locally from master_salt in shm so we can encrypt + * GOODBYE - the worker's cl->session_key is in a different process. */ + { + size_t plen = strlen(cl->password); + cl_ctr_hkdf_sha256((unsigned char *)cl->password, plen, + cl->peers->master_salt, CL_CTR_MASTER_SALT_SZ, + "cl_ctr_session", cl->session_key); + } + cl_ctr_send_pkt_with_ip(sock, CL_CTR_PKT_GOODBYE, cl, NULL, 0); + close(sock); + LM_INFO("clusterer_controller: [cluster %d] GOODBYE sent\n", + cl->cluster_id); + } + +cleanup: + for (i = 0; i < cl_ctr_cluster_count; i++) { + cl = &cl_ctr_clusters[i]; + if (!cl->peers) + continue; + if (cl->peers->lock) { + lock_destroy_rw(cl->peers->lock); + cl->peers->lock = NULL; + } + shm_free(cl->peers); + cl->peers = NULL; + } + LM_INFO("clusterer_controller: shut down\n"); +} diff --git a/modules/clusterer_controller/doc/clusterer_controller.xml b/modules/clusterer_controller/doc/clusterer_controller.xml new file mode 100644 index 00000000000..af2ee984ad0 --- /dev/null +++ b/modules/clusterer_controller/doc/clusterer_controller.xml @@ -0,0 +1,22 @@ + + + + + + +%docentities; +]> + + + CLUSTERER_CONTROLLER Module + &osipsname; + + + &admin; + &tests; + &contrib; + &docCopyrights; + ©right; 2026 VoIPLine Telecom + \ No newline at end of file diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml new file mode 100644 index 00000000000..51e0311a5dd --- /dev/null +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -0,0 +1,1931 @@ + + + + + &adminguide; + +
+ Overview + + The clusterer_controller module provides + automatic peer discovery and topology management for the + clusterer module via authenticated, encrypted + UDP multicast. It eliminates the need for static node configuration + or a database — nodes discover each other automatically at startup + and the cluster topology is maintained dynamically at runtime. + + + When a cluster is registered as controller-managed via + modparam("clusterer", "cluster_options", "cluster_id=N, + use_controller=1"), the + clusterer_controller module takes over all + topology management: it allocates unique node IDs, discovers peer + BIN socket addresses, and calls the clusterer internal API to add + or remove nodes as they join or leave the cluster. + + + By default (manage_shtags=1), the module also + provides fully automatic sharing tag failover. The controller + master node is the single decision point for which node holds the + active tag — no event routes, MI commands, or + seed_fallback_interval configuration is needed. + Sharing tags are forced to backup at startup regardless of the + =active config value, and the active tag is + claimed by the controller master automatically when the cluster + forms or when the active node departs. An operator can override this + automatic allocation and pin the active tag to a chosen node with the + cl_ctr_shtag_force MI command, reverting to automatic + allocation with cl_ctr_shtag_auto. + + + The minimal configuration per node is a single multicast group + address. No IP addresses, node IDs, BIN URLs, or sharing tag + management scripts need to be hardcoded or maintained. Any number + of nodes can join or leave without any configuration change on + the remaining nodes. + +
+ +
+ Discovery Protocol + + All traffic uses UDP multicast to the configured + multicast address and port. Clusters are kept + apart in two independent ways: by the multicast endpoint (two + clusters may use different IP addresses, or the same address with + different UDP ports), and by the id + (cluster_id) carried in the cleartext of every + packet. A node silently ignores any packet whose cluster_id differs + from its own, so several clusters can safely share one multicast + group and port. Two clusters merge only if they share + all of the multicast address, the UDP port, the + cluster_id and the password — i.e. they are configured identically, + which is the operator's responsibility to avoid. + + + Every packet is encrypted and authenticated with the XChaCha20-Poly1305 + AEAD (see the Security Architecture section). Two + distinct encryption keys are used depending on the communication phase: + + + Bootstrap key — derived from + the configured password with the + Argon2id memory-hard KDF (per-cluster salt), + so a password captured from a bootstrap packet cannot be + brute-forced cheaply offline. Derived once at startup. It is both + the outer AEAD key for the admission handshake (JOIN_REQ, + KEY_GRANT, JOIN_REJECT) and the split-brain MASTER_BEACON — traffic + that must be readable before a session key exists, or by masters + holding different session keys — and the pre-shared key for the + Noise join handshake. + + + Session key — derived via + HKDF-SHA256 from the password and a 32-byte master salt + generated once when the cluster first bootstraps. All normal + cluster traffic uses this key. It is preserved across master + changes (a new master reuses the key every member already + holds), so failover needs no re-keying. + + + + + Each packet wire format begins with a 2-byte magic value + identifying the key type and a 2-byte cluster_id, both in cleartext + (the magic and cluster_id must be readable before decryption to + select the key and to filter foreign clusters), followed by a random + 24-byte AEAD nonce and the ciphertext. The magic and cluster_id are additionally bound into + the authentication tag as AAD, so they cannot be altered undetected. + Only the payload is encrypted; the authenticated plaintext begins with + a 1-byte packet type and a 4-byte monotonic sequence number used for + replay protection. A 16-byte authentication tag follows the ciphertext. + Packets for a different cluster_id are dropped before decryption; + packets encrypted with a different password fail authentication and + are silently discarded. + + + The following packet types are defined: + + + ALIVE — periodic heartbeat + sent by every active node every query_time + seconds. Carries the sender IP, its X25519 public key (so peers + can prepare for key agreement) and a small descriptor of the + sender's consistency-critical settings + (manage_shtags, + master_stickiness, + query_time) used for configuration-drift + detection (see Security Architecture). Encrypted with the session + key. + + + JOIN_REQ — sent at startup + by a new node, carrying its IP, BIN socket list, and Noise + message 1 (the initiator's fresh ephemeral public key). + Encrypted with the bootstrap key so it can be sent before a + session key exists. + + + MEMBER_LIST — sent by the + master in response to a JOIN_REQ, carrying the member count, the + operator-forced sharing-tag holder node_id (0 = automatic), and + the full peer IP list so the joining node can participate in + elections. Only accepted from the current master (except during + initial join when no master is yet known). + + + NODE_ASSIGN — sent by the + master to multicast, allocating a node_id and BIN socket + record for a joining node. All cluster members receive and + apply it. + + + GOODBYE — sent on graceful + shutdown so peers can remove the node immediately without + waiting for timeout. Uses the sender's monotonic sequence + counter to prevent forgery. + + + MASTER_ALIVE — keepalive + sent by the master every 1 second (independent of + query_time). Used by all peers to detect + master failure quickly (3-second timeout). Encrypted with + the session key. + + + KEY_GRANT — the master's + response to a JOIN_REQ, addressed to the joining node. Carries + Noise message 2 (the master's ephemeral plus the AEAD-encrypted + master salt), completing the Noise_NNpsk0 handshake. Encrypted + with the bootstrap key. + + + KEY_HANDOFF — sent by the + outgoing master on graceful shutdown to the next-highest-IP peer, + delivering the master salt (as an anonymous crypto_box sealed to + that peer's long-lived X25519 key, learned from its ALIVE) so it + can become the new master without a full re-join cycle. Encrypted + with the session key. + + + JOIN_REJECT — sent by the + master to a joining node whose JOIN_REQ repeatedly fails + authentication (wrong password). After + CL_CTR_JOIN_FAIL_LIMIT (3) consecutive + bootstrap-key decryption failures from the same source IP the + master sends a JOIN_REJECT to that IP. Encrypted with the + bootstrap key so it cannot be forged by a node that does not + know the cluster password. The joining node logs a critical + error and shuts down OpenSIPS on receipt. + + + MASTER_BEACON — a master-only + announcement multicast every few MASTER_ALIVE ticks, carrying this + partition's member count. Unlike MASTER_ALIVE it is encrypted with + the bootstrap key, so it is readable even by a + master that holds a different session key. This is how a split + brain between two independently bootstrapped partitions is detected + and merged (see Master Election). + + + +
+ +
+ Master Election + + Each cluster has three roles: master + (the active coordinator), backup (the + standby promoted when the master fails, always the highest-IP + non-master) and member. The election + uses a quantized time window so that all nodes evaluate the same + eligible peer set and reach the same result deterministically. No NTP + synchronisation between nodes is required for correct election results. + + + The master_stickiness parameter (default 1) controls + whether a live master is kept when a higher-IP node joins. With + stickiness enabled, the master stays put and the higher-IP joiner + becomes the backup, minimising handovers; with stickiness disabled the + highest-IP node always becomes master. In either mode two live masters + are reconciled deterministically (see Split-brain + handling below). See the master_stickiness + parameter for details. + + + Only the master handles JOIN_REQ packets, allocates node_ids, and + sends NODE_ASSIGN and MEMBER_LIST packets. Non-master nodes are + passive during join events. A joining node receives the current + session key from the master (via KEY_GRANT) and joins as a member or + backup; it never seizes mastership during the join handshake. + + + Preserved session key: the session key + is generated once, when the first node bootstraps the cluster, and is + then preserved across every master change. A new master does not + re-key; because every member already holds the key (obtained when it + joined), master transitions require no re-keying and no re-JOIN cycle. + + + Fast master failure detection: + the master sends MASTER_ALIVE packets every 1 second. All non-master + peers maintain a 3-second watchdog timer that fires if no MASTER_ALIVE + is received. On expiry the silent master is aged out of the election + window and each peer immediately re-elects, promoting the backup + (highest-IP survivor) — which already holds the session key, so it + starts serving within one keepalive interval. + + + Graceful master handoff: when the + current master shuts down cleanly, it sends a KEY_HANDOFF packet + directly to the next-highest-IP peer before sending GOODBYE to + multicast. This confirms the master salt to the incoming master so it + can assume control immediately. + + + Split-brain handling. A split brain + (more than one node believing it is master) is prevented and, if it + still occurs, healed by three cooperating mechanisms: + + + Prevention at join time. When + several nodes start simultaneously they all exchange + (bootstrap-decryptable) JOIN_REQs and thus learn about each other. + At the join deadline, a node that has seen a higher-IP node also + still joining defers its own self-promotion (for a few bounded + rounds) and joins that node instead, so only the highest-IP starter + becomes master and no independent-key lone masters are created. + + + Same-key yield. Two masters that + share a session key (for example after a network partition heals) + can read each other's MASTER_ALIVE; the lower-IP master immediately + yields to the higher-IP one. + + + Divergent-key merge. Two masters + that were bootstrapped independently hold different session keys and + so cannot read each other's MASTER_ALIVE. Each therefore emits a + MASTER_BEACON encrypted with the shared bootstrap key. On hearing a + beacon from a superior partition — larger member count, ties broken + by higher IP — a node abandons its partition, re-joins the superior + master and adopts its session key, converging the whole cluster onto + a single master and key. + + + +
+ +
+ Security Architecture + + The module uses a two-phase key agreement to provide forward + secrecy and replay protection for all cluster traffic. + + + Payload encryption and header binding: + every packet's payload is sealed with an AEAD. The 2-byte magic (a key + selector that must be readable before decryption) and the 2-byte + cluster_id that precede the nonce are cleartext framing, but they are + bound into the AEAD tag as additional authenticated data (AAD): a + captured packet cannot be re-stamped with a different cluster_id and + still authenticate, which matters when two clusters share one multicast + group and password. A node also drops any packet whose cluster_id does + not match its own before attempting decryption, so + foreign-cluster traffic on the group never counts as an authentication + failure. + + + Crypto (all libsodium; a hard requirement): + the payload AEAD is XChaCha20-Poly1305 (24-byte + nonce, whose 192-bit nonce space removes any random-nonce collision + concern); the bootstrap-key KDF is Argon2id; and + X25519 / HKDF-SHA256 / RNG also come from libsodium. The active suite is + reported in the startup log (crypto=...). + + + Phase 1 — join handshake (JOIN_REQ / + KEY_GRANT): the join is a + Noise_NNpsk0_25519_ChaChaPoly_SHA256 handshake with + the pre-shared key set to the Argon2id bootstrap key: + -> psk, e JOIN_REQ carries Noise message 1 (a fresh ephemeral) +<- e, ee KEY_GRANT carries Noise message 2; its AEAD payload = master_salt + Authentication comes from the shared PSK, forward secrecy from the + ephemeral-ephemeral DH, and the Noise handshake hash binds the whole + transcript — so a stale KEY_GRANT for a superseded JOIN_REQ simply fails + to decrypt. Both handshake messages also travel inside the bootstrap-key + AEAD envelope, so a wrong-password node is rejected at the envelope + before the handshake is even reached. This replaces the earlier + hand-rolled ECDH-and-XOR salt wrap. + + + Phase 2 — session (all other + packets): once the master salt is known, all nodes + derive the session key as: + session_key = HKDF-SHA256(IKM=password, salt=master_salt, info="cc-session-key") + The session key is generated once, when the first node bootstraps the + cluster, and preserved across every master change: a new master reuses + the key that every member already holds, so master transitions require + no re-keying. All normal cluster traffic (ALIVE, MEMBER_LIST, + NODE_ASSIGN, GOODBYE, MASTER_ALIVE, KEY_HANDOFF) is encrypted with this + key. + + + Replay protection: each sender + maintains a monotonically increasing 32-bit sequence number + embedded in the authenticated plaintext of every session-key + packet. Each receiver tracks the last accepted sequence number + per source IP and rejects any packet whose sequence is not + strictly greater than the last accepted value. Sequence counters + are reset to zero whenever the session key is (re)derived — at cluster + bootstrap and when a joiner adopts the key via KEY_GRANT/KEY_HANDOFF — + and on peer restart detection (JOIN_REQ received from a known IP resets + that peer's counter; MEMBER_LIST upsert resets all listed peers). + This protection does not depend on clock synchronisation. + + + Rate limiting: a per-source rate + limiter (256 slots, 20 packets/second limit) is applied before any + decryption attempt. This prevents CPU exhaustion from packet floods + directed at the multicast group. + + + Join authentication and rejection: + the master tracks consecutive bootstrap-key decryption failures per + source IP in a small worker-local table + (CL_CTR_JOIN_FAIL_TABLE_SZ = 8 slots). When any + source IP accumulates CL_CTR_JOIN_FAIL_LIMIT (3) + consecutive failures — indicating a node attempting to join with the + wrong password — the master sends an encrypted JOIN_REJECT packet + and stops responding to further JOIN_REQs from that IP. + + + On the joining side, a received JOIN_REJECT is only acted on while + the node is still in the initial join phase + (CL_CTR_NODE_NEW state) and is addressed to this + node; an already-active cluster member ignores any JOIN_REJECT + unconditionally, so a node with the correct password can never be + evicted by a peer. + + + A node joining with the wrong password cannot + decrypt the JOIN_REJECT (it is encrypted with the master's bootstrap + key), so it relies on a self-contained signal instead: while joining + it counts packets received from other peers that it cannot decrypt. + If, at the join deadline, the node is still unjoined and has seen + CL_CTR_JOIN_FAIL_LIMIT or more such undecryptable + packets, it concludes that a cluster it cannot authenticate to exists + on the group and shuts down OpenSIPS with a critical log message — + rather than promoting itself into a lone, split-brain master (which, + with managed sharing tags, would create a duplicate active tag). This + counter is reset the moment a KEY_GRANT is successfully processed, so + a legitimate joiner that briefly saw an undecryptable packet before + receiving its key is never affected. + + + Rogue traffic isolation: a node + requests a re-key in response to an undecryptable session-key packet + only when that packet came from its current master (a legitimate key + rotation). Undecryptable session packets from any other source — for + example a wrong-password or malicious node broadcasting on the + multicast group — are ignored, so such traffic cannot drive the + cluster into a re-JOIN churn. + + + Peer table exhaustion defence: the + peer table is bounded at CL_CTR_MAX_PEERS (256) + entries. When the table is full, the master rejects JOIN_REQ + packets from unknown IPs with a JOIN_REJECT response. Known peers + that are reconnecting after a restart continue to be admitted + regardless of the table count, since they already own a slot. + This prevents an attacker with the cluster password from exhausting + the peer table by flooding JOIN_REQs from spoofed source addresses. + + + Configuration-consistency enforcement: + all nodes of a cluster must use identical consistency-critical settings + (manage_shtags, master_stickiness + and query_time); a per-node mismatch would otherwise + cause silent, inconsistent failover and sharing-tag behaviour (for + example, a master with manage_shtags=0 would leave no + node holding the active tag). Each node advertises these effective + settings in its ALIVE heartbeat and in its JOIN_REQ, so mismatches are + detected. What happens then is controlled by the + on_config_mismatch modparam: + + + reject (default) - when a node tries to + join an established cluster (a master is alive) with different + settings, the master logs the attempt and returns a JOIN_REJECT; + the joining node logs the offending settings and shuts down, so a + misconfigured node never joins. + + + warn - the node is allowed to join, but any + peer that observes a different value logs a single loud + CONFIG MISMATCH warning (repeated only if the + peer's advertised configuration changes, cleared once it matches). + + + adopt - the joining node adopts the running + cluster's (master's) settings at runtime and continues; the + adopted values are what cl_ctr_list_config + reports. + + + This turns an easy-to-miss misconfiguration into an obvious log line, a + refused join, or a self-correction rather than a hard-to-diagnose HA + failure. + + + Node identity and node_id allocation: + node_id values are allocated exclusively by the current + master, serialised under the peer-table lock, and a joining node never + picks its own id. The master hands out the lowest unused id by scanning + the live peer table, so a node that has failed but is not yet timed out + still occupies its slot and its id is never handed to a different joiner — + new nodes always receive a distinct id even during the failure-detection + window. A node that restarts and rejoins from the same address reuses its + previous id (and has its replay counter reset), so ids stay stable across + restarts. Because peers are keyed by source IP address, every node in a + cluster must present a stable, unique source IP: two distinct nodes that + appear behind the same address (for example through NAT) would share a + single peer slot and node_id. Deploy the cluster on a + network where each member has its own routable address on the + BIN/multicast interface. + + + Trust model — shared secret, not per-node + identity: the cluster is a single shared-secret trust domain. + Authentication proves only that a peer holds the cluster password; it does + not bind a cryptographic identity to an individual node, and there is no + per-node authorisation or revocation. Consequently any party in possession + of the password is a fully trusted member and can legitimately win the + highest-IP master election and assume the master role — there is no + distinction between "may be a member" and "may be master". An attacker + without the password cannot affect the election at + all: forged or replayed MASTER_ALIVE and beacon + packets fail AEAD authentication (or the strict per-source sequence check) + and are dropped before any election logic runs. The residual exposure is + therefore a malicious or compromised insider that + already holds the shared key. Protect the password accordingly, and rotate + it if a node is decommissioned or suspected compromised. Removing this + limitation — per-node keypairs with enrolment and revocation, so a single + node can be distrusted without re-keying the whole cluster — is planned + future work. + +
+ +
+ Dependencies +
+ &osips; Modules + + The following modules are required by this module: + + + + proto_bin — required so that + BIN listeners are registered and available for + discovery when clusterer_controller + initialises and scans the proto_bin listener list. + + + + + clusterer — required, and it + must register every controller-managed + cluster with + modparam("clusterer", "cluster_options", + "cluster_id=N, use_controller=1"). That per-cluster + parameter is what pre-creates the controller-managed cluster + stubs, marks them so they never touch the database, and arms + the guard that stops the controller from driving a native + cluster of the same id. The controller-managed ids declared here + and the cluster entries configured in + clusterer_controller must match + exactly: if either side names a cluster the + other does not, the controller refuses to start with an error + naming the offending id — a managed id with no controller config + has no BIN socket or crypto parameters, and a controller config + for an unmanaged id has nothing to drive. + + + + + + Both dependencies are declared in the module's + dep_export_t. &osips; will refuse to + start if either dependency is not satisfied. This does + not imply that the modules must appear + in a particular order in the configuration file — &osips; + resolves the dependency at runtime and will initialize the + required modules first regardless of + loadmodule order. + + + This is also checked from the other side: if a + cluster_options entry sets + use_controller=1 but the + clusterer_controller module is not loaded at + all, clusterer logs an error at startup, since the controller-managed + cluster stubs would never obtain a node identity or form. clusterer + itself keeps running, so any native or DB-backed clusters are + unaffected. + + + Hybrid deployments are unaffected. + use_controller is a per-cluster option carried in + cluster_options, defaulting to 0. + In a hybrid instance (native and controller-managed clusters side by + side), the controller-managed clusters each get a + cluster_options line with + use_controller=1, while the native ones are defined + the usual way (DB rows or static + my_node_info/neighbor_node_info) + and need no cluster_options line. The exact-match + check above compares only the controller-managed ids against the + controller's cluster entries, so it never fires on + a native cluster. + + + All other modules that use the clusterer interface + (tm, dialog, + dispatcher, usrloc + etc.) may be loaded in any order relative to + clusterer_controller. The clusterer + module automatically creates a cluster stub when a + cluster_options entry sets + use_controller=1 and a module attempts to register + a capability for that cluster. + + + Because a controller-managed node receives its + node_id at runtime (after it joins) rather than + from static configuration at startup, any module that stamps this + node's id into on-the-wire data must read it live per message + instead of caching it once at initialisation — a value read at + startup would be the not-yet-assigned placeholder, and it may also + change on a re-election. In particular tm's + anycast support (tm_replication_cluster / + t_anycast_replicate()) stamps this node's id + into the cid Via parameter so that a reply landing + on a different anycast member can be relayed to the node holding the + transaction; it renders that parameter from the current id on each + request, so anycast reply routing works unchanged under a + controller-managed cluster. + +
+ +
+ External Libraries or Applications + + libsodium is required: + + + + libsodium (required) — provides the + payload AEAD (XChaCha20-Poly1305), the bootstrap-key KDF + (Argon2id), the Noise_NNpsk0 join handshake, and X25519 / + HKDF-SHA256 / RNG. The build fails with a clear error if + libsodium development files are not found. It is linked + dynamically, so each target host also needs the libsodium + runtime package (for example libsodium23). + No other crypto library is used or linked. + + + + The active suite is printed in the startup log + (crypto=...). + +
+
+ +
+ Building the Module + + clusterer_controller is + excluded from the default build (it is + listed in exclude_modules in + Makefile.conf.template), like the other modules that + depend on external libraries. A stock &osips; build therefore does not + include it. To build it, add it to include_modules in + your Makefile.conf: + + +include_modules= clusterer_controller + + + then rebuild (make all / + make modules). libsodium development files must be + present on the build host or the build fails — see + External Libraries or Applications. + + + The clusterer module is unmodified when + clusterer_controller is not built. The controller integration + on the clusterer side (the + clusterer_ctrl API, the cluster_options + modparam and all controller hooks) is compiled only when + clusterer_controller is part of the build: the top-level Makefile detects + this and passes -DCLUSTERER_CTRL_SUPPORT to the + clusterer module. A build without clusterer_controller produces the + stock clusterer module, unchanged in behaviour and exported interface - + in particular the cluster_options parameter does not + exist and is rejected as unknown. Enabling clusterer_controller + automatically rebuilds clusterer with the support compiled in; the two + are always a matched pair. + +
+ +
+ Exported Parameters + +
+ <varname>cluster</varname> (string) + + Define a cluster to participate in. The value is a + comma-separated key=value string with the following fields: + + + id (required) — + positive integer cluster identifier, must match the + cluster_id used by clusterer + consumers (dialog, usrloc, dispatcher, etc.). + + + multicast (required) + — IPv4 multicast address and UDP port in the form + A.B.C.D:PORT. The address must + be in the 224.0.0.0/4 range. + + + password (optional) + — XChaCha20-Poly1305 encryption key material. All nodes in the + same cluster must use the same password. Falls back + to the global password modparam + if not set. + + + bin_socket + (optional) — BIN socket to advertise for this + cluster, in the form + bin:IP:PORT. Required when + multiple clusters are defined (so the controller knows + which listener to advertise), but it need + not be distinct — several clusters + may share the same BIN socket. When only one cluster + is defined and only one BIN socket exists, the + socket is auto-detected from the + proto_bin listeners. + + + manage_shtags + (optional) — per-cluster override for the global + manage_shtags modparam. Set to + 1 to enable automatic sharing + tag failover for this cluster, or + 0 to disable it. When omitted, + the global manage_shtags value + applies, regardless of the order in which + cluster and + manage_shtags modparams appear + in the config file. + + + + + This parameter may be set multiple times to participate in + multiple clusters simultaneously. Each cluster runs its own + independent worker process. + + + No default value. At least one cluster must be + defined. + + + Set <varname>cluster</varname> parameter — single cluster + +... +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +... + + + + Set <varname>cluster</varname> parameter — multiple clusters + on separate networks (one BIN socket per network) + +... +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.1.10:5566") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,bin_socket=bin:10.0.2.10:5566") +... + + + + Set <varname>cluster</varname> parameter — multiple clusters + sharing a single BIN socket (recommended default) + +... +# one proto_bin listener serves both clusters; the cluster_id in each +# BIN packet keeps their replication traffic separate +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,bin_socket=bin:10.0.0.10:5566") +... + + + + One BIN socket can serve any number of + clusters. Every clusterer BIN packet carries its + cluster_id, so a single proto_bin + listener demultiplexes traffic for all clusters. Unless you specifically + want different clusters on different interfaces or networks, set the + same bin_socket value on every + cluster entry (it is mandatory once more than one + cluster is defined, but need not be distinct). Using multiple BIN sockets + is for network/interface segregation, not throughput: + the replication data plane runs over BIN (TCP) and scales with the shared + TCP worker pool (tcp_children) and IPC dispatch, + independent of the number of BIN sockets. What does scale with the number + of clusters is the controller's own worker set — it spawns one lightweight + control worker (UDP multicast: discovery, election, keepalives) per + cluster — so it is the cluster count, not the + BIN-socket count, that adds processes. + + + Each cluster needs a distinct multicast + endpoint, but clusters may share one BIN socket. The two + sockets play opposite roles. The controller's multicast + endpoint is its control plane: every cluster defined in one instance must + use a different multicast address:port — two + cluster entries with the same + multicast value are rejected at startup + (duplicate multicast). Use a different port + (e.g. :3333 and :3334) or a + different group per cluster. The bin_socket is the + replication data plane and is the opposite: it may be freely shared + across clusters, since every BIN packet carries its + cluster_id. (The cluster_id filter + on the multicast wire header is for a different purpose: letting + separate deployments coexist on a shared multicast + group, not two clusters inside one instance.) + +
+ +
+ <varname>my_ip</varname> (string) + + Explicitly set the local IPv4 address used by the controller + for its own node identity and master election. This is the IP + that the controller advertises to peers in JOIN_REQ and + NODE_ASSIGN packets and uses for the highest-IP master + election algorithm. + + + Note: this parameter controls the controller's identity only. + The BIN socket address advertised to clusterer peers is + discovered separately from the proto_bin + listener list and is independent of this setting. Do not + confuse my_ip with the BIN socket IP + defined by the socket=bin:IP:PORT core + parameter. + + + When set, the module walks the interface list to find which + local interface owns this address and uses that interface for + multicast traffic. Startup fails if no local interface owns + the given address. + + + The module supports three identity resolution modes depending + on which modparams are provided: + + + Mode 1 — my_ip set: + The given IP is used directly. The owning interface is + resolved automatically from the system interface list. + Use this mode on multi-homed hosts where you want to + pin the controller identity to a specific IP. + + + Mode 2 — interface set, my_ip not set: + The first IPv4 address on the named interface is used as + the controller identity IP. A warning is logged if the + interface has multiple IPv4 addresses. + + + Mode 3 — neither set (default): + A throw-away UDP socket is connected to the multicast + group and getsockname() is called + to determine which source IP the kernel would select. + The interface name is resolved from the returned IP. + Suitable for single-homed hosts. + + + Default: auto-detected (Mode 3). + + + Set <varname>my_ip</varname> parameter + +... +modparam("clusterer_controller", "my_ip", "10.22.23.191") +... + + +
+ +
+ <varname>interface</varname> (string) + + Explicitly set the network interface name to use for + multicast traffic (e.g. eth0, + enp6s18). The module takes the first + IPv4 address assigned to this interface as the controller's + identity IP. This corresponds to Mode 2 described in the + my_ip parameter documentation above. + + + Like my_ip, this parameter affects the + controller's own identity only and has no effect on the + BIN socket addresses advertised to clusterer peers. + + + If the interface has more than one IPv4 address, a warning + is logged and the first address (in the order returned by + the kernel) is used. Set my_ip explicitly + to avoid ambiguity on multi-address interfaces. + + + Ignored if my_ip is also set — + my_ip takes precedence. + + + Default: auto-detected (Mode 3 — see + my_ip). + + + Set <varname>interface</varname> parameter + +... +modparam("clusterer_controller", "interface", "eth0") +... + + +
+ +
+ <varname>query_time</varname> (integer) + + How often (in seconds) each active node sends an ALIVE + heartbeat to the multicast group. This value also controls + the election window + (3 × query_time) and the peer purge + window (6 × query_time). + + + Smaller values mean faster failure detection but higher + multicast traffic. Valid range: 1–60. + + + Default value is 5. + + + Set <varname>query_time</varname> parameter + +... +modparam("clusterer_controller", "query_time", 5) +... + + +
+ +
+ <varname>password</varname> (string) + + Global default encryption password for all clusters. All + nodes in a cluster must use the same password. The password + serves two purposes: + + + Bootstrap key — + the password is stretched with Argon2id + (memory-hard, per-cluster salt); it is the pre-shared key for + the Noise join handshake (JOIN_REQ / KEY_GRANT) and the AEAD + key for bootstrap traffic before a session key exists. + + + Session key material — + the password is fed into HKDF-SHA256 together with the + master salt to derive the session key used for all + normal cluster traffic. + + + Can be overridden per cluster using the + password= key in the + cluster parameter. + + + Default value is 3eCrEt*5629. + Change this in production. Use a long, high-entropy + secret rather than a memorable phrase — Argon2id raises the cost of + an offline guess, but only a strong secret removes the risk. A + generated key is ideal, e.g. openssl rand -base64 32. + The module logs a startup warning if the configured password is the + default or has an estimated entropy below 80 bits. + + + Set <varname>password</varname> parameter + +... +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") +... + + +
+ +
+ <varname>manage_shtags</varname> (integer) + + When set to 1 (the default), + the controller master node automatically manages sharing tag + failover for all clusters. The controller becomes the single + decision point for which node holds the active tag, eliminating + races between nodes and requiring no script-level event routes + or MI commands to handle failover. While active, the + clusterer_set_tag_active MI command and + the $shtag() script variable setter are + blocked for controller-managed clusters, returning an error + to the caller. + + + Behaviour when manage_shtags=1: + + + Startup: all local sharing tags + are forced to backup state during module + initialisation, regardless of the =active + value in the clusterer sharing_tag modparam. + The deferred BIN broadcast flag is also cleared so no + SHTAG_ACTIVE packet is ever sent at + startup. This ensures that no node can steal the active tag from + an existing cluster member simply by restarting. + + + Bootstrap: when the first node + starts alone and no existing master responds within + query_time seconds (join deadline), it elects + itself master and activates all local backup tags exactly once. + Nodes that join an existing cluster are never eligible for this + bootstrap path and never self-activate. + + + Failover: when any node departs + (graceful shutdown via GOODBYE packet, or timeout-based removal), + the controller master activates its own backup tags for that + cluster. This covers all departure scenarios: last node standing, + master still present, and post re-election. + + + Rejoin: a node rejoining an + existing cluster always starts in backup state and never reclaims + the active tag from the current holder, even if + =active appears in its config. + + + When set to 0, the controller + does not touch sharing tag state at all. The + =active config value, + seed_fallback_interval, and external + MI/event-route scripts behave exactly as in stock clusterer + without the controller. Use this when you have existing tag + management scripts and want to opt out of automatic failover. + + + Default value is 1. + + + Global vs per-cluster scope: + This modparam sets a global default that applies to every cluster + defined via the cluster modparam. Individual + clusters can override it by including + manage_shtags=0 or + manage_shtags=1 directly in the cluster + string. The global default is resolved at startup after all + modparams are processed, so the order of + manage_shtags and cluster + lines in the config file does not matter. + + + Global <varname>manage_shtags</varname> — applies to all clusters + +... +# clusters 1 and 2 registered as controller-managed on the clusterer side +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +# Enable automatic failover for every cluster (this is also the default) +modparam("clusterer_controller", "manage_shtags", 1) +modparam("clusterer_controller", "cluster", "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "cluster", "id=2,multicast=239.0.90.2:3333") +... + + + + Per-cluster override — opt one cluster out of automatic failover + +... +# clusters 1 and 2 registered as controller-managed on the clusterer side +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +# manage_shtags=1 globally, but cluster 2 uses its own MI/event-route scripts +modparam("clusterer_controller", "manage_shtags", 1) +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,manage_shtags=0") +... + + + + Global opt-out with one cluster opting in + +... +# clusters 1 and 2 registered as controller-managed on the clusterer side +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +# Disable automatic failover globally; enable it only for cluster 1 +modparam("clusterer_controller", "manage_shtags", 0) +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,manage_shtags=1") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333") +... + + + + Typical full configuration with manage_shtags=1 (default) + +# All nodes use identical config — only the BIN socket IP differs per node. +# The =active tag value in sharing_tag is ignored by the controller; +# it is kept in the config only for compatibility with manage_shtags=0. + +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "sharing_tag", "vip1/1=active") +modparam("clusterer", "ping_interval", 4) +modparam("clusterer", "ping_timeout", 1500) + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") +# manage_shtags defaults to 1 — no need to set it explicitly + + +
+ +
+ <varname>master_stickiness</varname> (integer) + + Controls whether a live master keeps its role when a higher-IP + node joins. Default 1 (sticky). + + + The module recognises three roles per cluster: + master (the active coordinator), + backup (the standby that takes + over when the master fails), and + member (all other nodes). The + backup is always the highest-IP node that is not the master. + + + + master_stickiness=1 (default): + the master is sticky — a live master keeps + the role and is not displaced when a higher-IP node joins. The + newly joined node becomes the backup (replacing the previous + backup if it has a higher IP); the master only changes when the + current master actually fails, at which point the backup is + promoted. This minimises the number of master handovers. + + + master_stickiness=0: pure + highest-IP election — a higher-IP node takes over as master as + soon as it appears. This produces more handovers but always + keeps the highest-IP node as master. + + + + In both modes a split-brain (two nodes each believing they are + master, e.g. after a network partition heals) is resolved + deterministically: the lower-IP master yields to the higher-IP one. + + + Global vs per-cluster scope: + like manage_shtags, this sets a global default + that individual clusters can override with + master_stickiness=0 or + master_stickiness=1 in the + cluster string. Resolution happens at startup + regardless of modparam order. + + + Set <varname>master_stickiness</varname> parameter + +... +# both clusters registered as controller-managed on the clusterer side +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +# Global default (sticky) — omit entirely for the same effect +modparam("clusterer_controller", "master_stickiness", 1) + +# Per-cluster override: cluster 2 always promotes the highest-IP node +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.2:3333,master_stickiness=0") +... + + +
+ +
+ <varname>on_config_mismatch</varname> (string) + + Policy applied when a node's consistency-critical settings + (manage_shtags, + master_stickiness, query_time) + differ from those of the running cluster. All nodes of a cluster are + expected to use identical values; this parameter decides what happens + when they do not. One of: + + reject (default) - the + master refuses the join with a JOIN_REJECT and the joining node + shuts down after logging which settings differ. + warn - the node is allowed + to join; a single CONFIG MISMATCH warning is + logged per mismatching peer. + adopt - the joining node + adopts the running cluster's settings at runtime and + continues. + + Global only (not per-cluster). Default: reject. + + + Set <varname>on_config_mismatch</varname> parameter + +... +modparam("clusterer_controller", "on_config_mismatch", "reject") +... + + +
+ +
+ +
+ Exported MI Functions + +
+ <function moreinfo="none">cl_ctr_list_members</function> + + List all current cluster members with their node_id, status and + BIN socket addresses. Status is one of master, + backup (the standby that will be promoted if + the master fails) or member. Only peers within + the current election window are shown. + + Parameters: none + + <function>cl_ctr_list_members</function> usage + +opensips-cli -x mi cl_ctr_list_members +[ + { + "cluster_id": 1, + "members": [ + { + "ip": "10.22.23.191", + "node_id": 1, + "status": "master", + "bin_sockets": [ "bin:10.22.23.191:3857" ] + }, + { + "ip": "10.22.23.193", + "node_id": 2, + "status": "backup", + "bin_sockets": [ "bin:10.22.23.193:3857" ] + }, + { + "ip": "10.22.23.192", + "node_id": 3, + "status": "member", + "bin_sockets": [ "bin:10.22.23.192:3857" ] + } + ] + } +] + + +
+ +
+ <function moreinfo="none">cl_ctr_node_info</function> + + Return full information for a specific node identified by + its allocated node_id. + + Parameters: + + + node_id — the integer node_id to + look up. + + + + <function>cl_ctr_node_info</function> usage + +opensips-cli -x mi cl_ctr_node_info node_id=2 +{ + "node_id": 2, + "ip": "10.22.23.192", + "cluster_id": 1, + "status": "backup", + "bin_sockets": [ "bin:10.22.23.192:3857" ] +} + + +
+ +
+ <function moreinfo="none">cl_ctr_list_config</function> + + List all configured clusters and their resolved settings — the + effective values actually in use after global defaults and + per-cluster overrides have been applied. Useful for confirming that + a per-cluster master_stickiness or + manage_shtags override took effect. The cluster + password is never exposed. + + + The shtag_mode field reports the current + sharing-tag allocation policy: auto when the + active tag follows the master automatically, or + override:<node_id> when an operator has + pinned a fixed holder with cl_ctr_shtag_force. + + Parameters: none + + <function>cl_ctr_list_config</function> usage + +opensips-cli -x mi cl_ctr_list_config +[ + { + "cluster_id": 1, + "multicast": "239.0.90.1:3333", + "my_ip": "10.22.23.191", + "bin_socket": "bin:10.22.23.191:3857", + "query_time": 5, + "master_stickiness": 1, + "manage_shtags": 1, + "shtag_mode": "auto", + "member_count": 3 + } +] + + +
+ +
+ <function moreinfo="none">cl_ctr_shtag_force</function> + + Force a specific node to hold the active sharing tag, overriding the + normal master-driven allocation. This is useful for planned + maintenance or manual traffic steering: the chosen node becomes the + sole active shtag holder cluster-wide while every other node — + including the master — is put into backup for that tag. + + + The command must be issued on the current master + (it returns an error otherwise). The override is propagated to all + members in the MEMBER_LIST and survives master + fail-over: a newly elected master keeps honouring it rather + than reclaiming the tag. Automatic allocation stays suspended until + cl_ctr_shtag_auto is called. If the forced node + leaves the cluster or times out, the override is cleared + automatically and automatic allocation resumes. + + Parameters: + + cluster_id — the target cluster. + node_id — the node that must hold the active tag; it must be a current member of the cluster. + + + <function>cl_ctr_shtag_force</function> usage + +opensips-cli -x mi cl_ctr_shtag_force cluster_id=1 node_id=3 + + +
+ +
+ <function moreinfo="none">cl_ctr_shtag_auto</function> + + Clear any override set by cl_ctr_shtag_force and + resume automatic, master-driven sharing-tag allocation — the active + tag follows the master again. Must be issued on the current master. + + Parameters: + + cluster_id — the target cluster. + + + <function>cl_ctr_shtag_auto</function> usage + +opensips-cli -x mi cl_ctr_shtag_auto cluster_id=1 + + +
+ +
+ +
+ Exported Pseudo-Variables + + These read-only pseudo-variables expose live cluster state to the + routing script, so a decision such as only the master runs this + job can be made without an MI call. They read directly from + shared memory and are therefore available in every process (SIP + workers included). + + + Each variable optionally takes a cluster id as its argument, e.g. + $cl_ctr_role(2). The bare form + ($cl_ctr_role) resolves to the only configured + cluster; when several clusters are defined the bare form returns NULL + and logs a one-time warning, so the cluster must be named explicitly. + An unknown cluster id, or a value that is not currently known (e.g. no + master yet), returns NULL. All of them are read-only - assigning to + them fails. + + + + $cl_ctr_role — this node's role + in the cluster: master, + backup, member, or + joining (still authenticating / before the + first election). + + + $cl_ctr_is_master — 1 if this + node is the cluster master, 0 otherwise. A fast path for the most + common check. + + + $cl_ctr_master_ip — IP of the + current master (NULL if none is elected yet). + + + $cl_ctr_backup_ip — IP of the + current backup / standby master (NULL if none). + + + $cl_ctr_node_id — this node's + node_id within that cluster (NULL until + assigned). A node may hold different ids in different clusters. + + + $cl_ctr_my_ip — the controller + identity IP of this node. + + + $cl_ctr_members — number of live + members currently in the cluster. + + + $cl_ctr_shtag_mode — + auto (tags follow the elected master) or + forced (an operator pinned them with + cl_ctr_shtag_force). + + + $cl_ctr_forced_node — the + node_id the active sharing tag is pinned to + (NULL when in auto mode). + + + + Using <varname>$cl_ctr_*</varname> in the script + +# single cluster: run a periodic job only on the master +if ($cl_ctr_is_master) + route(do_master_only_work); + +# several clusters: name the one you mean +xlog("cluster 2 master is $cl_ctr_master_ip(2), I am $cl_ctr_role(2)\n"); + + + +
+ +
+ Exported Functions + + Per-peer lookups take two arguments (cluster_id, node_id) + and are therefore script functions, not + pseudo-variables (a comma inside a variable's parentheses is ambiguous when + the variable is used as a function argument). Boolean checks return + true/false for use directly in an if; value lookups + write into an output variable. All are usable from any route. + + + + cl_ctr_node_is_master(cluster_id, node_id) + — true if that node is the cluster master. + + + cl_ctr_node_present(cluster_id, node_id) + — true if that node_id is a live member. + + + cl_ctr_get_node_role(cluster_id, node_id, out_var) + — write that node's role + (master/backup/member) + into out_var; returns false if it is not a member. + + + cl_ctr_get_node_ip(cluster_id, node_id, out_var) + — write that node's IP into out_var. + + + + Per-peer lookup functions + +if (cl_ctr_node_present(1, 3) && cl_ctr_node_is_master(1, 3)) { + cl_ctr_get_node_ip(1, 3, $var(ip)); + xlog("node 3 ($var(ip)) leads cluster 1\n"); +} + + +
+ +
+ Multiple Clusters + + A single &osips; instance can participate in multiple clusters + simultaneously by repeating the cluster + modparam. Each cluster runs an independent worker process with + its own multicast socket, peer table, master election, and + node_id space. + + + Clusters are distinguished by the combination of their multicast + IP address and UDP port. Two useful topologies are possible: + + + + Different ports, same multicast IP + — convenient when all clusters share the same L2 segment. + The port number alone separates traffic for each cluster. + Each cluster's password should also + differ to provide an additional encryption barrier. + + + Different multicast IPs + — useful when clusters span different network segments or + when multicast routing is scoped differently per cluster. + + + + When multiple clusters are defined and the node has more than + one BIN socket, the bin_socket= key must be + specified in each cluster string to indicate which BIN socket + to advertise for that cluster. If only one BIN socket exists, + it is used for all clusters automatically. + + + Each cluster has its own independent clusterer + cluster_id, allowing different &osips; + subsystems to replicate on different clusters: + + + Multiple clusters — dialog on cluster 1, usrloc on cluster 2 + +# Two BIN sockets, one per cluster +socket=bin:10.0.1.10:5566 +socket=bin:10.0.2.10:5566 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +loadmodule "clusterer_controller.so" +# Cluster 1 — dialog replication group, LAN segment 10.0.1.0/24 +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.1.10:5566") +# Cluster 2 — usrloc replication group, LAN segment 10.0.2.0/24 +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,bin_socket=bin:10.0.2.10:5566") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) + +loadmodule "usrloc.so" +modparam("usrloc", "cluster_id", 2) + + + + Multiple clusters — same multicast IP, different ports + +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,password=ClusterOneSecret") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,password=ClusterTwoSecret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 2) + + +
+ +
+ Hybrid Topologies (native + controller clusters) + + A single &osips; instance can run native clusterer + clusters (defined in the database or statically via + my_node_info/neighbor_node_info) + and controller-managed clusters side by side. + This allows, for example, a fixed, DB-provisioned replication cluster + to coexist with a zero-config controller-driven HA cluster on the same + node. + + + Which kind a cluster is follows from how it is declared: + + + + Controller-managed — every + cluster_id registered with the clusterer module + via modparam("clusterer", "cluster_options", + "cluster_id=N, use_controller=1") (and matched by a + cluster entry in this module). + Its topology and this node's node_id are driven + at runtime by the controller; it never touches the database and + always behaves as db_mode=0, regardless of the + global db_mode. + + + Native — every cluster loaded + from the clusterer database (db_mode!=0) or + provisioned statically. It behaves exactly as classic clusterer: + fixed my_node_id, DB persistence (when + DB-backed), script/MI-managed sharing tags. + + + + The rules that keep the two kinds apart: + + + + A cluster_id is exclusively + controller-managed or native — declaring the same id both ways is + rejected at startup. + + + my_node_id identifies this node in its + native clusters only and is required only when + native clusters exist; controller clusters get their node id + assigned at runtime, and this node may well hold + different node ids in different clusters. + + + db_url/db_mode apply to + native clusters only. A controller-only deployment needs neither. + + + Sharing tags: only tags of controller-managed clusters are forced + to backup at startup and driven by the controller master; tags of + native clusters keep their configured state (=active + included) and stay script/MI-managed. + + + Native and controller clusters may share the same BIN socket - the + cluster_id carried in every BIN packet keeps + their traffic apart. + + + + Hybrid — DB-native cluster 10 + controller cluster 1 + +socket=bin:10.0.0.10:5566 + +loadmodule "db_mysql.so" +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +# native side: cluster 10 is defined entirely by rows in the clusterer DB +# table (one row per node, including a row whose node_id = my_node_id below +# for THIS node) - there is no cluster_id modparam for native clusters. +modparam("clusterer", "db_mode", 1) +modparam("clusterer", "db_url", "mysql://opensips:pass@localhost/opensips") +modparam("clusterer", "my_node_id", 5) # this node's id in its native/DB clusters (global) +# controller side: cluster 1 is dynamic (no DB, no static rows) +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566,password=S3cret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) # controller-managed + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 10) # DB-native + + + + Hybrid, no DB — static native cluster 7 + controller cluster 1 + +socket=bin:10.0.0.10:5566 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +# native side: cluster 7 is provisioned statically (no DB). +# db_mode=0 is REQUIRED, otherwise my_node_info/neighbor_node_info are ignored. +modparam("clusterer", "db_mode", 0) +modparam("clusterer", "my_node_id", 5) # this node's id in native cluster 7 +modparam("clusterer", "my_node_info", "cluster_id=7, url=bin:10.0.0.10:5566") +modparam("clusterer", "neighbor_node_info", "cluster_id=7, node_id=6, url=bin:10.0.0.11:5566") +# controller side: cluster 1 is dynamic +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:10.0.0.10:5566,password=S3cret") + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) # controller-managed (cluster 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 7) # native (cluster 7) + + + Here my_node_id=5 is this node's id in the native + cluster 7; in cluster 1 the controller assigns an id at runtime, which + may differ. + + +
+ +
+ Configuration Example + + The following example shows a minimal two-module configuration + for zero-config HA clustering with dialog replication. The + loadmodule order does not matter — the + dependency system enforces correct initialization order + automatically. + + + Minimal HA cluster configuration + +# Each node needs an explicit BIN socket (no wildcard) +socket=bin:10.22.23.191:3857 + +loadmodule "proto_bin.so" + +loadmodule "clusterer.so" +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "sharing_tag", "vip1/1=active") +modparam("clusterer", "ping_interval", 4) +modparam("clusterer", "ping_timeout", 1500) + +loadmodule "clusterer_controller.so" +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333") +modparam("clusterer_controller", "password", "MyStr0ngPassw0rd!") + +loadmodule "tm.so" +modparam("tm", "tm_replication_cluster", 1) + +loadmodule "dialog.so" +modparam("dialog", "dialog_replication_cluster", 1) +modparam("dialog", "cluster_auto_sync", 1) + +loadmodule "dispatcher.so" +modparam("dispatcher", "cluster_id", 1) +modparam("dispatcher", "cluster_probing_mode", "distributed") + + + + The same configuration file (with the node-specific + socket=bin:IP:PORT line changed per node) + is used on every node. No other per-node customization is + required. + +
+ +
+ Limitations + + + IPv4 only. IPv6 multicast is not currently supported. + + + The module pre-allocates approximately 152 KB of shared + memory (-m) per cluster and 6 KB of + private memory (-M) per worker process + at startup. If either allocation fails, OpenSIPS will refuse + to start with an error in the log. + + + Wildcard BIN sockets (bin:*:PORT) are + rejected at startup. An explicit IP address must be used in + the socket= line. + + + Node IDs are not persistent across a full cluster restart + (all nodes down simultaneously). IDs are reallocated + starting from 1 when the cluster reforms. This has no + operational impact as long as at least one node remains up + during rolling restarts. + + + The multicast network must support IP multicast routing + between all cluster nodes. Nodes on different L3 segments + require PIM or similar multicast routing. + + + PIM-DM networks: PIM Dense + Mode periodically re-floods multicast traffic (prune state + typically expires every 3 minutes). On slow or congested + networks this brief re-flood/prune cycle could cause a gap + in MASTER_ALIVE delivery and trigger a spurious master + re-election. If this occurs, increase the effective timeout + by raising CL_CTR_MASTER_KA_MISSED in the + source from 3 to 5 or higher. PIM Sparse Mode (PIM-SM) or + networks with IGMP snooping do not have this issue. + + + L2 overlay tunnels (Geneve, VXLAN, + GRE): if the overlay presents a flat L2 segment + with multicast support, the controller works transparently. + However, some VXLAN deployments disable multicast entirely + (no underlay multicast group, no BUM replication) — in that + case the controller will not function, as it has no unicast + fallback. Encapsulation overhead also adds latency and + jitter; on high-latency overlays consider raising + CL_CTR_MASTER_KA_MISSED to avoid spurious + re-elections. + + + IPsec-protected links: + native IPsec multicast requires GDOI/GET VPN (RFC 6407), + which is rarely deployed. The recommended approach is to + run multicast inside an inner tunnel (GRE-over-IPsec, + Geneve-over-IPsec) that presents a multicast-capable + interface. Running the controller over such a setup results + in double encryption (application-layer XChaCha20-Poly1305 plus + IPsec ESP), which is harmless but adds minor CPU overhead. + IPsec ESP tunnel mode also reduces the effective MTU by + approximately 50 bytes, which compounds the MEMBER_LIST + fragmentation issue described below. + + + MEMBER_LIST fragmentation: + the MEMBER_LIST packet grows with cluster size and reaches + approximately 4395 bytes at the maximum of 256 nodes. This + exceeds the 1472-byte UDP payload budget of a standard + 1500-byte MTU Ethernet link and requires IP fragmentation: + + Standard Ethernet (1500 MTU): 3 fragments + IPsec ESP tunnel (~1400 MTU): 4 fragments + GRE-over-IPsec (~1350 MTU): 4–5 fragments + + All other packet types (ALIVE, JOIN_REQ, KEY_GRANT, GOODBYE, + etc.) fit comfortably within a single datagram on any of + these links. The DF bit is not set, so IP fragmentation + occurs transparently where the network allows it. However, + firewalls or stateless middleboxes that silently drop + fragmented UDP will prevent new nodes from joining, since + MEMBER_LIST is required to complete the join sequence. + Verify that fragmented UDP is permitted on all paths between + cluster nodes, particularly over VPN tunnels and across + datacenter firewalls. + + +
+ +
+ Planned Features + + The following features are planned for future releases: + + + + Node maintenance mode — take a node + out of duty for a rolling upgrade while it stays in the cluster. A + node in maintenance keeps replicating and answering pings and stays + visible in cl_ctr_list_members, but is excluded + from election (never master or backup; if it is master it hands over + gracefully first) and sheds its sharing tags (a + cl_ctr_shtag_force pin on it auto-clears). Two + levels are planned: + + evicted — out of election + and tags; the routing script refuses new work while established + dialogs finish. + full — additionally marks + the node down for clusterer consumers so peers stop routing + replication work to it. + + The state is cluster-wide (carried in the ALIVE/MEMBER_LIST control + plane, so every node agrees and it survives master failover) and + runtime-only — a restart brings the node back in service. + + + Interfaces (following the read-only variables and MI commands above): + + + MI cl_ctr_maintenance (any node, the + master propagates it) sets a target node's state + evicted/full/off; + cl_ctr_list_members gains a maintenance + column. + + + Script function cl_ctr_set_maintenance() + (a verb - an action) for a node to put itself in or out of + maintenance from the routing logic. + + + Read-only variables $cl_ctr_maintenance + (this node: none/evicted/full) and + $cl_ctr_node_maint(cluster_id, node_id) + (any peer); $cl_ctr_role gains a + maintenance value. + + + Event E_CL_CTR_MAINTENANCE raised on + every node when any member's maintenance state changes, so an + event_route can react (e.g. shift + dispatcher weights). + + + + + Statistics — module statistics + (current role, member count, master changes, nodes joined/left, + JOIN_REJECTs, decrypt failures, config mismatches, split-brain + merges) exposed via get_statistics and + monitoring exporters. + + + Events — events raised on state + transitions (became master, demoted, node joined/left, split-brain + merged, config mismatch, authentication reject), named + E_CL_CTR_* and consumable from an + event_route or any event subscriber transport. + + + IPv6 multicast — the control plane + currently uses IPv4 multicast (groups in 224.0.0.0/4). + Add IPv6 multicast support (ff00::/8 groups, + AF_INET6 sockets and membership) so the controller + can run on IPv6-only or dual-stack deployments. + + + + Read-only script variables for cluster state + ($cl_ctr_role, + $cl_ctr_is_master, and the rest) are already + available - see . + +
+ +
diff --git a/modules/clusterer_controller/doc/clusterer_controller_tests.xml b/modules/clusterer_controller/doc/clusterer_controller_tests.xml new file mode 100644 index 00000000000..b09668c2c54 --- /dev/null +++ b/modules/clusterer_controller/doc/clusterer_controller_tests.xml @@ -0,0 +1,265 @@ + + + + HA Behaviour Tests + + The following tests were performed on a three-node cluster + (nodes A=10.22.23.191, + B=10.22.23.192, + C=10.22.23.193) to verify correct failover, + tag stability, and no-steal-on-join behaviour. The sharing tag + under test is vip1 in cluster 1. All nodes run + with manage_shtags=1. + + + Tag state was queried after each operation via: + opensips-cli -x mi clusterer_list_shtags + + +
+ Baseline + + All three nodes running. B holds the active tag; A and C are + backup. + + +A (10.22.23.191) svc=active tag=backup +B (10.22.23.192) svc=active tag=active +C (10.22.23.193) svc=active tag=backup + +
+ +
+ Test 1 — Stop the active node + + B (active) is stopped. The remaining nodes must elect a new + active holder. B must rejoin as backup and must not steal the + tag from whichever node became active. + + +# Stop B +A svc=active tag=backup +B svc=inactive tag=(down) +C svc=active tag=active <-- C promoted + +# Start B +A svc=active tag=backup +B svc=active tag=backup <-- rejoined as backup +C svc=active tag=active <-- C retains active + + + Result: PASS. + Failover within the dead-node detection window; rejoining node + did not steal the active tag. + +
+ +
+ Test 2 — Stop a backup node + + A (backup) is stopped. The active tag must remain on C without + any transition. A must rejoin as backup. + + +# Stop A +A svc=inactive tag=(down) +B svc=active tag=backup +C svc=active tag=active <-- unchanged + +# Start A +A svc=active tag=backup <-- rejoined as backup +B svc=active tag=backup +C svc=active tag=active <-- still active + + + Result: PASS. + Removing a backup node causes no tag movement; rejoining node + started in backup state. + +
+ +
+ Test 3 — Stop both backup nodes + + A and B (both backup) are stopped simultaneously. The lone + remaining node C must retain the active tag. A and B must + rejoin as backup. + + +# Stop A and B +A svc=inactive tag=(down) +B svc=inactive tag=(down) +C svc=active tag=active <-- unchanged, lone node + +# Start A, then B +A svc=active tag=backup <-- rejoined as backup +B svc=active tag=backup <-- rejoined as backup +C svc=active tag=active <-- still active + + + Result: PASS. + Active node remained stable while running alone; both rejoining + nodes came up in backup state. + +
+ +
+ Test 4 — Stop active node and one backup + + B (active) and C (backup) are stopped. The sole remaining node + A must become active. B and C must rejoin as backup. + + +# Stop B and C +A svc=active tag=active <-- A promoted, now lone node +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start B +A svc=active tag=active <-- retains active +B svc=active tag=backup <-- rejoined as backup +C svc=inactive tag=(down) + +# Start C +A svc=active tag=active <-- retains active +B svc=active tag=backup +C svc=active tag=backup <-- rejoined as backup + + + Result: PASS. + The surviving node correctly claimed the active tag; each + rejoining node started in backup state without challenging the + active holder. + +
+ +
+ Test 5 — Full cluster restart + + All three nodes are stopped (full outage). Nodes are then + started one at a time. The first node up must self-elect as + active (no peers available to sync from). Subsequent nodes must + join as backup and must not steal the active tag. + + +# All stopped +A svc=inactive tag=(down) +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start A first +A svc=active tag=active <-- first/lone seed, self-synced +B svc=inactive tag=(down) +C svc=inactive tag=(down) + +# Start B +A svc=active tag=active <-- retains active +B svc=active tag=backup <-- joined as backup, did not steal +C svc=inactive tag=(down) + +# Start C +A svc=active tag=active <-- retains active +B svc=active tag=backup +C svc=active tag=backup <-- joined as backup, did not steal + + + Result: PASS. + The first node to start elected itself active and no spurious + sync errors were logged. Each subsequent node joined as backup + without challenging the active holder. + +
+ +
+ Test 6 — Multiple clusters over one BIN socket + + Two controller clusters (id=1 and + id=2) are configured on all three nodes. Each + cluster has its own multicast endpoint (a distinct port is required - + two clusters on the same multicast address:port are rejected at + startup with duplicate multicast), but both + advertise the same BIN socket + (bin:IP:3857). Each cluster must form independently + and its replication must stay isolated over the shared socket. + + +modparam("clusterer", "cluster_options", "cluster_id=1, use_controller=1") +modparam("clusterer", "cluster_options", "cluster_id=2, use_controller=1") +modparam("clusterer_controller", "cluster", + "id=1,multicast=239.0.90.1:3333,bin_socket=bin:IP:3857") +modparam("clusterer_controller", "cluster", + "id=2,multicast=239.0.90.1:3334,bin_socket=bin:IP:3857") + +# both clusters converge, each with its own master/backup/member roles; +# clusterer_list shows cluster 1 and cluster 2 both using bin:IP:3857 + + + Result: PASS. + Both clusters formed independently (each member_count=3 + with consistent roles), the BIN links of both were Up + over the single shared socket, and there were zero decrypt / cross-talk / + foreign-cluster errors on any node - the cluster_id in + each BIN packet keeps the two clusters' replication traffic separate. A + control test confirmed that configuring both clusters on the same + multicast address:port is refused at startup. + +
+ +
+ Summary + + + + + + + + Test + Scenario + Result + + + + + 1 + Stop active node; rejoin + PASS + + + 2 + Stop backup node; rejoin + PASS + + + 3 + Stop both backups simultaneously; rejoin + PASS + + + 4 + Stop active + one backup; rejoin + PASS + + + 5 + Full cluster restart; sequential startup + PASS + + + 6 + Two clusters sharing one BIN socket (distinct multicast) + PASS + + + + + + In all five failover tests (1–5): exactly one node held the active + sharing tag at all times (including during the failure window), and no + rejoining node stole the active tag from the current holder. Test 6 + additionally confirmed that two clusters can share a single BIN socket + with fully isolated replication. + +
+ +
diff --git a/modules/clusterer_controller/doc/contributors.xml b/modules/clusterer_controller/doc/contributors.xml new file mode 100644 index 00000000000..7d19b210dc0 --- /dev/null +++ b/modules/clusterer_controller/doc/contributors.xml @@ -0,0 +1,25 @@ + + Contributors +
+ Contributors + + Definition, design and implementation of this module was made by: + + + Yury Kirsanov — VoIPLine Telecom + + + +
+
+ Documentation Contributors + + Documentation was written by: + + + Yury Kirsanov — VoIPLine Telecom + + + +
+
diff --git a/modules/clusterer_controller/test/cl_ctr_join_reject_test.py b/modules/clusterer_controller/test/cl_ctr_join_reject_test.py new file mode 100755 index 00000000000..f5f44a80593 --- /dev/null +++ b/modules/clusterer_controller/test/cl_ctr_join_reject_test.py @@ -0,0 +1,135 @@ +#!/usr/bin/env python3 +""" +cl_ctr_join_reject_test.py - rogue-joiner security test for clusterer_controller. + +Simulates an unauthorized node on the cluster's multicast group WITHOUT touching +any node config. It exercises two defences: + + 1. JOIN_REJECT path: send JOIN_REQ packets with the correct bootstrap magic but + a bogus (wrong-key) body. The master cannot decrypt them; after a few it is + expected to emit an encrypted JOIN_REJECT (a bootstrap-magic packet back on + the group). We can't decrypt that reject (wrong key), but observing a + bootstrap-magic packet from a cluster node in response confirms the master's + reject logic fired. + + 2. Rogue-traffic isolation (anti-churn): send bogus SESSION-magic packets + (a fake MASTER_ALIVE). A correct cluster must IGNORE these. We measure the + rate of real session traffic before vs during the flood: a large spike would + mean the flood pushed non-master nodes into a re-JOIN churn (the old bug). + +Run this on a host on the multicast segment that is NOT a live cluster member +(e.g. one node with `systemctl stop opensips`). Requires only python3. + + ./cl_ctr_join_reject_test.py --group 239.0.90.1 --port 3333 +""" +import argparse, collections, os, socket, struct, threading, time + +# --- wire constants (must match clusterer_controller.c) --- +BOOTSTRAP_MAGIC = bytes([0xCC, 0x01]) +SESSION_MAGIC = bytes([0xCC, 0x00]) +MAGIC_SZ, CLUSTER_ID_SZ, NONCE_SZ, TAG_SZ = 2, 2, 12, 16 + + +def bogus_packet(magic, cluster_id): + # [magic 2B][cluster_id 2B BE][nonce 12B][~60B random 'ciphertext'][tag 16B] + return (magic + struct.pack("!H", cluster_id & 0xFFFF) + + os.urandom(NONCE_SZ) + os.urandom(60) + os.urandom(TAG_SZ)) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--group", default="239.0.90.1") + ap.add_argument("--port", type=int, default=3333) + ap.add_argument("--count", type=int, default=6, help="packets per phase") + ap.add_argument("--interval", type=float, default=1.0, help="seconds between packets") + ap.add_argument("--cluster-id", type=int, default=1, + help="cluster_id to stamp on packets; use a value the target " + "cluster does NOT use to verify foreign packets are filtered") + args = ap.parse_args() + + # discover our own source IP (to filter our own loopback out) + p = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) + p.connect((args.group, args.port)); my_ip = p.getsockname()[0]; p.close() + + rx = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) + rx.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + rx.bind(("", args.port)) + rx.setsockopt(socket.IPPROTO_IP, socket.IP_ADD_MEMBERSHIP, + struct.pack("4s4s", socket.inet_aton(args.group), + socket.inet_aton(my_ip))) + rx.settimeout(0.3) + + tx = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) + tx.setsockopt(socket.IPPROTO_IP, socket.IP_MULTICAST_TTL, 1) + + boot_from = collections.Counter() # bootstrap-magic pkts from peers (JOIN_REJECT candidates) + sess_from = collections.Counter() # session-magic pkts from peers + stop = threading.Event() + + def receiver(): + while not stop.is_set(): + try: + data, addr = rx.recvfrom(65535) + except socket.timeout: + continue + if addr[0] == my_ip or len(data) < MAGIC_SZ: + continue + if data[:MAGIC_SZ] == BOOTSTRAP_MAGIC: + boot_from[addr[0]] += 1 + elif data[:MAGIC_SZ] == SESSION_MAGIC: + sess_from[addr[0]] += 1 + + threading.Thread(target=receiver, daemon=True).start() + print(f"[*] rogue joiner {my_ip} -> {args.group}:{args.port}") + print(f" cluster_id={args.cluster_id} bootstrap={BOOTSTRAP_MAGIC.hex()} session={SESSION_MAGIC.hex()}") + + # baseline: 3s of quiet listening to learn the normal session-traffic rate + print("[*] measuring baseline cluster traffic for 3s ...") + t0 = time.time(); base0 = sum(sess_from.values()); time.sleep(3.0) + base_rate = (sum(sess_from.values()) - base0) / (time.time() - t0) + print(f" baseline session rate: {base_rate:.1f} pkt/s") + + # phase 1 - JOIN_REJECT probe + boot_before = sum(boot_from.values()) + print(f"\n[*] PHASE 1: sending {args.count} bogus JOIN_REQ (bootstrap magic) ...") + for i in range(args.count): + tx.sendto(bogus_packet(BOOTSTRAP_MAGIC, args.cluster_id), (args.group, args.port)) + print(f" -> JOIN_REQ #{i+1}") + time.sleep(args.interval) + time.sleep(2.0) + rejects = sum(boot_from.values()) - boot_before + + # phase 2 - anti-churn probe (fake MASTER_ALIVE flood) + print(f"\n[*] PHASE 2: flooding {args.count*3} bogus SESSION packets (fake MASTER_ALIVE) ...") + t1 = time.time(); sess1 = sum(sess_from.values()) + for i in range(args.count * 3): + tx.sendto(bogus_packet(SESSION_MAGIC, args.cluster_id), (args.group, args.port)) + time.sleep(args.interval / 3.0) + flood_rate = (sum(sess_from.values()) - sess1) / (time.time() - t1) + + time.sleep(1.0); stop.set() + + # --- report --- + print("\n===================== RESULT =====================") + print(f"JOIN_REQ sent (phase 1): {args.count}") + print(f"JOIN_REJECT-candidate replies from master: {rejects} " + f"from {sorted(k for k,v in boot_from.items() if v)}") + print(f"real session rate baseline={base_rate:.1f}/s during-flood={flood_rate:.1f}/s") + print("--------------------------------------------------") + ok = True + if rejects > 0: + print("PASS master answered unauthenticated JOIN_REQ with a JOIN_REJECT") + else: + print("WARN no JOIN_REJECT seen (rejection may rely on joiner-side shutdown)") + # a >3x spike over baseline indicates the flood induced re-JOIN churn + if flood_rate <= max(base_rate * 3.0, base_rate + 5): + print("PASS cluster ignored the rogue session flood (no churn spike)") + else: + print("FAIL session traffic spiked under the flood -> cluster was disrupted"); ok = False + print("==================================================") + return 0 if ok else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/build/apt_requirements.txt b/scripts/build/apt_requirements.txt index 5866c6f33b8..aabed314a31 100644 --- a/scripts/build/apt_requirements.txt +++ b/scripts/build/apt_requirements.txt @@ -25,6 +25,7 @@ librabbitmq-dev libradcli-dev libsctp-dev libsnmp-dev +libsodium-dev libsqlite3-dev libwolfssl-dev libxml2-dev From 31a6d5c99cc3f4c737a3faa7cefd5fb4e0a77d8d Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Sun, 9 Aug 2026 22:20:42 +1000 Subject: [PATCH 04/32] clusterer_controller: a consumer messaging API over the encrypted plane The controller already owns an authenticated, encrypted, rate-limited UDP plane between the nodes of a cluster, plus the membership that says who is on it. Any other module wanting to exchange a message between nodes had to build that transport again from scratch. This exports it instead. A consumer registers a named channel pre-fork and sends on it: clctr_api_t clctr; if (load_clctr_api(&clctr) < 0) ... /* mod_init */ clctr.register_channel(&ch, my_cb); /* mod_init - PRE-FORK */ clctr.send_mcast(cluster_id, &ch, &payload, 0); clctr.send_ucast(cluster_id, node_id, &ch, &payload, 0); and inherits the XChaCha20-Poly1305 group session key and its rotation, the per-packet receive gauntlet (magic gate, cluster_id filter, size bound, per-source rate limiting) and the membership view, with no transport code of its own. The delivery contract is the part worth reading, because it is what a consumer gets wrong: - the receive callback runs in the CONTROLLER's worker process for that cluster, not in the process that called send. A consumer needing to wake a different process brings its own mechanism (shm + eventfd, ipc_send_rpc). Callbacks run on the cluster's receive path, so they must stay short. - sends are marshalled to that same worker over IPC, which is what keeps ordering per node and leaves the anti-replay sequence space single-writer. - CLCTR_SEND_TO_SELF dispatches locally rather than listening for our own packet, and a unicast to our own node id degenerates to exactly that, with nothing on the wire. - src_node_id is 0 when the sender had not been assigned an id yet. CLCTR_SEND_RELIABLE asks for acknowledgement and resend while unacknowledged. It is opt-in per send rather than per channel on purpose: it costs one ACK per recipient, so a reliable broadcast turns a single packet into N-1 packets back. Most consumer traffic is better served by being idempotent and retried by its own logic. Payload sizing has a deliberate split: CLCTR_MAX_PAYLOAD is a compile-time lower bound for consumers that size local buffers statically, while the real runtime limit is cc_max_payload, derived from the interface MTU at mod_init and larger on jumbo-frame links. Ships with tests for the consumer sequence, channel filtering, reliable broadcast, and the script-facing messaging surface. --- modules/clusterer_controller/api.h | 151 ++ .../clusterer_controller.c | 1255 ++++++++++++++++- .../doc/clusterer_controller_admin.xml | 232 +++ .../test/consumer_filter_test.py | 49 + .../test/consumer_seq_test.py | 53 + .../test/reliable_broadcast_test.py | 52 + .../test/script_messaging_rig.sh | 76 + .../test/script_messaging_test.py | 84 ++ .../test/script_send_list_test.py | 48 + 9 files changed, 1966 insertions(+), 34 deletions(-) create mode 100644 modules/clusterer_controller/api.h create mode 100644 modules/clusterer_controller/test/consumer_filter_test.py create mode 100644 modules/clusterer_controller/test/consumer_seq_test.py create mode 100644 modules/clusterer_controller/test/reliable_broadcast_test.py create mode 100755 modules/clusterer_controller/test/script_messaging_rig.sh create mode 100755 modules/clusterer_controller/test/script_messaging_test.py create mode 100644 modules/clusterer_controller/test/script_send_list_test.py diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h new file mode 100644 index 00000000000..5a3d8a5d970 --- /dev/null +++ b/modules/clusterer_controller/api.h @@ -0,0 +1,151 @@ +/* + * Copyright (C) 2026 VoIPcloud + * + * This file is part of opensips, a free SIP server. + * + * opensips is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * (at your option) any later version. + * + * opensips is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program; if not, write to the Free Software + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + */ + +/* + * Consumer messaging API - lets other modules (and, through them, the + * script) exchange messages over the controller's encrypted UDP plane + * instead of building their own transport. A consumer inherits the + * XChaCha20-Poly1305 group session key (with its rotation), the + * per-packet receive gauntlet (magic gate, cluster_id filter, size + * bound, per-source rate limiting) and the controller's membership, + * with zero transport code of its own. + * + * Usage: + * clctr_api_t clctr; + * if (load_clctr_api(&clctr) < 0) ... (mod_init) + * clctr.register_channel(&ch, my_cb); (mod_init - PRE-FORK) + * clctr.send_mcast(cluster_id, &ch, &payload, 0); + * clctr.send_ucast(cluster_id, node_id, &ch, &payload, 0); + * + * Delivery contract: + * - the receive callback runs in the CONTROLLER'S WORKER PROCESS for + * that cluster, not in the process that called send. A consumer + * that must wake another process brings its own mechanism (shm + + * eventfd, ipc_send_rpc, ...). Keep callbacks short - they run on + * the cluster's receive path. + * - sends are marshalled to the same worker over IPC, so ordering is + * preserved per node and the anti-replay sequence space stays + * single-writer. + * - CLCTR_SEND_TO_SELF delivers to this node by invoking the local + * callback directly - never by hearing our own packet back. A + * unicast addressed to our own node id degenerates to exactly that + * local dispatch, with no packet on the wire. + * - src_node_id is 0 when the sender had not been assigned an id yet. + */ + +#ifndef CL_CTR_API_H +#define CL_CTR_API_H + +#include "../../str.h" +#include "../../sr_module.h" + +/* deliver to this node too, via local callback dispatch (never off the wire) */ +#define CLCTR_SEND_TO_SELF (1 << 0) +/* Ask for the message to be acknowledged, and resent while it is not. + * + * Costs one ACK per recipient, so a reliable send to the whole cluster is + * N-1 packets back where a plain one was nothing - deliberately opt-in, and + * deliberately per send rather than per channel: most consumer traffic is + * better served by being idempotent and retried by its own logic, which is + * how the cross-node cache fetch works and why it asks for nothing here. */ +#define CLCTR_SEND_RELIABLE (1 << 1) + +/* limits a consumer can rely on */ +#define CLCTR_MAX_CHAN_LEN 31 +/* Compile-time lower bound for consumer payload; the actual runtime limit is + * cc_max_payload, which is derived from the interface MTU at mod_init and may + * be larger on jumbo-frame links. Consumers that size local buffers at + * compile time should use CLCTR_MAX_PAYLOAD; consumers that want to send the + * largest possible message at runtime should check cc_max_payload instead. */ +#define CLCTR_MAX_PAYLOAD 1300 +extern int cc_max_payload; + +typedef void (*clctr_msg_cb_f)(int cluster_id, int src_node_id, + str *channel, str *payload); + +/* register a named channel; PRE-FORK only (call from mod_init). + * Returns 0 on success, -1 on bad name / duplicate / table full. */ +typedef int (*clctr_register_channel_f)(str *channel, clctr_msg_cb_f cb); + +/* send to the cluster's multicast group / to one node by id. + * Returns 0 = accepted for sending, -1 = bad arguments or unknown + * cluster, -2 = cluster not ready (no worker / not joined yet). + * "Accepted" means handed to the cluster worker - UDP gives no + * delivery guarantee, by design (consumers must be loss-tolerant). */ +typedef int (*clctr_send_mcast_f)(int cluster_id, str *channel, + str *payload, int flags); +typedef int (*clctr_send_ucast_f)(int cluster_id, int node_id, + str *channel, str *payload, int flags); + +/* this node's controller-assigned id in the cluster; 0 = none yet */ +/* Send to a named set of nodes. Deliberately N unicasts rather than one + * multicast with the receivers filtering: a multicast is decrypted by every + * member, so "addressed to three of you" would still hand the payload to all + * of them. The cost is linear in the size of the list, which is the honest + * price of addressing a subset. + * + * Returns the number of nodes the message was sent to, or -1 on error. Nodes + * in the list that are not current members are skipped and counted in + * @unknown when it is not NULL. */ +typedef int (*clctr_send_list_f)(int cluster_id, const int *node_ids, int n, + str *channel, str *payload, int flags, + int *unknown); + +typedef int (*clctr_get_my_node_id_f)(int cluster_id); + +/* + * This node's own address on the cluster plane, as RESOLVED at startup - not + * the raw modparam. It comes from one of three places (explicit `my_ip`, the + * IPv4 of an explicit `interface`, or a default-route probe towards the + * multicast group), and which one won is not otherwise visible to a consumer + * module. @src, when non-NULL, receives a short constant describing that + * origin. Both outputs point at module-static storage, valid for the process + * lifetime; neither must be freed. Returns 0 on success, -1 before mod_init + * has resolved it. + * + * Worth exposing because a wrong answer here is not cosmetic: joining the + * multicast group on the wrong interface is exactly how a node ends up unable + * to decrypt its peers' traffic. + */ +typedef int (*clctr_get_my_ip_f)(const char **ip, const char **iface, + const char **src); + +typedef struct clctr_api { + clctr_register_channel_f register_channel; + clctr_send_mcast_f send_mcast; + clctr_send_ucast_f send_ucast; + clctr_send_list_f send_list; + clctr_get_my_node_id_f get_my_node_id; + clctr_get_my_ip_f get_my_ip; +} clctr_api_t; + +typedef int (*load_clctr_f)(clctr_api_t *api); + +static inline int load_clctr_api(clctr_api_t *api) +{ + load_clctr_f load_clctr; + + load_clctr = (load_clctr_f)(void *)find_export("load_clctr", 0); + if (!load_clctr) + return -1; + return load_clctr(api); +} + +#endif /* CL_CTR_API_H */ diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index de93fa22948..6691087ea6e 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -188,6 +188,7 @@ #include /* O_NONBLOCK, fcntl() */ #include /* getifaddrs(), freeifaddrs() */ #include /* IF_NAMESIZE, struct ifreq, SO_BINDTODEVICE */ +#include /* ioctl, SIOCGIFMTU */ #include "../../sr_module.h" /* module_exports, MODULE_VERSION, proc_export_t, PROC_FLAG_*, dep_export_t, DEP_ABORT, @@ -202,6 +203,7 @@ #include "../../net/api_proto.h" /* protos[] array */ #include "../../globals.h" /* process_no - this process's index */ #include "../../ipc.h" /* ipc_send_rpc() - cross-process job dispatch */ +#include "api.h" /* consumer messaging API contract */ #include "../../pvar.h" /* pv_export_t - read-only $cl_ctr_* variables */ #include "../clusterer/clusterer_ctrl.h" /* set_my_identity, add_node, remove_node */ @@ -230,6 +232,14 @@ #define CL_CTR_MAGIC_SZ 2 static const unsigned char CL_CTR_PACKET_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x00 }; static const unsigned char CL_CTR_BOOTSTRAP_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x01 }; +/* Consumer traffic carries its own magic. It is encrypted with the same + * session key as everything else - the distinction exists purely so the + * rate limiter, which runs BEFORE decryption and therefore cannot see a + * packet's type, can charge consumer data against its own budget instead + * of the control plane's. Without this, a consumer sending faster than + * the control-plane limit is silently throttled, and control packets and + * data compete for one allowance. */ +static const unsigned char CL_CTR_CONSUMER_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x02 }; /* Packet type bytes */ #define CL_CTR_PKT_ALIVE 0x01 @@ -243,6 +253,22 @@ static const unsigned char CL_CTR_BOOTSTRAP_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x0 #define CL_CTR_PKT_JOIN_REJECT 0x09 /* master -> joiner: authentication rejected */ #define CL_CTR_PKT_ACK 0x0B /* receiver -> sender: ack of a 1:1 handshake pkt */ #define CL_CTR_PKT_RESYNC 0x0C /* member -> master: my view differs, resend state */ +#define CL_CTR_PKT_CONSUMER 0x0D /* consumer API message: [src_id][chan][data] */ +/* Same payload, but the receiver acknowledges it and the sender retransmits + * until it does. A separate type rather than a flag in the payload so the + * receive path can tell them apart before it parses anything, and so an old + * build simply does not recognise it instead of half-understanding it. */ +#define CL_CTR_PKT_CONSUMER_REL 0x0E + +/* Consumer messaging (api.h): plaintext payload layout after type+seq is + * [src_node_id u16 BE][chan_len u8][channel bytes][consumer payload]. + * Bounded to one comfortably-under-MTU datagram; bulk data is the + * consumer's problem (the API contract says fall back to BIN for bulk). */ +#define CL_CTR_MAX_CHANNELS 8 +#define CL_CTR_CONSUMER_HDR_SZ (CL_CTR_NODE_ID_SZ + 1) +#define CL_CTR_CONSUMER_PKT_MAX (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ \ + + CL_CTR_CONSUMER_HDR_SZ + CLCTR_MAX_CHAN_LEN \ + + CLCTR_MAX_PAYLOAD + CL_CTR_TAG_SZ) #define CL_CTR_PKT_MASTER_BEACON 0x0A /* master-only announce (BOOTSTRAP key) so * masters with divergent session keys can * still discover each other and merge a @@ -371,11 +397,26 @@ static const unsigned char CL_CTR_BOOTSTRAP_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x0 * Tracks up to CL_CTR_RATE_TBL_SZ source IPs with a 1-second sliding window. */ #define CL_CTR_RATE_TBL_SZ 256 /* one slot per peer; matches max cluster size */ #define CL_CTR_RATE_LIMIT 20 /* max packets per second per source IP */ +/* Consumer data is bulk by nature - one pull is a packet each way - so it + * gets a far larger allowance than the handful of control packets a peer + * sends per second. Separate counters, so neither can starve the other: + * a flood of consumer traffic can never crowd out JOIN/ALIVE, and a busy + * consumer is not throttled at the control plane's rate. */ +#define CL_CTR_CONSUMER_RATE_DEFAULT 1000 +/* Reliable consumer delivery: how many times an unacknowledged message is + * sent again, and how long to wait between attempts. Deliberately modest - + * a consumer that needs more than this wants a different design, not a + * longer queue. Both are per-cluster overridable. */ +#define CL_CTR_CONSUMER_RETRIES_DEFAULT 2 +#define CL_CTR_CONSUMER_RETRY_MS_DEFAULT 40 typedef struct { uint32_t ip; /* network byte order; 0 = empty slot */ time_t window_start; int count; + /* consumer traffic is counted apart from control traffic, against its + * own budget, so neither class can exhaust the other's allowance */ + int consumer_count; } cl_ctr_rate_entry_t; /* Max packet sizes: wire(20) = magic(8) + nonce(12); plain(5) = type(1) + seq(4) @@ -524,6 +565,25 @@ typedef struct { struct sockaddr_storage dest; /* unicast destination */ socklen_t destlen; int pkt_len; /* sealed length */ + /* A broadcast that asked to be acknowledged is one entry, not one per + * member: the message goes out once as a multicast - which is the whole + * point of having a multicast - and only the nodes that fail to answer + * are repaired individually. Sending N unicasts up front would be about + * twice the packets for the same delivery, and at two hundred and fifty + * nodes that is the difference between a message and an event. + * + * Who still owes an ACK is a bitmap by node id. The addresses to repair + * to are not stored: they are looked up from current membership when the + * repair runs, so a node that left in the meantime is simply not chased, + * and one that joined is not expected to answer for something sent + * before it arrived. */ + int is_bcast; + uint8_t retries_cfg; /* the budget this started with, + * so a report can say how much of + * it was actually spent */ + uint16_t expect_n; /* members at send time */ + uint16_t acked_n; + unsigned char acked_map[(CL_CTR_MAX_PEERS + 7) / 8]; unsigned char pkt[CL_CTR_NODE_ASSIGN_MAX_SZ]; /* cached sealed bytes */ } cl_ctr_retx_entry_t; @@ -540,6 +600,13 @@ typedef struct cl_ctr_cluster_ { unsigned char key[32]; /* bootstrap key = SHA256(password); JOIN only */ unsigned char session_key[32]; /* group key = HKDF(password, master_salt) */ int manage_shtags; /* per-cluster override; defaults to global manage_shtags */ + /* Consumer-plane policy, per cluster: a fleet may run one cluster over a + * quiet management VLAN and another across a link where retries matter, + * and one global number cannot be right for both. -1 means "not set + * here", resolved to the global default in mod_init. */ + int consumer_retries; /* extra sends of an unacked message */ + int consumer_retry_ms; /* gap between those sends */ + int consumer_rate; /* per-source packets/s for consumers */ int master_stickiness; /* per-cluster override; -1 = inherit global */ cl_ctr_peers_t *peers; /* per-cluster peer table in shm */ /* BIN socket resolved at mod_init - advertised in JOIN_REQ/NODE_ASSIGN */ @@ -693,16 +760,80 @@ static int on_config_mismatch = CL_CTR_CFGMISMATCH_REJECT; /* resolved; defa /* Resolved at mod_init time - always valid after cl_ctr_resolve_local_identity() */ static char my_ip_buf[INET_ADDRSTRLEN]; +/* which of the three resolution paths produced my_ip - reported through + * clctr_api.get_my_ip() so a consumer can show it without guessing */ +static const char *my_ip_src = "unresolved"; static char my_interface_buf[IF_NAMESIZE]; +/* Maximum consumer payload in bytes, derived from the interface MTU at + * mod_init time. Replaces the compile-time CLCTR_MAX_PAYLOAD constant so + * jumbo-frame interfaces (MTU 9000) are not artificially limited to 1300 B. + * Read-only after mod_init; safe to access from any process. */ +int cc_max_payload = CLCTR_MAX_PAYLOAD; + /* Local node identity - populated at mod_init by scanning the config file */ static uint16_t my_node_id = 0; +/* ---- consumer messaging API (api.h) ----------------------------------- */ + +/* Channel registry: filled PRE-FORK by consumers' mod_init, inherited + * read-only by every process afterwards - no lock needed. */ +struct cl_ctr_channel { + char name[CLCTR_MAX_CHAN_LEN + 1]; + int len; + clctr_msg_cb_f cb; +}; +static struct cl_ctr_channel cl_ctr_channels[CL_CTR_MAX_CHANNELS]; +static int cl_ctr_nchannels; + +/* A send marshalled to the cluster worker over IPC. Routing every send + * through the worker keeps the anti-replay sequence space single-writer + * (cl_ctr_check_and_update_seq is strictly monotonic per sender IP, so + * concurrent senders in different processes would trip it) and reuses + * the worker's socket and session key, which are worker-local state. */ +struct cl_ctr_consumer_job { + cl_ctr_cluster_t *cl; + uint16_t dst_node_id; /* 0 = multicast */ + int flags; + int chan_len; + int payload_len; + char chan[CLCTR_MAX_CHAN_LEN]; + unsigned char payload[]; +}; + +static void cl_ctr_consumer_dispatch(cl_ctr_cluster_t *cl, int src_node_id, + const char *chan, int chan_len, const char *payload, int payload_len); +static void cl_ctr_handle_consumer(const char *payload, int payload_len, + const char *src_ip, cl_ctr_cluster_t *cl); +static void cl_ctr_rpc_consumer_send(int sender, void *param); +static int clctr_register_channel(str *channel, clctr_msg_cb_f cb); +static int cl_ctr_script_init(void); +static int cmd_cl_ctr_broadcast_req(struct sip_msg *msg, int *cluster_id, + str *gen_msg, str *tag, int *reliable); +static int cmd_cl_ctr_send_req(struct sip_msg *msg, int *cluster_id, + int *node_id, str *gen_msg, str *tag, int *reliable); +static int cmd_cl_ctr_send_rpl(struct sip_msg *msg, int *cluster_id, + int *node_id, str *gen_msg, str *tag); +static int cmd_cl_ctr_send_req_list(struct sip_msg *msg, int *cluster_id, + pv_spec_t *nodes, str *gen_msg, str *tag, pv_spec_t *out); +static int clctr_send_mcast(int cluster_id, str *channel, str *payload, + int flags); +static int clctr_send_ucast(int cluster_id, int node_id, str *channel, + str *payload, int flags); +static int clctr_send_list(int cluster_id, const int *node_ids, int n, + str *channel, str *payload, int flags, + int *unknown); +static int clctr_get_my_node_id(int cluster_id); +int cl_ctr_get_my_ip(const char **ip, const char **iface, const char **src); +int load_clctr(clctr_api_t *api); + /* clusterer integration - loaded at mod_init if clusterer use_controller=1 */ static clusterer_ctrl_binds_t clctl; static int clctl_loaded = 0; static int manage_shtags = 1; +static int consumer_retries = CL_CTR_CONSUMER_RETRIES_DEFAULT; +static int consumer_retry_ms = CL_CTR_CONSUMER_RETRY_MS_DEFAULT; /* master_stickiness (global default; per-cluster override via "cluster" string): * 1 (default) = the master is "sticky": a live master keeps the role and is * NOT displaced when a higher-IP node joins. The highest-IP @@ -712,6 +843,34 @@ static int manage_shtags = 1; * 0 = not sticky - pure highest-IP election, so a higher-IP node * takes over as master as soon as it appears (more handovers). */ static int master_stickiness = 1; +static int consumer_rate_limit = CL_CTR_CONSUMER_RATE_DEFAULT; + +/* ---- the script's own channel (S2c-API, second tier) ------------------- + * + * The messaging API's other half: a channel the module registers for + * itself, so a script can exchange messages with its peers without a + * module in between. It mirrors clusterer's generic messaging exactly - + * same three functions, same two events, same parameters - so a script + * moves between them by renaming, and the reason to move is what the + * transport underneath does: one multicast packet rather than a send per + * peer, over an encrypted channel rather than a plaintext one. + * + * Wire: [u8 kind][u8 tag_len][tag][message]. + */ +#define CL_CTR_SCRIPT_REQ 1 +#define CL_CTR_SCRIPT_RPL 2 + +static str cl_ctr_script_chan = str_init("_script"); +static str ei_req_name = str_init("E_CL_CTR_REQ_RECEIVED"); +static str ei_rpl_name = str_init("E_CL_CTR_RPL_RECEIVED"); +static event_id_t ei_req_id = EVI_ERROR; +static event_id_t ei_rpl_id = EVI_ERROR; +static evi_params_p ei_params; +static evi_param_p ei_clid_p, ei_srcid_p, ei_msg_p, ei_tag_p; +static str ei_clid_pname = str_init("cluster_id"); +static str ei_srcid_pname = str_init("src_id"); +static str ei_msg_pname = str_init("msg"); +static str ei_tag_pname = str_init("tag"); static char my_bin_sockets[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; static int my_bin_count = 0; @@ -746,6 +905,9 @@ static const param_export_t params[] = { {"interface", STR_PARAM, &my_interface}, {"query_time", INT_PARAM, &query_time}, {"password", STR_PARAM, &password}, + {"consumer_rate_limit", INT_PARAM, &consumer_rate_limit}, + {"consumer_retries", INT_PARAM, &consumer_retries}, + {"consumer_retry_ms", INT_PARAM, &consumer_retry_ms}, {"manage_shtags", INT_PARAM, &manage_shtags}, {"master_stickiness", INT_PARAM, &master_stickiness}, {"on_config_mismatch", STR_PARAM, &on_config_mismatch_s}, @@ -769,6 +931,13 @@ typedef struct cl_ctr_peer_ { unsigned char pubkey[CL_CTR_PUBKEY_SZ]; /* long-lived X25519 pubkey (from ALIVE); zero if unknown; used for KEY_HANDOFF */ uint32_t last_seq; /* highest seq accepted from this peer */ + /* Consumer traffic is counted separately from the control plane. They + * share a session key and a socket but not a sequence space: a consumer + * may send thousands of packets a second where the control plane sends a + * handful, and one counter for both means a reordered consumer packet can + * make a MASTER_ALIVE arriving behind it look like a replay - which is a + * missed liveness beacon, not a dropped cache reply. */ + uint32_t last_consumer_seq; /* Peer's advertised consistency-critical config (from ALIVE), used to warn * on accidental per-node config drift. cfg_known=0 until first advertised; * cfg_warned deduplicates the mismatch warning. */ @@ -796,6 +965,7 @@ struct cl_ctr_peers_ { * GOODBYE without needing the worker's private state. Reset to 0 on * every session key rotation so last_seq counters reset cleanly. */ uint32_t my_seq; + uint32_t my_consumer_seq; /* the consumer plane's own counter */ /* Sharing-tag override: 0 = automatic (master-driven) allocation; nonzero = * an operator has forced this node_id to be the active shtag holder for the * cluster (cl_ctr_shtag_force MI), suspending automatic allocation until @@ -849,6 +1019,16 @@ static void mod_destroy(void); static void cl_ctr_worker(int rank); static int cl_ctr_on_sock(int fd, void *param, int was_timeout); static void cl_ctr_retx_flush(cl_ctr_cluster_t *cl); +static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, + unsigned char type, const unsigned char *pkt, + int pkt_len, const struct sockaddr *dest, + socklen_t destlen); +static void cl_ctr_retx_enqueue_bcast(cl_ctr_cluster_t *cl, uint32_t seq, + const unsigned char *pkt, int pkt_len); +static void cl_ctr_retx_enqueue_consumer(cl_ctr_cluster_t *cl, uint32_t seq, + unsigned char type, const unsigned char *pkt, + int pkt_len, const struct sockaddr *dest, + socklen_t destlen); static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout); static void cl_ctr_membership_digest(cl_ctr_cluster_t *cl, uint16_t *count, uint64_t *hash); static void cl_ctr_send_resync(int sock, cl_ctr_cluster_t *cl, @@ -897,8 +1077,12 @@ static mi_response_t *mi_cl_ctr_shtag_auto(const mi_params_t *params, * ========================================================================= */ static proc_export_t procs[] = { + /* NEEDS_SCRIPT because a message arriving for the script's channel + * raises an event here, and an event route is script: without it the + * route structures are not set up in this process and running one + * crashes. */ {"clusterer_controller worker", 0, 0, cl_ctr_worker, 1, - PROC_FLAG_INITCHILD | PROC_FLAG_HAS_IPC}, + PROC_FLAG_INITCHILD | PROC_FLAG_HAS_IPC | PROC_FLAG_NEEDS_SCRIPT}, {0, 0, 0, 0, 0, 0} }; @@ -1241,6 +1425,20 @@ CL_CTR_PV_WRAP(cl_ctr_pv_is_master, CL_CTR_PV_IS_MASTER) CL_CTR_PV_WRAP(cl_ctr_pv_master_ip, CL_CTR_PV_MASTER_IP) CL_CTR_PV_WRAP(cl_ctr_pv_backup_ip, CL_CTR_PV_BACKUP_IP) CL_CTR_PV_WRAP(cl_ctr_pv_node_id, CL_CTR_PV_NODE_ID) +/* clctr_api.get_my_ip - see api.h */ +int cl_ctr_get_my_ip(const char **ip, const char **iface, const char **src) +{ + if (!my_ip) + return -1; /* mod_init has not resolved it yet */ + if (ip) + *ip = my_ip; + if (iface) + *iface = my_interface_buf[0] ? my_interface_buf : "(unknown)"; + if (src) + *src = my_ip_src; + return 0; +} + CL_CTR_PV_WRAP(cl_ctr_pv_my_ip, CL_CTR_PV_MY_IP) CL_CTR_PV_WRAP(cl_ctr_pv_members, CL_CTR_PV_MEMBERS) CL_CTR_PV_WRAP(cl_ctr_pv_shtag_mode, CL_CTR_PV_SHTAG_MODE) @@ -1278,6 +1476,32 @@ static const cmd_export_t cl_ctr_cmds[] = { {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_VAR, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, {"cl_ctr_get_node_ip", (cmd_function)w_cl_ctr_get_node_ip, { {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_INT, 0, 0}, {CMD_PARAM_VAR, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {"cl_ctr_broadcast_req", (cmd_function)cmd_cl_ctr_broadcast_req, { + {CMD_PARAM_INT,0,0}, + {CMD_PARAM_STR,0,0}, + {CMD_PARAM_STR|CMD_PARAM_OPT,0,0}, + {CMD_PARAM_INT|CMD_PARAM_OPT,0,0}, {0,0,0}}, + ALL_ROUTES}, + {"cl_ctr_send_req", (cmd_function)cmd_cl_ctr_send_req, { + {CMD_PARAM_INT,0,0}, + {CMD_PARAM_INT,0,0}, + {CMD_PARAM_STR,0,0}, + {CMD_PARAM_STR|CMD_PARAM_OPT,0,0}, + {CMD_PARAM_INT|CMD_PARAM_OPT,0,0}, {0,0,0}}, + ALL_ROUTES}, + {"cl_ctr_send_req_list", (cmd_function)cmd_cl_ctr_send_req_list, { + {CMD_PARAM_INT, 0, 0}, + {CMD_PARAM_VAR, 0, 0}, + {CMD_PARAM_STR, 0, 0}, + {CMD_PARAM_STR|CMD_PARAM_OPT, 0, 0}, + {CMD_PARAM_VAR|CMD_PARAM_OPT, 0, 0}, {0, 0, 0}}, ALL_ROUTES}, + {"cl_ctr_send_rpl", (cmd_function)cmd_cl_ctr_send_rpl, { + {CMD_PARAM_INT,0,0}, + {CMD_PARAM_INT,0,0}, + {CMD_PARAM_STR,0,0}, + {CMD_PARAM_STR|CMD_PARAM_OPT,0,0}, {0,0,0}}, + ALL_ROUTES}, + {"load_clctr", (cmd_function)load_clctr, {{0, 0, 0}}, 0}, {0, 0, {{0, 0, 0}}, 0} }; @@ -2070,8 +2294,11 @@ static int cl_ctr_derive_session_key(cl_ctr_cluster_t *cl) /* Reset sequence counters: old packets encrypted with the previous key * fail AEAD authentication, so starting from 0 is safe. */ cl->peers->my_seq = 0; - for (i = 0; i < cl->peers->count; i++) + cl->peers->my_consumer_seq = 0; + for (i = 0; i < cl->peers->count; i++) { cl->peers->entries[i].last_seq = 0; + cl->peers->entries[i].last_consumer_seq = 0; + } cl->have_session_key = 1; /* a valid group key now exists */ /* The salt (and my_seq) just changed, so any queued retransmit is now stale. */ cl_ctr_retx_flush(cl); @@ -2261,17 +2488,30 @@ static int cl_ctr_decrypt_pkt(char *buf, ssize_t n, const char *sender_ip, * @return 0 to accept, -1 to drop. */ static int cl_ctr_check_and_update_seq(const char *sender_ip, uint32_t pkt_seq, - cl_ctr_cluster_t *cl) + cl_ctr_cluster_t *cl, int is_consumer) { int i; for (i = 0; i < cl->peers->count; i++) { if (strcmp(cl->peers->entries[i].ip, sender_ip) == 0) { - if (pkt_seq <= cl->peers->entries[i].last_seq) { - LM_WARN("clusterer_controller: replay from %s seq=%u last=%u, dropping\n", - sender_ip, pkt_seq, cl->peers->entries[i].last_seq); + uint32_t *last = is_consumer + ? &cl->peers->entries[i].last_consumer_seq + : &cl->peers->entries[i].last_seq; + + if (pkt_seq <= *last) { + /* Debug for consumer traffic, warning for the control plane. + * A consumer sending at rate will reorder on any network with + * more than one path, and a warning per reordered packet says + * "attack" about something entirely ordinary. */ + if (is_consumer) + LM_DBG("clusterer_controller: consumer packet from %s out " + "of order seq=%u last=%u, dropping\n", + sender_ip, pkt_seq, *last); + else + LM_WARN("clusterer_controller: replay from %s seq=%u " + "last=%u, dropping\n", sender_ip, pkt_seq, *last); return -1; } - cl->peers->entries[i].last_seq = pkt_seq; + *last = pkt_seq; return 0; } } @@ -2517,6 +2757,86 @@ static void cl_ctr_retx_flush(cl_ctr_cluster_t *cl) * retransmit until ACKed. Best-effort: if the queue is full the packet still * went out once and the joiner's JOIN_REQ retry remains the backstop. */ +/* Track a reliable broadcast: one entry, the members we expect to hear from, + * and the sealed bytes to repair with. The expected count is a snapshot - a + * node that joins a moment later never saw the message and is not owed one. */ +static void cl_ctr_retx_enqueue_bcast(cl_ctr_cluster_t *cl, uint32_t seq, + const unsigned char *pkt, int pkt_len) +{ + cl_ctr_retx_entry_t *e = NULL; + int i, n = 0; + + if (pkt_len <= 0 || pkt_len > (int)sizeof(cl->retx_q[0].pkt)) + return; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) + if (cl->peers->entries[i].node_id > 0 && + cl->peers->entries[i].node_id <= CL_CTR_MAX_PEERS) + n++; + lock_stop_read(cl->peers->lock); + + if (n == 0) { + LM_DBG("clusterer_controller: [cluster %d] reliable broadcast with no " + "peers to acknowledge it - nothing to wait for\n", + cl->cluster_id); + return; + } + + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) + if (!cl->retx_q[i].used) { e = &cl->retx_q[i]; break; } + if (!e) { + LM_DBG("clusterer_controller: [cluster %d] retransmit queue full - the " + "broadcast went out once, unacknowledged\n", cl->cluster_id); + return; + } + + memset(e, 0, sizeof(*e)); + e->used = 1; + e->seq = seq; + e->type = CL_CTR_PKT_CONSUMER_REL; + e->is_bcast = 1; + e->expect_n = (uint16_t)n; + e->retries_left = cl->consumer_retries; + e->retries_cfg = (uint8_t)cl->consumer_retries; + e->next_due_us = get_uticks() + (utime_t)cl->consumer_retry_ms * 1000; + e->pkt_len = pkt_len; + memcpy(e->pkt, pkt, pkt_len); + cl->retx_count++; + /* First argument is the delay, and zero there means disarm - the value + * this once passed, which switched the retransmit timer off instead of + * on and left every reliable broadcast waiting for a repair that could + * never run. */ + cl_ctr_arm_tfd_us(cl->retx_tfd, + (utime_t)cl->consumer_retry_ms * 1000, 0); +} + +/* The consumer plane retransmits on its own schedule: the join handshake's + * budget is pinned to the JOIN_REQ retry interval, which has nothing to say + * about how long a consumer should wait. Both are per-cluster, because one + * cluster may cross a link where another does not. */ +static void cl_ctr_retx_enqueue_consumer(cl_ctr_cluster_t *cl, uint32_t seq, + unsigned char type, const unsigned char *pkt, + int pkt_len, const struct sockaddr *dest, + socklen_t destlen) +{ + int i; + + cl_ctr_retx_enqueue(cl, seq, type, pkt, pkt_len, dest, destlen); + /* Re-arm the entry we just made with this cluster's consumer budget. + * Enqueue does not take them as arguments because every other caller + * wants the handshake numbers. */ + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) + if (cl->retx_q[i].used && cl->retx_q[i].seq == seq && + cl->retx_q[i].type == type) { + cl->retx_q[i].retries_left = cl->consumer_retries; + cl->retx_q[i].retries_cfg = (uint8_t)cl->consumer_retries; + cl->retx_q[i].next_due_us = get_uticks() + + (utime_t)cl->consumer_retry_ms * 1000; + break; + } +} + static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned char type, const unsigned char *pkt, int pkt_len, const struct sockaddr *dest, socklen_t destlen) @@ -2554,7 +2874,8 @@ static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned cha } /* An ACK arrived: drop the queued packet whose seq it echoes. */ -static void cl_ctr_handle_ack(const char *payload, int payload_len, cl_ctr_cluster_t *cl) +static void cl_ctr_handle_ack(const char *payload, int payload_len, + cl_ctr_cluster_t *cl, const char *sender_ip) { uint32_t acked_be, acked; int i; @@ -2568,10 +2889,46 @@ static void cl_ctr_handle_ack(const char *payload, int payload_len, cl_ctr_clust cl_ctr_retx_entry_t *e = &cl->retx_q[i]; if (!e->used || e->seq != acked) continue; - /* seq is unique within a key epoch (a rekey flushes the queue), so this - * is the acknowledged packet; drop it. */ - LM_DBG("clusterer_controller: [cluster %d] ACK for 0x%02x seq %u\n", - cl->cluster_id, e->type, acked); + + if (e->is_bcast) { + /* Every member acknowledges the same sequence number, so this + * entry lives until they all have (or the budget runs out). */ + uint16_t nid = 0; + + lock_start_read(cl->peers->lock); + { + cl_ctr_peer_t *p = cl_ctr_peer_by_ip_locked(cl, sender_ip); + if (p) + nid = p->node_id; + } + lock_stop_read(cl->peers->lock); + + if (nid > 0 && nid <= CL_CTR_MAX_PEERS) { + int byte = (nid - 1) / 8, bit = 1 << ((nid - 1) % 8); + + if (!(e->acked_map[byte] & bit)) { + e->acked_map[byte] |= bit; + e->acked_n++; + } + } + if (e->acked_n < e->expect_n) { + LM_DBG("clusterer_controller: [cluster %d] broadcast seq %u " + "acknowledged by %u of %u\n", cl->cluster_id, acked, + e->acked_n, e->expect_n); + break; /* still waiting on others */ + } + LM_DBG("clusterer_controller: [cluster %d] broadcast seq %u " + "acknowledged by all %u member(s) after %u of the %u " + "retries configured for this cluster\n", cl->cluster_id, + acked, e->expect_n, + (unsigned)(e->retries_cfg - e->retries_left), + (unsigned)e->retries_cfg); + } else { + /* seq is unique within a key epoch (a rekey flushes the queue), so + * this is the acknowledged packet; drop it. */ + LM_DBG("clusterer_controller: [cluster %d] ACK for 0x%02x seq %u\n", + cl->cluster_id, e->type, acked); + } e->used = 0; cl->retx_count--; break; @@ -2599,16 +2956,63 @@ static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout) if (!e->used || now < e->next_due_us) continue; - if (sendto(cl->sock, e->pkt, e->pkt_len, 0, + if (e->is_bcast) { + /* Repair, not rebroadcast. Sending the multicast again would + * reach the members that already have it, and - because a + * duplicate that asked to be acknowledged is acknowledged again - + * every one of them would answer a second time. On a large + * cluster the repair for two missing nodes would cost a packet to + * everyone and an ACK back from everyone. So the repair is + * unicast, to exactly the nodes that still owe an answer. */ + int j, repaired = 0; + + lock_start_read(cl->peers->lock); + for (j = 0; j < cl->peers->count; j++) { + cl_ctr_peer_t *p = &cl->peers->entries[j]; + struct sockaddr_in d; + int byte, bit; + + if (p->node_id == 0 || p->node_id > CL_CTR_MAX_PEERS) + continue; + byte = (p->node_id - 1) / 8; + bit = 1 << ((p->node_id - 1) % 8); + if (e->acked_map[byte] & bit) + continue; /* this one answered */ + + memset(&d, 0, sizeof d); + d.sin_family = AF_INET; + d.sin_port = cl->mcast_dest.sin_port; + if (inet_pton(AF_INET, p->ip, &d.sin_addr) != 1) + continue; + if (sendto(cl->sock, e->pkt, e->pkt_len, 0, + (struct sockaddr *)&d, sizeof d) >= 0) + repaired++; + } + lock_stop_read(cl->peers->lock); + LM_DBG("clusterer_controller: [cluster %d] broadcast seq %u: " + "repairing %d node(s) that have not acknowledged\n", + cl->cluster_id, e->seq, repaired); + + } else if (sendto(cl->sock, e->pkt, e->pkt_len, 0, (struct sockaddr *)&e->dest, e->destlen) < 0 && - errno != EAGAIN && errno != EWOULDBLOCK) + errno != EAGAIN && errno != EWOULDBLOCK) { LM_DBG("clusterer_controller: [cluster %d] retransmit 0x%02x: %s\n", cl->cluster_id, e->type, strerror(errno)); + } if (--e->retries_left <= 0) { - LM_DBG("clusterer_controller: [cluster %d] 0x%02x (seq %u) unacked " - "after %d retransmits, giving up - joiner will re-JOIN_REQ\n", - cl->cluster_id, e->type, e->seq, CL_CTR_RETX_MAX_RETRIES); + if (e->is_bcast) + LM_INFO("clusterer_controller: [cluster %d] broadcast seq %u " + "reached %u of %u member(s), giving up after %u of the " + "%u retries configured for this cluster\n", + cl->cluster_id, e->seq, e->acked_n, e->expect_n, + (unsigned)e->retries_cfg, (unsigned)e->retries_cfg); + else + LM_DBG("clusterer_controller: [cluster %d] 0x%02x (seq %u) unacked " + "after %u retransmits, giving up - joiner will re-JOIN_REQ\n", + cl->cluster_id, e->type, e->seq, + (unsigned)(e->retries_cfg ? e->retries_cfg + : CL_CTR_RETX_MAX_RETRIES)); e->used = 0; cl->retx_count--; } else { @@ -3662,8 +4066,10 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le * ALIVE, not here - the JOIN_REQ now carries only an ephemeral Noise key. */ { cl_ctr_peer_t *e = cl_ctr_peer_by_ip_locked(cl, src_ip); - if (e) + if (e) { e->last_seq = 0; + e->last_consumer_seq = 0; + } } lock_stop_write(cl->peers->lock); @@ -3849,6 +4255,7 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, for (_j = 0; _j < cl->peers->count; _j++) { if (strcmp(cl->peers->entries[_j].ip, ip_buf) == 0) { cl->peers->entries[_j].last_seq = 0; + cl->peers->entries[_j].last_consumer_seq = 0; break; } } @@ -4491,26 +4898,84 @@ static void cl_ctr_handle_key_handoff(const char *payload, int payload_len, * Finds or creates a 1-second sliding-window counter for src_ip. * @return 0 if within CL_CTR_RATE_LIMIT packets/s, -1 to drop. */ -static int cl_ctr_rate_check(cl_ctr_cluster_t *cl, uint32_t src_ip) +/** + * cl_ctr_consumer_src_known() - is this address one of our cluster members? + * + * Consumer traffic only ever passes between nodes that have already joined: + * a module registers a channel and talks to its peers, and nothing legitimate + * sends consumer packets before it is a member. Control traffic is different + * - a JOIN_REQ necessarily comes from a stranger - so this filter is applied + * to consumer packets only. + * + * It runs before the rate limiter and therefore before any crypto, which is + * the point: a flood from an address we have never heard of costs one scan of + * a table bounded by the cluster size, and does not reach the AEAD, does not + * consume a rate-table slot, and cannot evict a real peer's counter from it. + * Our own address passes, because multicast comes back to its sender. + * + * A spoofed source that copies a member's address still gets through - this + * filter cannot fix address spoofing, and does not claim to. What it removes + * is the much easier attack of pointing a flood at the port from anywhere. + */ +static int cl_ctr_consumer_src_known(cl_ctr_cluster_t *cl, uint32_t src_ip) +{ + static uint32_t mine; /* my_ip in host order, resolved once */ + uint32_t src_num = ntohl(src_ip); /* peers keep host order */ + int i, known = 0; + + if (!mine && my_ip) + mine = ip_to_num(my_ip); + if (src_num == mine) + return 1; + + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) + if (cl->peers->entries[i].ip_num == src_num) { + known = 1; + break; + } + lock_stop_read(cl->peers->lock); + + return known; +} + +static int cl_ctr_rate_check(cl_ctr_cluster_t *cl, uint32_t src_ip, + int is_consumer) { time_t now = time(NULL); cl_ctr_rate_entry_t *oldest = NULL; + int limit = is_consumer ? cl->consumer_rate + : CL_CTR_RATE_LIMIT; int i; for (i = 0; i < CL_CTR_RATE_TBL_SZ; i++) { cl_ctr_rate_entry_t *e = &cl->rate_tbl[i]; + int *cnt; + if (e->ip == 0) { if (!oldest) oldest = e; /* prefer empty slot */ continue; } if (e->ip == src_ip) { + cnt = is_consumer ? &e->consumer_count : &e->count; if (now > e->window_start) { /* new second */ - e->window_start = now; - e->count = 1; + e->window_start = now; + e->count = 0; + e->consumer_count = 0; + *cnt = 1; return 0; } - if (++e->count > CL_CTR_RATE_LIMIT) + if (++(*cnt) > limit) { + /* Say so, once per window per source: a silent drop here + * looks exactly like packet loss to whatever was sending, + * which is a miserable thing to debug. */ + if (*cnt == limit + 1) + LM_WARN("clusterer_controller: [cluster %d] %s rate " + "limit of %d/s exceeded by a peer - dropping\n", + cl->cluster_id, + is_consumer ? "consumer" : "control", limit); return -1; + } return 0; } /* track oldest entry for eviction when table is full */ @@ -4519,9 +4984,14 @@ static int cl_ctr_rate_check(cl_ctr_cluster_t *cl, uint32_t src_ip) } /* new source IP - claim oldest/empty slot */ - oldest->ip = src_ip; - oldest->window_start = now; - oldest->count = 1; + oldest->ip = src_ip; + oldest->window_start = now; + oldest->count = 0; + oldest->consumer_count = 0; + if (is_consumer) + oldest->consumer_count = 1; + else + oldest->count = 1; return 0; } @@ -4709,7 +5179,8 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, } if (memcmp(buf, CL_CTR_PACKET_MAGIC, CL_CTR_MAGIC_SZ) != 0 && - memcmp(buf, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ) != 0) { + memcmp(buf, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ) != 0 && + memcmp(buf, CL_CTR_CONSUMER_MAGIC, CL_CTR_MAGIC_SZ) != 0) { LM_DBG("clusterer_controller: bad magic, dropping\n"); return; } @@ -4737,9 +5208,24 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, } } - /* Rate-limit before any crypto work to shed floods cheaply. */ - if (cl_ctr_rate_check(cl, src_addr.sin_addr.s_addr) < 0) - return; + /* Rate-limit before any crypto work to shed floods cheaply. Consumer + * traffic from an address that is not a member is dropped one step + * earlier still - it cannot be legitimate, and dropping it here keeps it + * out of the rate table as well as out of the cipher. */ + { + int is_consumer = (memcmp(buf, CL_CTR_CONSUMER_MAGIC, + CL_CTR_MAGIC_SZ) == 0); + + if (is_consumer && !cl_ctr_consumer_src_known(cl, + src_addr.sin_addr.s_addr)) { + if (cl->peers->count > 0) /* quiet while still forming */ + LM_DBG("clusterer_controller: [cluster %d] consumer packet " + "from a non-member, dropping\n", cl->cluster_id); + return; + } + if (cl_ctr_rate_check(cl, src_addr.sin_addr.s_addr, is_consumer) < 0) + return; + } /* Resolve sender IP once - used for HMAC warning and MEMBER_LIST dispatch */ { @@ -4753,6 +5239,7 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, * CL_CTR_PACKET_MAGIC -> session key (all normal traffic) * If session key decryption fails, schedule a re-JOIN to refresh it. */ int is_bootstrap = (memcmp(buf, CL_CTR_BOOTSTRAP_MAGIC, CL_CTR_MAGIC_SZ) == 0); + uint32_t pkt_seq = 0; /* needed again below, to acknowledge by seq */ dec_key = is_bootstrap ? cl->key : cl->session_key; if (cl_ctr_decrypt_pkt(buf, n, sender_ip_buf, dec_key, is_bootstrap) < 0) { @@ -4822,11 +5309,31 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, * without any special-casing. * Bootstrap packets (CL_CTR_BOOTSTRAP_MAGIC) use join_nonce instead. */ if (!is_bootstrap) { - uint32_t pkt_seq; + /* The type byte, not the cleartext magic: this runs after the AEAD, + * so the byte is authenticated and the magic is only the hint the + * rate limiter needed before there was anything to trust. */ + /* Both flavours belong to the consumer plane - the reliable one is + * still consumer traffic, and counting it against the control plane + * would put back exactly the mixing the split removed. */ + unsigned char _t = (unsigned char)buf[CL_CTR_WIRE_HDR_SZ]; + int is_consumer_pkt = (_t == CL_CTR_PKT_CONSUMER || + _t == CL_CTR_PKT_CONSUMER_REL); + memcpy(&pkt_seq, buf + CL_CTR_WIRE_HDR_SZ + 1, CL_CTR_SEQ_SZ); pkt_seq = ntohl(pkt_seq); - if (cl_ctr_check_and_update_seq(sender_ip_buf, pkt_seq, cl) < 0) + if (cl_ctr_check_and_update_seq(sender_ip_buf, pkt_seq, cl, + is_consumer_pkt) < 0) { + /* A duplicate of a message that asked to be acknowledged means + * our acknowledgement did not arrive: say it again. Without + * this the sender spends its whole retransmit budget against a + * receiver that has had the message all along and is dropping + * every copy in silence. */ + if ((unsigned char)buf[CL_CTR_WIRE_HDR_SZ] + == CL_CTR_PKT_CONSUMER_REL) + cl_ctr_send_ack(cl->sock, cl, pkt_seq, 0, + (const struct sockaddr *)&src_addr, src_len); return; + } } pkt_type = (unsigned char)buf[CL_CTR_WIRE_HDR_SZ]; @@ -4861,6 +5368,24 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, break; } + case CL_CTR_PKT_CONSUMER_REL: + /* Acknowledge first, deliver second: the sender is holding a + * retransmit slot on this, and delivery cannot fail in a way the + * acknowledgement should report - the consumer's own callback owns + * what happens next. */ + cl_ctr_send_ack(cl->sock, cl, pkt_seq, 0/*session key*/, + (const struct sockaddr *)&src_addr, src_len); + cl_ctr_handle_consumer(payload, payload_len, sender_ip_buf, cl); + break; + + case CL_CTR_PKT_CONSUMER: + /* session-key only: a consumer message under the bootstrap key + * would come from a node that has not even joined - drop it */ + if (is_bootstrap) + break; + cl_ctr_handle_consumer(payload, payload_len, sender_ip_buf, cl); + break; + case CL_CTR_PKT_JOIN_REQ: cl_ctr_handle_join_req(sock, payload, payload_len, cl, (const struct sockaddr *)&src_addr, src_len); @@ -4902,7 +5427,7 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, } case CL_CTR_PKT_ACK: - cl_ctr_handle_ack(payload, payload_len, cl); + cl_ctr_handle_ack(payload, payload_len, cl, sender_ip_buf); break; case CL_CTR_PKT_KEY_HANDOFF: @@ -4972,8 +5497,18 @@ static void cl_ctr_maybe_forward(const char *buf, int n, /* Shed floods before allocating: charge the source against our own limiter * (the target re-checks after it receives the forward). */ - if (cl_ctr_rate_check(from, src->sin_addr.s_addr) < 0) - return; + { + int is_consumer = (memcmp(buf, CL_CTR_CONSUMER_MAGIC, + CL_CTR_MAGIC_SZ) == 0); + + /* Judged against the target cluster's membership, not ours - the + * packet is about to be handed to that cluster's worker. */ + if (is_consumer && !cl_ctr_consumer_src_known(target, + src->sin_addr.s_addr)) + return; + if (cl_ctr_rate_check(from, src->sin_addr.s_addr, is_consumer) < 0) + return; + } /* worker_proc_no is written once at worker fork and stable thereafter. */ proc_no = target->peers ? target->peers->worker_proc_no : -1; @@ -5942,6 +6477,9 @@ static int cl_ctr_parse_cluster_str(const char *str, cl_ctr_cluster_t *cl) strncpy(cl->password, password, sizeof(cl->password) - 1); cl->password[sizeof(cl->password) - 1] = '\0'; cl->manage_shtags = -1; /* sentinel: inherit global default in mod_init */ + cl->consumer_retries = -1; + cl->consumer_retry_ms = -1; + cl->consumer_rate = -1; cl->master_stickiness = -1; /* sentinel: inherit global default in mod_init */ for (tok = strtok_r(buf, ",", &p); tok; tok = strtok_r(NULL, ",", &p)) { @@ -5995,6 +6533,30 @@ static int cl_ctr_parse_cluster_str(const char *str, cl_ctr_cluster_t *cl) strncpy(cl->bin_socket, val, CL_CTR_MAX_BIN_SOCK_LEN - 1); cl->bin_socket[CL_CTR_MAX_BIN_SOCK_LEN - 1] = '\0'; + } else if (strcmp(key, "consumer_retries") == 0) { + cl->consumer_retries = atoi(val); + if (cl->consumer_retries < 0 || cl->consumer_retries > 10) { + LM_ERR("clusterer_controller: consumer_retries must be 0..10 " + "in '%s'\n", str); + return -1; + } + + } else if (strcmp(key, "consumer_retry_ms") == 0) { + cl->consumer_retry_ms = atoi(val); + if (cl->consumer_retry_ms < 5 || cl->consumer_retry_ms > 5000) { + LM_ERR("clusterer_controller: consumer_retry_ms must be " + "5..5000 in '%s'\n", str); + return -1; + } + + } else if (strcmp(key, "consumer_rate_limit") == 0) { + cl->consumer_rate = atoi(val); + if (cl->consumer_rate < 1) { + LM_ERR("clusterer_controller: consumer_rate_limit must be " + "positive in '%s'\n", str); + return -1; + } + } else if (strcmp(key, "manage_shtags") == 0) { cl->manage_shtags = atoi(val) ? 1 : 0; @@ -6051,6 +6613,7 @@ static int cl_ctr_resolve_local_identity(void) freeifaddrs(ifap); return -1; } + my_ip_src = "my_ip modparam (explicit)"; LM_INFO("clusterer_controller: using IP %s on interface %s\n", my_ip, my_interface_buf); @@ -6089,6 +6652,7 @@ static int cl_ctr_resolve_local_identity(void) else LM_INFO("clusterer_controller: using IP %s on interface %s\n", my_ip, my_interface_buf); + my_ip_src = "interface modparam (IPv4 of that interface)"; } else { /* ---- Mode 3: neither - auto-detect via kernel routing table ---- */ @@ -6127,6 +6691,7 @@ static int cl_ctr_resolve_local_identity(void) inet_ntop(AF_INET, &local.sin_addr, my_ip_buf, sizeof(my_ip_buf)); my_ip = my_ip_buf; + my_ip_src = "auto-detected via default route to the multicast group"; /* Reverse-look up the interface name */ for (ifa = ifap; ifa; ifa = ifa->ifa_next) { @@ -6219,6 +6784,13 @@ static int mod_init(void) { struct in_addr addr; int i, j; + /* the script's messaging channel and its two events; done here so the + * reserved name is claimed before any consumer module registers */ + if (cl_ctr_script_init() < 0) { + LM_ERR("clusterer_controller: cannot set up script messaging\n"); + return -1; + } + LM_INFO("clusterer_controller: initialising\n"); @@ -6320,6 +6892,36 @@ static int mod_init(void) if (cl_ctr_resolve_local_identity() < 0) return -1; + /* Derive max consumer payload from the interface MTU so jumbo-frame + * links (e.g. MTU 9000) are not capped at the 1500-byte default. + * Overhead per consumer datagram: + * IP(20) + UDP(8) + wire_hdr(28) + plain_hdr(5) + + * consumer_hdr(3) + max_chan(31) + poly1305_tag(16) = 111 bytes. */ + if (my_interface_buf[0] != '\0') { + struct ifreq _mtu_ifr; + int _mtu_sock = socket(AF_INET, SOCK_DGRAM, 0); + if (_mtu_sock >= 0) { + memset(&_mtu_ifr, 0, sizeof(_mtu_ifr)); + memcpy(_mtu_ifr.ifr_name, my_interface_buf, strnlen(my_interface_buf, IF_NAMESIZE - 1)); + if (ioctl(_mtu_sock, SIOCGIFMTU, &_mtu_ifr) == 0) { + int _mtu = _mtu_ifr.ifr_mtu; + int _overhead = 20 + 8 + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + + CL_CTR_CONSUMER_HDR_SZ + CLCTR_MAX_CHAN_LEN + CL_CTR_TAG_SZ; + cc_max_payload = _mtu - _overhead; + if (cc_max_payload < CLCTR_MAX_PAYLOAD) + cc_max_payload = CLCTR_MAX_PAYLOAD; + LM_INFO("clusterer_controller: interface %s MTU=%d, " + "max consumer payload=%d bytes\n", + my_interface_buf, _mtu, cc_max_payload); + } else { + LM_WARN("clusterer_controller: SIOCGIFMTU on %s failed: %s - " + "using default max payload %d\n", + my_interface_buf, strerror(errno), cc_max_payload); + } + close(_mtu_sock); + } + } + if (cl_ctr_discover_bin_sockets() < 0) return -1; @@ -6360,6 +6962,12 @@ static int mod_init(void) cl->master_stickiness = master_stickiness ? 1 : 0; if (cl->manage_shtags == -1) cl->manage_shtags = manage_shtags ? 1 : 0; + if (cl->consumer_retries == -1) + cl->consumer_retries = consumer_retries; + if (cl->consumer_retry_ms == -1) + cl->consumer_retry_ms = consumer_retry_ms; + if (cl->consumer_rate == -1) + cl->consumer_rate = consumer_rate_limit; /* Resolve which BIN socket to use for this cluster. * Priority: explicit bin_socket= in cluster string > @@ -6607,3 +7215,582 @@ static void mod_destroy(void) } LM_INFO("clusterer_controller: shut down\n"); } + +/* ========================================================================= + * Consumer messaging API (api.h) + * + * Other modules ride the controller's encrypted UDP plane: named + * channels, multicast or per-node unicast, delivered to the registered + * callback in the cluster worker. See api.h for the delivery contract. + * ========================================================================= */ + +static cl_ctr_cluster_t *cl_ctr_cluster_by_id(int cluster_id) +{ + int i; + + for (i = 0; i < cl_ctr_cluster_count; i++) + if (cl_ctr_clusters[i].cluster_id == cluster_id) + return &cl_ctr_clusters[i]; + return NULL; +} + +/* Invoke the registered callback for @chan, if any. Runs in whichever + * process called it - for wire packets and IPC'd sends that is the + * cluster worker, which is the documented delivery context. */ +static void cl_ctr_consumer_dispatch(cl_ctr_cluster_t *cl, int src_node_id, + const char *chan, int chan_len, const char *payload, int payload_len) +{ + str ch, pl; + int i; + + for (i = 0; i < cl_ctr_nchannels; i++) { + if (cl_ctr_channels[i].len == chan_len && + memcmp(cl_ctr_channels[i].name, chan, chan_len) == 0) { + ch.s = (char *)chan; ch.len = chan_len; + pl.s = (char *)payload; pl.len = payload_len; + cl_ctr_channels[i].cb(cl->cluster_id, src_node_id, &ch, &pl); + return; + } + } + /* unknown channel: a mixed-version cluster is normal, drop quietly */ + LM_DBG("clusterer_controller: [cluster %d] no consumer for channel " + "'%.*s'\n", cl->cluster_id, chan_len, chan); +} + +/* Receive path (worker, via cl_ctr_recv_one): already through the magic + * gate, cluster_id filter, rate limiter, decrypt and seq check. */ +static void cl_ctr_handle_consumer(const char *payload, int payload_len, + const char *src_ip, cl_ctr_cluster_t *cl) +{ + uint16_t src_id_be; + int chan_len; + + /* our own multicast looped back: self-delivery, when asked for, was + * already done locally at send time - never accept it off the wire */ + if (strcmp(src_ip, my_ip) == 0) + return; + + if (payload_len < CL_CTR_CONSUMER_HDR_SZ) { + LM_DBG("clusterer_controller: [cluster %d] short consumer packet " + "(%d)\n", cl->cluster_id, payload_len); + return; + } + memcpy(&src_id_be, payload, CL_CTR_NODE_ID_SZ); + chan_len = (unsigned char)payload[CL_CTR_NODE_ID_SZ]; + if (chan_len == 0 || chan_len > CLCTR_MAX_CHAN_LEN || + payload_len < CL_CTR_CONSUMER_HDR_SZ + chan_len) { + LM_DBG("clusterer_controller: [cluster %d] bad consumer channel " + "length %d (payload %d)\n", cl->cluster_id, chan_len, + payload_len); + return; + } + + cl_ctr_consumer_dispatch(cl, ntohs(src_id_be), + payload + CL_CTR_CONSUMER_HDR_SZ, chan_len, + payload + CL_CTR_CONSUMER_HDR_SZ + chan_len, + payload_len - CL_CTR_CONSUMER_HDR_SZ - chan_len); +} + +/* Runs in the cluster worker. Builds, seals and sends the consumer + * packet - and/or dispatches locally for the self-delivery cases. */ +static void cl_ctr_rpc_consumer_send(int sender, void *param) +{ + struct cl_ctr_consumer_job *job = (struct cl_ctr_consumer_job *)param; + cl_ctr_cluster_t *cl = job->cl; + char pkt[CL_CTR_CONSUMER_PKT_MAX]; + char dst_ip[CL_CTR_MAX_IP_LEN + 1]; + uint32_t seq; + uint16_t id_be; + int plain_len, i, to_self = 0, on_wire = 1, reliable = 0; + + /* unicast to our own id never touches the wire; multicast with + * CLCTR_SEND_TO_SELF touches it AND dispatches locally */ + if (job->dst_node_id != 0 && job->dst_node_id == my_node_id) { + to_self = 1; + on_wire = 0; + } else if (job->flags & CLCTR_SEND_TO_SELF) { + to_self = 1; + } + + /* Reliability is per-recipient: it needs an address to retransmit to and + * an acknowledgement to stop on. A multicast has neither, and ACKing one + * from every member is the implosion the join handshake already refuses - + * so a reliable send to everyone is N unicasts, which is what send_list + * and the broadcast helper do before they get here. */ + reliable = (job->flags & CLCTR_SEND_RELIABLE) != 0; + + if (on_wire) { + if (!cl->have_session_key) { + LM_DBG("clusterer_controller: [cluster %d] consumer send before " + "session key - dropped\n", cl->cluster_id); + on_wire = 0; + } + } + + if (on_wire) { + seq = htonl(++cl->peers->my_consumer_seq); + id_be = htons(my_node_id); + + /* the consumer tag, so the receiver's pre-decrypt rate limiter + * charges this against the consumer budget rather than the much + * smaller control-plane one. Same session key either way - the + * tag is a routing hint, not a key selector here. */ + memcpy(pkt, CL_CTR_CONSUMER_MAGIC, CL_CTR_MAGIC_SZ); + pkt[CL_CTR_WIRE_HDR_SZ] = (char)(reliable ? CL_CTR_PKT_CONSUMER_REL + : CL_CTR_PKT_CONSUMER); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + 1, &seq, CL_CTR_SEQ_SZ); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ, &id_be, + CL_CTR_NODE_ID_SZ); + pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_NODE_ID_SZ] = + (char)job->chan_len; + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + + CL_CTR_CONSUMER_HDR_SZ, job->chan, job->chan_len); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + + CL_CTR_CONSUMER_HDR_SZ + job->chan_len, job->payload, + job->payload_len); + plain_len = CL_CTR_PLAIN_HDR_SZ + CL_CTR_CONSUMER_HDR_SZ + + job->chan_len + job->payload_len; + + if (job->dst_node_id == 0) { + cl_ctr_seal_and_send(cl->sock, cl, pkt, plain_len, + cl->session_key, + reliable ? CL_CTR_PKT_CONSUMER_REL : CL_CTR_PKT_CONSUMER); + if (reliable) + /* pkt is a char[] because cl_ctr_seal_and_send() writes + * into it in place; the retx queue takes it read-only as + * unsigned char, which is the natural type for wire bytes. + * The cast is the sign difference only - same address, same + * bytes - and silences -Wpointer-sign without retyping the + * buffer, which would only move the warning to the + * seal_and_send() calls that legitimately need char *. */ + cl_ctr_retx_enqueue_bcast(cl, ntohl(seq), + (const unsigned char *)pkt, plain_len); + } else { + struct sockaddr_in d; + + dst_ip[0] = '\0'; + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + if (cl->peers->entries[i].node_id == job->dst_node_id) { + strcpy(dst_ip, cl->peers->entries[i].ip); + break; + } + } + lock_stop_read(cl->peers->lock); + + if (!dst_ip[0]) { + LM_DBG("clusterer_controller: [cluster %d] consumer send to " + "unknown node %u - dropped\n", cl->cluster_id, + job->dst_node_id); + } else { + memset(&d, 0, sizeof d); + d.sin_family = AF_INET; + d.sin_port = cl->mcast_dest.sin_port; + if (inet_pton(AF_INET, dst_ip, &d.sin_addr) != 1) { + LM_ERR("clusterer_controller: [cluster %d] bad peer ip " + "'%s'\n", cl->cluster_id, dst_ip); + } else { + cl_ctr_seal_and_send_to(cl->sock, cl, pkt, plain_len, + cl->session_key, + reliable ? CL_CTR_PKT_CONSUMER_REL + : CL_CTR_PKT_CONSUMER, + (const struct sockaddr *)&d, sizeof d); + if (reliable) + /* same sign-only cast as the broadcast path above */ + cl_ctr_retx_enqueue_consumer(cl, ntohl(seq), + CL_CTR_PKT_CONSUMER_REL, + (const unsigned char *)pkt, plain_len, + (const struct sockaddr *)&d, sizeof d); + } + } + } + } + + if (to_self) + cl_ctr_consumer_dispatch(cl, my_node_id, job->chan, job->chan_len, + (const char *)job->payload, job->payload_len); + + shm_free(job); +} + +static int clctr_register_channel(str *channel, clctr_msg_cb_f cb) +{ + int i; + + if (!channel || !channel->s || channel->len <= 0 || + channel->len > CLCTR_MAX_CHAN_LEN || !cb) { + LM_ERR("bad consumer channel registration\n"); + return -1; + } + for (i = 0; i < cl_ctr_nchannels; i++) { + if (cl_ctr_channels[i].len == channel->len && + memcmp(cl_ctr_channels[i].name, channel->s, channel->len) == 0) { + LM_ERR("consumer channel '%.*s' already registered\n", + channel->len, channel->s); + return -1; + } + } + if (cl_ctr_nchannels == CL_CTR_MAX_CHANNELS) { + LM_ERR("consumer channel table full (%d)\n", CL_CTR_MAX_CHANNELS); + return -1; + } + memcpy(cl_ctr_channels[cl_ctr_nchannels].name, channel->s, channel->len); + cl_ctr_channels[cl_ctr_nchannels].name[channel->len] = '\0'; + cl_ctr_channels[cl_ctr_nchannels].len = channel->len; + cl_ctr_channels[cl_ctr_nchannels].cb = cb; + cl_ctr_nchannels++; + LM_DBG("consumer channel '%.*s' registered\n", channel->len, channel->s); + return 0; +} + +/* Common caller side: validate, marshal into shm, hand to the worker. */ +static int cl_ctr_consumer_submit(int cluster_id, int dst_node_id, + str *channel, str *payload, int flags) +{ + struct cl_ctr_consumer_job *job; + cl_ctr_cluster_t *cl; + int payload_len = payload ? payload->len : 0; + int proc_no; + + if (!channel || !channel->s || channel->len <= 0 || + channel->len > CLCTR_MAX_CHAN_LEN) + return -1; + if (payload_len < 0 || payload_len > cc_max_payload) { + LM_ERR("consumer payload of %d exceeds the %d-byte datagram bound - " + "use BIN for bulk data\n", payload_len, cc_max_payload); + return -1; + } + cl = cl_ctr_cluster_by_id(cluster_id); + if (!cl) { + LM_ERR("consumer send to unknown cluster %d\n", cluster_id); + return -1; + } + /* worker_proc_no is written once at worker fork and stable thereafter */ + proc_no = cl->peers ? cl->peers->worker_proc_no : -1; + if (proc_no < 0) + return -2; + + job = shm_malloc(sizeof *job + payload_len); + if (!job) { + LM_ERR("no shm for a consumer send\n"); + return -1; + } + job->cl = cl; + job->dst_node_id = (uint16_t)dst_node_id; + job->flags = flags; + job->chan_len = channel->len; + job->payload_len = payload_len; + memcpy(job->chan, channel->s, channel->len); + if (payload_len) + memcpy(job->payload, payload->s, payload_len); + + if (ipc_send_rpc(proc_no, cl_ctr_rpc_consumer_send, job) < 0) { + LM_ERR("cannot dispatch consumer send to the cluster %d worker\n", + cluster_id); + shm_free(job); + return -2; + } + return 0; +} + +static int clctr_send_mcast(int cluster_id, str *channel, str *payload, + int flags) +{ + return cl_ctr_consumer_submit(cluster_id, 0, channel, payload, flags); +} + +static int clctr_send_list(int cluster_id, const int *node_ids, int n, + str *channel, str *payload, int flags, int *unknown) +{ + int i, sent = 0, missing = 0; + + if (!node_ids || n <= 0) + return -1; + + /* One unicast per target rather than a multicast the receivers filter: + * a multicast is decrypted by every member, so "addressed to three of + * you" would still put the payload in front of all of them. The cost is + * linear in the list, which is the honest price of addressing a subset. */ + for (i = 0; i < n; i++) { + if (node_ids[i] <= 0) { + missing++; + continue; + } + if (clctr_send_ucast(cluster_id, node_ids[i], channel, payload, + flags) < 0) + missing++; + else + sent++; + } + if (unknown) + *unknown = missing; + if (missing) + LM_DBG("clusterer_controller: [cluster %d] list send reached %d of " + "%d node(s)\n", cluster_id, sent, n); + return sent; +} + +static int clctr_send_ucast(int cluster_id, int node_id, str *channel, + str *payload, int flags) +{ + if (node_id <= 0 || node_id > CL_CTR_MAX_PEERS) + return -1; + return cl_ctr_consumer_submit(cluster_id, node_id, channel, payload, + flags); +} + +/* Readable from ANY process: the worker-local my_node_id global is only + * authoritative inside the worker, so read our own entry from the shm + * peer table instead. */ +static int clctr_get_my_node_id(int cluster_id) +{ + cl_ctr_cluster_t *cl = cl_ctr_cluster_by_id(cluster_id); + int i, id = 0; + + if (!cl || !cl->peers) + return 0; + lock_start_read(cl->peers->lock); + for (i = 0; i < cl->peers->count; i++) { + if (strcmp(cl->peers->entries[i].ip, my_ip) == 0) { + id = cl->peers->entries[i].node_id; + break; + } + } + lock_stop_read(cl->peers->lock); + return id; +} + +/* ---- script tier: receive ---------------------------------------------- */ + +static void cl_ctr_script_recv(int cluster_id, int src_node_id, str *channel, + str *payload) +{ + str msg, tag; + unsigned char kind; + int taglen; + + if (payload->len < 2) + goto bad; + kind = (unsigned char)payload->s[0]; + taglen = (unsigned char)payload->s[1]; + if (payload->len < 2 + taglen) + goto bad; + tag.s = payload->s + 2; + tag.len = taglen; + msg.s = payload->s + 2 + taglen; + msg.len = payload->len - 2 - taglen; + + /* free when nobody is listening - the same gate the module's other + * events use */ + if (!evi_probe_event(kind == CL_CTR_SCRIPT_RPL ? ei_rpl_id : ei_req_id)) + return; + + if (evi_param_set_int(ei_clid_p, &cluster_id) < 0 || + evi_param_set_int(ei_srcid_p, &src_node_id) < 0 || + evi_param_set_str(ei_msg_p, &msg) < 0 || + evi_param_set_str(ei_tag_p, &tag) < 0) { + LM_ERR("clusterer_controller: cannot fill the message event\n"); + return; + } + if (evi_raise_event(kind == CL_CTR_SCRIPT_RPL ? ei_rpl_id : ei_req_id, + ei_params) < 0) + LM_ERR("clusterer_controller: cannot raise the message event\n"); + return; +bad: + LM_ERR("clusterer_controller: malformed script message from node %d\n", + src_node_id); +} + +/* ---- script tier: send -------------------------------------------------- */ + +static int cl_ctr_script_send(int cluster_id, int node_id, str *gen_msg, + str *tag, unsigned char kind, int flags) +{ + char buf[CLCTR_MAX_PAYLOAD]; + str pl; + int taglen = tag ? tag->len : 0; + + if (!gen_msg || gen_msg->len <= 0) + return -1; + if (taglen > 255) + taglen = 255; + if (2 + taglen + gen_msg->len > cc_max_payload) { + LM_ERR("clusterer_controller: message of %d bytes is more than the " + "%d a datagram carries\n", gen_msg->len, cc_max_payload); + return -1; + } + buf[0] = (char)kind; + buf[1] = (char)taglen; + if (taglen) + memcpy(buf + 2, tag->s, taglen); + memcpy(buf + 2 + taglen, gen_msg->s, gen_msg->len); + pl.s = buf; + pl.len = 2 + taglen + gen_msg->len; + + if (node_id > 0) + return clctr_send_ucast(cluster_id, node_id, &cl_ctr_script_chan, + &pl, flags) < 0 ? -1 : 1; + return clctr_send_mcast(cluster_id, &cl_ctr_script_chan, &pl, flags) < 0 + ? -1 : 1; +} + +static int cmd_cl_ctr_broadcast_req(struct sip_msg *msg, int *cluster_id, + str *gen_msg, str *tag, int *reliable) +{ + /* Reliability is asked for per send rather than per channel: most + * messages are better off cheap, and the ones that are not know it. */ + return cl_ctr_script_send(*cluster_id, 0, gen_msg, tag, + CL_CTR_SCRIPT_REQ, + (reliable && *reliable) ? CLCTR_SEND_RELIABLE : 0); +} + +static int cmd_cl_ctr_send_req(struct sip_msg *msg, int *cluster_id, + int *node_id, str *gen_msg, str *tag, int *reliable) +{ + return cl_ctr_script_send(*cluster_id, *node_id, gen_msg, tag, + CL_CTR_SCRIPT_REQ, + (reliable && *reliable) ? CLCTR_SEND_RELIABLE : 0); +} + +/** + * cl_ctr_send_req_list() - send one request to the nodes named in an AVP. + * + * The AVP is read as a list of node ids, in the order the script built it. + * Every target gets its own unicast: a multicast is decrypted by every + * member, so addressing a subset over one would still hand the payload to + * the nodes that were not addressed. + * + * Returns the number of nodes the message went to, so a script can compare + * it against the size of its own list; entries that name a node which is + * not a current member are skipped rather than failing the whole send. + */ +static int cmd_cl_ctr_send_req_list(struct sip_msg *msg, int *cluster_id, + pv_spec_t *nodes, str *gen_msg, str *tag, pv_spec_t *out) +{ + int ids[CL_CTR_MAX_PEERS]; + int n = 0, sent, unknown = 0; + struct usr_avp *avp = NULL; + int_str val; + + if (!cluster_id || !nodes || !gen_msg) + return -1; + if (nodes->type != PVT_AVP) { + LM_ERR("clusterer_controller: the node list must be an AVP\n"); + return -1; + } + + /* search_first_avp/search_next walk newest-first; the script's own + * order is not preserved and does not need to be - every named node + * gets the same message. */ + avp = search_first_avp(nodes->pvp.pvn.u.isname.type, + nodes->pvp.pvn.u.isname.name.n, &val, NULL); + while (avp && n < CL_CTR_MAX_PEERS) { + if (!(avp->flags & AVP_VAL_STR)) + ids[n++] = (int)val.n; + else + LM_DBG("clusterer_controller: skipping a non-numeric " + "entry in the node list\n"); + avp = search_next_avp(avp, &val); + } + if (n == 0) { + LM_WARN("clusterer_controller: the node list is empty - " + "nothing sent\n"); + return -1; + } + if (avp) + LM_WARN("clusterer_controller: node list longer than the %d a " + "cluster can hold - the rest is ignored\n", + CL_CTR_MAX_PEERS); + + { + char buf[CLCTR_MAX_PAYLOAD]; + str pl; + int tlen = tag ? tag->len : 0; + + if (tlen > 255) + tlen = 255; + if (2 + tlen + gen_msg->len > (int)sizeof(buf)) { + LM_ERR("clusterer_controller: message too large for the " + "cluster plane (%d bytes)\n", gen_msg->len); + return -1; + } + buf[0] = (char)CL_CTR_SCRIPT_REQ; + buf[1] = (char)tlen; + if (tlen) + memcpy(buf + 2, tag->s, tlen); + memcpy(buf + 2 + tlen, gen_msg->s, gen_msg->len); + pl.s = buf; + pl.len = 2 + tlen + gen_msg->len; + + sent = clctr_send_list(*cluster_id, ids, n, &cl_ctr_script_chan, + &pl, 0, &unknown); + } + /* How many nodes it reached goes out through a variable, never through + * the return value: the core stops the script when an action returns + * zero, and "reached nobody" is a real answer a script must be able to + * see rather than a reason to stop. The return value is therefore only + * whether the send could be attempted. */ + if (out) { + pv_value_t v; + + memset(&v, 0, sizeof v); + v.flags = PV_TYPE_INT | PV_VAL_INT; + v.ri = sent < 0 ? 0 : sent; + if (pv_set_value(msg, out, 0, &v) < 0) + LM_ERR("clusterer_controller: cannot write the delivery " + "count to the output variable\n"); + } + if (sent <= 0) + return -1; + if (unknown) + LM_DBG("clusterer_controller: [cluster %d] %d of %d target(s) " + "were not reachable members\n", *cluster_id, unknown, n); + return 1; +} + +static int cmd_cl_ctr_send_rpl(struct sip_msg *msg, int *cluster_id, + int *node_id, str *gen_msg, str *tag) +{ + return cl_ctr_script_send(*cluster_id, *node_id, gen_msg, tag, + CL_CTR_SCRIPT_RPL, 0); +} + +/* Publish the two events and claim the script's channel. Called from + * mod_init, before any consumer registers, so the reserved name cannot be + * taken by a module. */ +static int cl_ctr_script_init(void) +{ + ei_req_id = evi_publish_event(ei_req_name); + ei_rpl_id = evi_publish_event(ei_rpl_name); + if (ei_req_id == EVI_ERROR || ei_rpl_id == EVI_ERROR) { + LM_ERR("clusterer_controller: cannot publish the message events\n"); + return -1; + } + ei_params = pkg_malloc(sizeof *ei_params); + if (!ei_params) { + LM_ERR("clusterer_controller: no pkg for the event parameters\n"); + return -1; + } + memset(ei_params, 0, sizeof *ei_params); + if (!(ei_clid_p = evi_param_create(ei_params, &ei_clid_pname)) || + !(ei_srcid_p = evi_param_create(ei_params, &ei_srcid_pname)) || + !(ei_msg_p = evi_param_create(ei_params, &ei_msg_pname)) || + !(ei_tag_p = evi_param_create(ei_params, &ei_tag_pname))) { + LM_ERR("clusterer_controller: cannot create the event parameters\n"); + return -1; + } + return clctr_register_channel(&cl_ctr_script_chan, cl_ctr_script_recv); +} + +int load_clctr(clctr_api_t *api) +{ + if (!api) + return -1; + api->register_channel = clctr_register_channel; + api->send_mcast = clctr_send_mcast; + api->send_ucast = clctr_send_ucast; + api->send_list = clctr_send_list; + api->get_my_node_id = clctr_get_my_node_id; + api->get_my_ip = cl_ctr_get_my_ip; + return 0; +} diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml index 51e0311a5dd..ef99de39bad 100644 --- a/modules/clusterer_controller/doc/clusterer_controller_admin.xml +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -1193,6 +1193,75 @@ modparam("clusterer_controller", "on_config_mismatch", "reject") +
+ <varname>consumer_rate_limit</varname> (int) + + How many packets a second this node accepts from any one + source address on the messaging API described in + , before it starts dropping them. + Separate from the control plane's own much smaller budget, + which a busy consumer would otherwise exhaust: the two are + counted apart so neither can starve the other. + + + Set it per cluster in the cluster string + to override this default for that cluster alone. + + + Default value is 1000. + +
+ +
+ <varname>consumer_retries</varname> (int) + + How many times a message sent with acknowledgement + () is sent again while it + has not been acknowledged. Zero disables retransmission and + leaves only the acknowledgement itself, which is still worth + having: it is how the sender learns the message arrived. + + + Set it per cluster in the cluster string. + A node can belong to several clusters and they are not alike - + one may run over a quiet management network where a lost packet + is a curiosity, another across a link where it is routine - so + one number for all of them is rarely right. + + + When a reliable send finishes, the log says how much of this + budget it actually needed, so the configured value can be + judged against what the network does: + + +broadcast seq 41 acknowledged by all 3 member(s) after 0 of the 2 retries +configured for this cluster + + + and when it gives up, how far it got: + + +broadcast seq 44 reached 2 of 3 member(s), giving up after 2 of the 2 retries +configured for this cluster + + + Default value is 2. + +
+ +
+ <varname>consumer_retry_ms</varname> (int) + + Milliseconds between those attempts. Accepted range is 5 to + 5000. Set it per cluster in the cluster + string, for the same reason as + . + + + Default value is 40. + +
+
Exported MI Functions @@ -1479,6 +1548,169 @@ if (cl_ctr_node_present(1, 3) && cl_ctr_node_is_master(1, 3)) {
+
+ Messaging Between Nodes + + The channel the controller keeps for its own membership traffic is also + available to carry other people's messages: a script can talk to its + peers, and so can another module. The surface deliberately mirrors the + generic messaging the clusterer module offers - + the same shapes, the same event parameters - so moving between them is + familiar. What differs is underneath, and it differs in both directions: + one multicast packet instead of a send per peer, over an encrypted + channel rather than a plaintext one, but by datagram rather than over a + stream, which is the subject of + below. + + + Messages are limited to 1300 bytes so they fit a datagram, and a channel + name to 31 characters. The module reserves the channel it uses for + script traffic during startup, before any consumer can register, so a + module cannot claim it by accident. + + +
+ From the script + + + cl_ctr_broadcast_req(cluster_id, message + [, tag [, reliable]]) — send a request to every member of + the cluster, as one multicast packet. The sender does not receive + its own broadcast. The tag is carried through + to the receiving event route. Pass 1 for + reliable to ask for acknowledgement. + + + cl_ctr_send_req(cluster_id, node_id, message + [, tag [, reliable]]) — send a request to one node. + + + cl_ctr_send_req_list(cluster_id, nodes_avp, + message [, tag [, out_var]]) — send to the nodes named in + an AVP, one unicast each. out_var receives how + many nodes it was sent to; entries naming a node + that is no longer a member are skipped rather than failing the send + for the rest. + + + cl_ctr_send_rpl(cluster_id, node_id, message + [, tag]) — send a reply, normally from inside the request + event route, back to $param(src_id). + + + + Arriving messages are delivered as events: + E_CL_CTR_REQ_RECEIVED for requests and + E_CL_CTR_RPL_RECEIVED for replies, each carrying + cluster_id, src_id, + msg and tag. + + + Talking to the other nodes + +event_route[E_CL_CTR_REQ_RECEIVED] { + xlog("node $param(src_id) says: $param(msg)\n"); + cl_ctr_send_rpl($param(cluster_id), $param(src_id), "ack", $param(tag)); +} + +# to everyone, cheaply +cl_ctr_broadcast_req(1, "cache flushed"); + +# to everyone, with a tag, and ask to be told it arrived +cl_ctr_broadcast_req(1, "config changed", "cfg", 1); + +# to a few named nodes. Assigning to the same AVP again adds a value +# rather than replacing one, so this names two nodes - but note that the +# values are stored newest first, so they are read back 5 then 2. It +# makes no difference here, since every named node gets the same message. +$avp(nodes) = 2; +$avp(nodes) = 5; +if (cl_ctr_send_req_list(1, $avp(nodes), "just for you", "mytag", $var(sent))) + xlog("sent to $var(sent) node(s)\n"); + + + + Note where the count comes back. It is written into a variable, and the + function itself is simply true or false, which is why the call sits + inside the if above rather than being followed by a + test of its return code. It has to be that way round: &osips; stops + running the script when a function returns zero, so a function that + returned a count would halt the route on the day it reached nobody - + which is a real answer a script must be able to see, not a reason to + stop. + + + Note also what that count is, and is not. It is how many nodes the + message was sent to, which is known immediately. + How many acknowledged it is not knowable at that + point - the answers arrive afterwards - so for a reliable send that + result is reported to the log as it completes + () rather than handed back to + the script. Surfacing it to the script as an event is planned. + +
+ +
+ From another module + + A module binds the API with load_clctr() and gets + register_channel() to claim a name and receive what + arrives on it, send_mcast() to reach every member, + send_ucast() to reach one, + send_list() to reach a named set, and + get_my_node_id(). Sends are handed to the + controller's worker, so any process may call them. + + + Two flags. CLCTR_SEND_TO_SELF also delivers the + message to this node, without putting a packet on the wire, so a + consumer can treat itself the same as its peers. + CLCTR_SEND_RELIABLE is described next. + +
+ +
+ What delivery guarantees + + By default: best effort, unordered, and at most once. A message is sent + and not spoken of again. That suits anything the sender can afford to + repeat - a cache that will be asked a second time, a hint that is worth + having and not worth waiting for - and it is the cheapest thing the + network can do. + + + It is also weaker than what clusterer gives you + over a stream, and that is worth saying plainly rather than leaving to + be discovered: a script moved across without further thought has quietly + traded reliable, ordered delivery for neither. + + + CLCTR_SEND_RELIABLE, or a 1 in the last argument of + cl_ctr_broadcast_req(), asks for the message to be + acknowledged and sent again while it is not, up to + times at + intervals. It is chosen per + send rather than per channel, because most messages are better off cheap + and the ones that are not know who they are. + + + The cost is one acknowledgement per recipient. On a large cluster a + reliable broadcast is therefore one packet out and one back from every + member, where an ordinary one is a single packet and no answer at all. + It remains a multicast - sending it as one unicast per member would be + roughly twice the packets for the same delivery - and only the members + that fail to answer are then repaired individually. + + + Addressing a set of nodes is the other way round: those go out as one + unicast each, because a multicast is decrypted by every member, so a + message addressed to three of them over a multicast would + still be handed to all the others. Filtering after decryption is not + addressing. + +
+
+
Multiple Clusters diff --git a/modules/clusterer_controller/test/consumer_filter_test.py b/modules/clusterer_controller/test/consumer_filter_test.py new file mode 100644 index 00000000000..61b0adf3b97 --- /dev/null +++ b/modules/clusterer_controller/test/consumer_filter_test.py @@ -0,0 +1,49 @@ +#!/usr/bin/env python3 +"""Both fixes, on the live 3-node cluster.""" +import json, socket, subprocess, time +def mi(n,m,p=None): + c=socket.socket(socket.AF_INET,socket.SOCK_DGRAM); c.settimeout(8) + c.sendto(json.dumps({"jsonrpc":"2.0","id":1,"method":m,"params":p or []}).encode(), + ("10.94.0.1%d"%n,8787)) + r=json.loads(c.recv(262144)); c.close(); return r.get("result",r.get("error")) + +ok=fail=0 +def check(name,cond,detail=""): + global ok,fail + print(" %-56s %s %s"%(name,"PASS" if cond else "FAIL",detail)) + ok,fail=ok+(1 if cond else 0),fail+(0 if cond else 1) + +# 1. legitimate cross-node pull still works through the membership filter +mi(1,"perf_set",{"key":"th:filtertest","value":"still-works","ttl":600,"collection":"th"}) +r2=mi(2,"perf_pull",{"key":"th:filtertest","collection":"th"}) +check("a real pull between members still succeeds", + isinstance(r2,dict) and r2.get("value")=="still-works", r2) + +# 2. a flood of consumer-magic packets from a NON-member is dropped early +before=mi(1,"clusterer_controller:node_info") if False else None +s=socket.socket(socket.AF_INET,socket.SOCK_DGRAM) +s.bind(("10.94.0.1",0)) # host bridge IP - not a cluster member +pkt=bytes([0xCC,0x02])+ (9).to_bytes(2,"big") + b"\x00"*80 +t0=time.time(); n=0 +while time.time()-t0 < 3: + for _ in range(200): + s.sendto(pkt,("10.94.0.11",4499)); n+=1 +s.close() +print(" sent %d consumer-magic packets from a non-member in 3s"%n) +time.sleep(1) + +logs=subprocess.run(["nerdctl","logs","n1"],capture_output=True,text=True) +tail=logs.stdout[-4000:]+logs.stderr[-4000:] +check("the node did not die under the flood", + "SIGSEGV" not in tail and mi(1,"perf_stats") is not None) +check("no rate-limit warning (dropped before the limiter)", + "rate limit of" not in tail) +# the cluster must still be healthy afterwards +peers=[nd["node_id"] for cl in mi(1,"clusterer:list").get("Clusters",[]) + for nd in cl.get("Nodes",[])] +check("cluster membership intact after the flood", len(peers)==2, peers) +mi(1,"perf_set",{"key":"th:aftertest","value":"post-flood","ttl":600,"collection":"th"}) +r3=mi(3,"perf_pull",{"key":"th:aftertest","collection":"th"}) +check("pulls still work after the flood", + isinstance(r3,dict) and r3.get("value")=="post-flood", r3) +print("\n%d passed, %d failed"%(ok,fail)) diff --git a/modules/clusterer_controller/test/consumer_seq_test.py b/modules/clusterer_controller/test/consumer_seq_test.py new file mode 100644 index 00000000000..b5b50dda043 --- /dev/null +++ b/modules/clusterer_controller/test/consumer_seq_test.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python3 +"""The sequence split, under the two cases that matter: heavy consumer +traffic must not make control traffic look like a replay, and a node +restart (which rewinds both counters) must not wedge the consumer plane.""" +import json, socket, subprocess, sys, time +SRC = sys.argv[1] if len(sys.argv)>1 else "/dn/wt-cp15/modules/clusterer_controller/clusterer_controller.c" +def mi(n,m,p=None): + c=socket.socket(socket.AF_INET,socket.SOCK_DGRAM); c.settimeout(10) + c.sendto(json.dumps({"jsonrpc":"2.0","id":1,"method":m,"params":p or []}).encode(), + ("10.94.0.1%d"%n,8787)) + r=json.loads(c.recv(262144)); c.close(); return r.get("result",r.get("error")) +ok=fail=0 +def check(name,cond,detail=""): + global ok,fail + print(" %-58s %s %s"%(name,"PASS" if cond else "FAIL",detail)) + ok,fail=ok+(1 if cond else 0),fail+(0 if cond else 1) + +# heavy consumer traffic: 20k pulls, the rate our convergence storm produced +mi(1,"perf_del",{"glob":"th:*","collection":"th"}) +mi(2,"perf_del",{"glob":"th:*","collection":"th"}) +mi(3,"perf_del",{"glob":"th:*","collection":"th"}) +for i in range(3000): + mi(1,"perf_set",{"key":"th:seq%06d"%i,"value":"v"*40,"ttl":600,"collection":"th"}) +t0=time.time(); pulled=0 +for i in range(3000): + r=mi(2,"perf_pull",{"key":"th:seq%06d"%i,"collection":"th"}) + if isinstance(r,dict) and r.get("value"): pulled+=1 +dt=time.time()-t0 +print(" %d pulls in %.1fs (%.0f/s of consumer traffic)"%(pulled,dt,pulled/dt)) +check("every pull answered under sustained consumer load", pulled==3000, "%d/3000"%pulled) + +logs=subprocess.run(["nerdctl","logs","n2"],capture_output=True,text=True) +tail=logs.stdout+logs.stderr +check("no control packet mistaken for a replay", "replay from" not in tail, + [l for l in tail.splitlines() if "replay from" in l][:1]) +check("cluster still healthy after the load", + len([nd for cl in mi(2,"clusterer:list").get("Clusters",[]) + for nd in cl.get("Nodes",[])])==2) + +# The rewind path (session re-key / re-join) cannot be exercised from here: +# a restarted node currently fails to rejoin this cluster for reasons that +# predate this change - the identical failure reproduces on a build with none +# of it - so it is not what this test is for. The invariant is checked in the +# source instead: every site that rewinds a peer's last_seq must rewind +# last_consumer_seq beside it, or a receiver stays ahead of a sender that +# restarted at zero and silently drops everything it sends. +src = open(SRC).read() +parts = src.split("last_seq = 0")[:-1] +missed = [p for p in parts if "last_consumer_seq" not in p[-300:] + and "last_consumer_seq" not in src[src.find(p)+len(p):][:300]] +check("every last_seq rewind also rewinds last_consumer_seq", + not missed, "%d site(s) missing" % len(missed)) +print("\n%d passed, %d failed" % (ok, fail)) diff --git a/modules/clusterer_controller/test/reliable_broadcast_test.py b/modules/clusterer_controller/test/reliable_broadcast_test.py new file mode 100644 index 00000000000..8a99d2ebcbc --- /dev/null +++ b/modules/clusterer_controller/test/reliable_broadcast_test.py @@ -0,0 +1,52 @@ +#!/usr/bin/env python3 +"""Reliable broadcast: one multicast out, an ACK from every member, and +repair by unicast only to whoever did not answer.""" +import re, socket, subprocess, time +def sip(node,hdrs): + ip="10.94.0.1%d"%node + s=socket.socket(socket.AF_INET,socket.SOCK_DGRAM); s.bind(("10.94.0.1",0)) + p=s.getsockname()[1]; s.settimeout(15); t=int(time.time()*1000)%100000 + m=("OPTIONS sip:x@%s:5060 SIP/2.0\r\nVia: SIP/2.0/UDP 10.94.0.1:%d;branch=z9hG4bK-r%d\r\n" + "From: ;tag=r%d\r\nTo: \r\nCall-ID: rl-%d\r\nCSeq: 1 OPTIONS\r\n" + "%sMax-Forwards: 5\r\nContent-Length: 0\r\n\r\n")%(ip,p,t,t,t,hdrs) + s.sendto(m.encode(),(ip,5060)) + try: return s.recvfrom(2048)[0].split(b" ",2)[1].decode() + except socket.timeout: return "timeout" + finally: s.close() +def logs(n): + r=subprocess.run(["nerdctl","logs","n%d"%n],capture_output=True,text=True) + return r.stdout+r.stderr +ok=fail=0 +def check(name,cond,detail=""): + global ok,fail + print(" %-58s %s %s"%(name,"PASS" if cond else "FAIL",detail)) + ok,fail=ok+(1 if cond else 0),fail+(0 if cond else 1) + +base_got={n:logs(n).count("L-GOT") for n in (1,2,3)} +BR=subprocess.run("ip -o link show type bridge | grep -oE 'br-[a-z0-9]+' | head -1", + shell=True,capture_output=True,text=True).stdout.strip() +cap=subprocess.Popen(["timeout","12","tcpdump","-i",BR,"-nn","-q","udp port 4499"], + stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,text=True) +time.sleep(1.5) +print(" reply:", sip(1,"X-Rel: 1\r\n")) +time.sleep(4) +cap.terminate(); out=cap.stdout.read() + +after={n:logs(n).count("L-GOT") for n in (1,2,3)} +delta={n:after[n]-base_got[n] for n in (1,2,3)} +print(" received by:",delta) +check("both peers received the broadcast", delta[2]>=1 and delta[3]>=1, delta) +check("the sender did not receive its own broadcast", delta[1]==0) + +mc=len([l for l in out.splitlines() if "239.0.94.1.4499" in l and "10.94.0.11" in l]) +check("it went out as ONE multicast, not one packet per peer", mc==1, "%d multicast(s)"%mc) + +l1=logs(1) +acks=re.findall(r"broadcast seq (\d+) acknowledged by (\d+) of (\d+)", l1) +allack=re.findall(r"broadcast seq (\d+) acknowledged by all (\d+)", l1) +check("the sender counted acknowledgements from every member", + len(allack)>=1, allack[-1:] or acks[-2:]) +check("no repair was needed when nothing was lost", + "repairing" not in l1 or "repairing 0 node" in l1, + [l for l in l1.splitlines() if "repairing" in l][-1:]) +print("\n%d passed, %d failed"%(ok,fail)) diff --git a/modules/clusterer_controller/test/script_messaging_rig.sh b/modules/clusterer_controller/test/script_messaging_rig.sh new file mode 100755 index 00000000000..15d9169a319 --- /dev/null +++ b/modules/clusterer_controller/test/script_messaging_rig.sh @@ -0,0 +1,76 @@ +#!/bin/sh +# CP-15.13: script-to-script messaging over the controller's plane. +set -e +T=/dn/wt-cp15 +D=/dn/scripttier +mkdir -p $D +pkill -f "scripttier/n[12].cfg" 2>/dev/null || true +sleep 1 +for i in 1 2; do ip netns del s$i 2>/dev/null || true; ip link del x${i}p 2>/dev/null || true; done +ip link del br96 2>/dev/null || true +sleep 1 +rm -f $D/mi1.sock $D/mi2.sock + +ip link add br96 type bridge +ip link set br96 up +echo 0 > /sys/class/net/br96/bridge/multicast_snooping 2>/dev/null || true +for i in 1 2; do + ip netns add s$i + ip link add x$i type veth peer name x${i}p + ip link set x$i netns s$i + ip link set x${i}p master br96 up + ip netns exec s$i ip link set lo up + ip netns exec s$i ip addr add 10.96.0.$i/24 dev x$i + ip netns exec s$i ip link set x$i up + ip netns exec s$i ip route add 239.0.0.0/8 dev x$i + + cat > $D/n$i.cfg < $D/n$i.log 2>&1 & +done +echo "waiting for the cluster..." +sleep 14 +grep -ahc "roles:" $D/n1.log $D/n2.log diff --git a/modules/clusterer_controller/test/script_messaging_test.py b/modules/clusterer_controller/test/script_messaging_test.py new file mode 100755 index 00000000000..e4038a12a09 --- /dev/null +++ b/modules/clusterer_controller/test/script_messaging_test.py @@ -0,0 +1,84 @@ +#!/usr/bin/env python3 +"""CP-15.13: a script sends to its peers and receives as an event route, +the same surface clusterer offers, over the controller's plane.""" +import json, os, socket, subprocess, sys, time + +D = "/dn/scripttier" + +def sip(node, hdrs): + ip = "10.96.0.%d" % node + prog = ("import socket,time\n" + "s=socket.socket(socket.AF_INET,socket.SOCK_DGRAM)\n" + "s.bind(('%s',0)); p=s.getsockname()[1]\n" + "m=('OPTIONS sip:b@%s SIP/2.0\\r\\nVia: SIP/2.0/UDP %s:%%d;branch=z9hG4bK-%%f\\r\\n'\n" + " 'From: ;tag=b\\r\\nTo: \\r\\nCall-ID: sc-%%f\\r\\nCSeq: 1 OPTIONS\\r\\n%s'\n" + " 'Max-Forwards: 70\\r\\nContent-Length: 0\\r\\n\\r\\n')%%(p,time.time(),time.time())\n" + "s.settimeout(20); s.sendto(m.encode(),('%s',5060))\n" + "print(s.recvfrom(65535)[0].decode().split('\\r\\n')[0])\n" + ) % (ip, ip, ip, hdrs.replace("\r\n", "\\r\\n"), ip) + out = subprocess.run(["ip", "netns", "exec", "s%d" % node, "python3", "-c", prog], + capture_output=True, text=True, timeout=60) + return (out.stdout or out.stderr).strip() + +def log(n): + return open("%s/n%d.log" % (D, n), errors="replace").read() + +def mi(n, method, params=None): + c = socket.socket(socket.AF_UNIX, socket.SOCK_DGRAM) + c.bind("/tmp/st.%d.%f" % (os.getpid(), time.time())); c.settimeout(10) + c.sendto(json.dumps({"jsonrpc":"2.0","id":1,"method":method, + "params": params if params is not None else []}).encode(), + "%s/mi%d.sock" % (D, n)) + r = json.loads(c.recv(262144)); c.close() + return r.get("result", r.get("error")) + +def peer_id_seen_by(n): + """the id THIS node knows its peer by - which is the peer's real id, + because the controller assigns ids and clusterer adopts them. Never + assume the my_node_id in the config: the controller overrides it.""" + for cl in mi(n, "clusterer:list").get("Clusters", []): + for nd in cl.get("Nodes", []): + return nd["node_id"] + return 0 + +ok = fail = 0 +def check(name, cond, detail=""): + global ok, fail + print(" %-52s %s %s" % (name, "PASS" if cond else "FAIL", detail)) + ok, fail = ok + (1 if cond else 0), fail + (0 if cond else 1) + +# --- broadcast from n1: n2 must receive it as an event, and reply --- +print("broadcast from n1:", sip(1, "X-Bcast: 1\r\n")) +time.sleep(2) +l1, l2 = log(1), log(2) + +check("the peer received it as a script event", + "SCRIPT-REQ" in l2 and "hello-from-node1" in l2, + [x for x in l2.splitlines() if "SCRIPT-REQ" in x][:1]) +N1 = peer_id_seen_by(2) # what n2 calls n1 +N2 = peer_id_seen_by(1) # what n1 calls n2 +print(" runtime ids: n1=%d n2=%d (config said 1 and 2)" % (N1, N2)) +check("with the sender's real node id", + any("src=%d" % N1 in x for x in l2.splitlines() if "SCRIPT-REQ" in x), + "expected src=%d" % N1) +check("the sender did NOT receive its own broadcast", + "SCRIPT-REQ" not in l1) +check("the reply came back as the other event", + "SCRIPT-RPL" in l1 and "ack-from-node2" in l1, + [x for x in l1.splitlines() if "SCRIPT-RPL" in x][:1]) + +# --- directed send from n2 to node 1 --- +print("unicast n2 -> n1 (id %d):" % N1, + sip(2, "X-Ucast: 1\r\nX-To-Node: %d\r\n" % N1)) +time.sleep(2) +l1b, l2b = log(1), log(2) +check("a directed message reached only its target", + "direct-to-you" in l1b and "direct-to-you" not in l2b) + +for n, l in ((1, l1b), (2, l2b)): + bad = [x for x in l.splitlines() + if "CRITICAL:" in x or "core dumped" in x or "Segmentation" in x] + check("n%d: no crash" % n, not bad, bad[:1]) + +print("\n%d passed, %d failed" % (ok, fail)) +sys.exit(1 if fail else 0) diff --git a/modules/clusterer_controller/test/script_send_list_test.py b/modules/clusterer_controller/test/script_send_list_test.py new file mode 100644 index 00000000000..5a464db1be3 --- /dev/null +++ b/modules/clusterer_controller/test/script_send_list_test.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +"""A list send must reach exactly its targets - no more, no fewer.""" +import socket, subprocess, sys, time +def sip(node, hdrs): + ip="10.94.0.1%d"%node + s=socket.socket(socket.AF_INET,socket.SOCK_DGRAM); s.bind(("10.94.0.1",0)) + p=s.getsockname()[1]; s.settimeout(15) + m=("OPTIONS sip:x@%s:5060 SIP/2.0\r\nVia: SIP/2.0/UDP 10.94.0.1:%d;branch=z9hG4bK-l%d\r\n" + "From: ;tag=l%d\r\nTo: \r\nCall-ID: lt-%d\r\nCSeq: 1 OPTIONS\r\n" + "%sMax-Forwards: 5\r\nContent-Length: 0\r\n\r\n")%(ip,p,int(time.time()),int(time.time()),int(time.time()),hdrs) + s.sendto(m.encode(),(ip,5060)) + try: return s.recvfrom(2048)[0].split(b" ",2)[1].decode() + except socket.timeout: return "timeout" + finally: s.close() +def logs(n): + r=subprocess.run(["nerdctl","logs","n%d"%n],capture_output=True,text=True) + return r.stdout+r.stderr +ok=fail=0 +def check(name,cond,detail=""): + global ok,fail + print(" %-56s %s %s"%(name,"PASS" if cond else "FAIL",detail)); + ok,fail=ok+(1 if cond else 0),fail+(0 if cond else 1) + +base={n:logs(n).count("L-GOT") for n in (1,2,3)} +# node ids are assigned by the controller - ask a node what its peers are called +import json +def mi(n,m,p=None): + c=socket.socket(socket.AF_INET,socket.SOCK_DGRAM); c.settimeout(8) + c.sendto(json.dumps({"jsonrpc":"2.0","id":1,"method":m,"params":p or []}).encode(),("10.94.0.1%d"%n,8787)) + r=json.loads(c.recv(262144)); c.close(); return r.get("result",r.get("error")) +peers={} +for n in (1,2,3): + peers[n]=sorted(nd["node_id"] for cl in mi(n,"clusterer:list").get("Clusters",[]) for nd in cl.get("Nodes",[])) +print(" peer views:",peers) +# n1 sends to exactly one of its two peers +target=peers[1][0] +print(" n1 -> list containing only node %d"%target) +print(" reply:", sip(1,"X-List: 1\r\nX-N1: %d\r\n"%target)) +time.sleep(2) +after={n:logs(n).count("L-GOT") for n in (1,2,3)} +delta={n:after[n]-base[n] for n in (1,2,3)} +print(" who received:",delta) +got=[n for n in (1,2,3) if delta[n]>0] +check("exactly one node received the message", len(got)==1, got) +check("the sender did not receive its own list send", delta[1]==0) +sent=[l for l in logs(1).splitlines() if "L-SENT" in l][-1:] +check("the function reported one delivery", "n=1" in (sent[0] if sent else ""), sent) +print("\n%d passed, %d failed"%(ok,fail)) From 790e31ca7f28344bf181e63972731f0a83b10af6 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Sun, 9 Aug 2026 22:20:42 +1000 Subject: [PATCH 05/32] clusterer_controller: a member list says who belongs, not who is alive cl_ctr_handle_member_list() refreshed last_seen for every IP in the list, via cl_ctr_upsert_peer_locked(). A MEMBER_LIST is a membership announcement: the master lists a node until it prunes it, whether that node is reachable or not. Treating it as evidence of liveness makes every receiver believe it has just heard from peers it has never heard from. That is what made a dead member immortal. cl_ctr_alive_bitmap() is built from last_seen, so a backup holding refreshed timestamps for a node that is down asserts to the whole cluster that it is up the moment that backup becomes master - and from then on nobody can age it out. It survives its own funeral. Seen in production: 10.22.20.241 stayed in cluster 243 for about a day, across a master change, after being rolled back to a build with no controller at all. Nothing was listening on its BIN port, and the two surviving nodes spent ~3,500 failed connects a day each trying to reach it - roughly 10,668 of one node's 10,801 daily ERROR lines, which buried every other error on the box. The member-list path now inserts peers it did not know about and leaves last_seen alone for peers it already tracks. Liveness continues to come only from direct evidence: a packet sent BY the peer, or the master's alive bitmap, which the master derives from packets it received itself. A newly learned peer is still seeded with last_seen = now, so it gets one purge window to prove itself instead of being dropped on the next tick. This terminates rather than oscillating: once the master prunes a dead node it stops listing it, and every receiver ages it out one window later. --- .../clusterer_controller.c | 38 ++++++++++++++++++- 1 file changed, 37 insertions(+), 1 deletion(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 6691087ea6e..2ba30ba0266 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -1939,6 +1939,41 @@ static cl_ctr_peer_t *cl_ctr_peer_by_ip_locked(cl_ctr_cluster_t *cl, const char return NULL; } +static void cl_ctr_upsert_peer_locked(const char *src_ip, cl_ctr_cluster_t *cl); + +/** + * cl_ctr_learn_peer_locked() - membership WITHOUT a liveness claim. + * + * A MEMBER_LIST says who BELONGS to the cluster, not who is reachable now: the + * master lists a node until it prunes it, whether that node is up or not. + * Refreshing last_seen from such a list - which is what upsert does - makes + * every receiver believe it has just heard from peers it has never heard from. + * + * That is not cosmetic. cl_ctr_alive_bitmap() is built from last_seen, so a + * backup holding refreshed timestamps for a dead node asserts that node is + * alive to the entire cluster the moment it becomes master - and then nobody + * can age it out. A node survived in a live cluster for about a day that way, + * across a master change, while every TCP connect to it was refused. + * + * So: insert what we did not already know about, and leave last_seen of + * anything we do track to direct evidence only - cl_ctr_upsert_peer_locked() + * from a packet actually sent BY that peer, or the master's alive bitmap, + * which the master derives from its own direct evidence. + * + * A newly learned peer is still seeded with last_seen = now, deliberately: it + * gets one purge window to prove itself rather than being pruned on the next + * tick. This terminates - once the master prunes a dead node it stops listing + * it, and every receiver ages it out one window later. + * + * Must be called with cl->peers->lock held for write. + */ +static void cl_ctr_learn_peer_locked(const char *src_ip, cl_ctr_cluster_t *cl) +{ + if (cl_ctr_peer_by_ip_locked(cl, src_ip)) + return; /* already known - membership adds nothing */ + cl_ctr_upsert_peer_locked(src_ip, cl); +} + /** * cl_ctr_upsert_peer_locked() - insert or refresh a peer entry. * Does NOT call cl_ctr_elect_master(cl); callers do that explicitly. @@ -4251,7 +4286,8 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, ip_buf[CL_CTR_MAX_IP_LEN] = '\0'; if (ip_buf[0] == '\0') continue; - cl_ctr_upsert_peer_locked(ip_buf, cl); + /* membership, not liveness - see cl_ctr_learn_peer_locked() */ + cl_ctr_learn_peer_locked(ip_buf, cl); for (_j = 0; _j < cl->peers->count; _j++) { if (strcmp(cl->peers->entries[_j].ip, ip_buf) == 0) { cl->peers->entries[_j].last_seq = 0; From 5aaf72f5579dff30add854ca582a3602bc9373ad Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 01:06:40 +1000 Subject: [PATCH 06/32] clusterer_controller: make a co-bound port loud instead of silent A second controller process on the same host takes the same ip:port without complaint, because the socket needs SO_REUSEADDR and INADDR_ANY to receive multicast at all. The kernel then treats the two traffic types differently: a multicast datagram is delivered to EVERY co-bound socket, but a unicast is delivered to exactly ONE, chosen without reference to the destination address. Measured on 5.4, 200/200 trials, the winner being the most recently bound socket; SO_REUSEPORT does not help, it only changes which socket wins. Every 1:1 leg therefore lands in the wrong process. The join KEY_GRANT is unicast, so the joining node never authenticates and dies with cannot authenticate ... (wrong password, or a foreign cluster on cluster_id N). Shutting down. which sends the operator hunting a credential problem that does not exist. Reversing the start order moves the victim to the other instance, which is how the mechanism was confirmed: it follows bind order, not address. This does not make that topology work - one controller instance per host per port is what is supported, and production is unaffected because a node runs one. It makes the diagnosis available: - cl_ctr_setup_socket() probes the port WITHOUT SO_REUSEADDR before taking it. That bind fails EADDRINUSE precisely when another process already holds it, and succeeds when the port is free; a probe WITH SO_REUSEADDR would succeed either way, which is why the real bind cannot tell. One warning naming the actual constraint. Advisory, never fatal - a restart can briefly race the outgoing process, and a spurious warning is cheaper than refusing to start. - cl_ctr_maybe_forward()'s "no local cluster for this packet" drop was LM_DBG. L_DBG is 4 and deployments run 3, so a whole class of "the cluster does not converge" had no trace at any log level anyone uses. It is now LM_WARN, rate-limited to one per 30s carrying the suppressed count - a misdelivering peer can produce one per packet, and a warning that fires at line rate is its own outage. The in-process case is unchanged: several clusters in ONE process sharing a port are still recovered by cl_ctr_maybe_forward() through the shared cluster array, and that path never reaches the new warning. --- .../clusterer_controller.c | 78 ++++++++++++++++++- 1 file changed, 76 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 2ba30ba0266..f8fd870789c 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -520,6 +520,9 @@ typedef enum { * ========================================================================= */ #define CL_CTR_MAX_CLUSTERS 16 /* max cluster= entries */ +/* Minimum seconds between the shared-port misdelivery warnings; the drop + * they report can otherwise recur per packet. */ +#define CL_CTR_FWD_WARN_IVL 30 /* Forward-declared so cl_ctr_cluster_t can embed a pointer */ typedef struct cl_ctr_peers_ cl_ctr_peers_t; @@ -2592,6 +2595,47 @@ static int cl_ctr_setup_socket(cl_ctr_cluster_t *cl) local.sin_port = htons((uint16_t)cl->multicast_port); local.sin_addr.s_addr = htonl(INADDR_ANY); + /* Is somebody else already on this ip:port? + * + * We bind INADDR_ANY:multicast_port with SO_REUSEADDR because that is what + * multicast reception needs, and it is also what lets a SECOND process take + * the same port without any error. That second process is not harmless: + * the kernel fans multicast out to every co-bound socket, but delivers a + * UNICAST datagram to exactly ONE of them, chosen without reference to the + * destination address. Every 1:1 leg of the protocol - KEY_GRANT during + * the join, consumer replies, ACKs - can therefore land in the wrong + * process. The symptom is deeply misleading: the joiner never sees its + * KEY_GRANT and dies with "cannot authenticate ... wrong password", sending + * the operator after a credential problem that does not exist. + * + * Detect it by probing the port WITHOUT SO_REUSEADDR: that bind fails with + * EADDRINUSE precisely when someone already holds it, and succeeds when the + * port is free. (A probe WITH SO_REUSEADDR would succeed either way, which + * is why the real bind below cannot tell the difference.) + * + * Advisory only - never fatal. A restart can briefly race the outgoing + * process, and one spurious warning is a far smaller cost than refusing to + * start. cl_ctr_maybe_forward() still recovers the in-process case, where + * several clusters in THIS process share a port. */ + { + int probe = socket(AF_INET, SOCK_DGRAM, 0); + + if (probe >= 0) { + if (bind(probe, (struct sockaddr *)&local, sizeof(local)) < 0 && + errno == EADDRINUSE) + LM_WARN("clusterer_controller: [cluster %d] another process is " + "already bound to port %d - multicast will reach both, " + "but every unicast leg (KEY_GRANT, consumer replies) " + "goes to only ONE of them, chosen by the kernel and not " + "by address. A node that cannot join, or a consumer " + "that never gets replies, is explained by this and not " + "by a bad password. One controller instance per host " + "per port is what is supported.\n", + cl->cluster_id, cl->multicast_port); + close(probe); + } + } + if (bind(sock, (struct sockaddr *)&local, sizeof(local)) < 0) { LM_ERR("clusterer_controller: bind() port %d: %s\n", cl->multicast_port, strerror(errno)); @@ -5526,8 +5570,38 @@ static void cl_ctr_maybe_forward(const char *buf, int n, } } if (!target) { - LM_DBG("clusterer_controller: [cluster %d] no local cluster %u for " - "packet on shared port, dropping\n", from->cluster_id, pkt_cid); + /* Nothing in THIS process owns that cluster_id, so the datagram is + * unrecoverable here - it was almost certainly meant for a different + * process co-bound to the same port (see the probe in + * cl_ctr_setup_socket). This used to be LM_DBG, which meant it was + * invisible at every log_level anyone actually runs, and a whole class + * of "the cluster just does not converge" reports had no trace at all. + * + * Rate-limited because a misdelivering peer can produce one of these + * per packet, and a warning that can fire at line rate is its own + * outage. First occurrence prints immediately; after that at most one + * per CL_CTR_FWD_WARN_IVL seconds, carrying the count it stands for. */ + static time_t last_warn; + static unsigned long suppressed; + time_t now = time(NULL); + + if (last_warn == 0 || now - last_warn >= CL_CTR_FWD_WARN_IVL) { + if (suppressed) + LM_WARN("clusterer_controller: [cluster %d] no local cluster %u " + "for packet on shared port, dropping (%lu more in the " + "last %lds - another process is probably bound to this " + "port)\n", from->cluster_id, pkt_cid, suppressed, + (long)(now - last_warn)); + else + LM_WARN("clusterer_controller: [cluster %d] no local cluster %u " + "for packet on shared port, dropping - another process " + "is probably bound to this port\n", + from->cluster_id, pkt_cid); + last_warn = now; + suppressed = 0; + } else { + suppressed++; + } return; } From 6eae096da51d7d2739fcff0a179b162153ccf0df Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 08:48:49 +1000 Subject: [PATCH 07/32] clusterer_controller: NODE_ASSIGN is membership too, not liveness 678bc4c7ed stopped MEMBER_LIST from refreshing last_seen, on the grounds that a membership announcement says who BELONGS to the cluster, not who is reachable. NODE_ASSIGN carries exactly the same kind of claim and was still calling cl_ctr_upsert_peer_locked() on the IP in its payload, so the bug survived the fix. This half was worse, because it is a SELF-loop rather than a peer-to-peer echo. IP_MULTICAST_LOOP means the master receives its own NODE_ASSIGN - the handler's own doc comment says 'all nodes (including master via loopback) apply the assignment'. So every roster announcement refreshed last_seen for every node named in it, including a dead one, and since the master is the node that runs cl_ctr_prune_stale(), the corpse kept itself alive. Measured on the 3-node staging RGS cluster before this commit: a member whose controller plane was isolated with iptables, and which was confirmed silent by tcpdump on the master, was STILL a member on both survivors after four minutes against a 30 s purge deadline. Behaviour was identical with and without 678bc4c7ed, which is what showed the earlier fix was only half of it. Liveness still has two sound sources and both are untouched: the master learns it from the unicast ALIVE a settled non-master sends it, and backups learn it from the master's MASTER_ALIVE bitmap, which cl_ctr_apply_alive_bitmap() uses to set last_seen. A node the master no longer believes in is simply absent from that bitmap, so receivers age it out instead of being told to keep it. cl_ctr_rejoin_superior_master() was audited at the same time and deliberately left alone: it upserts sender_ip from a beacon actually sent by that node, which is direct evidence. (cherry picked from commit 9bd7d101bb45707324699d8fa68ce64088a6ce87) --- .../clusterer_controller/clusterer_controller.c | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index f8fd870789c..9b4083aac35 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -4569,7 +4569,8 @@ static void cl_ctr_handle_goodbye(int sock, const char *src_ip, cl_ctr_cluster_t * Payload: [node_id 2B BE][ip NUL][bin_count 1B][sock1 NUL]...[sockN NUL] * * All nodes (including master via loopback) apply the assignment: - * - Upsert the peer entry if not already present. + * - Learn the peer entry if not already present, WITHOUT claiming liveness + * (see the note at the call site - the loopback made this a self-refresh). * - Store node_id and BIN sockets. * - If ip == my_ip: record my_node_id. */ @@ -4615,7 +4616,19 @@ static void cl_ctr_handle_node_assign(const char *payload, int payload_len, } lock_start_write(cl->peers->lock); - cl_ctr_upsert_peer_locked(ip, cl); + /* Membership, not liveness - the same distinction cl_ctr_learn_peer_locked() + * exists for on the MEMBER_LIST path. 'ip' here comes from the PAYLOAD (the + * node being assigned), not from sender_ip, so it is not evidence that that + * node is reachable - only that the master still lists it. + * + * Upserting here was strictly worse than the member-list case, because it is + * a SELF-loop rather than a peer-to-peer echo: IP_MULTICAST_LOOP means the + * master receives its own NODE_ASSIGN, so every roster announcement refreshed + * last_seen for every node it named, including a dead one. The master is the + * node that must prune, so cl_ctr_prune_stale() could never fire and the + * corpse was immortal - measured on a 3-node staging cluster: a node isolated + * for 4 minutes against a 30 s deadline was still a member on every survivor. */ + cl_ctr_learn_peer_locked(ip, cl); cl_ctr_update_peer_bin_locked(ip, node_id, bin_cnt, (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, cl); From 7121f0b4993406d0085950e250a84b144f775187 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 09:49:04 +1000 Subject: [PATCH 08/32] clusterer: close a node's BIN connection when it leaves the cluster Removing a node from the topology did not touch the transport. clusterer sends with msg_send(send_sock, proto, &node->addr) and the core keeps the TCP connection in its own table keyed by destination, so nothing in clusterer_ctrl_remove_node() - delete_neighbour, remove_node_list, CLUSTER_NODE_DOWN, report_node_state - ever closed the socket. It survived until the TCP layer timed it out or the peer closed first. That is not academic. A node can be dead to the control plane and still perfectly reachable on BIN: hung, half-open, or partitioned on one plane only. Measured on a 3-node staging cluster, with the victim's control plane isolated and BIN deliberately left reachable: clusterer_controller purged the node at its 30 s deadline and both survivors still held ESTABLISHED connections to it, in both directions. The close lives here rather than in clusterer_controller on purpose. The BIN transport belongs to clusterer, and the controller drives it through this API instead of reaching into the core's TCP layer itself. It is exposed as close_node_conn() for the case where only the transport should be dropped; remove_node() now does it for you. Two ordering constraints, both load-bearing: - the url is copied out BEFORE remove_node_list(), which frees the node; - the close happens AFTER cl_list_lock is released and after the capability event_cb callbacks have run, because a callback may still send a final BIN packet to the departing node and pulling the socket first would only turn that into an error. get_node_by_id() walks node_list, which does not contain current_node, so this can never close our own listener. (cherry picked from commit d1d12a5025df4481d939fad5bfaff06307566026) --- modules/clusterer/clusterer_ctrl.c | 94 ++++++++++++++++++++++++++++++ modules/clusterer/clusterer_ctrl.h | 5 ++ 2 files changed, 99 insertions(+) diff --git a/modules/clusterer/clusterer_ctrl.c b/modules/clusterer/clusterer_ctrl.c index de9d5052107..e8e77fb693b 100644 --- a/modules/clusterer/clusterer_ctrl.c +++ b/modules/clusterer/clusterer_ctrl.c @@ -23,6 +23,7 @@ #include "clusterer.h" /* LS_DOWN, do_actions_node_ev, MAX_NO_CLUSTERS */ #include "sharing_tags.h" #include "topology.h" /* delete_neighbour */ /* shtag_event_handler */ +#include "../../net/net_tcp.h" /* tcp_close_connection */ #include "clusterer_ctrl.h" /* This whole API is compiled only when the clusterer_controller module is part @@ -152,6 +153,80 @@ int clusterer_ctrl_add_node(int cluster_id, int node_id, str *bin_url) return 0; } +/* Longest BIN url we will copy out of a node before it is freed. + * "bins:[]:" fits comfortably. */ +#define CL_CTRL_URL_BUF 128 + +/** + * close_node_bin_conn() - drop the BIN connection to a node's url. + * + * Removing a node from the topology does NOT touch the transport: clusterer + * sends via msg_send(send_sock, proto, &node->addr), and the core keeps the + * TCP connection in its own table keyed by destination. So a node that has + * been declared dead keeps a perfectly good socket open until the TCP layer + * times it out or the peer closes first - and a node that is dead to the + * control plane but still reachable on BIN (hung, or partitioned only on the + * control plane) holds it open indefinitely. + * + * This lives in clusterer rather than in clusterer_controller on purpose: the + * BIN transport is clusterer's, and the controller should drive it through + * this API rather than reaching into the core's TCP layer itself. + * + * Must be called WITHOUT cl_list_lock held - closing dispatches to the TCP + * main process, and the url must have been copied out of the node first, + * because remove_node_list() frees the node. + */ +static void close_node_bin_conn(int cluster_id, str *url) +{ + if (!url || !url->len) + return; + + LM_INFO("clusterer: [cluster %d] dropping BIN connection to %.*s\n", + cluster_id, url->len, url->s); + + /* 0 is also returned when there is simply no open connection, which is a + * perfectly normal state - only a parse/lookup failure is worth a word. */ + if (tcp_close_connection(url) < 0) + LM_DBG("clusterer: [cluster %d] could not close BIN connection to " + "%.*s\n", cluster_id, url->len, url->s); +} + +/** + * clusterer_ctrl_close_node_conn() - close a node's BIN connection, leaving + * the node in the topology. + * + * Exposed so the controller can drop a peer's transport explicitly. The + * usual path is clusterer_ctrl_remove_node(), which now does this for you. + */ +int clusterer_ctrl_close_node_conn(int cluster_id, int node_id) +{ + cluster_info_t *cl; + node_info_t *node; + char buf[CL_CTRL_URL_BUF]; + str url = {NULL, 0}; + + lock_start_read(cl_list_lock); + cl = get_cluster_by_id(cluster_id); + node = cl ? get_node_by_id(cl, node_id) : NULL; + if (node && node->url.s && node->url.len > 0 && + node->url.len < (int)sizeof buf) { + memcpy(buf, node->url.s, node->url.len); + buf[node->url.len] = '\0'; + url.s = buf; + url.len = node->url.len; + } + lock_stop_read(cl_list_lock); + + if (!url.len) { + LM_WARN("clusterer: close_node_conn: node %d not found in cluster %d, " + "or it has no usable url\n", node_id, cluster_id); + return -1; + } + + close_node_bin_conn(cluster_id, &url); + return 0; +} + /** * clusterer_ctrl_remove_node() - remove a departed peer at runtime. */ @@ -159,6 +234,9 @@ int clusterer_ctrl_remove_node(int cluster_id, int node_id) { cluster_info_t *cl; node_info_t *node; + /* copied out under the lock, used after the node has been freed */ + char url_buf[CL_CTRL_URL_BUF]; + str url = {NULL, 0}; lock_start_write(cl_list_lock); @@ -196,6 +274,16 @@ int clusterer_ctrl_remove_node(int cluster_id, int node_id) } } + /* remove_node_list() frees the node, so take the BIN url now - we need it + * after the lock is dropped to close the connection. */ + if (node->url.s && node->url.len > 0 && + node->url.len < (int)sizeof url_buf) { + memcpy(url_buf, node->url.s, node->url.len); + url_buf[node->url.len] = '\0'; + url.s = url_buf; + url.len = node->url.len; + } + /* Remove node from list, then fire callbacks outside the lock. * Callbacks (dialog rcv_cluster_event) call back into clusterer * to send BIN packets and need cl_list_lock for read. */ @@ -211,6 +299,11 @@ int clusterer_ctrl_remove_node(int cluster_id, int node_id) report_node_state(CLUSTER_NODE_DOWN, cluster_id, node_id); } + /* Last, and deliberately after the callbacks: a capability's event_cb may + * still want to send a final BIN packet to the departing node, and pulling + * the socket out from under it first would only turn that into an error. */ + close_node_bin_conn(cluster_id, &url); + LM_INFO("clusterer: [cluster %d] removed node_id=%d\n", cluster_id, node_id); return 0; @@ -392,6 +485,7 @@ int load_clusterer_ctrl_binds(clusterer_ctrl_binds_t *binds) binds->set_my_identity = clusterer_ctrl_set_identity; binds->add_node = clusterer_ctrl_add_node; binds->remove_node = clusterer_ctrl_remove_node; + binds->close_node_conn = clusterer_ctrl_close_node_conn; binds->update_identity = clusterer_ctrl_update_identity; binds->sync_current_id = clusterer_ctrl_sync_current_id; binds->activate_backup_shtags = clusterer_ctrl_activate_backup_shtags; diff --git a/modules/clusterer/clusterer_ctrl.h b/modules/clusterer/clusterer_ctrl.h index d5ba9f15289..67a8ec30e5b 100644 --- a/modules/clusterer/clusterer_ctrl.h +++ b/modules/clusterer/clusterer_ctrl.h @@ -88,6 +88,11 @@ typedef struct clusterer_ctrl_binds { * @return 0 on success, -1 if cluster or node not found */ int (*remove_node)(int cluster_id, int node_id); + + /* Close a node's BIN connection while leaving it in the topology. + * remove_node() already does this for you; use this only to drop the + * transport on its own. Returns -1 if the node is unknown. */ + int (*close_node_conn)(int cluster_id, int node_id); /** * update_identity() — correct this node's node_id after master assignment. * From 6cd3bfbec8b01b1000629c7e94fbe0858cbc1ee6 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 10:03:25 +1000 Subject: [PATCH 09/32] clusterer: close every BIN connection to a departing node, and say what happened Two defects in the previous commit, both found by measuring on a live cluster rather than by reading the code back. 1. It closed at most ONE connection. There are normally two per peer - the one we dialled out to its BIN port and the one it dialled in to ours - and both are reachable by the same address, because the core registers an alias for the peer's advertised port on an accepted connection as well. That aliasing is what lets an inbound connection be reused for outbound sends. tcp_close_connection() closes exactly one per call and flags it F_CONN_FORCE_CLOSED, which makes the next lookup skip it, so the fix is to loop until it reports nothing left. That terminates by construction: each pass removes one connection from the candidate set. The cap is paranoia against a future core change that stops setting the flag. 2. The logging announced intent, not outcome. It printed 'dropping BIN connection' before calling, and only said anything afterwards on a negative return - but tcp_close_connection() returns 1 for closed and 0 for nothing found, so the interesting case was silent and the log could not distinguish 'closed it' from 'there was nothing there'. On the first live test that made a no-op look like a success. It now reports the count it actually closed, and an outright failure is an error rather than a debug line. (cherry picked from commit e5838d3b8036fd5d5e7da51ac7b40a9980d85300) --- modules/clusterer/clusterer_ctrl.c | 42 +++++++++++++++++++++++++----- 1 file changed, 35 insertions(+), 7 deletions(-) diff --git a/modules/clusterer/clusterer_ctrl.c b/modules/clusterer/clusterer_ctrl.c index e8e77fb693b..31fac9dfe71 100644 --- a/modules/clusterer/clusterer_ctrl.c +++ b/modules/clusterer/clusterer_ctrl.c @@ -156,6 +156,8 @@ int clusterer_ctrl_add_node(int cluster_id, int node_id, str *bin_url) /* Longest BIN url we will copy out of a node before it is freed. * "bins:[]:" fits comfortably. */ #define CL_CTRL_URL_BUF 128 +/* paranoia bound on the close loop below */ +#define CL_CTRL_MAX_CONN_CLOSE 16 /** * close_node_bin_conn() - drop the BIN connection to a node's url. @@ -178,17 +180,43 @@ int clusterer_ctrl_add_node(int cluster_id, int node_id, str *bin_url) */ static void close_node_bin_conn(int cluster_id, str *url) { + int closed = 0, rc = 0; + if (!url || !url->len) return; - LM_INFO("clusterer: [cluster %d] dropping BIN connection to %.*s\n", - cluster_id, url->len, url->s); + /* There is usually MORE THAN ONE connection to a peer: the one we dialled + * out to its BIN port, and the one it dialled in to ours. Both are found + * by this address, because the core registers an alias for the peer's + * advertised port on an accepted connection too - that is how an inbound + * connection gets reused for outbound sends. + * + * tcp_close_connection() closes exactly ONE per call, and sets + * F_CONN_FORCE_CLOSED on it, which makes the next lookup skip it. So loop + * until it reports nothing left. This terminates: every pass takes one + * connection out of the candidate set. The cap is pure paranoia against a + * core change that stopped setting that flag. + * + * Return values are 1 = found and closed, 0 = nothing (left) to close, + * -1 = bad address or the close could not be dispatched. */ + while ((rc = tcp_close_connection(url)) == 1) + if (++closed >= CL_CTRL_MAX_CONN_CLOSE) { + LM_WARN("clusterer: [cluster %d] stopped after closing %d BIN " + "connections to %.*s - more may remain\n", + cluster_id, closed, url->len, url->s); + return; + } - /* 0 is also returned when there is simply no open connection, which is a - * perfectly normal state - only a parse/lookup failure is worth a word. */ - if (tcp_close_connection(url) < 0) - LM_DBG("clusterer: [cluster %d] could not close BIN connection to " - "%.*s\n", cluster_id, url->len, url->s); + if (rc < 0) + LM_ERR("clusterer: [cluster %d] failed to close a BIN connection to " + "%.*s (closed %d before the failure)\n", + cluster_id, url->len, url->s, closed); + else if (closed) + LM_INFO("clusterer: [cluster %d] closed %d BIN connection(s) to %.*s\n", + cluster_id, closed, url->len, url->s); + else + LM_DBG("clusterer: [cluster %d] no open BIN connection to %.*s\n", + cluster_id, url->len, url->s); } /** From da9516eeabd730a304a27d484619a84004edb341 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 10:36:31 +1000 Subject: [PATCH 10/32] clusterer_controller: never learn a peer from our own looped-back roster The master's MEMBER_LIST and NODE_ASSIGN are multicast, and IP_MULTICAST_LOOP delivers them back to the master itself. Both handlers ran the learn path on the payload IPs, so the master re-inserted every node its own announcement named - including a peer it had just purged. The re-seeded copy kept the roster naming the dead node, every receiver re-learned it one window later, and the purge could never converge. Observed live on a 3-node staging cluster: 'new peer ' every 35 s, in lockstep with the purge cycle, for a node whose control plane was provably blocked the whole time. An announcement we authored was built FROM this table; it cannot teach us anything. Receivers other than the author still learn membership from these packets exactly as before - that is the legitimate propagation path, and a node the master stops listing now ages out everywhere one window later, which restores the termination argument cl_ctr_learn_peer_locked() was written around. This also explains why a resurrected peer escaped removal forever: a MEMBER_LIST entry carries only IP + is_master, so the re-learned copy had node_id 0, and cl_ctr_prune_stale() only propagates removal to clusterer for node_id > 0. 'removed node_id' fired exactly once per incident and never again. (cherry picked from commit bbfc93cf57742fc89f11d531299d326f4dac3376) --- .../clusterer_controller.c | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 9b4083aac35..bd7cd11b0b0 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -4330,8 +4330,16 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, ip_buf[CL_CTR_MAX_IP_LEN] = '\0'; if (ip_buf[0] == '\0') continue; - /* membership, not liveness - see cl_ctr_learn_peer_locked() */ - cl_ctr_learn_peer_locked(ip_buf, cl); + /* Membership, not liveness - see cl_ctr_learn_peer_locked(). And never + * from our OWN echo: IP_MULTICAST_LOOP delivers the master's list back + * to the master, and learning from it re-inserts a peer the master just + * purged. That closed a loop: the re-seeded copy kept the roster + * naming the dead node, every receiver re-learned it each window, and + * the purge could never converge (observed live as 'new peer' every + * 35 s for a node whose control plane was provably blocked). The + * packet was built FROM this table; it cannot teach us anything. */ + if (strcmp(sender_ip, my_ip) != 0) + cl_ctr_learn_peer_locked(ip_buf, cl); for (_j = 0; _j < cl->peers->count; _j++) { if (strcmp(cl->peers->entries[_j].ip, ip_buf) == 0) { cl->peers->entries[_j].last_seq = 0; @@ -4628,7 +4636,11 @@ static void cl_ctr_handle_node_assign(const char *payload, int payload_len, * node that must prune, so cl_ctr_prune_stale() could never fire and the * corpse was immortal - measured on a 3-node staging cluster: a node isolated * for 4 minutes against a 30 s deadline was still a member on every survivor. */ - cl_ctr_learn_peer_locked(ip, cl); + /* Same self-echo rule as the member-list path: the master must not + * re-learn a peer from its own looped-back announcement. Receivers other + * than the author still learn membership here as before. */ + if (strcmp(sender_ip, my_ip) != 0) + cl_ctr_learn_peer_locked(ip, cl); cl_ctr_update_peer_bin_locked(ip, node_id, bin_cnt, (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, cl); From 2732d5d286da79f4c5d98b70eed6b8ce279ac264 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 10:36:31 +1000 Subject: [PATCH 11/32] clusterer: a controller-managed cluster never adds nodes from the wire Membership authority: in a controller-managed cluster the only way in is the controller's JOIN handshake, after which the controller calls clctl.add_node(). Wire self-discovery is the zero-config mode's mechanism and must not apply - but cl_db_mode() deliberately reads 0 for controller-managed clusters (so the controller's runtime add/remove works), and that same predicate gated the learn paths, which put controller-managed clusters on exactly the self-discovery behaviour they must not have. The consequence, measured on a 3-node staging cluster: a node the controller had expelled talked its way straight back in within one ping interval - PING from unknown -> UNKNOWN_ID -> NODE_DESCRIPTION -> add_node() - and the BIN connections that had just been closed were re-dialled (2 established grew to 4). A node dead to the control plane but alive on BIN is exactly the hung / one-plane-partitioned failure the purge exists for. New cl_ctr_owns_membership() (0 in non-controller builds, so default builds are unchanged) now gates all three wire-learn sites: - handle_full_top_update's two unknown-node learns, alongside cl_db_mode; - handle_internal_msg_unknown's NODE_DESCRIPTION add_node - which also no longer gossips the stranger's description onward via flood_message. The UNKNOWN_ID reply to a stranger's ping is kept: telling the node we do not know it is what prompts its controller to re-join properly. The ignore is logged at INFO, rate-limited to one line per 30 s, because a live expelled node re-announces on every ping cycle. (cherry picked from commit f2903b18b17fb0fd32698e081e913a9cda73158d) --- modules/clusterer/node_info.h | 17 +++++++++++++++++ modules/clusterer/topology.c | 29 +++++++++++++++++++++++++---- 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/modules/clusterer/node_info.h b/modules/clusterer/node_info.h index 953456bc790..b90db1a567a 100644 --- a/modules/clusterer/node_info.h +++ b/modules/clusterer/node_info.h @@ -222,8 +222,25 @@ static inline int cl_db_mode(const struct cluster_info *cl) { return (cl && cl->controller_managed) ? 0 : db_mode; } + +/* Membership authority. A controller-managed cluster must never ADD a node + * because of something that arrived on the wire - an unknown node's + * NODE_DESCRIPTION, or a neighbour's topology update naming a node we do not + * have. Wire self-discovery is the zero-config mode's mechanism; in a + * controller-managed cluster the ONLY way in is the controller's own JOIN + * handshake, after which the controller calls clctl.add_node(). Without this + * a node the controller had expelled talked its way straight back in: + * PING -> UNKNOWN_ID -> NODE_DESCRIPTION -> add_node(). (cl_db_mode() cannot + * express this: it deliberately reads 0 for controller-managed clusters so the + * controller's runtime add/remove works, which is exactly what put these + * clusters on the self-discovery paths.) */ +static inline int cl_ctr_owns_membership(const struct cluster_info *cl) +{ + return cl && cl->controller_managed; +} #else #define cl_db_mode(cl) (db_mode) +#define cl_ctr_owns_membership(cl) 0 #endif int update_db_state(int cluster_id, int node_id, int state); diff --git a/modules/clusterer/topology.c b/modules/clusterer/topology.c index b6baa9a413e..cc8f4bba9cd 100644 --- a/modules/clusterer/topology.c +++ b/modules/clusterer/topology.c @@ -1058,7 +1058,8 @@ void handle_full_top_update(bin_packet_t *packet, node_info_t *source, top_node = get_node_by_id(source->cluster, top_node_id[i]); if (!skip && !top_node) { - if (cl_db_mode(source->cluster)) { + if (cl_db_mode(source->cluster) || + cl_ctr_owns_membership(source->cluster)) { skip = 1; } else if (!top_node_info[i][0]) { LM_WARN("Unknown node id [%d] in topology update with " @@ -1101,7 +1102,8 @@ void handle_full_top_update(bin_packet_t *packet, node_info_t *source, for (j = 0; j < top_node_info[i][3]; j++) { top_neigh = get_node_by_id(source->cluster, top_node_info[i][j+4]); if (!top_neigh && top_node_info[i][j+4] != cluster_self_id(source->cluster)) { - if (cl_db_mode(source->cluster)) + if (cl_db_mode(source->cluster) || + cl_ctr_owns_membership(source->cluster)) continue; for (n_idx = 0; n_idx < no_nodes && top_node_info[i][j+4] != top_node_id[n_idx]; @@ -1206,9 +1208,28 @@ void handle_internal_msg_unknown(bin_packet_t *received, cluster_info_t *cl, bin_pop_str(received, &str_vals[STR_VALS_SIP_ADDR_COL]); bin_pop_int(received, &int_vals[INT_VALS_PRIORITY_COL]); bin_pop_int(received, &int_vals[INT_VALS_NO_PING_RETRIES_COL]); - add_node(received, cl, src_node_id, str_vals, int_vals); + if (!cl_ctr_owns_membership(cl)) { + add_node(received, cl, src_node_id, str_vals, int_vals); - flood_message(received, cl, src_node_id, 0); + flood_message(received, cl, src_node_id, 0); + } else { + /* Membership is the controller's; an expelled-but-alive node + * announcing itself over BIN does not get back in this way, + * and we do not gossip its description onward either. It + * must (re)JOIN through the controller, which will then + * clctl.add_node() it everywhere. Rate-limited: a live + * node in this state re-announces on every ping cycle. */ + static time_t desc_note; + time_t now_ts = time(NULL); + + if (now_ts - desc_note >= 30) { + desc_note = now_ts; + LM_INFO("Ignoring node description from node [%d]: " + "cluster %d membership is controller-managed - " + "the node must (re)join via the controller\n", + src_node_id, cl->cluster_id); + } + } break; default: LM_DBG("Ignoring message, type: %d from unknown source, id [%d]\n", From fc5d33257afa47372e5790059bb2545c6383c896 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 10:58:24 +1000 Subject: [PATCH 12/32] clusterer_controller: a yielding master must re-JOIN, not just concede Split-brain resolution had two merge paths that behaved differently. A master that learned of a superior via MASTER_BEACON went through cl_ctr_rejoin_superior_master() - demote, adopt the master, and send a JOIN_REQ. A master that learned of it via MASTER_ALIVE merely yielded: it recorded the winner and stopped asserting mastership, and that was all. Yielding alone is not a merge. The winner learned the yielding node only from that packet's sender-upsert - a peer entry with node_id 0 - and the ONLY place a node_id is ever assigned is handle_join_req(). Without a JOIN_REQ the yielded node sits in the winner's table as id 0 forever: it is never named in a NODE_ASSIGN, so it is never added to clusterer on any node, and its membership digest can never match the master's MASTER_ALIVE - which turns into a RESYNC-per-second livelock as the master keeps 're-broadcasting full state' that structurally cannot contain the missing node. Observed live on the 3-node staging cluster after a partition heal: the controller admitted the returning node as a member everywhere, but the master held it at node_id=0, clusterer never learned it, and RESYNC fired once a second indefinitely. This had been masked before the membership-authority change: clusterer's wire self-discovery quietly re-added the node at the BIN level, hiding the controller-level livelock. The yield path now calls cl_ctr_rejoin_superior_master() - identical to the beacon merge - which demotes, arms the dead watchdog, and sends the JOIN_REQ (join_pending-guarded, so an exchange already in flight is not stomped). The function takes the discovery vector as a string so the merge logs say which path found the superior. (cherry picked from commit 5084afa9219feb361a215312263de4a7b96b5864) --- .../clusterer_controller.c | 31 +++++++++++++------ 1 file changed, 22 insertions(+), 9 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index bd7cd11b0b0..aa809a060dd 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -1050,6 +1050,8 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_len, cl_ctr_cluster_t *cl, const struct sockaddr *src, socklen_t src_len); +static void cl_ctr_rejoin_superior_master(cl_ctr_cluster_t *cl, + const char *superior_ip, const char *via); static void cl_ctr_handle_node_assign(const char *payload, int payload_len, const char *sender_ip, cl_ctr_cluster_t *cl); static void cl_ctr_handle_goodbye(int sock, const char *src_ip, cl_ctr_cluster_t *cl); @@ -4701,9 +4703,10 @@ static void cl_ctr_handle_master_alive(const char *sender_ip, cl_ctr_cluster_t * if (i_am_master) { if (!from_self && ip_to_num(sender_ip) > ip_to_num(my_ip)) { - /* Split-brain: a higher-IP node also claims mastership - yield. */ - cl_ctr_upsert_peer_locked(sender_ip, cl); - cl_ctr_apply_master_from_list_locked(sender_ip, cl); + /* Split-brain: a higher-IP node also claims mastership - yield. + * The actual demotion happens below, OUTSIDE the lock, through + * cl_ctr_rejoin_superior_master(), which must not be called with + * the peers lock held. */ yielded = 1; } /* else: my own loopback, or a lower-IP claimant that will yield to us */ @@ -4720,7 +4723,16 @@ static void cl_ctr_handle_master_alive(const char *sender_ip, cl_ctr_cluster_t * LM_INFO("clusterer_controller: [cluster %d] yielding mastership to " "higher-IP master %s (split-brain resolution)\n", cl->cluster_id, sender_ip); - cl_ctr_arm_master_timers(cl, 0); /* stop MASTER_ALIVE, arm dead watchdog */ + /* Yielding alone is NOT a merge. The winner learned us only from this + * packet's sender-upsert, i.e. as a peer with node_id 0, and the ONLY + * place a node_id is ever assigned is its handle_join_req(). Without + * a JOIN_REQ we would sit in its table as id 0 forever: never named in + * a NODE_ASSIGN, so never added to clusterer anywhere, while our + * membership digest never matches its MASTER_ALIVE - a RESYNC-per- + * second livelock, observed live on the 3-node staging cluster after + * a partition heal. So re-join properly, exactly as the beacon merge + * path does; rejoin also demotes us and arms the dead watchdog. */ + cl_ctr_rejoin_superior_master(cl, sender_ip, "MASTER_ALIVE"); } /* Non-masters (including a node that just yielded) watch the keepalive. */ @@ -4772,7 +4784,8 @@ static void cl_ctr_handle_master_alive(const char *sender_ip, cl_ctr_cluster_t * * lands we can decrypt the superior partition's session traffic and are fully * merged. Call WITHOUT cl->peers->lock held. */ -static void cl_ctr_rejoin_superior_master(cl_ctr_cluster_t *cl, const char *superior_ip) +static void cl_ctr_rejoin_superior_master(cl_ctr_cluster_t *cl, const char *superior_ip, + const char *via) { int was_master; @@ -4785,12 +4798,12 @@ static void cl_ctr_rejoin_superior_master(cl_ctr_cluster_t *cl, const char *supe if (was_master) { LM_INFO("clusterer_controller: [cluster %d] superior master %s found via " - "beacon - demoting and merging (split-brain resolution)\n", - cl->cluster_id, superior_ip); + "%s - demoting and merging (split-brain resolution)\n", + cl->cluster_id, superior_ip, via); cl_ctr_arm_master_timers(cl, 0); /* stop MASTER_ALIVE, arm dead watchdog */ } else { LM_INFO("clusterer_controller: [cluster %d] moving to superior master %s " - "via beacon (split-brain merge)\n", cl->cluster_id, superior_ip); + "via %s (split-brain merge)\n", cl->cluster_id, superior_ip, via); } /* Drop any locally-held active shtag now that we are no longer master. */ @@ -4856,7 +4869,7 @@ static void cl_ctr_handle_master_beacon(const char *sender_ip, uint16_t sender_c if (!superior) return; /* we outrank the sender; it will yield to us on our beacon */ - cl_ctr_rejoin_superior_master(cl, sender_ip); + cl_ctr_rejoin_superior_master(cl, sender_ip, "beacon"); } /** From e8ae4f854f83c87f737e8fc412a2909c45687d2b Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Mon, 10 Aug 2026 17:42:48 +1000 Subject: [PATCH 13/32] clusterer_controller: document that RELIABLE needs a serialised channel The flag's note read like a cost trade-off - acks are not free, so opt in when you want them - and recommended idempotent-and-self-retrying consumers as the better default. Both true, and both beside the point: on a channel with concurrent traffic the flag does not work at all. Every packet carries a sequence number and the anti-replay guard drops anything not strictly greater than the last accepted from that peer. A retransmit re-sends cached bytes, so it carries its original seq. Serialised 1:1 - the join handshake this was built for - is safe, because nothing can overtake it. The moment a second message can, the retransmit lands behind a higher seq and is discarded, the receiver re-acks it so the sender stops trying, and neither end says anything above debug level. Measured on the cross-node cache pull channel: the flag made delivery WORSE, 61% -> 49% under 40% reply loss at two to three times the packets, with zero retransmitted payloads ever delivered. That change was reverted; this note is so the next consumer does not repeat it. Comment only - no functional change. (cherry picked from commit 321f8cfb5592a6ba91fc5560d0e961d7fd2be185) --- modules/clusterer_controller/api.h | 35 +++++++++++++++++++++++++----- 1 file changed, 30 insertions(+), 5 deletions(-) diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h index 5a3d8a5d970..c94ddb0db6f 100644 --- a/modules/clusterer_controller/api.h +++ b/modules/clusterer_controller/api.h @@ -60,11 +60,36 @@ #define CLCTR_SEND_TO_SELF (1 << 0) /* Ask for the message to be acknowledged, and resent while it is not. * - * Costs one ACK per recipient, so a reliable send to the whole cluster is - * N-1 packets back where a plain one was nothing - deliberately opt-in, and - * deliberately per send rather than per channel: most consumer traffic is - * better served by being idempotent and retried by its own logic, which is - * how the cross-node cache fetch works and why it asks for nothing here. */ + * ONLY USE THIS ON A CHANNEL WHERE ONE MESSAGE PER PEER IS IN FLIGHT AT A + * TIME. It is not a general-purpose reliability option, and on a channel + * with concurrent traffic it does not merely cost extra packets - it does + * not work, and it fails silently. + * + * Why: every packet carries a sequence number, and a receiver drops anything + * whose seq is not strictly greater than the last it accepted from that peer + * (cl_ctr_check_and_update_seq(), the anti-replay guard). A retransmit + * re-sends the cached bytes, so it carries its ORIGINAL seq. On a serialised + * 1:1 exchange - the join handshake this was built for - nothing else is in + * flight from that peer, so the retransmit is still the highest seq and is + * accepted. As soon as a second message can overtake it, the retransmit + * arrives behind a higher seq and is discarded as out of order. The receive + * path then re-ACKs it, deliberately, so the sender stops retransmitting and + * believes it delivered. Nothing is logged above debug level at either end. + * + * Measured, rather than reasoned about: applying this flag to the + * cross-node cache pull channel (concurrent by nature) made the delivery rate + * WORSE - 61% -> 49% under 40% reply loss, at two to three times the packet + * count - and not one retransmitted payload was ever delivered. See the + * clusterer_controller notes for the harness. + * + * So most consumer traffic is better served by being idempotent and retried + * by its own logic, which is how the cross-node cache fetch works and why it + * asks for nothing here. That was always the recommendation; the point of + * this note is that for a concurrent channel it is the only thing that works. + * + * The cost, where it IS applicable: one ACK per recipient, so a reliable send + * to the whole cluster is N-1 packets back where a plain one was nothing - + * hence opt-in, and per send rather than per channel. */ #define CLCTR_SEND_RELIABLE (1 << 1) /* limits a consumer can rely on */ From d7a5ca071672a4895af08ee4b1b0f64ede10a8bc Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Tue, 11 Aug 2026 23:12:16 +1000 Subject: [PATCH 14/32] clusterer_controller: size the transmit buffers from the bound that gates them cc_max_payload is derived from the interface MTU at mod_init, but the two buffers it gates were sized from the COMPILE-TIME CLCTR_MAX_PAYLOAD: cl_ctr_script_send() char buf[CLCTR_MAX_PAYLOAD]; /* 1300 */ guard: 2 + taglen + gen_msg->len > cc_max_payload cl_ctr_rpc_consumer_send() char pkt[CL_CTR_CONSUMER_PKT_MAX]; /* 83+1300 */ guard: cl_ctr_consumer_submit(), payload_len > cc_max_payload On any link above MTU 1411 the gate admits more than the buffer holds. This is not a corner case: build host 222 runs enp6s18 at MTU 9000, so cc_max_payload is 8889 against a 1300-byte stack array, and the prod gateways log "interface ens18 MTU=1500, max consumer payload=1389". A 1350-byte script message passes the gate and overruns by 52 bytes; on the 9000 link the ceiling is 7589. Fixed by making the buffers honour the bound, NOT by capping the bound. An upper clamp on cc_max_payload would have been a smaller diff and would have made the MTU derivation dead code - the capability is wanted, and the host this was found on is itself on a jumbo link. Both buffers are now pkg_malloc'd in mod_init from cc_max_payload, after the MTU probe has settled it. mod_init is pre-fork, so every worker inherits its own copy-on-write copy: no sharing between processes, and no per-call allocation on a path cachedb_perf's pull replies use for every message. Neither send function can re-enter itself - a script function runs to completion inside one route execution, and the consumer send runs in the cluster worker's own loop. CL_CTR_PKT_OVERHEAD is split out of CL_CTR_CONSUMER_PKT_MAX so the packet size can be computed at runtime; the macro stays for the compile-time contract in api.h. One real cap added, which does not touch jumbo frames: a datagram still has to fit a UDP payload, and loopback reports MTU 65536, which computed a payload one byte past what sendto() accepts. Clamped to CL_CTR_UDP_PAYLOAD_MAX minus the overhead. At MTU 9000 the bound is unchanged at 8889. Both sites also now bound against their own buffer as well as against cc_max_payload - the sibling cmd_cl_ctr_send_req_list() has always guarded with sizeof(buf), and the gap between the two forms is what this was. The consumer path drops with an error if the two ever disagree again. PROVEN fail-then-pass with /dn/task86b, one node, both prefixes built with -fstack-protector-all as a DETECTOR (a 52-byte overrun of a stack array otherwise may corrupt nothing observable and the run would prove nothing): before 1350 bytes -> *** stack smashing detected ***: terminated, workers alive 0 after 1350 bytes -> 200, X-T86B: SENT, workers alive 17 5000 bytes -> 200, X-T86B: SENT, workers alive 17 (the jumbo case: proof the headroom is usable, not merely safe) The rig needs one node because the overrun is in the memcpy that builds the frame, before anything is sent - no cluster has to form and no peer has to exist. Not exposed on the fleet today: none of .241/.242/.243 call cl_ctr_send_req, cl_ctr_broadcast_req or cl_ctr_send_rpl in their configs. The consumer path is reachable by any API consumer following the documented cc_max_payload contract in api.h:98-103. Full 131-module build, 0 errors, 0 stamp mismatches. (cherry picked from commit b3227013d03ddb06322ebed77706d705998a07be) --- .../clusterer_controller.c | 89 +++++++++++++++++-- 1 file changed, 83 insertions(+), 6 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index aa809a060dd..97ea39f4334 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -266,9 +266,19 @@ static const unsigned char CL_CTR_CONSUMER_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x0 * consumer's problem (the API contract says fall back to BIN for bulk). */ #define CL_CTR_MAX_CHANNELS 8 #define CL_CTR_CONSUMER_HDR_SZ (CL_CTR_NODE_ID_SZ + 1) -#define CL_CTR_CONSUMER_PKT_MAX (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ \ +/* Everything a consumer datagram carries BESIDES the payload. Split out from + * the old CL_CTR_CONSUMER_PKT_MAX because the packet buffer is now sized at + * runtime from cc_max_payload: a jumbo-frame link is meant to carry a bigger + * payload, and a buffer fixed at CLCTR_MAX_PAYLOAD could not - it just let the + * MTU-derived bound run off the end of it. */ +#define CL_CTR_PKT_OVERHEAD (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ \ + CL_CTR_CONSUMER_HDR_SZ + CLCTR_MAX_CHAN_LEN \ - + CLCTR_MAX_PAYLOAD + CL_CTR_TAG_SZ) + + CL_CTR_TAG_SZ) +#define CL_CTR_CONSUMER_PKT_MAX (CL_CTR_PKT_OVERHEAD + CLCTR_MAX_PAYLOAD) +/* A datagram cannot exceed the UDP payload maximum no matter what the MTU + * says - loopback reports 65536, which would otherwise compute a payload one + * byte past what sendto() can accept. */ +#define CL_CTR_UDP_PAYLOAD_MAX 65507 #define CL_CTR_PKT_MASTER_BEACON 0x0A /* master-only announce (BOOTSTRAP key) so * masters with divergent session keys can * still discover each other and merge a @@ -774,6 +784,27 @@ static char my_interface_buf[IF_NAMESIZE]; * Read-only after mod_init; safe to access from any process. */ int cc_max_payload = CLCTR_MAX_PAYLOAD; +/* The two transmit buffers, sized from cc_max_payload once the MTU is known. + * + * They were `char buf[CLCTR_MAX_PAYLOAD]` and `char pkt[CL_CTR_CONSUMER_PKT_MAX]` + * on the stack, while the length gates in front of them tested against the + * MTU-derived cc_max_payload - so on any link above MTU 1411 a caller could + * pass the gate and write past the end. A production gateway logs + * "interface ens18 MTU=1500, max consumer payload=1389" against a 1300-byte + * buffer: an 89-byte stack overrun with caller-supplied bytes. + * + * pkg, allocated in mod_init, which runs PRE-FORK: every worker inherits its + * own copy-on-write copy, so there is no sharing between processes and no + * per-call allocation on the send path (cachedb_perf's pull replies go through + * the consumer one for every message). Each is written only by the process + * that owns it, and neither send function can re-enter itself - script + * functions run to completion inside one route execution, and the consumer + * send runs in the cluster worker's own loop. */ +static char *cl_ctr_script_buf; +static int cl_ctr_script_buf_sz; +static char *cl_ctr_consumer_pkt; +static int cl_ctr_consumer_pkt_sz; + /* Local node identity - populated at mod_init by scanning the config file */ static uint16_t my_node_id = 0; @@ -7058,6 +7089,11 @@ static int mod_init(void) cc_max_payload = _mtu - _overhead; if (cc_max_payload < CLCTR_MAX_PAYLOAD) cc_max_payload = CLCTR_MAX_PAYLOAD; + /* However large the MTU claims to be, one datagram still has + * to fit a UDP payload. Loopback reports 65536, which lands a + * byte past what sendto() accepts. */ + if (cc_max_payload > CL_CTR_UDP_PAYLOAD_MAX - CL_CTR_PKT_OVERHEAD) + cc_max_payload = CL_CTR_UDP_PAYLOAD_MAX - CL_CTR_PKT_OVERHEAD; LM_INFO("clusterer_controller: interface %s MTU=%d, " "max consumer payload=%d bytes\n", my_interface_buf, _mtu, cc_max_payload); @@ -7070,6 +7106,22 @@ static int mod_init(void) } } + /* Size the transmit buffers from the bound that gates them, now that the + * MTU probe above has settled cc_max_payload. Doing it here rather than at + * compile time is the whole point: a jumbo-frame link raises + * cc_max_payload, and a CLCTR_MAX_PAYLOAD-sized buffer could not hold what + * that bound then admits. */ + cl_ctr_script_buf_sz = cc_max_payload; + cl_ctr_consumer_pkt_sz = CL_CTR_PKT_OVERHEAD + cc_max_payload; + cl_ctr_script_buf = pkg_malloc(cl_ctr_script_buf_sz); + cl_ctr_consumer_pkt = pkg_malloc(cl_ctr_consumer_pkt_sz); + if (!cl_ctr_script_buf || !cl_ctr_consumer_pkt) { + LM_ERR("clusterer_controller: no pkg memory for the %d/%d byte " + "transmit buffers\n", cl_ctr_script_buf_sz, + cl_ctr_consumer_pkt_sz); + return -1; + } + if (cl_ctr_discover_bin_sockets() < 0) return -1; @@ -7445,12 +7497,28 @@ static void cl_ctr_rpc_consumer_send(int sender, void *param) { struct cl_ctr_consumer_job *job = (struct cl_ctr_consumer_job *)param; cl_ctr_cluster_t *cl = job->cl; - char pkt[CL_CTR_CONSUMER_PKT_MAX]; + /* the per-process buffer sized from cc_max_payload at mod_init; + * a CL_CTR_CONSUMER_PKT_MAX-sized stack array could not hold what + * the MTU-derived bound in cl_ctr_consumer_submit() admits */ + char *pkt = cl_ctr_consumer_pkt; char dst_ip[CL_CTR_MAX_IP_LEN + 1]; uint32_t seq; uint16_t id_be; int plain_len, i, to_self = 0, on_wire = 1, reliable = 0; + /* Defence in depth against the two bounds drifting apart again. The + * admission gate in cl_ctr_consumer_submit() and the buffer allocated in + * mod_init are both derived from cc_max_payload, so this cannot fire + * today - which is exactly what was true of the old pair right up until + * the MTU probe was added and made them disagree. */ + if (!pkt || CL_CTR_PKT_OVERHEAD + job->payload_len > cl_ctr_consumer_pkt_sz) { + LM_ERR("clusterer_controller: consumer packet of %d bytes does not fit " + "the %d-byte transmit buffer - dropping\n", + CL_CTR_PKT_OVERHEAD + job->payload_len, cl_ctr_consumer_pkt_sz); + shm_free(job); + return; + } + /* unicast to our own id never touches the wire; multicast with * CLCTR_SEND_TO_SELF touches it AND dispatches locally */ if (job->dst_node_id != 0 && job->dst_node_id == my_node_id) { @@ -7754,7 +7822,7 @@ static void cl_ctr_script_recv(int cluster_id, int src_node_id, str *channel, static int cl_ctr_script_send(int cluster_id, int node_id, str *gen_msg, str *tag, unsigned char kind, int flags) { - char buf[CLCTR_MAX_PAYLOAD]; + char *buf = cl_ctr_script_buf; str pl; int taglen = tag ? tag->len : 0; @@ -7762,9 +7830,18 @@ static int cl_ctr_script_send(int cluster_id, int node_id, str *gen_msg, return -1; if (taglen > 255) taglen = 255; - if (2 + taglen + gen_msg->len > cc_max_payload) { + /* Bound against the BUFFER, not only against cc_max_payload. They are + * derived from one another at mod_init now, but this function writes + * into buf[] and buf[] is what it must answer to - the sibling + * cmd_cl_ctr_send_req_list() has always guarded with sizeof(buf), and + * the gap between the two forms is exactly how a 1500-MTU link turned a + * 1350-byte script message into an 89-byte stack overrun. */ + if (!buf || 2 + taglen + gen_msg->len > cl_ctr_script_buf_sz || + 2 + taglen + gen_msg->len > cc_max_payload) { LM_ERR("clusterer_controller: message of %d bytes is more than the " - "%d a datagram carries\n", gen_msg->len, cc_max_payload); + "%d a datagram carries\n", gen_msg->len, + cc_max_payload < cl_ctr_script_buf_sz ? cc_max_payload + : cl_ctr_script_buf_sz); return -1; } buf[0] = (char)kind; From f24e86f13bb822073ed8ca7f8452eefebc75cce3 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 11:49:41 +1000 Subject: [PATCH 15/32] clusterer_controller: gate CLCTR_SEND_RELIABLE behind enable_reliable_send The flag does not work on a channel carrying concurrent messages, and fails silently when it does not. cl_ctr_check_and_update_seq() requires pkt_seq > last_seq, and a retransmit reuses its original seq. On the serialised 1:1 exchange this was built for - the join handshake, KEY_GRANT - nothing else from that peer is in flight, so the retransmit still carries the highest seq and is accepted. On a concurrent channel, later messages have already advanced last_consumer_seq by the time the retransmit arrives, so it is dropped as out-of-order AND re-ACKed, which stops the sender retransmitting and leaves it believing it delivered. Measured, not argued: on the cross-node cache pull channel under 40% reply loss, three runs each, plain delivered 61.2% and RELIABLE 49.4% - worse, at two to three times the packets, with not one retransmitted payload ever delivered and pulls_late_stored at zero across all six runs. So the default now DROPS the flag rather than refusing the send: a plain send measurably out-delivers a reliable one here, making this an improvement rather than a consolation prize. One rate-limited warning per process per minute says so, and names the condition to check. Gated by a modparam rather than removed, because the question is answerable by exactly one party. The only callers that can reach the flag today are the script functions cmd_cl_ctr_send_req / cmd_cl_ctr_broadcast_req, and they all share one channel (cl_ctr_script_chan) - concurrent by definition if two routes fire at once. Whether that happens is a property of the deployment's own routes: the admin can answer it, a module author cannot be asked via config. modparam("clusterer_controller", "enable_reliable_send", 1) One gate, at cl_ctr_consumer_submit(), because both the script functions and the consumer API funnel through it - the flag cannot survive by another route. api.h now documents that the flag is gated, and why. This is MITIGATION, NOT REPAIR. It stops the broken path being reached by accident; it does not make RELIABLE reliable. Task #71 stays open for the real fix - a fresh seq per retransmit with message identity carried separately, or a small per-peer window instead of a single high-water mark - or for deleting the flag, which is defensible since the internal control traffic that genuinely depends on serialised delivery does not go through it. Verified on a live node, both directions: with the modparam absent a cl_ctr_send_req(..., reliable=1) logs the refusal once and sends plain; with enable_reliable_send=1 it is honoured and nothing is logged. The clctr send path is otherwise unchanged - the /dn/task86b rig still passes at 1350 and 5000 bytes with 17 workers alive. Full 131-module build, 0 stamp mismatches. (cherry picked from commit ab7851614c50214d48f95298d22ac7486b9eb25a) --- modules/clusterer_controller/api.h | 14 ++++- .../clusterer_controller.c | 58 +++++++++++++++++++ 2 files changed, 71 insertions(+), 1 deletion(-) diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h index c94ddb0db6f..23ca6ed3cb9 100644 --- a/modules/clusterer_controller/api.h +++ b/modules/clusterer_controller/api.h @@ -89,7 +89,19 @@ * * The cost, where it IS applicable: one ACK per recipient, so a reliable send * to the whole cluster is N-1 packets back where a plain one was nothing - - * hence opt-in, and per send rather than per channel. */ + * hence opt-in, and per send rather than per channel. + * + * GATED. Because of the above, the flag is IGNORED unless the deployment sets + * modparam("clusterer_controller", "enable_reliable_send", 1) + * A send that asks for it without that gets a plain send and one rate-limited + * warning - which is the better outcome, not a fallback: plain measurably + * out-delivers reliable on a concurrent channel. The gate is a modparam + * because the only callers that can reach the flag today are the script + * functions, which all share one channel, and whether two of their messages + * are ever in flight to the same peer at once is a property of the + * deployment's own routes - a question the admin can answer and this module + * cannot. A consumer module that genuinely has a serialised 1:1 exchange still + * needs the admin to enable it. */ #define CLCTR_SEND_RELIABLE (1 << 1) /* limits a consumer can rely on */ diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 97ea39f4334..5b17e3bc122 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -275,6 +275,8 @@ static const unsigned char CL_CTR_CONSUMER_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x0 + CL_CTR_CONSUMER_HDR_SZ + CLCTR_MAX_CHAN_LEN \ + CL_CTR_TAG_SZ) #define CL_CTR_CONSUMER_PKT_MAX (CL_CTR_PKT_OVERHEAD + CLCTR_MAX_PAYLOAD) +/* how often one process may repeat the reliable-send refusal */ +#define CL_CTR_RELIABLE_WARN_IVL 60 /* A datagram cannot exceed the UDP payload maximum no matter what the MTU * says - loopback reports 65536, which would otherwise compute a payload one * byte past what sendto() can accept. */ @@ -868,6 +870,32 @@ static int clctl_loaded = 0; static int manage_shtags = 1; static int consumer_retries = CL_CTR_CONSUMER_RETRIES_DEFAULT; static int consumer_retry_ms = CL_CTR_CONSUMER_RETRY_MS_DEFAULT; +/* + * CLCTR_SEND_RELIABLE is OFF unless the admin turns it on, because it does not + * work on a channel that carries concurrent messages and fails SILENTLY when + * it does not. + * + * cl_ctr_check_and_update_seq() requires pkt_seq > last_seq, and a retransmit + * reuses its original seq. On a serialised 1:1 exchange - the join handshake + * this was built for - nothing else from that peer is in flight, so the + * retransmit still carries the highest seq and is accepted. On a concurrent + * channel, later messages have already advanced last_consumer_seq by the time + * the retransmit arrives, so it is dropped as out-of-order AND re-ACKed, which + * stops the sender retransmitting and leaves it believing it delivered. + * + * Measured on the cross-node cache pull channel under 40% reply loss, three + * runs each: plain 61.2% delivered, RELIABLE 49.4% - worse, at two to three + * times the packets, with not one retransmitted payload ever delivered. So + * refusing the flag is an IMPROVEMENT, not a degradation, which is why the + * default drops it rather than failing the send. + * + * The only callers that can reach the flag today are the script functions, + * and they all share one channel (cl_ctr_script_chan) - concurrent by + * definition if two routes fire at once. Whether that happens is a property of + * the admin's own routes, which is exactly why this is a modparam: the admin + * can answer the question, and a module author cannot be asked via config. + */ +static int enable_reliable_send = 0; /* master_stickiness (global default; per-cluster override via "cluster" string): * 1 (default) = the master is "sticky": a live master keeps the role and is * NOT displaced when a higher-IP node joins. The highest-IP @@ -942,6 +970,7 @@ static const param_export_t params[] = { {"consumer_rate_limit", INT_PARAM, &consumer_rate_limit}, {"consumer_retries", INT_PARAM, &consumer_retries}, {"consumer_retry_ms", INT_PARAM, &consumer_retry_ms}, + {"enable_reliable_send", INT_PARAM, &enable_reliable_send}, {"manage_shtags", INT_PARAM, &manage_shtags}, {"master_stickiness", INT_PARAM, &master_stickiness}, {"on_config_mismatch", STR_PARAM, &on_config_mismatch_s}, @@ -7676,6 +7705,35 @@ static int cl_ctr_consumer_submit(int cluster_id, int dst_node_id, "use BIN for bulk data\n", payload_len, cc_max_payload); return -1; } + + /* Single gate for every sender - the script functions and the consumer API + * both funnel through here, so the flag cannot survive by another route. + * Dropped rather than refused: a plain send measurably OUT-DELIVERS a + * reliable one on a concurrent channel (61.2% vs 49.4% under 40% loss), so + * ignoring the request is the better outcome, not a consolation prize. See + * the enable_reliable_send comment above for why it fails. + * + * The warning is process-local and rate-limited: several workers may each + * emit one per interval, which is the same trade pull_send_failed() makes - + * an occasional duplicate line costs less than a lock on a send path. */ + if ((flags & CLCTR_SEND_RELIABLE) && !enable_reliable_send) { + static unsigned int last_warn; /* per process, deliberately */ + unsigned int now = get_ticks(); + + flags &= ~CLCTR_SEND_RELIABLE; + if (last_warn == 0 || now - last_warn >= CL_CTR_RELIABLE_WARN_IVL) { + last_warn = now; + LM_WARN("clusterer_controller: reliable delivery was requested on " + "channel '%.*s' but enable_reliable_send is 0 - sending " + "plain. That is deliberate: a retransmit reuses its seq, " + "so on a channel carrying concurrent messages it is " + "dropped as out-of-order AND re-ACKed, which silently " + "stops the sender retransmitting. Enable it only if this " + "channel never has two messages in flight to the same peer " + "at once.\n", channel->len, channel->s); + } + } + cl = cl_ctr_cluster_by_id(cluster_id); if (!cl) { LM_ERR("consumer send to unknown cluster %d\n", cluster_id); From 26697ceb1b250bac10a0056f1f16a3a1b7bbfb44 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:50:24 +1000 Subject: [PATCH 16/32] clusterer_controller: retransmit the sealed packet, not its plaintext length Both consumer enqueue sites cached and re-sent plain_len for a packet that cl_ctr_seal_and_send() had encrypted in place. What goes on the wire is CL_CTR_WIRE_HDR_SZ + plain_len + CL_CTR_TAG_SZ, so every repair left 44 bytes short and every receiver discarded it as "short packet, dropping" before it could be decrypted. Consumer retransmission had never delivered a single byte. The control plane's own enqueue (KEY_GRANT) always passed the sealed length, which is exactly why the join handshake's ARQ worked and this did not. Measured on a two-node rig under 40% loss: the receiver logged 113 x "short packet (25 bytes)", 25 being precisely the plain_len of the messages being repaired. (cherry picked from commit 7e69636bfe6ae352cabda7b6496b1176e4dfae2d) --- .../clusterer_controller/clusterer_controller.c | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 5b17e3bc122..7db26642a0f 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -7608,8 +7608,18 @@ static void cl_ctr_rpc_consumer_send(int sender, void *param) * bytes - and silences -Wpointer-sign without retyping the * buffer, which would only move the warning to the * seal_and_send() calls that legitimately need char *. */ + /* The SEALED length. cl_ctr_seal_and_send() encrypted in + * place, so what is on the wire is the wire header, the + * ciphertext and the AEAD tag - 44 bytes more than plain_len. + * Caching plain_len re-sent a packet truncated by exactly that + * much, which every receiver discarded as "short" before it + * could be decrypted: consumer retransmission had never + * delivered a single byte. The control plane's own enqueue + * (KEY_GRANT) always passed the sealed length, which is why + * the join handshake's ARQ worked and this did not. */ cl_ctr_retx_enqueue_bcast(cl, ntohl(seq), - (const unsigned char *)pkt, plain_len); + (const unsigned char *)pkt, + CL_CTR_WIRE_HDR_SZ + plain_len + CL_CTR_TAG_SZ); } else { struct sockaddr_in d; @@ -7642,9 +7652,11 @@ static void cl_ctr_rpc_consumer_send(int sender, void *param) (const struct sockaddr *)&d, sizeof d); if (reliable) /* same sign-only cast as the broadcast path above */ + /* sealed length, as above */ cl_ctr_retx_enqueue_consumer(cl, ntohl(seq), CL_CTR_PKT_CONSUMER_REL, - (const unsigned char *)pkt, plain_len, + (const unsigned char *)pkt, + CL_CTR_WIRE_HDR_SZ + plain_len + CL_CTR_TAG_SZ, (const struct sockaddr *)&d, sizeof d); } } From ed2cc413c59a8215aedbdde5a799ee1597b44a66 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:51:10 +1000 Subject: [PATCH 17/32] clusterer_controller: accept each sequence number exactly once cl_ctr_check_and_update_seq() was a bare high-water mark, so it could not tell "already delivered" from "delivered out of order" and had to reject both. Two consequences: a retransmit carries its ORIGINAL seq, so once any later message had been accepted the repair was discarded - and then re-ACKed, which stopped the sender retransmitting and left it believing it had delivered; and ordinary reordering, which any multipath network produces, silently dropped consumer payloads that were never duplicated at all. Replace it with the standard construction (IPsec RFC 4303 A.2, DTLS RFC 6347 4.1.2.6): the mark plus a 1024-slot circular bitmap of what has already been accepted at or below it. A seq below the mark whose bit is clear was never delivered, so its repair is accepted; a seq whose bit is set is a true duplicate, re-ACKed but not delivered twice. No seq is ever accepted twice, so replay protection is unchanged. Below the window the receiver cannot know whether it ever had the message, so it drops it and deliberately does NOT acknowledge it - an acknowledgement it cannot justify is the silent failure this is meant to remove. mod_init reports the per-peer message rate the window spans, since both terms of that bound (consumer_retries x consumer_retry_ms) are configurable. (cherry picked from commit f3dded5b6d0dba7709ade6902fa47b1980c02bf6) --- .../clusterer_controller.c | 325 ++++++++++++++---- 1 file changed, 255 insertions(+), 70 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 7db26642a0f..9312de97e22 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -567,9 +567,11 @@ static inline const char *cl_ctr_role_name(int r) /* * One outstanding 1:1 handshake packet awaiting an ACK. The full sealed bytes * are cached so a retransmit is a single sendto() with no re-encryption: the - * same seq/nonce is resent, which a peer that already got it (and ACKed) will - * not see again, while one that lost it still has an older last_seq and accepts - * it. Worker-local (only the controller worker touches the queue), so no lock. + * same seq/nonce is resent, and the receiver's replay window sorts out which + * is which - a peer that already got it (and ACKed) finds the bit set and only + * re-ACKs, while one that lost it finds the bit clear and accepts it, however + * many of its peer's later messages arrived in between. Worker-local (only + * the controller worker touches the queue), so no lock. */ typedef struct { int used; @@ -875,19 +877,20 @@ static int consumer_retry_ms = CL_CTR_CONSUMER_RETRY_MS_DEFAU * work on a channel that carries concurrent messages and fails SILENTLY when * it does not. * - * cl_ctr_check_and_update_seq() requires pkt_seq > last_seq, and a retransmit - * reuses its original seq. On a serialised 1:1 exchange - the join handshake - * this was built for - nothing else from that peer is in flight, so the - * retransmit still carries the highest seq and is accepted. On a concurrent - * channel, later messages have already advanced last_consumer_seq by the time - * the retransmit arrives, so it is dropped as out-of-order AND re-ACKed, which - * stops the sender retransmitting and leaves it believing it delivered. - * - * Measured on the cross-node cache pull channel under 40% reply loss, three - * runs each: plain 61.2% delivered, RELIABLE 49.4% - worse, at two to three - * times the packets, with not one retransmitted payload ever delivered. So - * refusing the flag is an IMPROVEMENT, not a degradation, which is why the - * default drops it rather than failing the send. + * ON by default, which it was NOT when it was first added hours earlier: the + * flag then could not work on a channel carrying concurrent messages, and + * failed silently when it did not. A retransmit reuses its original seq, and + * the receiver's anti-replay guard was a bare high-water mark, so once any + * later message had been accepted the repair copy was rejected as + * out-of-order AND re-ACKed - which stopped the sender retransmitting and left + * it believing it had delivered. Measured on the cross-node cache pull channel + * under 40% reply loss: plain 61.2% delivered, RELIABLE 49.4%, at two to three + * times the packets, with not one retransmitted payload ever delivered. + * + * The replay window (cl_ctr_replay_t) removes that: a seq below the mark whose + * bit is clear was never delivered, so its retransmit is accepted, while a seq + * whose bit is set is a true duplicate and is re-ACKed without being delivered + * twice. The flag now does what it says on any channel. * * The only callers that can reach the flag today are the script functions, * and they all share one channel (cl_ctr_script_chan) - concurrent by @@ -981,6 +984,117 @@ static const param_export_t params[] = { * Peer table (shared memory) * ========================================================================= */ +/* + * Anti-replay window - a high-water mark PLUS a bitmap of what has already + * been accepted at or below it, which is the standard construction (IPsec + * RFC 4303 A.2, DTLS RFC 6347 4.1.2.6). + * + * A bare high-water mark cannot tell "already delivered" from "delivered out + * of order", so it has to reject both, and that breaks two things at once: + * + * - a retransmit carries its ORIGINAL seq, so once any later message has + * been accepted the repair copy is discarded - the reliable-send defect; + * - ordinary reordering, which any multipath network produces, silently + * drops consumer payloads that were never duplicated at all. + * + * The bitmap separates the two. A seq below the mark is accepted exactly + * once: the first copy is delivered and its bit set, and every later copy of + * that same seq finds the bit set and is a genuine duplicate. Replay + * protection is unchanged - no seq is ever accepted twice - while loss repair + * and reordering now work. + * + * The bitmap is circular: the slot for seq s is s % CL_CTR_REPLAY_WIN_BITS, + * so advancing the mark clears only the slots the window moved onto (one, in + * the in-order case) instead of shifting 128 bytes per packet. + * + * SIZING. The window has to span the longest interval over which a repair for + * seq s can still arrive, expressed in messages from that peer on that plane: + * + * window > peak per-peer send rate x retransmit horizon + * + * The consumer horizon is consumer_retries x consumer_retry_ms = 80 ms with + * the defaults, so 1024 slots hold out to ~12,800 msg/s from a single peer - + * two orders above anything this plane carries (the cross-node cache pull + * channel measures ~0.3/s). The control plane's horizon is longer, 3 x 250 ms, + * but it sends a handful of packets a second. mod_init logs the derived + * ceiling because both horizon terms are admin-settable and a config that + * narrows it should say so rather than wait to be discovered as loss. + * + * Cost is 128 bytes per peer per plane: 64 KB of shm for a full 256-peer table. + * + * NOT addressed here, and unchanged from the bare counter this replaces: the + * comparison is plain unsigned, so if a sender's 32-bit counter ever wraps + * inside one key epoch the receiver rejects everything after the wrap until + * the next rotation resets both sides. Fail-closed, but silent. Left alone + * deliberately - widening the accept test to serial-number arithmetic would + * quietly enlarge what counts as a replay, which is not a change to make in + * the same commit as a delivery fix. Task #89. + */ +#define CL_CTR_REPLAY_WIN_BITS 1024 +#define CL_CTR_REPLAY_WIN_WORDS (CL_CTR_REPLAY_WIN_BITS / 64) + +typedef struct { + uint32_t last; /* highest seq accepted so far */ + uint64_t win[CL_CTR_REPLAY_WIN_WORDS]; /* circular, slot = seq % BITS */ +} cl_ctr_replay_t; + +#define CL_CTR_RP_WORD(s) ((((uint32_t)(s)) % CL_CTR_REPLAY_WIN_BITS) / 64) +#define CL_CTR_RP_MASK(s) (1ULL << ((((uint32_t)(s)) % CL_CTR_REPLAY_WIN_BITS) % 64)) + +/* Verdicts from cl_ctr_check_and_update_seq(). Three, not two: "seen this + * exact packet before" and "too old to have an opinion about" call for + * opposite handling on the reliable path - one must be re-ACKed, the other + * must not be. */ +#define CL_CTR_SEQ_OK 0 /* fresh: accept and deliver */ +#define CL_CTR_SEQ_DUP (-1) /* inside the window, bit already set: drop */ +#define CL_CTR_SEQ_OLD (-2) /* below the window: drop, and we cannot know */ + /* whether we ever had it */ + +static inline void cl_ctr_replay_reset(cl_ctr_replay_t *r) +{ + r->last = 0; + memset(r->win, 0, sizeof r->win); +} + +/* + * Accept @seq at most once. Not thread-safe by design: the only caller is the + * cluster's single worker reactor, same as the counter it replaces. + */ +static inline int cl_ctr_replay_check(cl_ctr_replay_t *r, uint32_t seq) +{ + uint32_t d, s; + + if (seq > r->last) { /* forward: the common case */ + d = seq - r->last; + if (d >= CL_CTR_REPLAY_WIN_BITS) { + memset(r->win, 0, sizeof r->win); + } else { + /* Clear the slots the window has just moved onto. These hold + * bits for seqs a full window older, which must not be mistaken + * for the fresh ones now mapping to the same slots. */ + for (s = r->last + 1; s != seq; s++) + r->win[CL_CTR_RP_WORD(s)] &= ~CL_CTR_RP_MASK(s); + r->win[CL_CTR_RP_WORD(seq)] &= ~CL_CTR_RP_MASK(seq); + } + r->last = seq; + r->win[CL_CTR_RP_WORD(seq)] |= CL_CTR_RP_MASK(seq); + return CL_CTR_SEQ_OK; + } + + /* At or below the mark. Note seq == last == 0 with an empty window is the + * untouched state, and falls out here as a legitimate first accept - it + * costs nothing to allow, and every sender in fact starts at 1 + * (++my_seq). */ + d = r->last - seq; + if (d >= CL_CTR_REPLAY_WIN_BITS) + return CL_CTR_SEQ_OLD; + if (r->win[CL_CTR_RP_WORD(seq)] & CL_CTR_RP_MASK(seq)) + return CL_CTR_SEQ_DUP; + + r->win[CL_CTR_RP_WORD(seq)] |= CL_CTR_RP_MASK(seq); + return CL_CTR_SEQ_OK; +} + typedef struct cl_ctr_peer_ { char ip[CL_CTR_MAX_IP_LEN + 1]; unsigned int ip_num; @@ -993,14 +1107,14 @@ typedef struct cl_ctr_peer_ { char bin_sockets[CL_CTR_MAX_BIN_SOCKETS][CL_CTR_MAX_BIN_SOCK_LEN]; unsigned char pubkey[CL_CTR_PUBKEY_SZ]; /* long-lived X25519 pubkey (from ALIVE); zero if unknown; used for KEY_HANDOFF */ - uint32_t last_seq; /* highest seq accepted from this peer */ + cl_ctr_replay_t replay; /* control-plane anti-replay */ /* Consumer traffic is counted separately from the control plane. They * share a session key and a socket but not a sequence space: a consumer * may send thousands of packets a second where the control plane sends a * handful, and one counter for both means a reordered consumer packet can * make a MASTER_ALIVE arriving behind it look like a replay - which is a * missed liveness beacon, not a dropped cache reply. */ - uint32_t last_consumer_seq; + cl_ctr_replay_t replay_consumer; /* Peer's advertised consistency-critical config (from ALIVE), used to warn * on accidental per-node config drift. cfg_known=0 until first advertised; * cfg_warned deduplicates the mismatch warning. */ @@ -1026,7 +1140,7 @@ struct cl_ctr_peers_ { unsigned char master_salt[CL_CTR_MASTER_SALT_SZ]; /* my_seq: monotonic send counter; in shm so mod_destroy can use it for * GOODBYE without needing the worker's private state. Reset to 0 on - * every session key rotation so last_seq counters reset cleanly. */ + * every session key rotation so peers' replay windows reset cleanly. */ uint32_t my_seq; uint32_t my_consumer_seq; /* the consumer plane's own counter */ /* Sharing-tag override: 0 = automatic (master-driven) allocation; nonzero = @@ -2396,8 +2510,8 @@ static int cl_ctr_derive_session_key(cl_ctr_cluster_t *cl) cl->peers->my_seq = 0; cl->peers->my_consumer_seq = 0; for (i = 0; i < cl->peers->count; i++) { - cl->peers->entries[i].last_seq = 0; - cl->peers->entries[i].last_consumer_seq = 0; + cl_ctr_replay_reset(&cl->peers->entries[i].replay); + cl_ctr_replay_reset(&cl->peers->entries[i].replay_consumer); } cl->have_session_key = 1; /* a valid group key now exists */ /* The salt (and my_seq) just changed, so any queued retransmit is now stale. */ @@ -2578,44 +2692,50 @@ static int cl_ctr_decrypt_pkt(char *buf, ssize_t n, const char *sender_ip, } /** - * cl_ctr_check_and_update_seq() - reject replayed or reordered packets. - * Looks up sender_ip in the peer table; requires pkt_seq > last_seq. - * Updates last_seq on accept. Unknown senders (new nodes not yet in - * the peer table) are accepted so their first packet (ALIVE/JOIN_REQ) - * can populate the table. + * cl_ctr_check_and_update_seq() - accept each sequence number exactly once. + * Looks up sender_ip in the peer table and runs @pkt_seq through that peer's + * replay window for the requested plane. Unknown senders (new nodes not yet + * in the peer table) are accepted so their first packet (ALIVE/JOIN_REQ) can + * populate the table. * Only called for CL_CTR_PACKET_MAGIC packets; bootstrap packets use join_nonce. * Single-threaded caller (cl_ctr_worker reactor); no lock needed for the check. - * @return 0 to accept, -1 to drop. + * @return CL_CTR_SEQ_OK to accept, CL_CTR_SEQ_DUP / CL_CTR_SEQ_OLD to drop. */ static int cl_ctr_check_and_update_seq(const char *sender_ip, uint32_t pkt_seq, cl_ctr_cluster_t *cl, int is_consumer) { - int i; + int i, rc; for (i = 0; i < cl->peers->count; i++) { if (strcmp(cl->peers->entries[i].ip, sender_ip) == 0) { - uint32_t *last = is_consumer - ? &cl->peers->entries[i].last_consumer_seq - : &cl->peers->entries[i].last_seq; - - if (pkt_seq <= *last) { - /* Debug for consumer traffic, warning for the control plane. - * A consumer sending at rate will reorder on any network with - * more than one path, and a warning per reordered packet says - * "attack" about something entirely ordinary. */ - if (is_consumer) - LM_DBG("clusterer_controller: consumer packet from %s out " - "of order seq=%u last=%u, dropping\n", - sender_ip, pkt_seq, *last); - else - LM_WARN("clusterer_controller: replay from %s seq=%u " - "last=%u, dropping\n", sender_ip, pkt_seq, *last); - return -1; - } - *last = pkt_seq; - return 0; + cl_ctr_replay_t *r = is_consumer + ? &cl->peers->entries[i].replay_consumer + : &cl->peers->entries[i].replay; + + rc = cl_ctr_replay_check(r, pkt_seq); + if (rc == CL_CTR_SEQ_OK) + return rc; + + /* Debug for consumer traffic, warning for the control plane. A + * consumer sending at rate will duplicate on any path that retries, + * and a warning per copy says "attack" about something entirely + * ordinary. Reordering no longer reaches here at all - the window + * accepts it - so what is left really is a repeat or something + * older than the window can vouch for. */ + if (is_consumer) + LM_DBG("clusterer_controller: consumer packet from %s seq=%u " + "last=%u %s, dropping\n", sender_ip, pkt_seq, r->last, + rc == CL_CTR_SEQ_DUP ? "already accepted" + : "older than the replay window"); + else + LM_WARN("clusterer_controller: %s from %s seq=%u last=%u, " + "dropping\n", + rc == CL_CTR_SEQ_DUP ? "replay" + : "packet older than the replay window", + sender_ip, pkt_seq, r->last); + return rc; } } - return 0; /* unknown sender: accept, handler will upsert into peer table */ + return CL_CTR_SEQ_OK; /* unknown sender: accept, handler will upsert into peer table */ } /* ========================================================================= @@ -4200,16 +4320,17 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le (const char (*)[CL_CTR_MAX_BIN_SOCK_LEN])bin_socks, cl); - /* Reset last_seq in the peer table. Essential: a restarted node begins its - * seq counter from 0, and without this reset peers would permanently reject - * its new packets (old last_seq > new seq) until the next key rotation. The + /* Reset the replay window in the peer table. Essential: a restarted node + * begins its seq counter from 0, and without this reset peers would + * permanently reject its new packets (old high-water mark far above the new + * low seq) until the next key rotation. The * joiner's long-lived pubkey (for a future KEY_HANDOFF) is learned from its * ALIVE, not here - the JOIN_REQ now carries only an ephemeral Noise key. */ { cl_ctr_peer_t *e = cl_ctr_peer_by_ip_locked(cl, src_ip); if (e) { - e->last_seq = 0; - e->last_consumer_seq = 0; + cl_ctr_replay_reset(&e->replay); + cl_ctr_replay_reset(&e->replay_consumer); } } @@ -4380,11 +4501,11 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, lock_start_write(cl->peers->lock); - /* Second pass: upsert all peers and reset their last_seq. - * Resetting last_seq here covers the case where a peer restarted and + /* Second pass: upsert all peers and reset their replay windows. + * Resetting them here covers the case where a peer restarted and * sent JOIN_REQ: the MEMBER_LIST is the broadcast announcement that a * join event occurred. Without the reset, non-master peers would reject - * the restarted node's new packets (old last_seq > new low seq). */ + * the restarted node's new packets (old mark > new low seq). */ for (i = 0; i < (int)count; i++, p += CL_CTR_IP_ENTRY_SZ) { char ip_buf[CL_CTR_MAX_IP_LEN + 1]; int _j; @@ -4404,8 +4525,8 @@ static void cl_ctr_handle_member_list(const char *payload, int payload_len, cl_ctr_learn_peer_locked(ip_buf, cl); for (_j = 0; _j < cl->peers->count; _j++) { if (strcmp(cl->peers->entries[_j].ip, ip_buf) == 0) { - cl->peers->entries[_j].last_seq = 0; - cl->peers->entries[_j].last_consumer_seq = 0; + cl_ctr_replay_reset(&cl->peers->entries[_j].replay); + cl_ctr_replay_reset(&cl->peers->entries[_j].replay_consumer); break; } } @@ -5496,20 +5617,57 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, unsigned char _t = (unsigned char)buf[CL_CTR_WIRE_HDR_SZ]; int is_consumer_pkt = (_t == CL_CTR_PKT_CONSUMER || _t == CL_CTR_PKT_CONSUMER_REL); + int _seq_rc; memcpy(&pkt_seq, buf + CL_CTR_WIRE_HDR_SZ + 1, CL_CTR_SEQ_SZ); pkt_seq = ntohl(pkt_seq); - if (cl_ctr_check_and_update_seq(sender_ip_buf, pkt_seq, cl, - is_consumer_pkt) < 0) { - /* A duplicate of a message that asked to be acknowledged means - * our acknowledgement did not arrive: say it again. Without - * this the sender spends its whole retransmit budget against a - * receiver that has had the message all along and is dropping - * every copy in silence. */ - if ((unsigned char)buf[CL_CTR_WIRE_HDR_SZ] - == CL_CTR_PKT_CONSUMER_REL) - cl_ctr_send_ack(cl->sock, cl, pkt_seq, 0, - (const struct sockaddr *)&src_addr, src_len); + _seq_rc = cl_ctr_check_and_update_seq(sender_ip_buf, pkt_seq, cl, + is_consumer_pkt); + if (_seq_rc != CL_CTR_SEQ_OK) { + if (_t == CL_CTR_PKT_CONSUMER_REL) { + if (_seq_rc == CL_CTR_SEQ_DUP) { + /* A duplicate of a message that asked to be + * acknowledged means our acknowledgement did not + * arrive: say it again. Without this the sender spends + * its whole retransmit budget against a receiver that + * has had the message all along and is dropping every + * copy in silence. */ + cl_ctr_send_ack(cl->sock, cl, pkt_seq, 0, + (const struct sockaddr *)&src_addr, + src_len); + } else { + /* Below the window. We genuinely do not know whether + * this payload was ever delivered, so we must not claim + * it was - an ACK here is exactly the silent lie that + * made reliable delivery unreliable before the window + * existed. Withholding it costs the sender the rest of + * its budget and one honest "unacked" line. + * + * Reaching this at all means the window is too small + * for the offered rate, which is a sizing problem the + * operator can act on - hence a warning, not a debug + * line, rate limited per process because a burst that + * outruns the window outruns it for many packets. */ + static unsigned int last_old_warn; + unsigned int now_t = get_ticks(); + + if (last_old_warn == 0 || + now_t - last_old_warn >= CL_CTR_RELIABLE_WARN_IVL) { + last_old_warn = now_t; + LM_WARN("clusterer_controller: reliable message from " + "%s seq=%u fell more than %d messages behind " + "the replay window and cannot be confirmed - " + "NOT acknowledging. The sender is offering " + "more than the window spans within its " + "retransmit horizon (consumer_retries x " + "consumer_retry_ms = %d ms); lower that " + "horizon or expect losses.\n", + sender_ip_buf, pkt_seq, + CL_CTR_REPLAY_WIN_BITS, + cl->consumer_retries * cl->consumer_retry_ms); + } + } + } return; } } @@ -7198,6 +7356,33 @@ static int mod_init(void) if (cl->consumer_rate == -1) cl->consumer_rate = consumer_rate_limit; + /* The replay window is a fixed number of MESSAGES, but what it has to + * span is a fixed amount of TIME - the retransmit horizon, which is + * configurable and can legally be set as high as 10 x 5000 ms. Report + * the resulting per-peer rate ceiling rather than leaving an operator + * to meet it as unexplained loss. Above the ceiling, a repair copy can + * arrive after its slot has been reused and is refused (and, on the + * reliable path, deliberately not acknowledged). */ + { + int horizon_ms = cl->consumer_retries * cl->consumer_retry_ms; + int ceiling = horizon_ms > 0 + ? (CL_CTR_REPLAY_WIN_BITS * 1000) / horizon_ms : 0; + + if (horizon_ms > 0 && ceiling < 1000) + LM_WARN("clusterer_controller: [cluster %d] consumer retransmit " + "horizon is %d ms (%d retries x %d ms), so the %d-slot " + "replay window only spans ~%d consumer messages/s from " + "any one peer; above that, retransmits arrive too late " + "to be judged and are dropped\n", cl->cluster_id, + horizon_ms, cl->consumer_retries, cl->consumer_retry_ms, + CL_CTR_REPLAY_WIN_BITS, ceiling); + else + LM_DBG("clusterer_controller: [cluster %d] replay window spans " + "%d messages, ~%d msg/s per peer at a %d ms retransmit " + "horizon\n", cl->cluster_id, CL_CTR_REPLAY_WIN_BITS, + ceiling, horizon_ms); + } + /* Resolve which BIN socket to use for this cluster. * Priority: explicit bin_socket= in cluster string > * sole discovered socket > From 3453976a25f61b9ffbc2f90dd6384666f88397d4 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:52:41 +1000 Subject: [PATCH 18/32] clusterer_controller: arm the retransmit timer for the earliest deadline cl_ctr_retx_enqueue_bcast() armed the shared timerfd unconditionally to now + retry_ms on every send. timerfd_settime() REPLACES a pending expiry, so a steady stream of reliable sends pushed the deadline forward faster than it could arrive: at fifty messages a second with a 40 ms gap the timer was re-armed every 20 ms to fire 40 ms later, and never fired at all while traffic continued. Measured on the rig: 45 lost messages produced 2 retransmits. Two related defects in the same timer. The sweep armed and re-armed to a flat CL_CTR_RETX_INTERVAL_US (250 ms) rather than to the earliest pending deadline, so a consumer entry due in 40 ms waited 250; and it reset next_due_us to now + 250 ms on every retry, discarding the per-cluster cadence that cl_ctr_retx_enqueue_consumer() had just set. Give each entry its own retry_ivl_us and route every arm through cl_ctr_retx_rearm(), which arms for the earliest deadline in the queue and disarms when it is empty. Never arm 0 us - that is timerfd's disarm, not "fire immediately". (cherry picked from commit c1028110d8dcf7790257f502900f32bafb9e4271) --- .../clusterer_controller.c | 91 +++++++++++++++---- 1 file changed, 73 insertions(+), 18 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 9312de97e22..d7e3c038d5a 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -579,6 +579,13 @@ typedef struct { unsigned char type; /* packet type, for logging */ int retries_left; utime_t next_due_us; /* get_uticks() deadline for next send */ + /* Gap to the NEXT send after this one. Per entry, not a constant, because + * the control plane and the consumer plane share this queue and want + * different cadences - the handshake's is pinned to the JOIN_REQ retry, + * the consumer's is the per-cluster consumer_retry_ms. Holding it here is + * what lets one sweep serve both without either inheriting the other's + * timing. */ + utime_t retry_ivl_us; struct sockaddr_storage dest; /* unicast destination */ socklen_t destlen; int pkt_len; /* sealed length */ @@ -2995,6 +3002,51 @@ static void cl_ctr_arm_tfd_us(int tfd, uint64_t usec_value, uint64_t usec_interv LM_WARN("clusterer_controller: timerfd_settime(us): %s\n", strerror(errno)); } +/* + * Arm the shared retransmit timer for the EARLIEST deadline in the queue, or + * disarm it when the queue is empty. + * + * Every arm has to go through here. Arming it directly to "now + one + * interval" on each enqueue - which is what the broadcast path used to do - + * makes a steady stream of reliable sends push the deadline forward faster + * than it can arrive: at fifty messages a second and a 40 ms retry gap, the + * timer was re-armed every 20 ms to fire 40 ms later and therefore never fired + * at all while traffic continued. Measured on the two-node rig: 45 lost + * messages produced 2 retransmits. A reliable send is at its least reliable + * exactly when the channel is busy, which is the opposite of what it promises. + * + * Deadline-driven rather than fixed-interval for the second reason too: with a + * flat 250 ms sweep an entry due in 40 ms waited 250, so a consumer budget of + * 2 x 40 ms was spent as 2 x 250 ms and every repair landed long after the + * consumer had given up on it. + */ +static void cl_ctr_retx_rearm(cl_ctr_cluster_t *cl) +{ + utime_t now, first = 0; + int i; + + if (cl->retx_count == 0) { + cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); /* nothing pending */ + return; + } + + for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) + if (cl->retx_q[i].used && + (first == 0 || cl->retx_q[i].next_due_us < first)) + first = cl->retx_q[i].next_due_us; + + if (first == 0) { /* count/queue disagree */ + cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); + return; + } + + now = get_uticks(); + /* Never 0 microseconds: that is timerfd's disarm, not "fire immediately", + * and an already-due entry would then wait for the next enqueue to be + * noticed at all. */ + cl_ctr_arm_tfd_us(cl->retx_tfd, first > now ? (uint64_t)(first - now) : 1, 0); +} + /* * Drop every outstanding retransmit. Called on loss of mastership and on * session-key rotation, so a demoted or re-keyed node never keeps delivering a @@ -3060,16 +3112,15 @@ static void cl_ctr_retx_enqueue_bcast(cl_ctr_cluster_t *cl, uint32_t seq, e->expect_n = (uint16_t)n; e->retries_left = cl->consumer_retries; e->retries_cfg = (uint8_t)cl->consumer_retries; - e->next_due_us = get_uticks() + (utime_t)cl->consumer_retry_ms * 1000; + e->retry_ivl_us = (utime_t)cl->consumer_retry_ms * 1000; + e->next_due_us = get_uticks() + e->retry_ivl_us; e->pkt_len = pkt_len; memcpy(e->pkt, pkt, pkt_len); cl->retx_count++; - /* First argument is the delay, and zero there means disarm - the value - * this once passed, which switched the retransmit timer off instead of - * on and left every reliable broadcast waiting for a repair that could - * never run. */ - cl_ctr_arm_tfd_us(cl->retx_tfd, - (utime_t)cl->consumer_retry_ms * 1000, 0); + /* Through the helper, never straight at the timer: this used to arm it + * unconditionally to now + one interval, which under continuous broadcasts + * pushed the sweep past every deadline it was meant to serve. */ + cl_ctr_retx_rearm(cl); } /* The consumer plane retransmits on its own schedule: the join handshake's @@ -3092,10 +3143,14 @@ static void cl_ctr_retx_enqueue_consumer(cl_ctr_cluster_t *cl, uint32_t seq, cl->retx_q[i].type == type) { cl->retx_q[i].retries_left = cl->consumer_retries; cl->retx_q[i].retries_cfg = (uint8_t)cl->consumer_retries; + cl->retx_q[i].retry_ivl_us = (utime_t)cl->consumer_retry_ms * 1000; cl->retx_q[i].next_due_us = get_uticks() - + (utime_t)cl->consumer_retry_ms * 1000; + + cl->retx_q[i].retry_ivl_us; break; } + /* The deadline just moved in (consumer gaps are shorter than the + * handshake's), so the timer enqueue armed has to be pulled forward. */ + cl_ctr_retx_rearm(cl); } static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned char type, @@ -3103,7 +3158,7 @@ static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned cha const struct sockaddr *dest, socklen_t destlen) { cl_ctr_retx_entry_t *e; - int i, slot = -1, was_empty; + int i, slot = -1; if (pkt_len <= 0 || pkt_len > (int)sizeof(cl->retx_q[0].pkt) || destlen == 0 || destlen > (socklen_t)sizeof(cl->retx_q[0].dest)) @@ -3117,21 +3172,20 @@ static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned cha return; } - was_empty = (cl->retx_count == 0); e = &cl->retx_q[slot]; memset(e, 0, sizeof(*e)); e->used = 1; e->seq = seq; e->type = type; e->retries_left = CL_CTR_RETX_MAX_RETRIES; - e->next_due_us = get_uticks() + CL_CTR_RETX_INTERVAL_US; + e->retry_ivl_us = CL_CTR_RETX_INTERVAL_US; + e->next_due_us = get_uticks() + e->retry_ivl_us; memcpy(&e->dest, dest, destlen); e->destlen = destlen; memcpy(e->pkt, pkt, pkt_len); e->pkt_len = pkt_len; cl->retx_count++; - if (was_empty) - cl_ctr_arm_tfd_us(cl->retx_tfd, CL_CTR_RETX_INTERVAL_US, 0); + cl_ctr_retx_rearm(cl); } /* An ACK arrived: drop the queued packet whose seq it echoes. */ @@ -3194,8 +3248,9 @@ static void cl_ctr_handle_ack(const char *payload, int payload_len, cl->retx_count--; break; } - if (cl->retx_count == 0) - cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); /* nothing pending - disarm */ + /* Dropping an entry can make a LATER one the earliest, so re-arm rather + * than only disarming on empty. */ + cl_ctr_retx_rearm(cl); } /* @@ -3277,12 +3332,12 @@ static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout) e->used = 0; cl->retx_count--; } else { - e->next_due_us = now + CL_CTR_RETX_INTERVAL_US; + e->next_due_us = now + (e->retry_ivl_us ? e->retry_ivl_us + : CL_CTR_RETX_INTERVAL_US); } } - if (cl->retx_count > 0) - cl_ctr_arm_tfd_us(cl->retx_tfd, CL_CTR_RETX_INTERVAL_US, 0); + cl_ctr_retx_rearm(cl); return 0; } From 2ef79ffd1e8e232847c2f4f52bba17896e2a2371 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:52:42 +1000 Subject: [PATCH 19/32] clusterer_controller: size the retransmit cache from the MTU, per entry The cache was one inline CL_CTR_NODE_ASSIGN_MAX_SZ array per entry (~580 bytes), sized for the largest control-plane packet. Any reliable consumer send above roughly 540 bytes was therefore never queued for repair at all - well under CLCTR_MAX_PAYLOAD, the size api.h tells consumers they may rely on - and it degraded in silence, on an LM_DBG. Sizing it to the compile-time CLCTR_MAX_PAYLOAD instead would cap repair at 1300 bytes on a jumbo-frame link, a seventh of what that link and cc_max_payload happily carry: the same trap the transmit buffers had before they were sized from the MTU. So the bound is cl_ctr_retx_pkt_max, derived in mod_init from cc_max_payload beside cl_ctr_script_buf_sz, and each entry allocates for the packet it actually holds - the cost then tracks outstanding reliable messages instead of reserving queue x MTU up front, which at a 9000 MTU would be 1.1 MB per cluster sitting idle. cl_ctr_retx_release() becomes the only way an entry is cleared, so a buffer cannot be dropped by a path that merely clears `used`; the flush path releases each entry rather than memset-ing the array over live pointers. A send too large to cache still goes out once, but now says so with a rate-limited warning instead of a debug line. (cherry picked from commit dd2befdaf0efa0d0fb0c6d34c126ed70327e01c9) --- .../clusterer_controller.c | 109 ++++++++++++++++-- 1 file changed, 99 insertions(+), 10 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index d7e3c038d5a..4210f26a487 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -375,6 +375,20 @@ static const unsigned char CL_CTR_CONSUMER_MAGIC[CL_CTR_MAGIC_SZ] = { 0xCC, 0x0 #define CL_CTR_RETX_INTERVAL_US (CL_CTR_JOIN_REQ_MIN_US / 2) /* 250 ms */ #define CL_CTR_RETX_MAX_RETRIES 3 /* then give up */ #define CL_CTR_RETX_QUEUE_SZ 128 /* max outstanding unacked 1:1 packets */ +/* The cached packet is either a control-plane handshake packet, whose largest + * is NODE_ASSIGN, or a consumer message, whose largest is set by the MTU at + * mod_init. So the bound is a RUNTIME value (cl_ctr_retx_pkt_max), not a + * constant: a fixed cache would either be sized for NODE_ASSIGN (~580 B, which + * silently excluded any reliable consumer send above ~540 B) or for the + * compile-time CLCTR_MAX_PAYLOAD (1300), which on a jumbo-frame link would cap + * repair at a seventh of what the link and cc_max_payload happily carry. Same + * trap the transmit buffers had before they were sized from cc_max_payload. + * + * Each entry's buffer is allocated for the packet actually being held and + * freed when the entry is released, so the cost tracks outstanding reliable + * messages rather than reserving queue x MTU up front - which at a 9000 MTU + * would be 1.1 MB per cluster sitting idle, and far more on a link that + * reports more. */ #if (CL_CTR_RETX_MAX_RETRIES * CL_CTR_RETX_INTERVAL_US) >= (CL_CTR_JOIN_DEFER_SECS * 1000000) #error "retransmit budget must stay shorter than the JOIN_REQ retry interval" #endif @@ -608,7 +622,8 @@ typedef struct { uint16_t expect_n; /* members at send time */ uint16_t acked_n; unsigned char acked_map[(CL_CTR_MAX_PEERS + 7) / 8]; - unsigned char pkt[CL_CTR_NODE_ASSIGN_MAX_SZ]; /* cached sealed bytes */ + unsigned char *pkt; /* cached sealed bytes, pkg, owned by + * the entry; NULL when released */ } cl_ctr_retx_entry_t; /** @@ -794,6 +809,10 @@ static char my_interface_buf[IF_NAMESIZE]; * jumbo-frame interfaces (MTU 9000) are not artificially limited to 1300 B. * Read-only after mod_init; safe to access from any process. */ int cc_max_payload = CLCTR_MAX_PAYLOAD; +/* Largest packet the retransmit cache will hold, raised in mod_init once the + * MTU is known. Provisional value covers the control plane, which is all that + * can be sent before then anyway. */ +static int cl_ctr_retx_pkt_max = CL_CTR_NODE_ASSIGN_MAX_SZ; /* The two transmit buffers, sized from cc_max_payload once the MTU is known. * @@ -3002,6 +3021,46 @@ static void cl_ctr_arm_tfd_us(int tfd, uint64_t usec_value, uint64_t usec_interv LM_WARN("clusterer_controller: timerfd_settime(us): %s\n", strerror(errno)); } +/* + * Release an entry and give its packet buffer back. Every path that clears + * `used` goes through here - an entry cleared without freeing would leak the + * buffer, and the next enqueue into that slot memsets the pointer away. + */ +static void cl_ctr_retx_release(cl_ctr_cluster_t *cl, cl_ctr_retx_entry_t *e) +{ + if (e->pkt) { + pkg_free(e->pkt); + e->pkt = NULL; + } + if (e->used) { + e->used = 0; + if (cl->retx_count > 0) + cl->retx_count--; + } +} + +/* + * A reliable send too large to cache goes out once and is never repaired. + * That is a downgrade of the guarantee the caller asked for, so it is a + * warning rather than the debug line it used to be - rate limited per process, + * because whatever is oversized will be oversized every time. + */ +static void cl_ctr_retx_too_big(cl_ctr_cluster_t *cl, int pkt_len) +{ + static unsigned int last_warn; + unsigned int now = get_ticks(); + + if (last_warn == 0 || now - last_warn >= CL_CTR_RELIABLE_WARN_IVL) { + last_warn = now; + LM_WARN("clusterer_controller: [cluster %d] a %d-byte packet asked for " + "reliable delivery but the retransmit cache holds %d - it was " + "sent once, best-effort, and will not be repaired. The bound " + "follows this interface's MTU, so a payload within " + "cc_max_payload (%d) always fits.\n", + cl->cluster_id, pkt_len, cl_ctr_retx_pkt_max, cc_max_payload); + } +} + /* * Arm the shared retransmit timer for the EARLIEST deadline in the queue, or * disarm it when the queue is empty. @@ -3060,6 +3119,12 @@ static void cl_ctr_retx_flush(cl_ctr_cluster_t *cl) return; LM_DBG("clusterer_controller: [cluster %d] flushing %d pending retransmit(s)\n", cl->cluster_id, cl->retx_count); + { /* release each, then clear: a bare memset over the array would drop + * every entry's packet pointer on the floor. */ + int _i; + for (_i = 0; _i < CL_CTR_RETX_QUEUE_SZ; _i++) + cl_ctr_retx_release(cl, &cl->retx_q[_i]); + } memset(cl->retx_q, 0, sizeof(cl->retx_q)); cl->retx_count = 0; cl_ctr_arm_tfd_us(cl->retx_tfd, 0, 0); /* disarm */ @@ -3079,8 +3144,10 @@ static void cl_ctr_retx_enqueue_bcast(cl_ctr_cluster_t *cl, uint32_t seq, cl_ctr_retx_entry_t *e = NULL; int i, n = 0; - if (pkt_len <= 0 || pkt_len > (int)sizeof(cl->retx_q[0].pkt)) + if (pkt_len <= 0 || pkt_len > cl_ctr_retx_pkt_max) { + cl_ctr_retx_too_big(cl, pkt_len); return; + } lock_start_read(cl->peers->lock); for (i = 0; i < cl->peers->count; i++) @@ -3104,7 +3171,14 @@ static void cl_ctr_retx_enqueue_bcast(cl_ctr_cluster_t *cl, uint32_t seq, return; } - memset(e, 0, sizeof(*e)); + memset(e, 0, sizeof(*e)); /* pkt was NULLed by the release that freed it */ + e->pkt = pkg_malloc(pkt_len); + if (!e->pkt) { + LM_ERR("clusterer_controller: [cluster %d] no pkg for a %d-byte " + "retransmit cache entry - broadcast sent once, best-effort\n", + cl->cluster_id, pkt_len); + return; + } e->used = 1; e->seq = seq; e->type = CL_CTR_PKT_CONSUMER_REL; @@ -3160,9 +3234,12 @@ static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned cha cl_ctr_retx_entry_t *e; int i, slot = -1; - if (pkt_len <= 0 || pkt_len > (int)sizeof(cl->retx_q[0].pkt) || - destlen == 0 || destlen > (socklen_t)sizeof(cl->retx_q[0].dest)) + if (pkt_len <= 0 || pkt_len > cl_ctr_retx_pkt_max || + destlen == 0 || destlen > (socklen_t)sizeof(cl->retx_q[0].dest)) { + if (pkt_len > cl_ctr_retx_pkt_max) + cl_ctr_retx_too_big(cl, pkt_len); return; + } for (i = 0; i < CL_CTR_RETX_QUEUE_SZ; i++) if (!cl->retx_q[i].used) { slot = i; break; } @@ -3173,7 +3250,14 @@ static void cl_ctr_retx_enqueue(cl_ctr_cluster_t *cl, uint32_t seq, unsigned cha } e = &cl->retx_q[slot]; - memset(e, 0, sizeof(*e)); + memset(e, 0, sizeof(*e)); /* pkt was NULLed by the release that freed it */ + e->pkt = pkg_malloc(pkt_len); + if (!e->pkt) { + LM_ERR("clusterer_controller: [cluster %d] no pkg for a %d-byte " + "retransmit cache entry - 0x%02x sent once, best-effort\n", + cl->cluster_id, pkt_len, type); + return; + } e->used = 1; e->seq = seq; e->type = type; @@ -3244,8 +3328,7 @@ static void cl_ctr_handle_ack(const char *payload, int payload_len, LM_DBG("clusterer_controller: [cluster %d] ACK for 0x%02x seq %u\n", cl->cluster_id, e->type, acked); } - e->used = 0; - cl->retx_count--; + cl_ctr_retx_release(cl, e); break; } /* Dropping an entry can make a LATER one the earliest, so re-arm rather @@ -3329,8 +3412,7 @@ static int cl_ctr_on_retx_tfd(int fd, void *param, int was_timeout) cl->cluster_id, e->type, e->seq, (unsigned)(e->retries_cfg ? e->retries_cfg : CL_CTR_RETX_MAX_RETRIES)); - e->used = 0; - cl->retx_count--; + cl_ctr_retx_release(cl, e); } else { e->next_due_us = now + (e->retry_ivl_us ? e->retry_ivl_us : CL_CTR_RETX_INTERVAL_US); @@ -7355,6 +7437,13 @@ static int mod_init(void) * that bound then admits. */ cl_ctr_script_buf_sz = cc_max_payload; cl_ctr_consumer_pkt_sz = CL_CTR_PKT_OVERHEAD + cc_max_payload; + /* The retransmit cache follows the same bound, for the same reason: a + * reliable send that the transmit buffer accepts must also be one the + * repair path can hold, or reliability quietly stops at a size the API + * never mentions. Whichever is larger - the control plane's biggest + * handshake packet, or a full-MTU consumer message. */ + cl_ctr_retx_pkt_max = cl_ctr_consumer_pkt_sz > CL_CTR_NODE_ASSIGN_MAX_SZ + ? cl_ctr_consumer_pkt_sz : CL_CTR_NODE_ASSIGN_MAX_SZ; cl_ctr_script_buf = pkg_malloc(cl_ctr_script_buf_sz); cl_ctr_consumer_pkt = pkg_malloc(cl_ctr_consumer_pkt_sz); if (!cl_ctr_script_buf || !cl_ctr_consumer_pkt) { From ab13b84d8df0cd7fc8a8a269283d9d9c06982997 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:52:43 +1000 Subject: [PATCH 20/32] clusterer_controller: cl_ctr_send_req_list must use the MTU-derived buffer It built its payload in a CLCTR_MAX_PAYLOAD stack array while its two siblings, cl_ctr_send_req() and cl_ctr_broadcast_req(), use the cc_max_payload-sized cl_ctr_script_buf. On a jumbo-frame link a script could therefore broadcast an 8 KB message but could not send that same message to a list of nodes: the list form stopped at a seventh of what the other two carried, and blamed "the cluster plane" rather than its own buffer. Share the buffer and the guard. (cherry picked from commit 25eafda958d0ee8dbaa1c7950dfec3a67baab49d) --- .../clusterer_controller.c | 26 +++++++++++++------ 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 4210f26a487..69a1fdc044c 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -8231,10 +8231,10 @@ static int cl_ctr_script_send(int cluster_id, int node_id, str *gen_msg, taglen = 255; /* Bound against the BUFFER, not only against cc_max_payload. They are * derived from one another at mod_init now, but this function writes - * into buf[] and buf[] is what it must answer to - the sibling - * cmd_cl_ctr_send_req_list() has always guarded with sizeof(buf), and - * the gap between the two forms is exactly how a 1500-MTU link turned a - * 1350-byte script message into an 89-byte stack overrun. */ + * into buf[] and buf[] is what it must answer to - the gap between the + * two forms is exactly how a 1500-MTU link turned a 1350-byte script + * message into an 89-byte stack overrun. cmd_cl_ctr_send_req_list() + * now shares this buffer and this guard. */ if (!buf || 2 + taglen + gen_msg->len > cl_ctr_script_buf_sz || 2 + taglen + gen_msg->len > cc_max_payload) { LM_ERR("clusterer_controller: message of %d bytes is more than the " @@ -8327,15 +8327,25 @@ static int cmd_cl_ctr_send_req_list(struct sip_msg *msg, int *cluster_id, CL_CTR_MAX_PEERS); { - char buf[CLCTR_MAX_PAYLOAD]; + /* The same MTU-derived buffer its two siblings use, not a + * CLCTR_MAX_PAYLOAD stack array. A compile-time 1300 here meant + * that on a jumbo link a script could broadcast an 8 KB message + * but could not send the same message to a LIST of nodes - the + * list form silently stopped at a seventh of what the other two + * carried, with an error that blamed "the cluster plane" rather + * than the buffer. */ + char *buf = cl_ctr_script_buf; str pl; int tlen = tag ? tag->len : 0; if (tlen > 255) tlen = 255; - if (2 + tlen + gen_msg->len > (int)sizeof(buf)) { - LM_ERR("clusterer_controller: message too large for the " - "cluster plane (%d bytes)\n", gen_msg->len); + if (!buf || 2 + tlen + gen_msg->len > cl_ctr_script_buf_sz || + 2 + tlen + gen_msg->len > cc_max_payload) { + LM_ERR("clusterer_controller: message of %d bytes is more " + "than the %d a datagram carries\n", gen_msg->len, + cc_max_payload < cl_ctr_script_buf_sz + ? cc_max_payload : cl_ctr_script_buf_sz); return -1; } buf[0] = (char)CL_CTR_SCRIPT_REQ; From bcc5fb61ad6a31e0cc0d720dab85c6749c04620f Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 16:53:03 +1000 Subject: [PATCH 21/32] clusterer_controller: enable_reliable_send defaults on; document the contract The switch was added hours earlier to gate a mechanism that could not work: a retransmit reused its seq, the receiver's anti-replay guard was a bare high-water mark, and the repair was sent truncated to its plaintext length - so on any channel it was rejected, and on a concurrent one silently re-ACKed. Measured on the cross-node cache pull channel under 40% reply loss: plain 61.2% delivered, reliable 49.4%, at two to three times the packets. With the sealed length, the replay window and the deadline-driven retransmit timer fixed, the flag does what it says. Interleaved A/B on a two-node rig, 40% loss on the consumer data leg, 300 messages at 50/s, three runs each: 57.0/59.0/60.0% before, 93.3/94.0/92.7% after - both matching closed form, for no working ARQ (1-p) and for two retries (1-p^3). Packet counts corroborate it independently: before, exactly 300 packets went out for 300 messages, so not one retransmit ever reached the wire. Separately, with a clean data leg and 35% ACK loss, 128 duplicates arrived and none was delivered twice. So the default becomes 1. The switch stays as a kill switch, because ARQ has a cost an operator may not want: a reliable broadcast draws one ACK back from every member, so on a large cluster one message becomes an event. Turning it off degrades a reliable send to a plain one - the payload still goes out - and warns. api.h and the admin guide are rewritten accordingly: the at-most-once contract, the fact that consumers are not asked to deduplicate, and the one real limit - the window spans messages rather than time, so a peer that outruns it gets a repair that is dropped and deliberately not acknowledged. (cherry picked from commit 93b1aaef0d08f939adb225fa3ef96325310d46e9) --- modules/clusterer_controller/api.h | 67 ++++++++----------- .../clusterer_controller.c | 37 +++++----- .../doc/clusterer_controller_admin.xml | 44 ++++++++++++ 3 files changed, 88 insertions(+), 60 deletions(-) diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h index 23ca6ed3cb9..164d8159749 100644 --- a/modules/clusterer_controller/api.h +++ b/modules/clusterer_controller/api.h @@ -60,48 +60,39 @@ #define CLCTR_SEND_TO_SELF (1 << 0) /* Ask for the message to be acknowledged, and resent while it is not. * - * ONLY USE THIS ON A CHANNEL WHERE ONE MESSAGE PER PEER IS IN FLIGHT AT A - * TIME. It is not a general-purpose reliability option, and on a channel - * with concurrent traffic it does not merely cost extra packets - it does - * not work, and it fails silently. + * Valid on any channel, including one carrying concurrent messages. That was + * NOT true before the receiver grew a replay window, and code or notes + * predating it may still say the flag is for serialised 1:1 exchanges only. * - * Why: every packet carries a sequence number, and a receiver drops anything - * whose seq is not strictly greater than the last it accepted from that peer - * (cl_ctr_check_and_update_seq(), the anti-replay guard). A retransmit - * re-sends the cached bytes, so it carries its ORIGINAL seq. On a serialised - * 1:1 exchange - the join handshake this was built for - nothing else is in - * flight from that peer, so the retransmit is still the highest seq and is - * accepted. As soon as a second message can overtake it, the retransmit - * arrives behind a higher seq and is discarded as out of order. The receive - * path then re-ACKs it, deliberately, so the sender stops retransmitting and - * believes it delivered. Nothing is logged above debug level at either end. + * How it behaves. Every packet carries a sequence number and the receiver + * accepts each one exactly once. A retransmit re-sends the cached bytes, so + * it carries its ORIGINAL seq; the receiver looks that seq up in a window of + * the last CL_CTR_REPLAY_WIN_BITS it has seen from that peer and can tell the + * two cases apart - never delivered (accept it now, ACK) versus already + * delivered (do not deliver twice, but re-ACK, because a duplicate means our + * previous ACK was lost). Delivery is therefore at-most-once, and the sender + * retransmits until acknowledged or its budget runs out. * - * Measured, rather than reasoned about: applying this flag to the - * cross-node cache pull channel (concurrent by nature) made the delivery rate - * WORSE - 61% -> 49% under 40% reply loss, at two to three times the packet - * count - and not one retransmitted payload was ever delivered. See the - * clusterer_controller notes for the harness. + * The one limit worth knowing: the window spans a number of MESSAGES, so a + * peer sending faster than the window divided by the retransmit horizon + * (consumer_retries x consumer_retry_ms) can outrun it. A repair arriving + * that late is dropped and deliberately NOT acknowledged - the sender is told, + * by silence, that delivery is unconfirmed, and the receiver logs it. With + * the defaults that ceiling is ~12,800 msg/s from one peer; mod_init logs the + * value in force. * - * So most consumer traffic is better served by being idempotent and retried - * by its own logic, which is how the cross-node cache fetch works and why it - * asks for nothing here. That was always the recommendation; the point of - * this note is that for a concurrent channel it is the only thing that works. + * The cost: one ACK per recipient, so a reliable send to the whole cluster is + * N-1 packets back where a plain one was nothing - hence opt-in, and per send + * rather than per channel. Traffic that can simply be made idempotent and + * retried by its own logic should still prefer that, which is what the + * cross-node cache fetch does and why it asks for nothing here. * - * The cost, where it IS applicable: one ACK per recipient, so a reliable send - * to the whole cluster is N-1 packets back where a plain one was nothing - - * hence opt-in, and per send rather than per channel. - * - * GATED. Because of the above, the flag is IGNORED unless the deployment sets - * modparam("clusterer_controller", "enable_reliable_send", 1) - * A send that asks for it without that gets a plain send and one rate-limited - * warning - which is the better outcome, not a fallback: plain measurably - * out-delivers reliable on a concurrent channel. The gate is a modparam - * because the only callers that can reach the flag today are the script - * functions, which all share one channel, and whether two of their messages - * are ever in flight to the same peer at once is a property of the - * deployment's own routes - a question the admin can answer and this module - * cannot. A consumer module that genuinely has a serialised 1:1 exchange still - * needs the admin to enable it. */ + * KILL SWITCH. The flag is honoured unless the deployment sets + * modparam("clusterer_controller", "enable_reliable_send", 0) + * which degrades every reliable send to a plain one - the payload still goes + * out, it is simply neither acknowledged nor retransmitted - and logs one + * rate-limited warning. That exists for fleets that would rather not pay the + * ACK traffic, not because the mechanism is in doubt. */ #define CLCTR_SEND_RELIABLE (1 << 1) /* limits a consumer can rely on */ diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 69a1fdc044c..6bbac68fd67 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -899,9 +899,7 @@ static int manage_shtags = 1; static int consumer_retries = CL_CTR_CONSUMER_RETRIES_DEFAULT; static int consumer_retry_ms = CL_CTR_CONSUMER_RETRY_MS_DEFAULT; /* - * CLCTR_SEND_RELIABLE is OFF unless the admin turns it on, because it does not - * work on a channel that carries concurrent messages and fails SILENTLY when - * it does not. + * enable_reliable_send - kill switch for the consumer ARQ (ACK + retransmit). * * ON by default, which it was NOT when it was first added hours earlier: the * flag then could not work on a channel carrying concurrent messages, and @@ -918,13 +916,13 @@ static int consumer_retry_ms = CL_CTR_CONSUMER_RETRY_MS_DEFAU * whose bit is set is a true duplicate and is re-ACKed without being delivered * twice. The flag now does what it says on any channel. * - * The only callers that can reach the flag today are the script functions, - * and they all share one channel (cl_ctr_script_chan) - concurrent by - * definition if two routes fire at once. Whether that happens is a property of - * the admin's own routes, which is exactly why this is a modparam: the admin - * can answer the question, and a module author cannot be asked via config. + * The switch survives the fix rather than being deleted with it, because ARQ + * still has a cost the operator may not want on a given fleet: a reliable + * broadcast draws one ACK per member, so on a large cluster a single message + * becomes an event. Turning this off degrades every reliable send to a plain + * one - lossy, but cheap and never silent about it. */ -static int enable_reliable_send = 0; +static int enable_reliable_send = 1; /* master_stickiness (global default; per-cluster override via "cluster" string): * 1 (default) = the master is "sticky": a live master keeps the role and is * NOT displaced when a higher-IP node joins. The highest-IP @@ -8049,14 +8047,12 @@ static int cl_ctr_consumer_submit(int cluster_id, int dst_node_id, /* Single gate for every sender - the script functions and the consumer API * both funnel through here, so the flag cannot survive by another route. - * Dropped rather than refused: a plain send measurably OUT-DELIVERS a - * reliable one on a concurrent channel (61.2% vs 49.4% under 40% loss), so - * ignoring the request is the better outcome, not a consolation prize. See - * the enable_reliable_send comment above for why it fails. * - * The warning is process-local and rate-limited: several workers may each - * emit one per interval, which is the same trade pull_send_failed() makes - - * an occasional duplicate line costs less than a lock on a send path. */ + * enable_reliable_send=0 degrades a reliable send to a plain one rather + * than failing it: the payload still goes out, it just is not chased. The + * warning is process-local and rate-limited - several workers may each emit + * one per interval, the same trade pull_send_failed() makes, because an + * occasional duplicate line costs less than a lock on a send path. */ if ((flags & CLCTR_SEND_RELIABLE) && !enable_reliable_send) { static unsigned int last_warn; /* per process, deliberately */ unsigned int now = get_ticks(); @@ -8066,12 +8062,9 @@ static int cl_ctr_consumer_submit(int cluster_id, int dst_node_id, last_warn = now; LM_WARN("clusterer_controller: reliable delivery was requested on " "channel '%.*s' but enable_reliable_send is 0 - sending " - "plain. That is deliberate: a retransmit reuses its seq, " - "so on a channel carrying concurrent messages it is " - "dropped as out-of-order AND re-ACKed, which silently " - "stops the sender retransmitting. Enable it only if this " - "channel never has two messages in flight to the same peer " - "at once.\n", channel->len, channel->s); + "plain, so this message is not acknowledged and not " + "retransmitted if it is lost.\n", + channel->len, channel->s); } } diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml index ef99de39bad..43c38fa1da4 100644 --- a/modules/clusterer_controller/doc/clusterer_controller_admin.xml +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -1262,6 +1262,29 @@ configured for this cluster
+
+ <varname>enable_reliable_send</varname> (int) + + Whether the CLCTR_SEND_RELIABLE flag is + honoured. On by default. Set it to 0 to turn the + acknowledge-and-retransmit machinery off fleet-wide: a send that + asks for reliability then gets an ordinary best-effort one - the + payload still goes out, it is simply not acknowledged and not + repeated if it is lost - and the module logs one rate-limited + warning. + + + The reason to want that is cost rather than doubt. A reliable + broadcast draws one acknowledgement back from every member, so + on a large cluster one message becomes an event; a fleet that + would rather not pay for it can decline here without editing the + consumers that ask. + + + Default value is 1. + +
+
Exported MI Functions @@ -1693,6 +1716,27 @@ if (cl_ctr_send_req_list(1, $avp(nodes), "just for you", "mytag", $var(sent))) send rather than per channel, because most messages are better off cheap and the ones that are not know who they are. + + What that buys is at-most-once delivery, on any channel - including one + carrying several messages to the same peer at the same time. The receiver + accepts each sequence number exactly once and keeps a window of the recent + ones, so it can tell a repair (never delivered: accept it, acknowledge it) + from a duplicate (already delivered: acknowledge again, because our first + answer evidently went missing, but do not hand the payload up twice). + Consumers are not asked to deduplicate. + + + The window spans a fixed number of messages rather than a fixed time, so a + peer sending faster than that window divided by its retransmit horizon - + multiplied by + - can outrun it. A repair that + arrives that late is dropped and deliberately not + acknowledged, because at that point the receiver genuinely cannot say + whether it ever had the message, and an acknowledgement it cannot justify + is worse than none. Both ends log it. With the default timings the ceiling + is around 12,800 messages a second from any one peer, and the module + reports the figure in force at startup. + The cost is one acknowledgement per recipient. On a large cluster a reliable broadcast is therefore one packet out and one back from every From 2f778c20521cefa10e468d2c94da9d1ff948bfcf Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Wed, 12 Aug 2026 20:53:32 +1000 Subject: [PATCH 22/32] clusterer_controller: survive a sequence-number wrap without going deaf The replay window compared sequence numbers with plain unsigned arithmetic, so when a peer's 32-bit counter wrapped, every packet after the wrap read as 2^32-ish behind and was rejected until the next key rotation reset both sides. Fail-closed, but silent - and the rotation it waits for may never come: a fresh master_salt is generated only in cl_ctr_on_became_master(), so a cluster with a stable master never rekeys on its own. Nor is the wrap remote. Every ACK bumps my_seq, so the control plane advances at the rate of the consumer traffic it acknowledges, and at the window's own 12,800 msg/s ceiling 2^32 is under four days. Compare with serial-number arithmetic instead. On its own that would trade the outage for something worse - a packet captured 2^31 or more messages ago has a difference that reads back as a large POSITIVE and would be accepted as fresh - so the forward step is bounded too: CL_CTR_REPLAY_MAX_JUMP (2^24) or more is refused as implausible rather than believed, with its own verdict and log line so a real partition is not mistaken for an attack. A peer that genuinely exceeds it is recovered by the window reset that already runs when it rejoins or reappears in a MEMBER_LIST. Note the old code accepted that 2^31-behind packet as well - unsigned comparison called it a forward jump - so the bound closes a hole rather than opening one. What remains is inherent to a 32-bit counter: a packet held across a full 2^32 cycle lands back inside the window. Proven by /dn/task89, which extracts the window logic from a given revision so it always tests the real code: against the parent commit the four wrap cases and the 2^31 replay fail; against this one all fourteen pass. (cherry picked from commit 09740850b0de9b4dff171de71d0ad5dc34513a44) --- .../clusterer_controller.c | 102 ++++++++++++------ 1 file changed, 71 insertions(+), 31 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 6bbac68fd67..c401c8e246e 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -1046,16 +1046,37 @@ static const param_export_t params[] = { * * Cost is 128 bytes per peer per plane: 64 KB of shm for a full 256-peer table. * - * NOT addressed here, and unchanged from the bare counter this replaces: the - * comparison is plain unsigned, so if a sender's 32-bit counter ever wraps - * inside one key epoch the receiver rejects everything after the wrap until - * the next rotation resets both sides. Fail-closed, but silent. Left alone - * deliberately - widening the accept test to serial-number arithmetic would - * quietly enlarge what counts as a replay, which is not a change to make in - * the same commit as a delivery fix. Task #89. + * WRAP. The comparison is serial-number arithmetic - (int32_t)(seq - last) - + * not plain unsigned, because the 32-bit counters do wrap inside one key + * epoch and a plain comparison rejects everything after the wrap until the + * next rotation. That is fail-closed but silent, and the rotation it waits + * for may never come: a fresh master_salt is generated only in + * cl_ctr_on_became_master(), so a cluster with a stable master never rekeys + * on its own. Nor is the wrap far off - every ACK bumps my_seq, so the + * control plane advances at the rate of the CONSUMER traffic it acknowledges, + * and at the window's own 12,800 msg/s ceiling 2^32 is under four days. + * + * Serial arithmetic alone would trade that outage for something worse: a + * packet captured 2^31 or more messages ago has a difference that reads back + * as a large POSITIVE, so it would be accepted as fresh. Hence the forward + * jump is bounded as well - a step of CL_CTR_REPLAY_MAX_JUMP or more is + * refused as implausible rather than believed. Real gaps stay far below it + * (it is 16.7M messages, ~22 minutes of one peer's output at that ceiling), + * and a peer that somehow exceeds it is recovered by the window reset that + * already runs when it re-joins or reappears in a MEMBER_LIST. + * + * What remains, and is inherent to a 32-bit counter: a packet held across a + * FULL 2^32 cycle lands back inside the window and is accepted. That needs + * 2^32 messages in one key epoch with the attacker holding the packet + * throughout. */ #define CL_CTR_REPLAY_WIN_BITS 1024 #define CL_CTR_REPLAY_WIN_WORDS (CL_CTR_REPLAY_WIN_BITS / 64) +/* Largest forward step believed to be a real gap rather than a wrapped-around + * replay. Must sit well above any credible burst of loss from one peer and + * well below 2^31, where serial arithmetic stops being able to tell the two + * apart. */ +#define CL_CTR_REPLAY_MAX_JUMP (1u << 24) typedef struct { uint32_t last; /* highest seq accepted so far */ @@ -1073,6 +1094,9 @@ typedef struct { #define CL_CTR_SEQ_DUP (-1) /* inside the window, bit already set: drop */ #define CL_CTR_SEQ_OLD (-2) /* below the window: drop, and we cannot know */ /* whether we ever had it */ +#define CL_CTR_SEQ_JUMP (-3) /* implausibly far ahead: a wrapped-around */ + /* replay, or a peer we have lost far too much */ + /* of - refuse either way */ static inline void cl_ctr_replay_reset(cl_ctr_replay_t *r) { @@ -1086,10 +1110,13 @@ static inline void cl_ctr_replay_reset(cl_ctr_replay_t *r) */ static inline int cl_ctr_replay_check(cl_ctr_replay_t *r, uint32_t seq) { + int32_t fwd = (int32_t)(seq - r->last); /* serial-number difference */ uint32_t d, s; - if (seq > r->last) { /* forward: the common case */ - d = seq - r->last; + if (fwd > 0) { /* forward: the common case */ + d = (uint32_t)fwd; + if (d >= CL_CTR_REPLAY_MAX_JUMP) + return CL_CTR_SEQ_JUMP; if (d >= CL_CTR_REPLAY_WIN_BITS) { memset(r->win, 0, sizeof r->win); } else { @@ -1105,10 +1132,11 @@ static inline int cl_ctr_replay_check(cl_ctr_replay_t *r, uint32_t seq) return CL_CTR_SEQ_OK; } - /* At or below the mark. Note seq == last == 0 with an empty window is the - * untouched state, and falls out here as a legitimate first accept - it - * costs nothing to allow, and every sender in fact starts at 1 - * (++my_seq). */ + /* At or below the mark. Unsigned subtraction gives the exact backward + * distance even across a wrap, and avoids negating INT32_MIN. Note + * seq == last == 0 with an empty window is the untouched state and falls + * out here as a legitimate first accept - it costs nothing to allow, and + * every sender in fact starts at 1 (++my_seq). */ d = r->last - seq; if (d >= CL_CTR_REPLAY_WIN_BITS) return CL_CTR_SEQ_OLD; @@ -2728,6 +2756,7 @@ static int cl_ctr_decrypt_pkt(char *buf, ssize_t n, const char *sender_ip, static int cl_ctr_check_and_update_seq(const char *sender_ip, uint32_t pkt_seq, cl_ctr_cluster_t *cl, int is_consumer) { + const char *why; int i, rc; for (i = 0; i < cl->peers->count; i++) { if (strcmp(cl->peers->entries[i].ip, sender_ip) == 0) { @@ -2745,17 +2774,18 @@ static int cl_ctr_check_and_update_seq(const char *sender_ip, uint32_t pkt_seq, * ordinary. Reordering no longer reaches here at all - the window * accepts it - so what is left really is a repeat or something * older than the window can vouch for. */ + why = rc == CL_CTR_SEQ_DUP ? "already accepted" + : rc == CL_CTR_SEQ_JUMP ? "implausibly far ahead - a wrapped " + "replay, or this peer's traffic has " + "been lost wholesale" + : "older than the replay window"; if (is_consumer) LM_DBG("clusterer_controller: consumer packet from %s seq=%u " "last=%u %s, dropping\n", sender_ip, pkt_seq, r->last, - rc == CL_CTR_SEQ_DUP ? "already accepted" - : "older than the replay window"); + why); else - LM_WARN("clusterer_controller: %s from %s seq=%u last=%u, " - "dropping\n", - rc == CL_CTR_SEQ_DUP ? "replay" - : "packet older than the replay window", - sender_ip, pkt_seq, r->last); + LM_WARN("clusterer_controller: packet from %s seq=%u last=%u " + "%s, dropping\n", sender_ip, pkt_seq, r->last, why); return rc; } } @@ -5789,17 +5819,27 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, if (last_old_warn == 0 || now_t - last_old_warn >= CL_CTR_RELIABLE_WARN_IVL) { last_old_warn = now_t; - LM_WARN("clusterer_controller: reliable message from " - "%s seq=%u fell more than %d messages behind " - "the replay window and cannot be confirmed - " - "NOT acknowledging. The sender is offering " - "more than the window spans within its " - "retransmit horizon (consumer_retries x " - "consumer_retry_ms = %d ms); lower that " - "horizon or expect losses.\n", - sender_ip_buf, pkt_seq, - CL_CTR_REPLAY_WIN_BITS, - cl->consumer_retries * cl->consumer_retry_ms); + if (_seq_rc == CL_CTR_SEQ_JUMP) + LM_WARN("clusterer_controller: reliable message " + "from %s seq=%u is implausibly far ahead " + "of that peer's last accepted sequence - " + "refused and NOT acknowledged. Either a " + "replay of a very old packet, or this " + "peer's traffic has been lost wholesale; " + "membership will reset the window when it " + "rejoins.\n", sender_ip_buf, pkt_seq); + else + LM_WARN("clusterer_controller: reliable message " + "from %s seq=%u fell more than %d messages " + "behind the replay window and cannot be " + "confirmed - NOT acknowledging. The sender " + "is offering more than the window spans " + "within its retransmit horizon " + "(consumer_retries x consumer_retry_ms = " + "%d ms); lower that horizon or expect " + "losses.\n", sender_ip_buf, pkt_seq, + CL_CTR_REPLAY_WIN_BITS, + cl->consumer_retries * cl->consumer_retry_ms); } } } From ee2e13c10f71e764210462274e3b419b536d9e4e Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 13:32:17 +1000 Subject: [PATCH 23/32] clusterer_controller: honour the interface MTU instead of padding up to CLCTR_MAX_PAYLOAD mod_init derived the maximum consumer payload from the interface MTU and then raised it back to CLCTR_MAX_PAYLOAD whenever the link was smaller than that. The effect was the exact inverse of why the MTU is consulted at all: on any interface under about 1411 - a VPN, a GRE or IPIP tunnel, PPPoE, the usual 1400/1420/1436 - the module advertised a bound ABOVE what the path carries. Every full-size consumer datagram then IP-fragmented, and a fragmented datagram is lost entirely if any one fragment is, so the padding manufactured precisely the losses the retransmit machinery then had to repair. Honour the link and say so instead. CLCTR_MAX_PAYLOAD stays what a consumer sizes a compile-time buffer with, but it stops being a promise the network cannot keep, and a consumer that cares about the real bound now asks for it. (cherry picked from commit f0ff9a2dfb, clusterer_controller portion only - the cachedb_perf half belongs to the cachedb_perf PR) --- modules/clusterer_controller/api.h | 18 ++++++++--- .../clusterer_controller.c | 31 ++++++++++++++++++- 2 files changed, 43 insertions(+), 6 deletions(-) diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h index 164d8159749..38ae709e290 100644 --- a/modules/clusterer_controller/api.h +++ b/modules/clusterer_controller/api.h @@ -97,11 +97,19 @@ /* limits a consumer can rely on */ #define CLCTR_MAX_CHAN_LEN 31 -/* Compile-time lower bound for consumer payload; the actual runtime limit is - * cc_max_payload, which is derived from the interface MTU at mod_init and may - * be larger on jumbo-frame links. Consumers that size local buffers at - * compile time should use CLCTR_MAX_PAYLOAD; consumers that want to send the - * largest possible message at runtime should check cc_max_payload instead. */ +/* A constant to size compile-time buffers with - nothing more. It is NOT a + * guaranteed payload size. + * + * The runtime limit is cc_max_payload, derived from the interface MTU at + * mod_init: LARGER on a jumbo-frame link, and SMALLER on a VPN, tunnel or + * PPPoE link whose MTU is under about 1411. It used to be padded up to this + * constant when the link was smaller, which made the constant a promise the + * network could not keep and fragmented every full-size datagram to pretend + * otherwise; the module now honours the link and warns at startup instead. + * + * So: size a stack buffer with CLCTR_MAX_PAYLOAD - that is always safe, since + * a buffer can only be too big - but test the length you actually intend to + * send against cc_max_payload, which is the only bound that will be enforced. */ #define CLCTR_MAX_PAYLOAD 1300 extern int cc_max_payload; diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index c401c8e246e..e653d448e4b 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -7449,13 +7449,42 @@ static int mod_init(void) int _overhead = 20 + 8 + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_CONSUMER_HDR_SZ + CLCTR_MAX_CHAN_LEN + CL_CTR_TAG_SZ; cc_max_payload = _mtu - _overhead; + /* Do NOT pad this back up to CLCTR_MAX_PAYLOAD when the link is + * smaller than that. It used to, and the effect was that any + * interface under ~1411 MTU - a VPN, a GRE/IPIP tunnel, PPPoE, + * the usual 1400/1420/1436 - reported a bound ABOVE what the + * path carries, which is the exact inverse of why the MTU is + * consulted at all. Every full-size consumer datagram then IP + * fragments, and a fragmented datagram is lost entirely if any + * one fragment is, so the padding manufactured precisely the + * losses the retransmit machinery then has to repair. + * + * Honour the link and say so instead: the compile-time constant + * stays what a consumer sizes a buffer with, but it stops being + * a promise the network cannot keep. */ if (cc_max_payload < CLCTR_MAX_PAYLOAD) - cc_max_payload = CLCTR_MAX_PAYLOAD; + LM_WARN("clusterer_controller: interface %s has MTU %d, so a " + "consumer datagram carries at most %d bytes - below " + "the %d that CLCTR_MAX_PAYLOAD leads consumers to " + "expect. Sends above %d will be refused on this " + "link. Raise the MTU, or keep consumer payloads " + "under it; the bound is deliberately NOT padded up " + "to the constant, because that only fragments and a " + "lost fragment loses the whole message.\n", + my_interface_buf, _mtu, cc_max_payload, + (int)CLCTR_MAX_PAYLOAD, cc_max_payload); /* However large the MTU claims to be, one datagram still has * to fit a UDP payload. Loopback reports 65536, which lands a * byte past what sendto() accepts. */ if (cc_max_payload > CL_CTR_UDP_PAYLOAD_MAX - CL_CTR_PKT_OVERHEAD) cc_max_payload = CL_CTR_UDP_PAYLOAD_MAX - CL_CTR_PKT_OVERHEAD; + if (cc_max_payload <= 0) { + LM_ERR("clusterer_controller: interface %s has MTU %d, which " + "cannot carry even the %d bytes of per-datagram " + "overhead - the cluster plane cannot run on it\n", + my_interface_buf, _mtu, _overhead); + return -1; + } LM_INFO("clusterer_controller: interface %s MTU=%d, " "max consumer payload=%d bytes\n", my_interface_buf, _mtu, cc_max_payload); From 7adcdd6919a51b80305bc894a79aff89072c9cc4 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 13:32:17 +1000 Subject: [PATCH 24/32] clusterer_controller: publish the runtime payload bound through the API A consumer needs the MTU-derived bound, not the compile-time constant, and the only way to reach it was to declare `extern int cc_max_payload`. That is defined in clusterer_controller.c, and OpenSIPS modules are dlopen'd, so the reference leaves an undefined symbol in the consumer's .so that resolves only if clusterer_controller happens to have been loaded first. It does not fail at build time - it fails at startup with cachedb_perf.so: undefined symbol: cc_max_payload and takes the whole config down with it. Every other cross-module call here goes through the bound API for exactly this reason, so add get_max_payload() alongside them and say plainly in api.h not to declare the extern. (cherry picked from commit bdb52a3ba3, clusterer_controller portion only - the cachedb_perf half belongs to the cachedb_perf PR) --- modules/clusterer_controller/api.h | 20 +++++++++++++++++-- .../clusterer_controller.c | 9 +++++++++ 2 files changed, 27 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/api.h b/modules/clusterer_controller/api.h index 38ae709e290..4ab18e30e71 100644 --- a/modules/clusterer_controller/api.h +++ b/modules/clusterer_controller/api.h @@ -109,9 +109,18 @@ * * So: size a stack buffer with CLCTR_MAX_PAYLOAD - that is always safe, since * a buffer can only be too big - but test the length you actually intend to - * send against cc_max_payload, which is the only bound that will be enforced. */ + * send against the runtime bound, which is the only one that will be enforced, + * and which you obtain through get_max_payload() below. + * + * Do NOT declare `extern int cc_max_payload` in a consumer. It is defined in + * clusterer_controller.c, and OpenSIPS modules are dlopen'd, so referencing it + * directly makes the consumer's .so carry an undefined symbol that resolves + * only if clusterer_controller happens to have been loaded first. It does not + * fail at build time - it fails at startup with + * "cachedb_perf.so: undefined symbol: cc_max_payload" + * and takes the whole config down with it. Every other cross-module call here + * goes through the bound API for exactly this reason. */ #define CLCTR_MAX_PAYLOAD 1300 -extern int cc_max_payload; typedef void (*clctr_msg_cb_f)(int cluster_id, int src_node_id, str *channel, str *payload); @@ -146,6 +155,12 @@ typedef int (*clctr_send_list_f)(int cluster_id, const int *node_ids, int n, typedef int (*clctr_get_my_node_id_f)(int cluster_id); +/* The runtime maximum consumer payload, derived from the interface MTU at + * mod_init. May be LARGER than CLCTR_MAX_PAYLOAD on a jumbo link or SMALLER on + * one under ~1411 MTU, so a consumer that cares must ask rather than assume. + * Safe to call any time after the controller's mod_init. */ +typedef int (*clctr_get_max_payload_f)(void); + /* * This node's own address on the cluster plane, as RESOLVED at startup - not * the raw modparam. It comes from one of three places (explicit `my_ip`, the @@ -170,6 +185,7 @@ typedef struct clctr_api { clctr_send_list_f send_list; clctr_get_my_node_id_f get_my_node_id; clctr_get_my_ip_f get_my_ip; + clctr_get_max_payload_f get_max_payload; } clctr_api_t; typedef int (*load_clctr_f)(clctr_api_t *api); diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index e653d448e4b..bc6162d6d17 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -891,6 +891,7 @@ static int clctr_send_list(int cluster_id, const int *node_ids, int n, static int clctr_get_my_node_id(int cluster_id); int cl_ctr_get_my_ip(const char **ip, const char **iface, const char **src); int load_clctr(clctr_api_t *api); +static int clctr_get_max_payload(void); /* clusterer integration - loaded at mod_init if clusterer use_controller=1 */ static clusterer_ctrl_binds_t clctl; @@ -8478,6 +8479,13 @@ static int cl_ctr_script_init(void) return clctr_register_channel(&cl_ctr_script_chan, cl_ctr_script_recv); } +/* Consumers must reach cc_max_payload through the bound API, never as an extern + * - see the note in api.h. */ +static int clctr_get_max_payload(void) +{ + return cc_max_payload; +} + int load_clctr(clctr_api_t *api) { if (!api) @@ -8488,5 +8496,6 @@ int load_clctr(clctr_api_t *api) api->send_list = clctr_send_list; api->get_my_node_id = clctr_get_my_node_id; api->get_my_ip = cl_ctr_get_my_ip; + api->get_max_payload = clctr_get_max_payload; return 0; } From 9eaf9e2358ef740d97c26c5fca61c3d63a799f67 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 02:13:15 +1000 Subject: [PATCH 25/32] clusterer_controller: detect the cluster-plane MTU from the interface The MTU is read from the interface the plane runs on and kept in cc_mtu (what we joined at) and cc_mtu_now (the live reading). There is deliberately no modparam for it: a configured value is a second source of truth that can disagree with the kernel, so the config says 1500 while the link says 9000 and the node refuses to start while being perfectly consistent with its own segment. That is the same defect class as the CLCTR_MAX_PAYLOAD floor. Because the MTU now decides whether a node may join at all, two places that used to shrug become fatal: - auto-detection that cannot name the interface owning the resolved IP used to warn and carry on with the compile-time default. Both explicit modes (my_ip, interface) always yield a name, so this is only reachable from the default-route probe, and naming either modparam is the fix. - a failed SIOCGIFMTU leaves the node unable to establish whether it belongs in the cluster, which is not a state it can serve SIP from. The value is exported through cl_ctr_list_members, cl_ctr_node_info and cl_ctr_list_config, because a log line is not something an operator can alert on - and one gateway in this fleet has no journal at all. (cherry picked from commit ccfdc3e40f6da9283e3e80ea5372f969b7358eba) --- .../clusterer_controller.c | 136 +++++++++++++++++- 1 file changed, 129 insertions(+), 7 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index bc6162d6d17..05e81a30c73 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -476,6 +476,32 @@ typedef struct { * accidental per-node config drift for the same cluster: * manage_shtags(1B) + master_stickiness(1B) + query_time(2B BE). */ #define CL_CTR_CONFIG_SZ 4 +/* The node's cluster-plane MTU, advertised in JOIN_REQ and ALIVE as 2 bytes BE. + * + * Deliberately NOT folded into CL_CTR_CONFIG_SZ above. That block is governed + * by the on_config_mismatch modparam, and two of its three modes would be wrong + * here: "warn" would quietly demote MTU enforcement to a log line, and "adopt" + * is meaningless - a node cannot adopt a peer's MTU, the kernel owns it. There + * is deliberately NO modparam for the MTU at all, because a configured value is + * a second source of truth that can disagree with the interface: the config + * says 1500, the link says 9000, and the node refuses to start while being + * perfectly consistent with its own segment. So the MTU gets its own field and + * its own unconditional check. + * + * 0 means "not advertised" - a peer built before this field existed. Such a + * peer is admitted with a warning rather than rejected, because rejecting it + * would make a rolling upgrade impossible: every not-yet-upgraded node would be + * refused by the first upgraded master and would self-terminate. */ +#define CL_CTR_MTU_SZ 2 +/* Consecutive polls that must disagree with the MTU we joined on before this + * node accepts that its own link really changed. A single reading is not + * enough: bond failover, a driver reset or a VLAN parent bounce can show a + * transient value, and acting on the first sample would kill a healthy node + * over a blip. At query_time=5 the default is a ~15 s confirmation. */ +#define CL_CTR_MTU_DRIFT_STRIKES 3 +/* Distinct peers a master may refuse on MTU before it says out loud that IT is + * the likely misconfiguration. See cl_ctr_mtu_note_reject(). */ +#define CL_CTR_MTU_SUSPECT_PEERS 2 /* Noise handshake message sizes (NNpsk0, X25519, ChaChaPoly, SHA-256): * msg 1 = e(32) + tag over empty payload(16) = 48 * msg 2 = e(32) + AEAD(master_salt 32 + tag 16) = 80 */ @@ -809,6 +835,62 @@ static char my_interface_buf[IF_NAMESIZE]; * jumbo-frame interfaces (MTU 9000) are not artificially limited to 1300 B. * Read-only after mod_init; safe to access from any process. */ int cc_max_payload = CLCTR_MAX_PAYLOAD; +/* This node's cluster-plane MTU as read from my_interface_buf at mod_init. + * 0 only before mod_init has resolved it; after that it is always a real + * kernel-reported value, because mod_init refuses to start otherwise. */ +static int cc_mtu = 0; +/* The interface's CURRENT MTU, refreshed by the drift poll. Distinct from + * cc_mtu on purpose, and the distinction is load-bearing: + * + * cc_mtu - what we joined at. The cluster's MTU, and therefore the only + * thing admission and drift are judged against. Never changes. + * cc_mtu_now - what the kernel says right now. This is what we ADVERTISE. + * + * Advertising the joined-at value instead would make the peer-drift warning + * unreachable: a node whose link changed would keep announcing the old number, + * so no peer could ever observe the change. Announcing the live reading means + * a healthy peer sees the drift on the next ALIVE - which matters, because the + * drifting node is about to remove itself and its own log is the one least + * likely to be watched. */ +static int cc_mtu_now = 0; +/* Consecutive polls that disagreed with cc_mtu (see CL_CTR_MTU_DRIFT_STRIKES). */ +static int cc_mtu_drift_strikes = 0; + +/** + * cl_ctr_read_iface_mtu() - current MTU of the interface we run the plane on. + * + * Returns the kernel's value, or -1 if it cannot be read. This is the ONLY + * source of MTU truth in the module; there is no configured alternative to + * fall back to, by design. + */ +static int cl_ctr_read_iface_mtu(void) +{ + struct ifreq ifr; + int s, mtu = -1; + + if (my_interface_buf[0] == '\0') + return -1; + if ((s = socket(AF_INET, SOCK_DGRAM, 0)) < 0) + return -1; + memset(&ifr, 0, sizeof(ifr)); + memcpy(ifr.ifr_name, my_interface_buf, + strnlen(my_interface_buf, IF_NAMESIZE - 1)); + if (ioctl(s, SIOCGIFMTU, &ifr) == 0) + mtu = ifr.ifr_mtu; + close(s); + return mtu; +} + +/* The MTU travels the wire as an unsigned 16-bit value. Loopback reports + * 65536, one past what that holds, so clamp - two nodes both on loopback still + * agree, and no real cluster link is anywhere near it. */ +static inline uint16_t cl_ctr_mtu_wire(int mtu) +{ + if (mtu <= 0) return 0; + if (mtu > 0xFFFF) return 0xFFFF; + return (uint16_t)mtu; +} + /* Largest packet the retransmit cache will hold, raised in mod_init once the * MTU is known. Provisional value covers the control plane, which is all that * can be sent before then anyway. */ @@ -1176,6 +1258,10 @@ typedef struct cl_ctr_peer_ { int cfg_master_stickiness; int cfg_query_time; int cfg_warned; + /* Peer's advertised cluster-plane MTU (0 = never advertised, i.e. a build + * older than the field). mtu_warned deduplicates the drift warning. */ + int mtu; + int mtu_warned; } cl_ctr_peer_t; struct cl_ctr_peers_ { @@ -6631,7 +6717,8 @@ static mi_response_t *mi_cl_ctr_members(const mi_params_t *params, lock_stop_read(cl->peers->lock); goto error; } - if (add_mi_string(peer_obj, MI_SSTR("ip"), + if (add_mi_number(peer_obj, MI_SSTR("mtu"), e->mtu) < 0 || + add_mi_string(peer_obj, MI_SSTR("ip"), e->ip, strlen(e->ip)) < 0 || add_mi_number(peer_obj, MI_SSTR("node_id"), e->node_id) < 0 || add_mi_string(peer_obj, MI_SSTR("status"), @@ -6701,7 +6788,8 @@ static mi_response_t *mi_cl_ctr_node_info(const mi_params_t *params, lock_stop_read(cl->peers->lock); return NULL; } - if (add_mi_number(root, MI_SSTR("node_id"), e->node_id) < 0 || + if (add_mi_number(root, MI_SSTR("mtu"), e->mtu) < 0 || + add_mi_number(root, MI_SSTR("node_id"), e->node_id) < 0 || add_mi_string(root, MI_SSTR("ip"), e->ip, strlen(e->ip)) < 0 || add_mi_number(root, MI_SSTR("cluster_id"), cl->cluster_id) < 0 || add_mi_string(root, MI_SSTR("status"), @@ -6802,6 +6890,10 @@ static mi_response_t *mi_cl_ctr_config(const mi_params_t *params, add_mi_string(cl_obj, MI_SSTR("my_ip"), my_ip, strlen(my_ip)) < 0 || add_mi_string(cl_obj, MI_SSTR("bin_socket"), cl->bin_socket, strlen(cl->bin_socket)) < 0 || + add_mi_number(cl_obj, MI_SSTR("mtu"), cc_mtu) < 0 || + add_mi_string(cl_obj, MI_SSTR("interface"), + my_interface_buf[0] ? my_interface_buf : "(unknown)", + my_interface_buf[0] ? strlen(my_interface_buf) : 9) < 0 || add_mi_number(cl_obj, MI_SSTR("query_time"), eff_qt) < 0 || add_mi_number(cl_obj, MI_SSTR("master_stickiness"), eff_stick) < 0 || add_mi_number(cl_obj, MI_SSTR("manage_shtags"), eff_manage) < 0 || @@ -7247,12 +7339,24 @@ static int cl_ctr_resolve_local_identity(void) break; } } - if (found) + if (found) { LM_INFO("clusterer_controller: auto-detected IP %s on " "interface %s\n", my_ip, my_interface_buf); - else - LM_WARN("clusterer_controller: auto-detected IP %s but could " - "not determine interface name\n", my_ip); + } else { + /* Fatal, where it used to warn and continue. The interface name is + * the only route to the MTU, and the MTU is what admission to the + * cluster is decided on - a node that cannot name its own interface + * cannot know whether it belongs, and would previously have run on + * the compile-time default as if it did. Both explicit modes + * (my_ip, interface) always yield a name, so this is only reachable + * from auto-detection, and naming either one is the fix. */ + LM_ERR("clusterer_controller: auto-detected IP %s but no interface " + "owns it, so its MTU cannot be read - and the cluster admits " + "only nodes whose MTU matches. Set the 'interface' modparam " + "(or 'my_ip') so the link is unambiguous\n", my_ip); + freeifaddrs(ifap); + return -1; + } } freeifaddrs(ifap); @@ -7439,6 +7543,22 @@ static int mod_init(void) * Overhead per consumer datagram: * IP(20) + UDP(8) + wire_hdr(28) + plain_hdr(5) + * consumer_hdr(3) + max_chan(31) + poly1305_tag(16) = 111 bytes. */ + { + int _probe_mtu = cl_ctr_read_iface_mtu(); + if (_probe_mtu <= 0) { + /* No modparam can supply this: the MTU is detected, never + * configured, precisely so it cannot disagree with the kernel. + * Without it this node cannot know whether it may join, and a node + * that cannot participate must not serve SIP. */ + LM_ERR("clusterer_controller: cannot read the MTU of %s (%s) - the " + "cluster admits only nodes whose MTU matches, so this node " + "cannot establish whether it belongs and will not start\n", + my_interface_buf[0] ? my_interface_buf : "(no interface)", + my_interface_buf[0] ? strerror(errno) : "interface unresolved"); + return -1; + } + cc_mtu = cc_mtu_now = _probe_mtu; + } if (my_interface_buf[0] != '\0') { struct ifreq _mtu_ifr; int _mtu_sock = socket(AF_INET, SOCK_DGRAM, 0); @@ -7487,7 +7607,9 @@ static int mod_init(void) return -1; } LM_INFO("clusterer_controller: interface %s MTU=%d, " - "max consumer payload=%d bytes\n", + "max consumer payload=%d bytes - every cluster member " + "must run at this MTU; one that does not is refused at " + "join\n", my_interface_buf, _mtu, cc_max_payload); } else { LM_WARN("clusterer_controller: SIOCGIFMTU on %s failed: %s - " From 1739f484b2abd022b698e88bed67704f0cf61d5d Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 02:13:15 +1000 Subject: [PATCH 26/32] clusterer_controller: refuse a joiner whose MTU differs from the cluster's JOIN_REQ and ALIVE now carry the sender's MTU, and the master admits only nodes whose value equals its own. That single rule is the whole mechanism: by induction every member carries the master's MTU, so the backup that eventually becomes master necessarily carries it too and a handover cannot change it. Nothing has to be inherited, agreed or carried in MEMBER_LIST - membership is itself the proof of agreement. A partition inherits the property on both sides, and a master that drifts removes itself while the backup keeps the original value, so a drifting node cannot redefine the cluster's MTU on its way out. The check has to happen here because the failure is undetectable later: an oversized datagram is dropped by the small-MTU node's NIC with no ICMP, since every node is on one L2 segment and there is no router to generate one, and multicast has no path-MTU discovery in any topology. The receiver therefore can never observe the packet it failed to receive. At join both sides are still exchanging small handshake packets that do arrive. It refuses BEFORE KEY_GRANT for key custody, not speed: the joiner holds no session key until then, so a refused node never obtains the group key at all, where admit-then-eject would have handed it multicast decryption and needed a rotation to take it back. The refusal also drops the peer from the table - while still NEW a node upserts every joiner it hears, so refusing alone would leave a node we just turned away being counted as a member. Unconditional, unlike the config gate: there is no policy knob, because a node that cannot participate must not serve SIP. Without shared usrloc, dialog replication and shtag coordination it answers REGISTERs into a local store and routes on a partial view while the load balancer keeps feeding it calls. A joiner that advertises no MTU at all is an older build. It is admitted with a warning rather than refused, or a rolling upgrade could never complete: every not-yet-upgraded node would be turned away by the first upgraded master. Master election is by highest IP and completely MTU-blind, so on a simultaneous cold start a single misconfigured host that wins it will refuse an otherwise healthy fleet. cl_ctr_mtu_note_reject() says so out loud once the master has turned away several distinct peers while holding no members of its own. It only warns - a master that stepped down because OTHERS disagreed would break the rule that a node acts on its own condition alone, and would hand anyone holding the bootstrap key a way to force re-elections. A refused node self-terminates, naming both MTUs and the interface, but only after checking the claim against its own kernel reading: a reject quoting our own MTU back at us contradicts itself and is ignored. It acts whenever an admission request is outstanding, not merely while NEW - on a simultaneous cold start both nodes reach the join deadline, both self-promote, and the loser merges into the winner, so it is ACTIVE rather than NEW when its JOIN_REQ is refused. A settled member with no request outstanding still ignores rejects, which is what stops one member from evicting another. (cherry picked from commit ab56c508c22292681bf64582985f107774725563) --- .../clusterer_controller.c | 231 +++++++++++++++++- 1 file changed, 223 insertions(+), 8 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 05e81a30c73..3dfc9568bc0 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -510,7 +510,7 @@ typedef struct { /* JOIN_REQ: [ip NUL][bin_count 1B][sockets...][noise_msg1 48B][config 4B] */ #define CL_CTR_JOIN_PKT_MAX_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 \ + CL_CTR_BIN_INFO_MAX_SZ + CL_CTR_NOISE_MSG1_SZ \ - + CL_CTR_CONFIG_SZ + CL_CTR_TAG_SZ) + + CL_CTR_CONFIG_SZ + CL_CTR_MTU_SZ + CL_CTR_TAG_SZ) /* KEY_GRANT: [target_ip NUL][noise_msg2 80B] */ #define CL_CTR_KEY_GRANT_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 \ + CL_CTR_NOISE_MSG2_SZ + CL_CTR_TAG_SZ) @@ -770,6 +770,12 @@ typedef struct cl_ctr_cluster_ { * so a peer that keeps sending JOIN_REQ yet never becomes master cannot * defer us forever. Worker-local; reset once we leave the NEW state. */ int join_defer_total; + /* Distinct peers we have refused on MTU while acting as master, and whether + * we have already said we are probably the one at fault. Worker-local and + * diagnostic only - see cl_ctr_mtu_note_reject(). */ + uint32_t mtu_reject_ips[CL_CTR_MTU_SUSPECT_PEERS * 2]; + int mtu_reject_n; + int mtu_suspect_said; /* utime (us since start) of the last JOIN_REQ we transmitted, for a * minimum-interval throttle so a key-mismatch/split-brain burst cannot * flood the group with JOIN_REQs. 0 = never sent. Worker-local. */ @@ -820,6 +826,7 @@ static char *password = CL_CTR_DEFAULT_PASSWORD; /* default; falls back /* JOIN_REJECT reason codes (1 byte after the target IP in the payload). */ #define CL_CTR_REJECT_GENERIC 0 /* wrong password / unauthorized / table full */ #define CL_CTR_REJECT_CONFIG 1 /* different cluster settings (reject policy) */ +#define CL_CTR_REJECT_MTU 2 /* different cluster-plane MTU (never optional) */ static char *on_config_mismatch_s = NULL; /* raw modparam string */ static int on_config_mismatch = CL_CTR_CFGMISMATCH_REJECT; /* resolved; default reject */ @@ -3572,8 +3579,9 @@ static void cl_ctr_send_ack(int sock, cl_ctr_cluster_t *cl, uint32_t acked_seq, static void cl_ctr_send_pkt_with_ip(int sock, unsigned char type, cl_ctr_cluster_t *cl, const struct sockaddr *dest, socklen_t destlen) { - /* Sized for ALIVE which carries an extra pubkey + config descriptor */ - char pkt[CL_CTR_SMALL_PKT_SZ + CL_CTR_PUBKEY_SZ + CL_CTR_CONFIG_SZ]; + /* Sized for ALIVE which carries an extra pubkey + config descriptor + MTU */ + char pkt[CL_CTR_SMALL_PKT_SZ + CL_CTR_PUBKEY_SZ + + CL_CTR_CONFIG_SZ + CL_CTR_MTU_SZ]; uint32_t seq = htonl(++cl->peers->my_seq); int ip_len = (int)strlen(my_ip); int plain_len; @@ -3604,6 +3612,14 @@ static void cl_ctr_send_pkt_with_ip(int sock, unsigned char type, cl_ctr_cluster memcpy(c + 2, &qt, 2); plain_len += CL_CTR_CONFIG_SZ; } + /* And the MTU of the link this plane runs on, so peers can notice that + * one of them has drifted. Separate from the config block above + * because it is detected, not configured - see CL_CTR_MTU_SZ. */ + { + uint16_t m = htons(cl_ctr_mtu_wire(cc_mtu_now)); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + plain_len, &m, CL_CTR_MTU_SZ); + plain_len += CL_CTR_MTU_SZ; + } } if (dest) @@ -3755,6 +3771,17 @@ static void cl_ctr_send_join_req_pkt(int sock, cl_ctr_cluster_t *cl) p += 2; } + /* Advertise the MTU of our cluster link. The master admits only nodes + * whose MTU equals its own, which is what keeps the whole cluster uniform: + * every member matched at this point, so the backup that eventually + * becomes master necessarily carries the same value and a handover cannot + * change it. */ + { + uint16_t m = htons(cl_ctr_mtu_wire(cc_mtu_now)); + memcpy(p, &m, CL_CTR_MTU_SZ); + p += CL_CTR_MTU_SZ; + } + plain_len = (int)(p - (pkt + CL_CTR_WIRE_HDR_SZ)); if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->key, CL_CTR_PKT_JOIN_REQ) == 0) LM_DBG("clusterer_controller: [cluster %d] sent JOIN_REQ bin=%s\n", @@ -4301,6 +4328,91 @@ static void cl_ctr_adopt_config(cl_ctr_cluster_t *cl, int new_manage, int new_st } } +/** + * cl_ctr_drop_peer_locked() - remove a peer from the table by IP. + * + * Needed because refusing a JOIN_REQ is not by itself enough to keep a node + * out: while this node was still NEW it upserts every joiner it hears (the + * split-brain defer path), so a peer can already be in the table by the time + * we become master and refuse it. Without this the master goes on counting a + * node it just told to go away - and, worse, cl_ctr_mtu_note_reject() would + * never see the "holding no members" condition that makes it speak up. + * + * Same swap-with-last removal cl_ctr_prune_stale uses. Call WITH the write + * lock held. + */ +static void cl_ctr_drop_peer_locked(const char *ip, cl_ctr_cluster_t *cl) +{ + int i; + + for (i = 0; i < cl->peers->count; i++) { + uint16_t dropped_id; + if (strcmp(cl->peers->entries[i].ip, ip) != 0) + continue; + dropped_id = cl->peers->entries[i].node_id; + cl->peers->count--; + if (i < cl->peers->count) + cl->peers->entries[i] = cl->peers->entries[cl->peers->count]; + memset(&cl->peers->entries[cl->peers->count], 0, sizeof(cl_ctr_peer_t)); + if (clctl_loaded && dropped_id > 0) + clctl.remove_node(cl->cluster_id, dropped_id); + return; + } +} + +/** + * cl_ctr_mtu_note_reject() - remember that we refused a peer on MTU, and if we + * have now refused several DIFFERENT peers while holding no members of our own, + * say plainly that WE are the likely misconfiguration. + * + * Why this exists. Master election is highest-IP (cl_ctr_elect_master) and + * entirely MTU-blind, so on a simultaneous cold start the winner is decided by + * something with no relationship to which MTU is correct. If six nodes run at + * 9000 and the one 1500 host happens to hold the highest IP, it becomes master + * and refuses all six - one misconfigured host takes down the whole fleet, + * where that same host booting into an already-formed cluster would only have + * killed itself. Nothing else in the design catches that, so this warning is + * the safety net rather than a nicety. + * + * It WARNS and nothing more. A master that stepped down or terminated because + * OTHERS disagreed would break the rule that a node acts only on its own + * condition, and would hand anyone holding the bootstrap key a way to force + * re-elections at will. + */ +static void cl_ctr_mtu_note_reject(const char *src_ip, cl_ctr_cluster_t *cl) +{ + uint32_t ip_num = ip_to_num(src_ip); + int i, members; + + if (ip_num == 0 || cl->mtu_suspect_said) + return; + for (i = 0; i < cl->mtu_reject_n; i++) + if (cl->mtu_reject_ips[i] == ip_num) + return; /* same peer retrying, not a new one */ + if (cl->mtu_reject_n < (int)(sizeof(cl->mtu_reject_ips) + / sizeof(cl->mtu_reject_ips[0]))) + cl->mtu_reject_ips[cl->mtu_reject_n++] = ip_num; + + /* "Holding no members" means nobody but ourselves is in the table. */ + lock_start_read(cl->peers->lock); + members = cl->peers->count; + lock_stop_read(cl->peers->lock); + + if (cl->mtu_reject_n >= CL_CTR_MTU_SUSPECT_PEERS && members <= 1) { + cl->mtu_suspect_said = 1; + LM_CRIT("clusterer_controller: [cluster %d] THIS NODE IS PROBABLY THE " + "MISCONFIGURED ONE: it is master at MTU %d on %s, has refused %d " + "different peers for not matching, and holds no members of its " + "own. Master election is by highest IP and ignores the MTU, so a " + "single wrongly-configured host that wins it will refuse an " + "otherwise healthy fleet. Check this node's MTU before changing " + "any of the others\n", + cl->cluster_id, cc_mtu, + my_interface_buf[0] ? my_interface_buf : "(unknown)", + cl->mtu_reject_n); + } +} + /** * cl_ctr_handle_alive() - process a CL_CTR_PKT_ALIVE packet. * @@ -4430,6 +4542,7 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le uint8_t bin_cnt = 0; int ip_len, was_master, i; int j_cfg_present = 0, j_manage = 0, j_stick = 0, j_qt = 0; + int j_mtu = 0; /* 0 = joiner advertised none (older build) */ uint16_t new_id; /* --- Parse IP --- */ @@ -4478,6 +4591,14 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le p += CL_CTR_CONFIG_SZ; } + /* --- Parse the joiner's cluster-plane MTU (absent on older builds) --- */ + if (p + CL_CTR_MTU_SZ <= end) { + uint16_t m_be; + memcpy(&m_be, p, CL_CTR_MTU_SZ); + j_mtu = ntohs(m_be); + p += CL_CTR_MTU_SZ; + } + LM_INFO("clusterer_controller: [cluster %d] JOIN_REQ from %s " "(%d BIN socket(s))\n", cl->cluster_id, src_ip, bin_cnt); @@ -4530,6 +4651,51 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le } } + /* MTU gate (master side). A cluster must have ONE MTU: an oversized + * datagram is dropped by the small-MTU node's NIC with no ICMP to report + * it (same L2 segment, so there is no router to generate one), and a + * multicast sender has no path to discover. So the receiver can never + * observe the packet it failed to receive - the mismatch has to be caught + * here, while both sides are still exchanging small handshake packets that + * do arrive. + * + * Refuse BEFORE KEY_GRANT, which is key custody rather than speed: the + * joiner holds no session key until KEY_GRANT hands it the ECDH-wrapped + * master_salt, so refusing now means a mismatched node never obtains the + * group key at all. Admitting it and ejecting it later would have handed + * it multicast decryption and required a key rotation to take that back. + * + * Unconditional - there is no policy modparam here, unlike the config gate + * above. A node that cannot participate must not serve SIP: without shared + * usrloc, dialog replication and shtag coordination it would answer + * REGISTERs into a local store and route on a partial view while the load + * balancer kept feeding it calls. Down is the honest state. */ + if (j_mtu > 0 && cc_mtu > 0 && j_mtu != cc_mtu) { + /* Drop it before releasing the lock: while we were ourselves NEW we + * upserted every joiner we heard, so this peer may already be in the + * table and refusing the join alone would leave it there. */ + cl_ctr_drop_peer_locked(src_ip, cl); + lock_stop_write(cl->peers->lock); + LM_WARN("clusterer_controller: [cluster %d] rejecting JOIN_REQ from %s: " + "its cluster-plane MTU is %d, this cluster runs at %d. Every " + "member must agree, because an oversized datagram is dropped " + "silently by the smaller link.\n", + cl->cluster_id, src_ip, j_mtu, cc_mtu); + cl_ctr_send_join_reject(sock, src_ip, cl, CL_CTR_REJECT_MTU); + cl_ctr_mtu_note_reject(src_ip, cl); + return; + } + if (j_mtu == 0) { + /* Older build: it cannot tell us, so uniformity is unenforceable for + * this peer. Admitting it is deliberate - refusing would make a + * rolling upgrade impossible, since every not-yet-upgraded node would + * be turned away by the first upgraded master. */ + LM_WARN("clusterer_controller: [cluster %d] %s did not advertise an " + "MTU (older build) - admitting it, but MTU uniformity is NOT " + "enforced for this peer until it is upgraded\n", + cl->cluster_id, src_ip); + } + /* Reject JOIN_REQ from an unknown IP when the peer table is full. * Known peers (reconnecting after restart) are still allowed through * since they already occupy a slot. */ @@ -5610,7 +5776,10 @@ static int cl_ctr_join_fail_check(const char *src_ip, cl_ctr_cluster_t *cl) static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_cluster_t *cl, int reason) { - char pkt[CL_CTR_SMALL_PKT_SZ + 1]; /* +1 for the reason byte */ + /* +1 reason byte, +2 the master's MTU (diagnostic only - it is NOT cluster + * state the joiner adopts; it exists so the log line can name both + * numbers instead of leaving an operator to guess which side is wrong) */ + char pkt[CL_CTR_SMALL_PKT_SZ + 1 + CL_CTR_MTU_SZ]; uint32_t seq = htonl(++cl->peers->my_seq); int ip_len, plain_len; @@ -5623,12 +5792,19 @@ static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_clus pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + ip_len] = '\0'; /* reason byte follows the NUL-terminated target IP */ pkt[CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + ip_len + 1] = (char)reason; + /* then our MTU, so a rejected node can log what it should have been */ + { + uint16_t m = htons(cl_ctr_mtu_wire(cc_mtu)); + memcpy(pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + ip_len + 2, + &m, CL_CTR_MTU_SZ); + } - plain_len = CL_CTR_PLAIN_HDR_SZ + ip_len + 1 + 1; + plain_len = CL_CTR_PLAIN_HDR_SZ + ip_len + 1 + 1 + CL_CTR_MTU_SZ; if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->key, CL_CTR_PKT_JOIN_REJECT) == 0) LM_WARN("clusterer_controller: [cluster %d] sent JOIN_REJECT to %s (%s)\n", cl->cluster_id, target_ip, - reason == CL_CTR_REJECT_CONFIG ? "different cluster settings" + reason == CL_CTR_REJECT_CONFIG ? "different cluster settings" : + reason == CL_CTR_REJECT_MTU ? "different cluster-plane MTU" : "repeated auth failure - wrong password?"); } @@ -5645,7 +5821,18 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, lock_start_read(cl->peers->lock); still_new = (cl->peers->node_state == CL_CTR_NODE_NEW); lock_stop_read(cl->peers->lock); - if (!still_new) return; + /* cl->join_pending is set whenever we have an admission request in flight - + * including the split-brain merge, where cl_ctr_rejoin_superior_master() + * sends a JOIN_REQ from the ACTIVE state to fetch the superior master's + * key. Testing node_state alone missed exactly that case: on a + * simultaneous cold start BOTH nodes reach the join deadline, BOTH + * self-promote, and the loser then merges - so it is ACTIVE, not NEW, when + * its JOIN_REQ is refused, and it went on running with the reject + * discarded. A node that ASKED to be admitted must honour the refusal of + * that request whatever state it thinks it is in. A settled member with + * no request outstanding still ignores rejects, which is what stops one + * member from evicting another. */ + if (!still_new && !cl->join_pending) return; l = (int)strnlen(payload, CL_CTR_MAX_IP_LEN); if (l >= payload_len) return; @@ -5658,9 +5845,37 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, /* reason byte follows the NUL-terminated target IP (older senders omit it) */ { int reason = CL_CTR_REJECT_GENERIC; + int their_mtu = 0; if (payload_len > l + 1) reason = (unsigned char)payload[l + 1]; - if (reason == CL_CTR_REJECT_CONFIG) + if (payload_len >= l + 2 + CL_CTR_MTU_SZ) { + uint16_t m_be; + memcpy(&m_be, payload + l + 2, CL_CTR_MTU_SZ); + their_mtu = ntohs(m_be); + } + /* Only act if the refusal is self-consistent with what WE measure: the + * master says the cluster runs at X, and our own interface really is + * not X. A reject quoting our own MTU back at us is bogus and is + * ignored - we check the kernel, not the sender's assertion. */ + if (reason == CL_CTR_REJECT_MTU && their_mtu > 0 && their_mtu == cc_mtu) { + LM_WARN("clusterer_controller: [cluster %d] ignoring an MTU " + "JOIN_REJECT from %s that quotes %d - which is exactly this " + "node's own MTU, so the refusal contradicts itself\n", + cl->cluster_id, sender_ip, their_mtu); + return; + } + if (reason == CL_CTR_REJECT_MTU) + LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " + "cluster runs at MTU %d, this node has %d on %s. Every member " + "must agree, because an oversized datagram is dropped silently " + "by the smaller link. Change the MTU of %s to %d (and persist " + "it in the boot config) or move this node to a matching " + "segment; shutting down\n", + cl->cluster_id, sender_ip, their_mtu, cc_mtu, + my_interface_buf[0] ? my_interface_buf : "(unknown interface)", + my_interface_buf[0] ? my_interface_buf : "the cluster interface", + their_mtu); + else if (reason == CL_CTR_REJECT_CONFIG) LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " "running cluster has different settings than this node; fix the " "local config (manage_shtags/master_stickiness/query_time) to " From 953502579037aefafb1a384d0a4a4e5c5e31eefa Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 02:13:15 +1000 Subject: [PATCH 27/32] clusterer_controller: detect an MTU change under a running node Nothing notified the module when its link changed - the MTU was read once at mod_init and never revisited - so a node could go on believing it matched a cluster it no longer did. The ALIVE timer now re-reads it. Poll rather than netlink RTM_NEWLINK: netlink drops events on ENOBUFS, so a reconcile poll would have to exist anyway and a missed MTU change is exactly the silent failure this removes, and it fires on transients a poll harmlessly misses - a bond failover, a driver reset, a VLAN parent bounce. Acting on the first event would kill a healthy node over a blip, hence the strike count as well. One ioctl per query_time costs nothing. Own drift terminates the node rather than warning, because shrinking an MTU breaks RECEIVE, not send: the kernel still fragments on egress, so the node keeps emitting heartbeats and looks healthy to every peer while silently no longer receiving anything larger than its new MTU. MEMBER_LIST on a real cluster is well over 1500, so it would sit on stale membership and an incomplete BIN mesh, invisible to its peers, while the LB kept handing it calls. This is the node acting on its OWN reading, which is what makes acting safe at all - a switch-side change does not move /sys/class/net/X/mtu. Peer drift only warns, and must: terminating because someone else changed would turn one `ip link` command into a fleet outage. That peer detects its own change and removes itself. What is advertised is the live reading, not the value we joined at. Announcing the joined-at value instead made the peer warning unreachable - a node whose link changed kept announcing the old number, so no peer could ever observe the change, and the only account of it was in the log of the node that was about to disappear. (cherry picked from commit 5d7f0130257d6d98c7438db0b3e56231c08d1bfc) --- .../clusterer_controller.c | 89 ++++++++++++++++++- 1 file changed, 88 insertions(+), 1 deletion(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 3dfc9568bc0..04d9dcb4524 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -4424,10 +4424,12 @@ static void cl_ctr_handle_alive(const char *src_ip, const unsigned char *pubkey, /* may be NULL */ int cfg_present, int peer_manage, int peer_stick, int peer_qt, + int peer_mtu, cl_ctr_cluster_t *cl) { int prev_master, now_master; int warn = 0, adopt = 0, is_active = 0, ent = -1; + int mtu_warn = 0, mtu_seen = 0; char warn_ip[CL_CTR_MAX_IP_LEN + 1] = ""; int loc_manage = cl->manage_shtags ? 1 : 0; int loc_stick = cl->master_stickiness ? 1 : 0; @@ -4460,6 +4462,21 @@ static void cl_ctr_handle_alive(const char *src_ip, e->cfg_master_stickiness = peer_stick; e->cfg_query_time = peer_qt; } + /* A peer whose MTU no longer matches ours: WARN ONLY, never act. + * That peer's own drift poll will notice and remove itself, which + * is the node terminating on its OWN condition. Terminating here + * instead would mean one `ip link set mtu` on any single host + * taking down every other node that heard about it. */ + if (peer_mtu > 0) { + if (cc_mtu > 0 && peer_mtu != cc_mtu && !e->mtu_warned) { + e->mtu_warned = 1; + mtu_warn = 1; + mtu_seen = peer_mtu; + } else if (cc_mtu > 0 && peer_mtu == cc_mtu) { + e->mtu_warned = 0; /* re-arm once it agrees again */ + } + e->mtu = peer_mtu; + } break; } } @@ -4499,6 +4516,14 @@ static void cl_ctr_handle_alive(const char *src_ip, "values cause inconsistent failover/sharing-tag behaviour (%s)\n", cl->cluster_id, warn_ip, diff); } + if (mtu_warn) + LM_WARN("clusterer_controller: [cluster %d] peer %s now advertises MTU %d " + "but this cluster runs at %d. Traffic larger than the smaller of " + "the two is dropped silently in that direction. NOT acting on it " + "here - that peer detects its own change and removes itself; " + "terminating on someone else's reading would turn one `ip link` " + "command into a fleet outage\n", + cl->cluster_id, src_ip, mtu_seen, cc_mtu); if (adopt) cl_ctr_adopt_config(cl, peer_manage, peer_stick, peer_qt, is_active); /* Defer acting as master until we hold the cluster key. In normal @@ -6162,6 +6187,7 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, int ip_len = (int)strnlen(payload, CL_CTR_MAX_IP_LEN); const unsigned char *pubkey = NULL; int cfg_present = 0, p_manage = 0, p_stick = 0, p_qt = 0; + int p_mtu = 0; memcpy(ip_buf, payload, ip_len); ip_buf[ip_len] = '\0'; /* Pubkey appended after NUL-terminated IP */ @@ -6177,7 +6203,16 @@ static void cl_ctr_recv_one(int sock, cl_ctr_cluster_t *cl, p_qt = ntohs(qt_be); cfg_present = 1; } - cl_ctr_handle_alive(ip_buf, pubkey, cfg_present, p_manage, p_stick, p_qt, cl); + /* MTU appended after the config block (optional - older builds) */ + if (payload_len >= ip_len + 1 + (int)CL_CTR_PUBKEY_SZ + + CL_CTR_CONFIG_SZ + CL_CTR_MTU_SZ) { + uint16_t m_be; + memcpy(&m_be, payload + ip_len + 1 + CL_CTR_PUBKEY_SZ + + CL_CTR_CONFIG_SZ, CL_CTR_MTU_SZ); + p_mtu = ntohs(m_be); + } + cl_ctr_handle_alive(ip_buf, pubkey, cfg_present, p_manage, p_stick, p_qt, + p_mtu, cl); break; } @@ -6408,6 +6443,58 @@ static int cl_ctr_on_alive_tfd(int fd, void *param, int was_timeout) cl_ctr_cluster_t *cl = (cl_ctr_cluster_t *)param; int prev_master, now_master; cl_ctr_drain_tfd(fd); + + /* --- Has our OWN link changed under us? --------------------------------- + * Nothing notifies this module: cc_mtu is read once at mod_init. Poll it + * here rather than subscribing to netlink RTM_NEWLINK, because netlink + * DROPS events on ENOBUFS - so a reconcile poll would have to exist anyway, + * and a missed MTU change is exactly the silent failure this is meant to + * remove. One ioctl per query_time is free. + * + * Terminate rather than warn, because SHRINKING an MTU breaks RECEIVE, not + * send: the kernel still fragments on egress, so this node keeps emitting + * ALIVEs and looks perfectly healthy to everyone while silently no longer + * receiving anything larger than its new MTU. CL_CTR_LIST_PKT_MAX_SZ + * scales with CL_CTR_MAX_PEERS, so MEMBER_LIST on a real cluster is well + * over 1500 - the node would sit on stale membership and an incomplete BIN + * mesh, invisible to its peers, while the LB kept handing it calls. + * + * This is the node acting on its OWN condition, which is what makes it safe + * to act at all - it reads its own interface, and a switch-side change does + * not move /sys/class/net/X/mtu; only a host-local action does. */ + if (cc_mtu > 0) { + int now_mtu = cl_ctr_read_iface_mtu(); + if (now_mtu > 0) + cc_mtu_now = now_mtu; /* what we advertise: the live reading */ + if (now_mtu > 0 && now_mtu != cc_mtu) { + if (++cc_mtu_drift_strikes >= CL_CTR_MTU_DRIFT_STRIKES) { + LM_CRIT("clusterer_controller: [cluster %d] the MTU of %s changed " + "from %d to %d and stayed there for %d checks. This node " + "joined at %d and every other member still uses it, so it " + "can no longer RECEIVE full-size cluster traffic - it would " + "keep sending heartbeats and look healthy while silently " + "missing membership updates. Shutting down; restore the MTU " + "to %d and restart\n", + cl->cluster_id, + my_interface_buf[0] ? my_interface_buf : "(unknown)", + cc_mtu, now_mtu, cc_mtu_drift_strikes, cc_mtu, cc_mtu); + exit(-1); + } + LM_WARN("clusterer_controller: [cluster %d] MTU of %s reads %d, this " + "node joined at %d (%d/%d consecutive) - confirming before " + "acting, since a bond failover or driver reset can report a " + "transient value\n", + cl->cluster_id, + my_interface_buf[0] ? my_interface_buf : "(unknown)", + now_mtu, cc_mtu, cc_mtu_drift_strikes, + CL_CTR_MTU_DRIFT_STRIKES); + } else if (cc_mtu_drift_strikes) { + LM_INFO("clusterer_controller: [cluster %d] MTU of %s is back to %d - " + "the earlier reading was a transient\n", cl->cluster_id, + my_interface_buf[0] ? my_interface_buf : "(unknown)", cc_mtu); + cc_mtu_drift_strikes = 0; + } + } /* ALIVE transport: a settled non-master unicasts its heartbeat to the master, * which relays liveness to the whole group via the MASTER_ALIVE bitmap - so * the old all-to-all O(N^2) becomes O(N). During formation (no settled master From 3bf3cfe8c22aea53d86641a55d221300ba8f0d06 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 12:55:35 +1000 Subject: [PATCH 28/32] clusterer_controller: a failed MTU read is not a recovery The drift poll had two outcomes where it needed three. cl_ctr_read_iface_mtu() returns -1 when the ioctl fails, and that fell into the same branch as "the link agrees again", which did two wrong things at once. It cleared the strike count, so a read failing at roughly the confirmation cadence could hold a genuinely drifted node alive indefinitely - the exact silent failure the poll exists to remove. And it logged that the MTU was "back to" a value it had never read: a measurement that did not happen, which is worse than no line at all, because an operator reading it concludes the link recovered. A failed read is neither drift nor recovery. It is an absence of evidence, so it now leaves the strike count exactly where it stood and says so - on the first failure and rarely after, since a poll that has gone blind must not be able to look like a quiet healthy one. It is deliberately not fatal, even though mod_init refuses to START without an MTU. The asymmetry is the point: at startup the node holds no membership and loses nothing by refusing, while here it is an established member carrying calls, and an ioctl we could not complete is evidence about our own visibility rather than about the link. The drift this guards against - a host-local `ip link` change - leaves the interface perfectly readable; a read that fails means it was renamed or removed, which announces itself far more loudly elsewhere. The decision moves into cl_ctr_mtu_step() so it can be exercised directly: the sequences that matter (a failure landing mid-drift, a true transient, a drift that persists) are awkward to stage against a real interface and trivial to enumerate. /dn/task97 extracts the function straight from this file and runs both the old and new logic over the same inputs - the old one never confirms a drift once a failed read interrupts it, including when failures alternate with drift readings forever. (cherry picked from commit 984e55502008f08759a11af58769b53ae4d3babc) --- .../clusterer_controller.c | 135 ++++++++++++++---- 1 file changed, 111 insertions(+), 24 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index 04d9dcb4524..bc1336e01e7 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -502,6 +502,10 @@ typedef struct { /* Distinct peers a master may refuse on MTU before it says out loud that IT is * the likely misconfiguration. See cl_ctr_mtu_note_reject(). */ #define CL_CTR_MTU_SUSPECT_PEERS 2 +/* Consecutive unreadable MTU polls between complaints. The first failure is + * always reported; after that this keeps a blind poll from being silent + * without letting it flood. At query_time=5 this is about once a minute. */ +#define CL_CTR_MTU_READ_FAIL_LOUD 12 /* Noise handshake message sizes (NNpsk0, X25519, ChaChaPoly, SHA-256): * msg 1 = e(32) + tag over empty payload(16) = 48 * msg 2 = e(32) + AEAD(master_salt 32 + tag 16) = 80 */ @@ -862,6 +866,10 @@ static int cc_mtu = 0; static int cc_mtu_now = 0; /* Consecutive polls that disagreed with cc_mtu (see CL_CTR_MTU_DRIFT_STRIKES). */ static int cc_mtu_drift_strikes = 0; +/* Consecutive polls that could not read the MTU at all. Tracked separately + * from the drift strikes because a failed read is not evidence of anything - + * see cl_ctr_mtu_step(). */ +static int cc_mtu_read_fails = 0; /** * cl_ctr_read_iface_mtu() - current MTU of the interface we run the plane on. @@ -898,6 +906,50 @@ static inline uint16_t cl_ctr_mtu_wire(int mtu) return (uint16_t)mtu; } +/* Outcome of a single MTU poll. */ +#define CL_CTR_MTU_STEP_STEADY 0 /* reads what we joined at */ +#define CL_CTR_MTU_STEP_UNREADABLE 1 /* the read failed - nothing learned */ +#define CL_CTR_MTU_STEP_DRIFTING 2 /* differs, not yet confirmed */ +#define CL_CTR_MTU_STEP_CONFIRMED 3 /* differs, confirmed - stand down */ +#define CL_CTR_MTU_STEP_RECOVERED 4 /* agrees again after >=1 strike */ + +/** + * cl_ctr_mtu_step() - decide what one MTU reading means. + * + * Split out of the timer callback so the decision can be exercised directly: + * the interesting sequences (a failed read landing in the middle of a drift, a + * transient that recovers, a drift that persists) are awkward to produce + * against a real interface and trivial to enumerate here. + * + * THE FAILED READ IS THE WHOLE POINT. It used to fall into the same branch as + * "the link is fine again", which did two wrong things at once: it cleared the + * strike count, so an ioctl failing at roughly the confirmation cadence could + * hold a genuinely drifted node alive indefinitely; and it logged that the MTU + * was "back to" a value it had never read - a measurement that did not happen, + * which is the one kind of log line worse than no log line at all. + * + * A failed read is neither drift nor recovery. It is an absence of evidence, + * so it leaves the strike count exactly where it was. + */ +static int cl_ctr_mtu_step(int now_mtu, int joined_mtu, int *strikes, int *fails) +{ + if (now_mtu <= 0) { + (*fails)++; + return CL_CTR_MTU_STEP_UNREADABLE; + } + *fails = 0; + if (now_mtu == joined_mtu) { + if (*strikes) { + *strikes = 0; + return CL_CTR_MTU_STEP_RECOVERED; + } + return CL_CTR_MTU_STEP_STEADY; + } + if (++(*strikes) >= CL_CTR_MTU_DRIFT_STRIKES) + return CL_CTR_MTU_STEP_CONFIRMED; + return CL_CTR_MTU_STEP_DRIFTING; +} + /* Largest packet the retransmit cache will hold, raised in mod_init once the * MTU is known. Provisional value covers the control plane, which is all that * can be sent before then anyway. */ @@ -6464,35 +6516,70 @@ static int cl_ctr_on_alive_tfd(int fd, void *param, int was_timeout) * not move /sys/class/net/X/mtu; only a host-local action does. */ if (cc_mtu > 0) { int now_mtu = cl_ctr_read_iface_mtu(); + const char *ifn = my_interface_buf[0] ? my_interface_buf : "(unknown)"; + + /* Advertise only a reading we actually got. A failed read must not + * push a zero out to peers as though the link had changed. */ if (now_mtu > 0) - cc_mtu_now = now_mtu; /* what we advertise: the live reading */ - if (now_mtu > 0 && now_mtu != cc_mtu) { - if (++cc_mtu_drift_strikes >= CL_CTR_MTU_DRIFT_STRIKES) { - LM_CRIT("clusterer_controller: [cluster %d] the MTU of %s changed " - "from %d to %d and stayed there for %d checks. This node " - "joined at %d and every other member still uses it, so it " - "can no longer RECEIVE full-size cluster traffic - it would " - "keep sending heartbeats and look healthy while silently " - "missing membership updates. Shutting down; restore the MTU " - "to %d and restart\n", - cl->cluster_id, - my_interface_buf[0] ? my_interface_buf : "(unknown)", - cc_mtu, now_mtu, cc_mtu_drift_strikes, cc_mtu, cc_mtu); - exit(-1); - } + cc_mtu_now = now_mtu; + + switch (cl_ctr_mtu_step(now_mtu, cc_mtu, &cc_mtu_drift_strikes, + &cc_mtu_read_fails)) { + case CL_CTR_MTU_STEP_CONFIRMED: + LM_CRIT("clusterer_controller: [cluster %d] the MTU of %s changed " + "from %d to %d and stayed there for %d checks. This node " + "joined at %d and every other member still uses it, so it " + "can no longer RECEIVE full-size cluster traffic - it would " + "keep sending heartbeats and look healthy while silently " + "missing membership updates. Shutting down; restore the MTU " + "to %d and restart\n", + cl->cluster_id, ifn, cc_mtu, now_mtu, + cc_mtu_drift_strikes, cc_mtu, cc_mtu); + exit(-1); + case CL_CTR_MTU_STEP_DRIFTING: LM_WARN("clusterer_controller: [cluster %d] MTU of %s reads %d, this " "node joined at %d (%d/%d consecutive) - confirming before " "acting, since a bond failover or driver reset can report a " "transient value\n", - cl->cluster_id, - my_interface_buf[0] ? my_interface_buf : "(unknown)", - now_mtu, cc_mtu, cc_mtu_drift_strikes, - CL_CTR_MTU_DRIFT_STRIKES); - } else if (cc_mtu_drift_strikes) { - LM_INFO("clusterer_controller: [cluster %d] MTU of %s is back to %d - " - "the earlier reading was a transient\n", cl->cluster_id, - my_interface_buf[0] ? my_interface_buf : "(unknown)", cc_mtu); - cc_mtu_drift_strikes = 0; + cl->cluster_id, ifn, now_mtu, cc_mtu, + cc_mtu_drift_strikes, CL_CTR_MTU_DRIFT_STRIKES); + break; + case CL_CTR_MTU_STEP_RECOVERED: + LM_INFO("clusterer_controller: [cluster %d] MTU of %s reads %d again, " + "matching what this node joined at - the earlier readings " + "were a transient\n", cl->cluster_id, ifn, cc_mtu); + break; + case CL_CTR_MTU_STEP_UNREADABLE: + /* Deliberately NOT fatal, and deliberately not a reset. + * + * Not a reset because a read that failed is not a link that + * recovered - the strike count is left exactly where it was, so a + * drift already under way still confirms on the next real reading. + * + * Not fatal, even though mod_init refuses to START without an MTU, + * and the asymmetry is intentional: at startup the node has no + * membership and nothing to lose by refusing, while here it is an + * established member carrying calls, and an ioctl we could not + * complete is evidence about our own visibility, not about the + * link. The drift this poll exists to catch - a host-local `ip + * link` change - leaves the interface perfectly readable. A reading + * that fails instead means the interface was renamed or removed, + * which announces itself far more loudly elsewhere. + * + * Say so on the first failure and then rarely, so a persistently + * blind poll cannot masquerade as a quiet healthy one. */ + if (cc_mtu_read_fails == 1 + || cc_mtu_read_fails % CL_CTR_MTU_READ_FAIL_LOUD == 0) + LM_WARN("clusterer_controller: [cluster %d] cannot read the MTU " + "of %s (%s) - %d consecutive attempts. This node is not " + "verifying that it still matches the cluster's %d; it " + "keeps running on its last good reading, and any drift " + "already being confirmed is NOT cancelled by this\n", + cl->cluster_id, ifn, strerror(errno), + cc_mtu_read_fails, cc_mtu); + break; + default: + break; /* steady - the common case, silent */ } } /* ALIVE transport: a settled non-master unicasts its heartbeat to the master, From 17efcf64efe964d2dbeb32cfb505eccf7c12cb18 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 12:55:35 +1000 Subject: [PATCH 29/32] clusterer_controller: say what the DF bit actually does The fragmentation note claimed "the DF bit is not set so fragmentation occurs transparently where the network allows it". That was never measured and is false. The socket sets no IP_MTU_DISCOVER at all, so Linux's default applies - IP_PMTUDISC_WANT - and that default DOES set DF, on unicast and on multicast alike, confirmed on the wire at 2000, 4000 and 8900 bytes. What actually carries a 4395-byte MEMBER_LIST across a 1500-byte link is that it EXCEEDS THE LOCAL MTU: a datagram the kernel must fragment locally cannot also be doing path-MTU discovery, so those fragments leave with DF clear. The old wording described the right outcome by the wrong mechanism, and the mechanism matters - a datagram that FITS the local MTU goes out with DF SET, so meeting a smaller-MTU hop it is dropped with an ICMP "fragmentation needed" that this topology routinely filters, rather than being fragmented and delivered. The behaviour is left exactly as it is. Clearing DF would let a jumbo node reach a peer across a narrow routed hop, but this module's contract is the interface it runs on, not how the operator carries multicast between segments: enforcing one MTU across the cluster is what makes the local bound mean anything, and moving packets between segments is the network's job. (cherry picked from commit 3c65e429ef88e6613f52ee0008c9f23ecef6225e) --- .../clusterer_controller.c | 23 +++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index bc1336e01e7..e186c391aac 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -469,8 +469,27 @@ typedef struct { * - IPsec ESP tunnel (~1400 MTU): 4 fragments * - GRE-over-IPsec (~1350 MTU): 4-5 fragments * Firewalls that block fragmented UDP packets will silently drop MEMBER_LIST, - * preventing new nodes from joining. The DF bit is not set so fragmentation - * occurs transparently where the network allows it. */ + * preventing new nodes from joining. + * + * WHY THAT FRAGMENTATION IS ALLOWED TO HAPPEN, precisely - an earlier version + * of this comment said "the DF bit is not set", which is false and was never + * measured. The socket sets no IP_MTU_DISCOVER at all, so Linux's default + * applies (IP_PMTUDISC_WANT, per-route), and that default DOES set DF - on + * unicast AND on multicast alike, verified on the wire. What actually carries + * MEMBER_LIST across a 1500-byte link is that 4395 bytes EXCEEDS THE LOCAL MTU, + * and a datagram the kernel must fragment locally cannot also be doing path-MTU + * discovery, so those fragments leave with DF clear. The old wording happened + * to describe the right outcome for the wrong reason, and the reason matters: + * a datagram that FITS the local MTU goes out with DF SET, so if it later meets + * a smaller-MTU hop it is dropped with an ICMP "fragmentation needed" that is + * routinely filtered - it is not fragmented and does not arrive. + * + * That is left exactly as it is, deliberately. Clearing DF would let a jumbo + * node reach a peer across a narrow ROUTED hop, but this module's contract is + * about the interface it runs on, not about how the operator has chosen to + * carry multicast between segments. Enforcing a uniform MTU across the cluster + * is what makes the local bound meaningful; getting the packets between + * segments is the network's job. */ #define CL_CTR_SMALL_PKT_SZ (CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + CL_CTR_MAX_IP_LEN + 1 + CL_CTR_TAG_SZ) /* Consistency-critical settings advertised in ALIVE so peers can detect * accidental per-node config drift for the same cluster: From eeb553bcfe7c4f02b4b96628209097f7a72294ff Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Thu, 13 Aug 2026 20:50:34 +1000 Subject: [PATCH 30/32] clusterer_controller: document the uniform-MTU rule, the join_pending guard and what DF actually does --- modules/clusterer_controller/README | 90 +++++++++++++++-- .../doc/clusterer_controller_admin.xml | 97 +++++++++++++++++-- 2 files changed, 171 insertions(+), 16 deletions(-) diff --git a/modules/clusterer_controller/README b/modules/clusterer_controller/README index a0fe9651bdc..e15d782b4b9 100644 --- a/modules/clusterer_controller/README +++ b/modules/clusterer_controller/README @@ -389,11 +389,71 @@ on-key") JOIN_REJECT packet and stops responding to further JOIN_REQs from that IP. - On the joining side, a received JOIN_REJECT is only acted on - while the node is still in the initial join phase (CL_CTR_NODE_NEW - state) and is addressed to this node; an already-active cluster - member ignores any JOIN_REJECT unconditionally, so a node with - the correct password can never be evicted by a peer. + On the joining side, a received JOIN_REJECT is acted on only if + it is addressed to this node and this node currently has an + admission request outstanding (join_pending). A member that is + not asking to be admitted ignores JOIN_REJECT unconditionally, + so a node with the correct password can never be evicted by a + peer. + + The guard is deliberately the pending request and not the + CL_CTR_NODE_NEW state. After a simultaneous cold start both + nodes can reach the join deadline and self-promote, and the + loser then merges into the winner by sending a fresh JOIN_REQ + from the active state. Keying on the state label would have + that node discard the refusal of a join it had just made, which + is precisely the case the MTU and config checks below exist for. + + Uniform cluster MTU: every member of a cluster runs at the same + cluster-plane MTU, and the master enforces it at admission. The + value is detected from the interface that owns the cluster-plane + IP (SIOCGIFMTU) and there is deliberately no modparam for it: + the kernel already owns that fact, and a configured copy could + only ever disagree with it. A node that cannot read its own MTU + refuses to start. + + Each node advertises its current reading in JOIN_REQ and ALIVE. + A master that receives a JOIN_REQ whose MTU differs from its own + answers with a JOIN_REJECT carrying reason CL_CTR_REJECT_MTU, + and the joiner stands down. This is not covered by + on_config_mismatch and is never optional - that parameter + governs genuinely configured settings, whereas a mismatched MTU + means the two nodes cannot exchange full-size packets at all. (A + node old enough not to advertise an MTU is admitted unchanged.) + + The rule needs nothing on the wire beyond that check, by + induction: the master admits only equal-MTU nodes, so every + member carries the master's MTU, so the backup that is promoted + on handover already carries it too. A cluster's MTU therefore + cannot change through an election, and there is no inherited MTU + field in MEMBER_LIST. + + After joining, each node polls its own interface. What it joined + at is what it is judged against; what the kernel says now is + what it advertises - the two are tracked separately, or a node + whose link changed would keep announcing its old value and no + peer could ever see the difference. CL_CTR_MTU_DRIFT_STRIKES (3) + consecutive disagreeing readings confirm a local change, and the + node then logs a critical message and shuts itself down: it can + no longer receive full-size cluster traffic, and it would + otherwise keep sending heartbeats and look healthy while + silently missing membership updates. A single odd reading, or a + reading that agrees again, clears the strikes. A failed + SIOCGIFMTU is not a strike, is not advertised, and does not + reset one - it is reported on the first failure and then every + CL_CTR_MTU_READ_FAIL_LOUD (12) failures, so a persistently blind + poll cannot pass for a quiet healthy one. + + A node never acts on a peer's reading. Seeing a peer advertise a + different MTU produces a warning only - that peer detects its + own change and removes itself, and terminating on someone else's + reading would turn one `ip link` command into a fleet outage. + One case does need a louder signal: master election is by + highest IP and ignores the MTU, so a single wrongly-configured + host that wins it will refuse an otherwise healthy fleet. A + master that has refused CL_CTR_MTU_SUSPECT_PEERS (2) distinct + peers while holding no members of its own logs a critical + message naming itself as the probable misconfiguration. A node joining with the wrong password cannot decrypt the JOIN_REJECT (it is encrypted with the master's bootstrap key), @@ -1500,9 +1560,23 @@ modparam("dispatcher", "cluster_probing_mode", "distributed") + GRE-over-IPsec (~1350 MTU): 4–5 fragments All other packet types (ALIVE, JOIN_REQ, KEY_GRANT, GOODBYE, etc.) fit comfortably within a single datagram on - any of these links. The DF bit is not set, so IP - fragmentation occurs transparently where the network allows - it. However, firewalls or stateless middleboxes that + any of these links. Because every member runs at the same + MTU (see the uniform-MTU rule above), the fragment count is + the same on every node, and a host on a narrower link is + refused admission rather than left fragmenting differently + from the rest. The socket sets no IP_MTU_DISCOVER option, so + Linux's default (IP_PMTUDISC_WANT) applies and DF IS set - + on multicast as well as unicast, verified on the wire. What + lets MEMBER_LIST through is that it exceeds the LOCAL MTU: a + datagram the kernel must fragment locally is not also doing + path-MTU discovery, so those fragments leave with DF clear. + A datagram that fits the local MTU goes out with DF set, and + if it later meets a narrower routed hop it is dropped with + an ICMP "fragmentation needed" that is routinely filtered. + This is left as it is deliberately: the module's contract is + the interface it runs on, not how the operator chose to + carry multicast between segments. However, firewalls or + stateless middleboxes that silently drop fragmented UDP will prevent new nodes from joining, since MEMBER_LIST is required to complete the join sequence. Verify that fragmented UDP is permitted on all diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml index 43c38fa1da4..5d8556da2e6 100644 --- a/modules/clusterer_controller/doc/clusterer_controller_admin.xml +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -360,12 +360,77 @@ and stops responding to further JOIN_REQs from that IP. - On the joining side, a received JOIN_REJECT is only acted on while - the node is still in the initial join phase - (CL_CTR_NODE_NEW state) and is addressed to this - node; an already-active cluster member ignores any JOIN_REJECT - unconditionally, so a node with the correct password can never be - evicted by a peer. + On the joining side, a received JOIN_REJECT is acted on only if it is + addressed to this node and this node currently has + an admission request outstanding + (join_pending). A member that is not asking to be + admitted ignores JOIN_REJECT unconditionally, so a node with the correct + password can never be evicted by a peer. + + + The guard is deliberately the pending request and not the + CL_CTR_NODE_NEW state. After a simultaneous cold + start both nodes can reach the join deadline and self-promote, and the + loser then merges into the winner by sending a fresh JOIN_REQ + from the active state. Keying on the state label + would have that node discard the refusal of a join it had just made, + which is precisely the case the MTU and config checks below exist for. + + + Uniform cluster MTU: + every member of a cluster runs at the same cluster-plane MTU, and the + master enforces it at admission. The value is detected + from the interface that owns the cluster-plane IP + (SIOCGIFMTU) and there is deliberately no modparam + for it: the kernel already owns that fact, and a configured copy could + only ever disagree with it. A node that cannot read its own MTU refuses + to start. + + + Each node advertises its current reading in JOIN_REQ + and ALIVE. A master that receives a JOIN_REQ whose MTU differs from its + own answers with a JOIN_REJECT carrying reason + CL_CTR_REJECT_MTU, and the joiner stands down. This + is not covered by on_config_mismatch and is never + optional - that parameter governs genuinely configured settings, whereas + a mismatched MTU means the two nodes cannot exchange full-size packets at + all. (A node old enough not to advertise an MTU is admitted unchanged.) + + + The rule needs nothing on the wire beyond that check, by induction: the + master admits only equal-MTU nodes, so every member carries the master's + MTU, so the backup that is promoted on handover already carries it too. A + cluster's MTU therefore cannot change through an election, and there is + no inherited MTU field in MEMBER_LIST. + + + After joining, each node polls its own interface. What it joined at is + what it is judged against; what the kernel says now is what it + advertises - the two are tracked separately, or a node whose link changed + would keep announcing its old value and no peer could ever see the + difference. CL_CTR_MTU_DRIFT_STRIKES (3) consecutive + disagreeing readings confirm a local change, and the node then logs a + critical message and shuts itself down: it can no longer receive + full-size cluster traffic, and it would otherwise keep sending + heartbeats and look healthy while silently missing membership updates. A + single odd reading, or a reading that agrees again, clears the strikes. A + failed SIOCGIFMTU is not a strike, is not + advertised, and does not reset one - it is reported on the first failure + and then every CL_CTR_MTU_READ_FAIL_LOUD (12) + failures, so a persistently blind poll cannot pass for a quiet healthy + one. + + + A node never acts on a peer's reading. Seeing a peer + advertise a different MTU produces a warning only - that peer detects its + own change and removes itself, and terminating on someone else's reading + would turn one ip link command into a fleet outage. + One case does need a louder signal: master election is by highest IP and + ignores the MTU, so a single wrongly-configured host that wins it will + refuse an otherwise healthy fleet. A master that has refused + CL_CTR_MTU_SUSPECT_PEERS (2) distinct peers while + holding no members of its own logs a critical message naming + itself as the probable misconfiguration. A node joining with the wrong password cannot @@ -2105,8 +2170,24 @@ modparam("dispatcher", "cluster_probing_mode", "distributed") All other packet types (ALIVE, JOIN_REQ, KEY_GRANT, GOODBYE, etc.) fit comfortably within a single datagram on any of - these links. The DF bit is not set, so IP fragmentation - occurs transparently where the network allows it. However, + these links. Because every member runs at the same MTU (see + the uniform-MTU rule above), the fragment count is the same + on every node, and a host on a narrower link is refused + admission rather than left fragmenting differently from the + rest. The socket sets no + IP_MTU_DISCOVER option, so Linux's + default (IP_PMTUDISC_WANT) applies and + DF is set - on multicast as well as + unicast, verified on the wire. What lets MEMBER_LIST + through is that it exceeds the LOCAL MTU: a datagram the + kernel must fragment locally is not also doing path-MTU + discovery, so those fragments leave with DF clear. A + datagram that fits the local MTU goes out with DF set, and + if it later meets a narrower routed hop it is dropped with + an ICMP "fragmentation needed" that is routinely filtered. + This is left as it is deliberately: the module's contract is + the interface it runs on, not how the operator chose to + carry multicast between segments. However, firewalls or stateless middleboxes that silently drop fragmented UDP will prevent new nodes from joining, since MEMBER_LIST is required to complete the join sequence. From 2f23e3d090945fa6352539e1504281a9883da701 Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Fri, 14 Aug 2026 11:50:11 +1000 Subject: [PATCH 31/32] clusterer_controller: carry the master's config in the JOIN_REJECT, and refuse a refusal that contradicts itself --- modules/clusterer_controller/README | 19 +++++ .../clusterer_controller.c | 82 ++++++++++++++++--- .../doc/clusterer_controller_admin.xml | 21 +++++ 3 files changed, 111 insertions(+), 11 deletions(-) diff --git a/modules/clusterer_controller/README b/modules/clusterer_controller/README index e15d782b4b9..3676142a269 100644 --- a/modules/clusterer_controller/README +++ b/modules/clusterer_controller/README @@ -404,6 +404,25 @@ on-key") that node discard the refusal of a join it had just made, which is precisely the case the MTU and config checks below exist for. + What a JOIN_REJECT carries, and why. Beyond the target IP and a + reason byte, the refusal carries the master's own cluster-plane + MTU and its consistency-critical settings (manage_shtags, + master_stickiness, query_time), encoded exactly as JOIN_REQ + carries the joiner's. Both are diagnostic, not state the joiner + adopts, and they buy two things. First, the node that is about + to stop can name both sides in its own log - it is the node an + operator looks at first, and "the settings differ" without + values sends them off to read two config files by hand. Second, + the joiner can check the refusal against what it holds and + refuse a refusal that contradicts itself: a reject quoting this + node's own MTU, or its own config triple, is either a stale + packet or a peer asserting something untrue, since a real master + sends those reasons only when the values differ. Without that + check any holder of the cluster password could end a merging + master's process by assertion alone. Older senders omit both + fields; the parse is by length, and a reject without them is + still honoured, just with the generic message. + Uniform cluster MTU: every member of a cluster runs at the same cluster-plane MTU, and the master enforces it at admission. The value is detected from the interface that owns the cluster-plane diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index e186c391aac..bfdd8e596c3 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -5872,10 +5872,15 @@ static int cl_ctr_join_fail_check(const char *src_ip, cl_ctr_cluster_t *cl) static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_cluster_t *cl, int reason) { - /* +1 reason byte, +2 the master's MTU (diagnostic only - it is NOT cluster - * state the joiner adopts; it exists so the log line can name both - * numbers instead of leaving an operator to guess which side is wrong) */ - char pkt[CL_CTR_SMALL_PKT_SZ + 1 + CL_CTR_MTU_SZ]; + /* +1 reason byte, +2 the master's MTU, +CL_CTR_CONFIG_SZ the master's + * config triple. All three are diagnostic only - NOT cluster state the + * joiner adopts; they exist so the refused node's own log can name both + * sides instead of leaving an operator to guess which is wrong, and so it + * can refuse a refusal that contradicts itself (see + * cl_ctr_handle_join_reject). The MTU came first; the config triple is + * the same idea applied to the reason an operator actually meets. */ + char pkt[CL_CTR_SMALL_PKT_SZ + 1 + CL_CTR_MTU_SZ + + CL_CTR_CONFIG_SZ]; uint32_t seq = htonl(++cl->peers->my_seq); int ip_len, plain_len; @@ -5895,7 +5900,18 @@ static void cl_ctr_send_join_reject(int sock, const char *target_ip, cl_ctr_clus &m, CL_CTR_MTU_SZ); } - plain_len = CL_CTR_PLAIN_HDR_SZ + ip_len + 1 + 1 + CL_CTR_MTU_SZ; + /* then our config triple, encoded exactly as JOIN_REQ carries the + * joiner's: [manage 1B][stickiness 1B][query_time 2B BE] */ + { + char *c = pkt + CL_CTR_WIRE_HDR_SZ + CL_CTR_PLAIN_HDR_SZ + + ip_len + 2 + CL_CTR_MTU_SZ; + uint16_t qt = htons((uint16_t)(query_time & 0xFFFF)); + c[0] = (char)(cl->manage_shtags ? 1 : 0); + c[1] = (char)(cl->master_stickiness ? 1 : 0); + memcpy(c + 2, &qt, 2); + } + plain_len = CL_CTR_PLAIN_HDR_SZ + ip_len + 1 + 1 + CL_CTR_MTU_SZ + + CL_CTR_CONFIG_SZ; if (cl_ctr_seal_and_send(sock, cl, pkt, plain_len, cl->key, CL_CTR_PKT_JOIN_REJECT) == 0) LM_WARN("clusterer_controller: [cluster %d] sent JOIN_REJECT to %s (%s)\n", cl->cluster_id, target_ip, @@ -5942,6 +5958,8 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, { int reason = CL_CTR_REJECT_GENERIC; int their_mtu = 0; + /* the master's config triple, absent on older senders */ + int cfg_present = 0, r_manage = 0, r_stick = 0, r_qt = 0; if (payload_len > l + 1) reason = (unsigned char)payload[l + 1]; if (payload_len >= l + 2 + CL_CTR_MTU_SZ) { @@ -5949,6 +5967,15 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, memcpy(&m_be, payload + l + 2, CL_CTR_MTU_SZ); their_mtu = ntohs(m_be); } + if (payload_len >= l + 2 + CL_CTR_MTU_SZ + CL_CTR_CONFIG_SZ) { + const char *c = payload + l + 2 + CL_CTR_MTU_SZ; + uint16_t qt_be; + r_manage = (unsigned char)c[0]; + r_stick = (unsigned char)c[1]; + memcpy(&qt_be, c + 2, 2); + r_qt = ntohs(qt_be); + cfg_present = 1; + } /* Only act if the refusal is self-consistent with what WE measure: the * master says the cluster runs at X, and our own interface really is * not X. A reject quoting our own MTU back at us is bogus and is @@ -5960,6 +5987,23 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, cl->cluster_id, sender_ip, their_mtu); return; } + /* Same self-consistency rule the MTU branch uses, for the same + * reason: act on a refusal only if it disagrees with what WE hold. A + * CONFIG reject quoting our own triple back at us is bogus - a real + * master only sends this reason when the values DIFFER - so it is + * either a stale packet or a peer asserting something untrue, and + * acting on it would let any PSK holder end a merging master. */ + if (reason == CL_CTR_REJECT_CONFIG && cfg_present + && r_manage == (cl->manage_shtags ? 1 : 0) + && r_stick == (cl->master_stickiness ? 1 : 0) + && r_qt == (query_time & 0xFFFF)) { + LM_WARN("clusterer_controller: [cluster %d] ignoring a CONFIG " + "JOIN_REJECT from %s that quotes manage_shtags=%d " + "master_stickiness=%d query_time=%d - which is exactly what " + "this node runs, so the refusal contradicts itself\n", + cl->cluster_id, sender_ip, r_manage, r_stick, r_qt); + return; + } if (reason == CL_CTR_REJECT_MTU) LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " "cluster runs at MTU %d, this node has %d on %s. Every member " @@ -5971,12 +6015,28 @@ static void cl_ctr_handle_join_reject(const char *payload, int payload_len, my_interface_buf[0] ? my_interface_buf : "(unknown interface)", my_interface_buf[0] ? my_interface_buf : "the cluster interface", their_mtu); - else if (reason == CL_CTR_REJECT_CONFIG) - LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " - "running cluster has different settings than this node; fix the " - "local config (manage_shtags/master_stickiness/query_time) to " - "match and restart; shutting down\n", - cl->cluster_id, sender_ip); + else if (reason == CL_CTR_REJECT_CONFIG) { + /* Name the values when the master sent them. The node that dies is + * the one an operator looks at first, and "settings differ" without + * numbers sends them to read two config files by hand. */ + char diff[160]; + if (cfg_present) { + cl_ctr_fmt_cfg_diff(diff, sizeof(diff), "cluster", "this node", + r_manage, cl->manage_shtags ? 1 : 0, + r_stick, cl->master_stickiness ? 1 : 0, + r_qt, query_time & 0xFFFF); + LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - " + "the running cluster has different settings than this node " + "(%s); fix the local config to match and restart; shutting " + "down\n", cl->cluster_id, sender_ip, diff); + } else { + LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - the " + "running cluster has different settings than this node; fix the " + "local config (manage_shtags/master_stickiness/query_time) to " + "match and restart; shutting down\n", + cl->cluster_id, sender_ip); + } + } else LM_CRIT("clusterer_controller: [cluster %d] JOIN_REJECT from %s - " "wrong password or unauthorized node; shutting down\n", diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml index 5d8556da2e6..00c99e6391d 100644 --- a/modules/clusterer_controller/doc/clusterer_controller_admin.xml +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -376,6 +376,27 @@ would have that node discard the refusal of a join it had just made, which is precisely the case the MTU and config checks below exist for. + + What a JOIN_REJECT carries, and why. + Beyond the target IP and a reason byte, the refusal carries the + master's own cluster-plane MTU and its + consistency-critical settings + (manage_shtags, master_stickiness, + query_time), encoded exactly as JOIN_REQ carries the + joiner's. Both are diagnostic, not state the joiner adopts, and they buy + two things. First, the node that is about to stop can name + both sides in its own log - it is the node an + operator looks at first, and "the settings differ" without values sends + them off to read two config files by hand. Second, the joiner can check + the refusal against what it holds and refuse a refusal that + contradicts itself: a reject quoting this node's own MTU, or + its own config triple, is either a stale packet or a peer asserting + something untrue, since a real master sends those reasons only when the + values differ. Without that check any holder of the cluster password + could end a merging master's process by assertion alone. Older senders + omit both fields; the parse is by length, and a reject without them is + still honoured, just with the generic message. + Uniform cluster MTU: every member of a cluster runs at the same cluster-plane MTU, and the From b11a3ea950922677c914595f5e77de3b86b273dc Mon Sep 17 00:00:00 2001 From: Yury Kirsanov Date: Fri, 14 Aug 2026 12:10:13 +1000 Subject: [PATCH 32/32] clusterer_controller: warn when THIS master is the config outlier, mirroring the MTU case --- modules/clusterer_controller/README | 8 ++- .../clusterer_controller.c | 59 +++++++++++++++++++ .../doc/clusterer_controller_admin.xml | 8 ++- 3 files changed, 73 insertions(+), 2 deletions(-) diff --git a/modules/clusterer_controller/README b/modules/clusterer_controller/README index 3676142a269..32b17572cfb 100644 --- a/modules/clusterer_controller/README +++ b/modules/clusterer_controller/README @@ -472,7 +472,13 @@ on-key") host that wins it will refuse an otherwise healthy fleet. A master that has refused CL_CTR_MTU_SUSPECT_PEERS (2) distinct peers while holding no members of its own logs a critical - message naming itself as the probable misconfiguration. + message naming itself as the probable misconfiguration. The + same safety net exists for the CONFIG gate + (CL_CTR_CFG_SUSPECT_PEERS, also 2): election ignores + manage_shtags / master_stickiness / query_time exactly as it + ignores the MTU, so a master that has refused two distinct + peers on settings while holding no members names itself the + probable culprit too. A node joining with the wrong password cannot decrypt the JOIN_REJECT (it is encrypted with the master's bootstrap key), diff --git a/modules/clusterer_controller/clusterer_controller.c b/modules/clusterer_controller/clusterer_controller.c index bfdd8e596c3..adc88909be5 100644 --- a/modules/clusterer_controller/clusterer_controller.c +++ b/modules/clusterer_controller/clusterer_controller.c @@ -521,6 +521,11 @@ typedef struct { /* Distinct peers a master may refuse on MTU before it says out loud that IT is * the likely misconfiguration. See cl_ctr_mtu_note_reject(). */ #define CL_CTR_MTU_SUSPECT_PEERS 2 +/* Same idea for the config gate: a master refusing this many DIFFERENT peers on + * settings while holding no members is probably the misconfigured one, because + * election ignores the settings just as it ignores the MTU. See + * cl_ctr_cfg_note_reject(). */ +#define CL_CTR_CFG_SUSPECT_PEERS 2 /* Consecutive unreadable MTU polls between complaints. The first failure is * always reported; after that this keeps a blind poll from being silent * without letting it flood. At query_time=5 this is about once a minute. */ @@ -799,6 +804,11 @@ typedef struct cl_ctr_cluster_ { uint32_t mtu_reject_ips[CL_CTR_MTU_SUSPECT_PEERS * 2]; int mtu_reject_n; int mtu_suspect_said; + /* Same, for peers we have refused on config settings. See + * cl_ctr_cfg_note_reject() - the config analogue of the MTU one above. */ + uint32_t cfg_reject_ips[CL_CTR_CFG_SUSPECT_PEERS * 2]; + int cfg_reject_n; + int cfg_suspect_said; /* utime (us since start) of the last JOIN_REQ we transmitted, for a * minimum-interval throttle so a key-mismatch/split-brain burst cannot * flood the group with JOIN_REQs. 0 = never sent. Worker-local. */ @@ -4484,6 +4494,49 @@ static void cl_ctr_mtu_note_reject(const char *src_ip, cl_ctr_cluster_t *cl) } } +/** + * cl_ctr_cfg_note_reject() - the config analogue of cl_ctr_mtu_note_reject(). + * + * Election is highest-IP and just as blind to manage_shtags / master_stickiness + * / query_time as it is to the MTU, so the same failure exists: one host booted + * with the wrong settings that happens to win the election becomes master and + * refuses an otherwise-agreeing fleet, where that same host joining an already + * formed cluster would only have shut itself down. Warn, and nothing more, for + * the same reasons spelled out above the MTU version. + */ +static void cl_ctr_cfg_note_reject(const char *src_ip, cl_ctr_cluster_t *cl) +{ + uint32_t ip_num = ip_to_num(src_ip); + int i, members; + + if (ip_num == 0 || cl->cfg_suspect_said) + return; + for (i = 0; i < cl->cfg_reject_n; i++) + if (cl->cfg_reject_ips[i] == ip_num) + return; /* same peer retrying, not a new one */ + if (cl->cfg_reject_n < (int)(sizeof(cl->cfg_reject_ips) + / sizeof(cl->cfg_reject_ips[0]))) + cl->cfg_reject_ips[cl->cfg_reject_n++] = ip_num; + + lock_start_read(cl->peers->lock); + members = cl->peers->count; + lock_stop_read(cl->peers->lock); + + if (cl->cfg_reject_n >= CL_CTR_CFG_SUSPECT_PEERS && members <= 1) { + cl->cfg_suspect_said = 1; + LM_CRIT("clusterer_controller: [cluster %d] THIS NODE IS PROBABLY THE " + "MISCONFIGURED ONE: it is master with manage_shtags=%d " + "master_stickiness=%d query_time=%d, has refused %d different " + "peers for not matching, and holds no members of its own. Master " + "election is by highest IP and ignores these settings, so a " + "single wrongly-configured host that wins it will refuse an " + "otherwise healthy fleet. Check this node's config before " + "changing any of the others\n", + cl->cluster_id, cl->manage_shtags ? 1 : 0, + cl->master_stickiness ? 1 : 0, query_time, cl->cfg_reject_n); + } +} + /** * cl_ctr_handle_alive() - process a CL_CTR_PKT_ALIVE packet. * @@ -4735,6 +4788,11 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le int loc_stick = cl->master_stickiness ? 1 : 0; if (j_manage != loc_manage || j_stick != loc_stick || j_qt != query_time) { char diff[160]; + /* Drop before releasing the lock: while we were ourselves NEW we may + * have upserted this joiner, and refusing alone would leave it in the + * table, so cl_ctr_cfg_note_reject() would count it as a member we + * hold and never fire. Mirrors the MTU refusal below. */ + cl_ctr_drop_peer_locked(src_ip, cl); lock_stop_write(cl->peers->lock); cl_ctr_fmt_cfg_diff(diff, sizeof(diff), "cluster", "node", loc_manage, j_manage, loc_stick, j_stick, @@ -4743,6 +4801,7 @@ static void cl_ctr_handle_join_req(int sock, const char *payload, int payload_le "different settings than the running cluster (%s)\n", cl->cluster_id, src_ip, diff); cl_ctr_send_join_reject(sock, src_ip, cl, CL_CTR_REJECT_CONFIG); + cl_ctr_cfg_note_reject(src_ip, cl); return; } } diff --git a/modules/clusterer_controller/doc/clusterer_controller_admin.xml b/modules/clusterer_controller/doc/clusterer_controller_admin.xml index 00c99e6391d..97dce78b17e 100644 --- a/modules/clusterer_controller/doc/clusterer_controller_admin.xml +++ b/modules/clusterer_controller/doc/clusterer_controller_admin.xml @@ -451,7 +451,13 @@ refuse an otherwise healthy fleet. A master that has refused CL_CTR_MTU_SUSPECT_PEERS (2) distinct peers while holding no members of its own logs a critical message naming - itself as the probable misconfiguration. + itself as the probable misconfiguration. The same + safety net exists for the CONFIG gate + (CL_CTR_CFG_SUSPECT_PEERS, also 2): election ignores + manage_shtags / master_stickiness / + query_time exactly as it ignores the MTU, so a master + that has refused two distinct peers on settings while holding no members + names itself the probable culprit too. A node joining with the wrong password cannot