From 8583d4216f3ef4d9094d29b3da474bbbbaa94667 Mon Sep 17 00:00:00 2001 From: evgeny Date: Tue, 11 Aug 2026 23:30:31 +0300 Subject: [PATCH] fix auto_socket: handle_nl_link silently skipped on if_indextoname failure for deleted interfaces When ip link del removes an interface, the kernel sends RTM_DELLINK after the interface is unregistered from the name table. if_indextoname fails and handle_nl_link silently returned without calling remove_iface_sockets, causing old ETCP sockets to accumulate indefinitely. Fix: fall through to remove_iface_sockets(ifindex) even when if_indextoname fails. Test: added diagnostics (diag_dump_state, per-category debug levels to file) to verify socket/links cleanup. --- src/transport_layer/auto_socket.c | 7 +++- tests/test_auto_socket_dynamic.c | 58 ++++++++++++++++++++++++++++++- 2 files changed, 63 insertions(+), 2 deletions(-) diff --git a/src/transport_layer/auto_socket.c b/src/transport_layer/auto_socket.c index e233fc17..3e68dba2 100644 --- a/src/transport_layer/auto_socket.c +++ b/src/transport_layer/auto_socket.c @@ -932,7 +932,12 @@ static void handle_nl_addr(struct AUTO_SOCKET* as, struct ifaddrmsg* ifa, int ms * - UP (новый или поднялся) → запускаем reconcile */ static void handle_nl_link(struct AUTO_SOCKET* as, struct ifinfomsg* ifi, int msg_type) { char ifname[IFNAMSIZ]; - if (!if_indextoname(ifi->ifi_index, ifname)) return; + if (!if_indextoname(ifi->ifi_index, ifname)) { + DEBUG_WARN(DEBUG_CATEGORY_AS, "[as] handle_nl_link: if_indextoname(%u) failed, msg=%d — removing by index", + ifi->ifi_index, msg_type); + remove_iface_sockets(as, ifi->ifi_index); + return; + } if (ifi->ifi_flags & IFF_LOOPBACK) return; int is_up = (ifi->ifi_flags & IFF_UP) != 0; diff --git a/tests/test_auto_socket_dynamic.c b/tests/test_auto_socket_dynamic.c index f5e7f3c5..54ae6450 100644 --- a/tests/test_auto_socket_dynamic.c +++ b/tests/test_auto_socket_dynamic.c @@ -183,6 +183,51 @@ static void start_traffic(void) { timeout_handle = uasync_set_timeout(ctx.ua, TRAF_MON_TB, NULL, traffic_monitor_timer, "traf_mon"); } +/* ── diagnostics ── */ +static void diag_dump_state(const char* label) { + fprintf(stderr, "--- DIAG [%s] r=%d s=%d ---\n", label, ctx.round, ctx.step); + struct ll_entry* e = ctx.client->connections->head; + int conn_n = 0; + while (e) { conn_n++; e = e->next; } + fprintf(stderr, " connections=%d\n", conn_n); + fprintf(stderr, " current_iface=%s prev_iface=%s\n", ctx.cur_iface, ctx.prev_iface[0] ? ctx.prev_iface : "(none)"); + + /* list sockets */ + fprintf(stderr, " sockets:\n"); + struct ETCP_SOCKET* s = ctx.client->etcp_sockets; + while (s) { + char ifname[IFNAMSIZ] = "?"; + if (s->netif_index) if_indextoname(s->netif_index, ifname); + fprintf(stderr, " %-30s fd=%d if=%s(%u) fam=%d tcp=%d\n", + s->name, (int)s->fd, ifname, s->netif_index, + s->local_addr.ss_family, s->is_tcp); + s = s->next; + } + + /* list links for the server connection */ + fprintf(stderr, " links to srv 0x%016llx:\n", (unsigned long long)ctx.srv_node_id); + e = ctx.client->connections->head; + while (e) { + struct conn_queue_entry* ce = (struct conn_queue_entry*)e->data; + if (ce->conn->peer_node_id == ctx.srv_node_id) { + int nlink = 0; for (struct ETCP_LINK* tl = ce->conn->links; tl; tl = tl->next) nlink++; + fprintf(stderr, " conn=%s state=%d links=%d\n", + ce->conn->log_name, ce->conn->state, nlink); + struct ETCP_LINK* l = ce->conn->links; + while (l) { + char ifname[IFNAMSIZ] = "?"; + if (l->conn->netif_index) if_indextoname(l->conn->netif_index, ifname); + fprintf(stderr, " link=%d init=%d status=%d is_tcp=%d sock=%s if=%s\n", + l->local_link_id, l->initialized, l->link_status, l->is_tcp, + l->conn ? l->conn->name : "?", ifname); + l = l->next; + } + } + e = e->next; + } + fflush(stderr); +} + /* ── ip addr add/del via system ── */ static int ip_addr_add(const char* ifname, const char* cidr) { char cmd[256]; snprintf(cmd, sizeof(cmd), "ip addr add %s dev %s 2>/dev/null", cidr, ifname); @@ -229,6 +274,7 @@ static void phase_del_last(void* arg) { (void)arg; if (ctx.result) return; ctx.step = 11; + diag_dump_state("before del_last"); ip_link_del(rounds[N_ROUNDS-1].ifname); uasync_set_timeout(ctx.ua, STEP_TB, NULL, phase_check_del_last, "chk_del_last"); } @@ -256,15 +302,19 @@ static void phase_check_del_last(void* arg) { static void phase_check_ip(void* arg) { (void)arg; if (ctx.result) return; static uint64_t wait_start = 0; + static int diag_cnt = 0; ctx.step = 6; if (count_links_to_srv(ctx.cur_iface) == 0 || ctx.total_recv <= ctx.recv_at_ip_change) { + if (++diag_cnt <= 3) diag_dump_state("wait chk_ip"); if (!wait_start) wait_start = get_time_tb(); if (get_time_tb() - wait_start < (uint64_t)STEP_TB * 10) { uasync_set_timeout(ctx.ua, STEP_TB/3, NULL, phase_check_ip, "chk_ip"); return; } + diag_dump_state("timeout chk_ip"); fail("traffic not recovered after IP change"); return; } + diag_cnt = 0; wait_start = 0; do_check(); if (ctx.result) return; @@ -422,7 +472,13 @@ static void setup(void) { static void cleanup(void) { test_unlink(scf); test_unlink(ccf); test_rmdir(tdir); } int main(void) { - debug_config_init(); debug_set_level(DEBUG_LEVEL_WARN); + debug_config_init(); + debug_enable_file_output("/tmp/as_dyn_diag.log", 1); + debug_set_level(DEBUG_LEVEL_WARN); + debug_set_category_level(DEBUG_CATEGORY_SOCKET, DEBUG_LEVEL_INFO); + debug_set_category_level(DEBUG_CATEGORY_CONNECTION, DEBUG_LEVEL_INFO); + debug_set_category_level(DEBUG_CATEGORY_ETCP, DEBUG_LEVEL_INFO); + debug_set_category_level(DEBUG_CATEGORY_DEBUG, DEBUG_LEVEL_INFO); utun_instance_set_tun_init_enabled(0); setup();