RE: [RFC net-next 1/2] tipc: abort pending connects when the peer node is lost
From: Tung Quang Nguyen
Date: Thu Oct 08 2026 - 02:47:58 EST
>Subject: [RFC net-next 1/2] tipc: abort pending connects when the peer node is
>lost
>
>From: liushike <liushike@xxxxxxxxxxxxx>
>
>A client can send a SYN and block in connect() while the listening server has
>not yet called accept(). If the last usable link to that server is lost, the client may
>remain asleep until its connection timeout expires.
>Only established connections are in the peer node's conn_sks list, so
>node_lost_contact() cannot notify the pending connection.
Do retrying calling connect() and setting connection timeout to link tolerance (default: 1500 ms) help your use case ?
>
>Register an active open before sending SYN and check node availability under
>the node write lock. Let the existing node-loss notification abort the pending
>handshake with EHOSTUNREACH and wake the connect waiter.
>Losing one link does not abort the connect while another link survives.
>
>Update the listening port to the accepted port in place on receipt of ACK,
>preserving the node-loss registration. Reject a late ACK if node loss already
>removed the registration, even if the node has recovered.
>Transfer the registration when a named SYN is rerouted to another node.
>Avoid registering active opens twice; remove registrations on transmit failure
>and terminal rejection, but retain them across overload retries and
>connection-wait timeouts, which do not cancel the handshake.
>
>Signed-off-by: liushike <liushike@xxxxxxxxxxxxx>
>---
> net/tipc/node.c | 37 ++++++++++++++++++++++++++++++++++++-
> net/tipc/node.h | 1 +
> net/tipc/socket.c | 38 +++++++++++++++++++++++++++++++++++---
> 3 files changed, 72 insertions(+), 4 deletions(-)
>
>diff --git a/net/tipc/node.c b/net/tipc/node.c index
>bd91378b7540..31f2ef44ad37 100644
>--- a/net/tipc/node.c
>+++ b/net/tipc/node.c
>@@ -713,13 +713,48 @@ int tipc_node_add_conn(struct net *net, u32 dnode,
>u32 port, u32 peer_port)
> conn->peer_port = peer_port;
>
> tipc_node_write_lock(node);
>- list_add_tail(&conn->list, &node->conn_sks);
>+ if (node_is_up(node)) {
>+ list_add_tail(&conn->list, &node->conn_sks);
>+ } else {
>+ err = -EHOSTUNREACH;
>+ kfree(conn);
>+ }
> tipc_node_write_unlock(node);
> exit:
> tipc_node_put(node);
> return err;
> }
>
>+/* Replace the listening port with the accepted port without losing the
>+ * node loss subscription. A missing entry means node_lost_contact()
>+has
>+ * already aborted this connection attempt, even if the node is up again.
>+ */
>+int tipc_node_update_conn(struct net *net, u32 dnode, u32 port, u32
>+peer_port) {
>+ struct tipc_sock_conn *conn;
>+ struct tipc_node *node;
>+ int err = -EHOSTUNREACH;
>+
>+ if (in_own_node(net, dnode))
>+ return 0;
>+
>+ node = tipc_node_find(net, dnode);
>+ if (!node)
>+ return err;
>+
>+ tipc_node_write_lock(node);
>+ list_for_each_entry(conn, &node->conn_sks, list) {
Traversing the list of thousands of connections look bad.
Have you tried to measure performance regression when one socket of this node is sending/receiving messages and 4000 thousand other connections are initiated simultaneously ?
>+ if (conn->port != port)
>+ continue;
>+ conn->peer_port = peer_port;
>+ err = 0;
>+ break;
>+ }
>+ tipc_node_write_unlock(node);
>+ tipc_node_put(node);
>+ return err;
>+}
>+
> void tipc_node_remove_conn(struct net *net, u32 dnode, u32 port) {
> struct tipc_node *node;
>diff --git a/net/tipc/node.h b/net/tipc/node.h index
>154a5bbb0d29..48693a050aa8 100644
>--- a/net/tipc/node.h
>+++ b/net/tipc/node.h
>@@ -107,6 +107,7 @@ void tipc_node_subscribe(struct net *net, struct
>list_head *subscr, u32 addr); void tipc_node_unsubscribe(struct net *net,
>struct list_head *subscr, u32 addr); void tipc_node_broadcast(struct net *net,
>struct sk_buff *skb, int rc_dests); int tipc_node_add_conn(struct net *net, u32
>dnode, u32 port, u32 peer_port);
>+int tipc_node_update_conn(struct net *net, u32 dnode, u32 port, u32
>+peer_port);
> void tipc_node_remove_conn(struct net *net, u32 dnode, u32 port); int
>tipc_node_get_mtu(struct net *net, u32 addr, u32 sel, bool connected); bool
>tipc_node_is_up(struct net *net, u32 addr); diff --git a/net/tipc/socket.c
>b/net/tipc/socket.c index d5d70eb230b5..191579636351 100644
>--- a/net/tipc/socket.c
>+++ b/net/tipc/socket.c
>@@ -1511,6 +1511,16 @@ static int __tipc_sendmsg(struct socket *sock, struct
>msghdr *m, size_t dlen)
> return -ENOMEM;
> }
>
>+ /* Subscribe before sending SYN so node loss also aborts pending
>connects. */
>+ if (syn) {
>+ rc = tipc_node_add_conn(net, skaddr.node, tsk->portid,
>skaddr.ref);
>+ if (rc) {
>+ __skb_queue_purge(&pkts);
>+ __skb_queue_purge(&sk->sk_write_queue);
>+ return rc;
>+ }
>+ }
>+
> /* Send message */
> trace_tipc_sk_sendmsg(sk, skb_peek(&pkts), TIPC_DUMP_SK_SNDQ, "
>");
> rc = tipc_node_xmit(net, &pkts, skaddr.node, tsk->portid); @@ -1520,6
>+1530,11 @@ static int __tipc_sendmsg(struct socket *sock, struct msghdr *m,
>size_t dlen)
> rc = 0;
> }
>
>+ if (unlikely(syn && rc)) {
>+ tipc_node_remove_conn(net, skaddr.node, tsk->portid);
>+ __skb_queue_purge(&sk->sk_write_queue);
>+ }
>+
> if (unlikely(syn && !rc)) {
> tipc_set_sk_state(sk, TIPC_CONNECTING);
> if (dlen && timeout) {
>@@ -1674,8 +1689,10 @@ static void tipc_sk_finish_conn(struct tipc_sock
>*tsk, u32 peer_port,
> msg_set_hdr_sz(msg, SHORT_H_SIZE);
>
> sk_reset_timer(sk, &sk->sk_timer, jiffies + CONN_PROBING_INTV);
>+ /* Active opens already subscribed before sending SYN. */
>+ if (sk->sk_state != TIPC_CONNECTING)
>+ tipc_node_add_conn(net, peer_node, tsk->portid, peer_port);
> tipc_set_sk_state(sk, TIPC_ESTABLISHED);
>- tipc_node_add_conn(net, peer_node, tsk->portid, peer_port);
> tsk->max_pkt = tipc_node_get_mtu(net, peer_node, tsk->portid, true);
> tsk->peer_caps = tipc_node_get_capabilities(net, peer_node);
> tsk_set_nagle(tsk);
>@@ -2199,12 +2216,14 @@ static bool tipc_sk_filter_connect(struct tipc_sock
>*tsk, struct sk_buff *skb,
> struct net *net = sock_net(sk);
> struct tipc_msg *hdr = buf_msg(skb);
> bool con_msg = msg_connected(hdr);
>+ bool reject_ack = false;
> u32 pport = tsk_peer_port(tsk);
> u32 pnode = tsk_peer_node(tsk);
> u32 oport = msg_origport(hdr);
> u32 onode = msg_orignode(hdr);
> int err = msg_errcode(hdr);
> unsigned long delay;
>+ int rc;
>
> if (unlikely(msg_mcast(hdr)))
> return false;
>@@ -2216,6 +2235,18 @@ static bool tipc_sk_filter_connect(struct tipc_sock
>*tsk, struct sk_buff *skb,
> if (likely(con_msg)) {
> if (err)
> break;
>+ rc = tipc_node_update_conn(net, pnode, tsk->portid,
>oport);
>+ /* A named SYN may have been rerouted to another
>node. */
>+ if (!rc && onode != pnode) {
>+ tipc_node_remove_conn(net, pnode, tsk-
>>portid);
>+ rc = tipc_node_add_conn(net, onode, tsk-
>>portid, oport);
>+ }
>+ if (rc) {
>+ err = TIPC_ERR_NO_NODE;
>+ /* Return a late ACK to its sender, including any
>data. */
>+ reject_ack = true;
>+ break;
>+ }
> tipc_sk_finish_conn(tsk, oport, onode);
> msg_set_importance(&tsk->phdr,
>msg_importance(hdr));
> /* ACK+ message with data is added to receive queue
>*/ @@ -2285,9 +2316,10 @@ static bool tipc_sk_filter_connect(struct
>tipc_sock *tsk, struct sk_buff *skb,
> }
> /* Abort connection setup attempt */
> tipc_set_sk_state(sk, TIPC_DISCONNECTING);
>- sk->sk_err = ECONNREFUSED;
>+ tipc_node_remove_conn(net, pnode, tsk->portid);
>+ sk->sk_err = err == TIPC_ERR_NO_NODE ? EHOSTUNREACH :
>ECONNREFUSED;
> sk->sk_state_change(sk);
>- return true;
>+ return !reject_ack;
> }
>
> /**
>--
>2.34.1