From: Hangbin Liu A later patch widens the bonding TLB tx counters from u32 to u64. The unbalanced_load counter sits in the transmit hot path, and cross-CPU synchronization of a u64 would introduce measurable overhead. Convert unbalanced_load to a per-cpu counter first so that the subsequent widening only touches per-cpu data local to each CPU. Introduce struct unbalanced_load_stats to hold the per-cpu counter, and move the aggregation into a helper, reset_unbalanced_load(), which sums and clears all per-cpu instances. Signed-off-by: Hangbin Liu --- drivers/net/bonding/bond_alb.c | 25 ++++++++++++++++++------- drivers/net/bonding/bond_main.c | 9 +++++++++ include/net/bond_alb.h | 6 +++++- 3 files changed, 32 insertions(+), 8 deletions(-) diff --git a/drivers/net/bonding/bond_alb.c b/drivers/net/bonding/bond_alb.c index 839f7482dc18..d54d834cf72b 100644 --- a/drivers/net/bonding/bond_alb.c +++ b/drivers/net/bonding/bond_alb.c @@ -1345,7 +1345,7 @@ static netdev_tx_t bond_do_alb_xmit(struct sk_buff *skb, struct bonding *bond, /* unbalanced or unassigned, send through primary */ tx_slave = rcu_dereference(bond->curr_active_slave); if (bond->params.tlb_dynamic_lb) - bond_info->unbalanced_load += skb->len; + this_cpu_add(bond_info->unbalanced_load->tx_bytes, skb->len); } if (tx_slave && bond_slave_can_tx(tx_slave)) { @@ -1529,6 +1529,21 @@ netdev_tx_t bond_alb_xmit(struct sk_buff *skb, struct net_device *bond_dev) return bond_do_alb_xmit(skb, bond, tx_slave); } +static u32 reset_unbalanced_load(struct alb_bond_info *bond_info) +{ + struct unbalanced_load_stats *p; + u32 total_bytes = 0; + int i; + + for_each_possible_cpu(i) { + p = per_cpu_ptr(bond_info->unbalanced_load, i); + total_bytes += READ_ONCE(p->tx_bytes); + WRITE_ONCE(p->tx_bytes, 0); + } + + return total_bytes / BOND_TLB_REBALANCE_INTERVAL; +} + void bond_alb_monitor(struct work_struct *work) { struct bonding *bond = container_of(work, struct bonding, @@ -1570,12 +1585,8 @@ void bond_alb_monitor(struct work_struct *work) if (atomic_read(&bond_info->tx_rebalance_counter) >= BOND_TLB_REBALANCE_TICKS) { bond_for_each_slave_rcu(bond, slave, iter) { tlb_clear_slave(bond, slave, 1); - if (slave == rcu_access_pointer(bond->curr_active_slave)) { - SLAVE_TLB_INFO(slave).load = - bond_info->unbalanced_load / - BOND_TLB_REBALANCE_INTERVAL; - bond_info->unbalanced_load = 0; - } + if (slave == rcu_access_pointer(bond->curr_active_slave)) + SLAVE_TLB_INFO(slave).load = reset_unbalanced_load(bond_info); } atomic_set(&bond_info->tx_rebalance_counter, 0); } diff --git a/drivers/net/bonding/bond_main.c b/drivers/net/bonding/bond_main.c index 522eab060f9e..9fb44e0031c8 100644 --- a/drivers/net/bonding/bond_main.c +++ b/drivers/net/bonding/bond_main.c @@ -5995,6 +5995,7 @@ static void bond_destructor(struct net_device *bond_dev) destroy_workqueue(bond->wq); free_percpu(bond->rr_tx_counter); + free_percpu(bond->alb_info.unbalanced_load); } void bond_setup(struct net_device *bond_dev) @@ -6494,6 +6495,10 @@ static int bond_init(struct net_device *bond_dev) if (!bond->wq) return -ENOMEM; + bond->alb_info.unbalanced_load = alloc_percpu(struct unbalanced_load_stats); + if (!bond->alb_info.unbalanced_load) + goto wq_out; + bond->notifier_ctx = false; spin_lock_init(&bond->stats_lock); @@ -6511,6 +6516,10 @@ static int bond_init(struct net_device *bond_dev) eth_hw_addr_random(bond_dev); return 0; + +wq_out: + destroy_workqueue(bond->wq); + return -ENOMEM; } unsigned int bond_get_num_tx_queues(void) diff --git a/include/net/bond_alb.h b/include/net/bond_alb.h index e5945427f38d..3fabf4714dec 100644 --- a/include/net/bond_alb.h +++ b/include/net/bond_alb.h @@ -123,9 +123,13 @@ struct tlb_slave_info { */ }; +struct unbalanced_load_stats { + u32 tx_bytes; +}; + struct alb_bond_info { struct tlb_client_info *tx_hashtbl; /* Dynamically allocated */ - u32 unbalanced_load; + struct unbalanced_load_stats __percpu *unbalanced_load; atomic_t tx_rebalance_counter; int lp_counter; /* -------- rlb parameters -------- */ -- 2.55.0