bpf_tcp_ops.{parse_hdr,hdr_opt_len} are called for every incoming / outgoing skb. bpf_tcp_ops.rtt is called once per RTT, which is every incoming skb in ping-pong workloads like tcp_rr. Even attaching NULL callbacks in the fast path hurts performance. Let's guard them (and write_hdr_opt) with the new per-socket flags. __bpf_tcp_ops_call() and bpf_tcp_ops_call_flag() are added to check cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) first and avoid accessing tp->bpf_tcp_ops_flags when no bpf_tcp_ops is attached. Note that bpf_tcp_ops_hdr_opt_len() has 3 callers and previously checked the static key both in tcp_established_options() and inside bpf_tcp_ops_call(). Now the static key is checked once at the beginning of bpf_tcp_ops_hdr_opt_len(), and the redundant check in tcp_established_options() is removed. Signed-off-by: Kuniyuki Iwashima --- v4: Check static key first and then flags --- include/net/tcp.h | 44 ++++++++++++++++++++++++++++--------------- net/ipv4/tcp_input.c | 12 +++++++++++- net/ipv4/tcp_output.c | 34 ++++++++++++++++----------------- 3 files changed, 57 insertions(+), 33 deletions(-) diff --git a/include/net/tcp.h b/include/net/tcp.h index 55ad9db99596..b7c0f1a8797a 100644 --- a/include/net/tcp.h +++ b/include/net/tcp.h @@ -3064,22 +3064,33 @@ struct bpf_tcp_ops { u32 opt_off); }; -#define bpf_tcp_ops_call(op, sk, ...) \ +#define __bpf_tcp_ops_call(op, sk, ...) \ do { \ - if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { \ - const struct bpf_prog_array_item *item; \ - const struct bpf_tcp_ops *tcp_ops; \ - struct cgroup *cgrp; \ + const struct bpf_prog_array_item *item; \ + const struct bpf_tcp_ops *tcp_ops; \ + struct cgroup *cgrp; \ \ - cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data); \ - rcu_read_lock_dont_migrate(); \ - bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp, \ - CGROUP_TCP_SOCK_OPS) { \ - if (tcp_ops->op) \ - tcp_ops->op(sk, ##__VA_ARGS__); \ - } \ - rcu_read_unlock_migrate(); \ + cgrp = sock_cgroup_ptr(&sk->sk_cgrp_data); \ + rcu_read_lock_dont_migrate(); \ + bpf_cgroup_struct_ops_foreach(tcp_ops, item, cgrp, \ + CGROUP_TCP_SOCK_OPS) { \ + if (tcp_ops->op) \ + tcp_ops->op(sk, ##__VA_ARGS__); \ } \ + rcu_read_unlock_migrate(); \ +} while (0) + +#define bpf_tcp_ops_call(op, sk, ...) \ +do { \ + if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) \ + __bpf_tcp_ops_call(op, sk, ##__VA_ARGS__); \ +} while (0) + +#define bpf_tcp_ops_call_flag(op, flag, sk, ...) \ +do { \ + if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) && \ + BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), flag)) \ + __bpf_tcp_ops_call(op, sk, ##__VA_ARGS__); \ } while (0) #define bpf_tcp_ops_call_int(op, init_retval, sk, ...) \ @@ -3115,7 +3126,9 @@ do { \ }) #else -#define bpf_tcp_ops_call(op, sk, ...) do { } while (0) +#define __bpf_tcp_ops_call(op, sk, ...) do { } while (0) +#define bpf_tcp_ops_call(op, sk, ...) do { } while (0) +#define bpf_tcp_ops_call_flag(op, flag, sk, ...) do { } while (0) #define bpf_tcp_ops_call_int(op, init_retval, sk, ...) (init_retval) #endif @@ -3150,7 +3163,8 @@ static inline void tcp_bpf_rtt(struct sock *sk, long mrtt, u32 srtt) { if (BPF_SOCK_OPS_TEST_FLAG(tcp_sk(sk), BPF_SOCK_OPS_RTT_CB_FLAG)) tcp_call_bpf_2arg(sk, BPF_SOCK_OPS_RTT_CB, mrtt, srtt); - bpf_tcp_ops_call(rtt, sk, mrtt, srtt); + + bpf_tcp_ops_call_flag(rtt, RTT, sk, mrtt, srtt); } #if IS_ENABLED(CONFIG_SMC) diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c index 209db8effcc4..4478d3f3d4b0 100644 --- a/net/ipv4/tcp_input.c +++ b/net/ipv4/tcp_input.c @@ -210,6 +210,11 @@ static void bpf_skops_established(struct sock *sk, int bpf_op, static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb) { + const struct tcp_sock *tp; + + if (!cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) + return; + switch (sk->sk_state) { case TCP_SYN_RECV: case TCP_SYN_SENT: @@ -217,7 +222,12 @@ static void bpf_tcp_ops_parse_hdr(struct sock *sk, struct sk_buff *skb) return; } - bpf_tcp_ops_call(parse_hdr, sk, skb); + tp = tcp_sk(sk); + + if ((tp->rx_opt.saw_unknown && + BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_UNKNOWN)) || + BPF_TCP_OPS_TEST_FLAG(tp, PARSE_HDR_OPT_ALL)) + __bpf_tcp_ops_call(parse_hdr, sk, skb); } static __cold void tcp_gro_dev_warn(const struct sock *sk, const struct sk_buff *skb, diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c index 3ac465513bf5..8770f3084efe 100644 --- a/net/ipv4/tcp_output.c +++ b/net/ipv4/tcp_output.c @@ -581,8 +581,8 @@ static void bpf_skops_write_hdr_opt(struct sock *sk, struct sk_buff *skb, * writer's bytes). The writer finds the append point by scanning from * first_opt_off + nr_written to the first NOP. */ - bpf_tcp_ops_call(write_hdr_opt, sk, skb, req, syn_skb, synack_type, - first_opt_off + nr_written); + bpf_tcp_ops_call_flag(write_hdr_opt, WRITE_HDR_OPT, sk, skb, req, + syn_skb, synack_type, first_opt_off + nr_written); } #else static u32 bpf_skops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, @@ -613,11 +613,13 @@ static u32 bpf_tcp_ops_hdr_opt_len(struct sock *sk, struct sk_buff *skb, { unsigned int remaining_out = remaining, reserved; - if (!remaining) - return 0; + if (!cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS) || + !BPF_TCP_OPS_TEST_FLAG(tcp_sk(sk), WRITE_HDR_OPT)|| + !remaining) + return remaining; /* bpf_tcp_ops_reserve_hdr_opt() reserves space via remaining_out */ - bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out); + __bpf_tcp_ops_call(hdr_opt_len, sk, skb, req, syn_skb, synack_type, &remaining_out); reserved = remaining - remaining_out; if (!reserved) @@ -1193,8 +1195,9 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb struct tcp_key *key) { struct tcp_sock *tp = tcp_sk(sk); - unsigned int size = 0; unsigned int eff_sacks; + unsigned int remaining; + unsigned int size = 0; opts->options = 0; opts->bpf_opt_len = 0; @@ -1224,10 +1227,10 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb * left. */ if (sk_is_mptcp(sk)) { - unsigned int remaining = MAX_TCP_OPTION_SPACE - size; bool has_ts = opts->options & OPTION_TS; int opt_size; + remaining = MAX_TCP_OPTION_SPACE - size; opts->mptcp.drop_ts = 0; opt_size = mptcp_established_options(sk, skb, remaining, has_ts, @@ -1244,7 +1247,8 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb eff_sacks = tp->rx_opt.num_sacks + tp->rx_opt.dsack; if (unlikely(eff_sacks)) { - const unsigned int remaining = MAX_TCP_OPTION_SPACE - size; + remaining = MAX_TCP_OPTION_SPACE - size; + if (likely(remaining >= TCPOLEN_SACK_BASE_ALIGNED + TCPOLEN_SACK_PERBLOCK)) { opts->num_sack_blocks = @@ -1277,22 +1281,18 @@ static unsigned int tcp_established_options(struct sock *sk, struct sk_buff *skb if (unlikely(BPF_SOCK_OPS_TEST_FLAG(tp, BPF_SOCK_OPS_WRITE_HDR_OPT_CB_FLAG))) { - unsigned int remaining = MAX_TCP_OPTION_SPACE - size; - + remaining = MAX_TCP_OPTION_SPACE - size; remaining = bpf_skops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, remaining); size = MAX_TCP_OPTION_SPACE - remaining; } - if (cgroup_bpf_enabled(CGROUP_TCP_SOCK_OPS)) { - unsigned int remaining = MAX_TCP_OPTION_SPACE - size; - - remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, - remaining); + remaining = MAX_TCP_OPTION_SPACE - size; + remaining = bpf_tcp_ops_hdr_opt_len(sk, skb, NULL, NULL, 0, opts, + remaining); - size = MAX_TCP_OPTION_SPACE - remaining; - } + size = MAX_TCP_OPTION_SPACE - remaining; return size; } -- 2.56.0.360.g66cac248cb-goog