bpf_tcp_ops.{enqueue,dequeue}_rcvq() were added to parse skb and adjust sk->sk_rcvlowat dynamically to suppress unnecessary wakeups. Let's add a new kfunc to set sk->sk_rcvlowat. Negative values are clamped to INT_MAX, consistent with SO_RCVLOWAT. For enqueue_rcvq(), wakeup is set to false because: * tcp_data_ready() is always called after the hooks in tcp_queue_rcv() and tcp_ofo_queue(). * when tcp_fastopen_add_skb() is called for TFO SYN, the socket is not yet accept()ed, and when called for TFO SYN+ACK, the socket is woken up by sk->sk_state_change() anyway. For dequeue_rcvq(), wakeup is set to true because tcp_data_ready() is not called in that path. An alternative would be to support bpf_setsockopt() for these hooks. However, that approach involves excessive conditionals and an unnecessary memcpy(), costs we do not want to pay for every skb in the TCP fast path. Signed-off-by: Kuniyuki Iwashima Acked-by: Stanislav Fomichev Tested-by: Clément Léger Reviewed-by: Emil Tsalapatis Reviewed-by: Amery Hung --- net/ipv4/bpf_tcp_ops.c | 46 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c index 8182037c4269..2ba73dd6c52c 100644 --- a/net/ipv4/bpf_tcp_ops.c +++ b/net/ipv4/bpf_tcp_ops.c @@ -361,12 +361,31 @@ __bpf_kfunc int bpf_tcp_ops_set_flags(struct tcp_sock *tp, u32 enable, u32 disab return 0; } +__bpf_kfunc int bpf_tcp_ops_set_rcvlowat(struct sock *sk, int rcvlowat, + const struct bpf_prog_aux *aux) +{ + u32 moff = aux->attach_st_ops_member_off; + bool wakeup = false; + + if (moff == offsetof(struct bpf_tcp_ops, dequeue_rcvq)) + wakeup = true; + + if (rcvlowat < 0) + rcvlowat = INT_MAX; + + return __tcp_set_rcvlowat(sk, rcvlowat, wakeup); +} + __bpf_kfunc_end_defs(); BTF_KFUNCS_START(bpf_tcp_ops_set_flags_kfunc_set) BTF_ID_FLAGS(func, bpf_tcp_ops_set_flags) BTF_KFUNCS_END(bpf_tcp_ops_set_flags_kfunc_set) +BTF_KFUNCS_START(bpf_tcp_ops_set_rcvlowat_kfunc_set) +BTF_ID_FLAGS(func, bpf_tcp_ops_set_rcvlowat, KF_IMPLICIT_ARGS) +BTF_KFUNCS_END(bpf_tcp_ops_set_rcvlowat_kfunc_set) + static int bpf_tcp_ops_set_flags_kfunc_filter(const struct bpf_prog *prog, u32 kfunc_id) { @@ -390,6 +409,31 @@ static const struct btf_kfunc_id_set bpf_tcp_ops_set_flags_kfunc_id_set = { .filter = bpf_tcp_ops_set_flags_kfunc_filter, }; +static int bpf_tcp_ops_set_rcvlowat_kfunc_filter(const struct bpf_prog *prog, + u32 kfunc_id) +{ + u32 moff; + + if (!btf_id_set8_contains(&bpf_tcp_ops_set_rcvlowat_kfunc_set, kfunc_id)) + return 0; + + if (prog->aux->st_ops != &bpf_tcp_ops) + return -EACCES; + + moff = prog->aux->attach_st_ops_member_off; + if (moff != offsetof(struct bpf_tcp_ops, enqueue_rcvq) && + moff != offsetof(struct bpf_tcp_ops, dequeue_rcvq)) + return -EACCES; + + return 0; +} + +static const struct btf_kfunc_id_set bpf_tcp_ops_set_rcvlowat_kfunc_id_set = { + .owner = THIS_MODULE, + .set = &bpf_tcp_ops_set_rcvlowat_kfunc_set, + .filter = bpf_tcp_ops_set_rcvlowat_kfunc_filter, +}; + static int __init __bpf_tcp_ops_init(void) { int ret; @@ -399,6 +443,8 @@ static int __init __bpf_tcp_ops_init(void) /* BPF_PROG_TYPE_CGROUP_{SOCKOPT,SKB} share BTF_KFUNC_HOOK_CGROUP. */ ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_CGROUP_SOCKOPT, &bpf_tcp_ops_set_flags_kfunc_id_set); + ret = ret ?: register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, + &bpf_tcp_ops_set_rcvlowat_kfunc_id_set); ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops); return ret; -- 2.56.0.360.g66cac248cb-goog