bpf_tcp_ops.{enqueue,dequeue}_rcvq() were added to parse skb and adjust sk->sk_rcvlowat dynamically to suppress unnecessary wakeups. Let's add a new kfunc to set sk->sk_rcvlowat. Negative values are clamped to INT_MAX, consistent with SO_RCVLOWAT. For enqueue_rcvq(), wakeup is set to false because: * tcp_data_ready() is always called after the hooks in tcp_queue_rcv() and tcp_ofo_queue(). * when tcp_fastopen_add_skb() is called for TFO SYN, the socket is not yet accept()ed, and when called for TFO SYN+ACK, the socket is woken up by sk->sk_state_change() anyway. For dequeue_rcvq(), wakeup is set to true because tcp_data_ready() is not called in that path. An alternative would be to support bpf_setsockopt() for these hooks. However, that approach involves excessive conditionals and an unnecessary memcpy(), costs we do not want to pay for every skb in the TCP fast path. Signed-off-by: Kuniyuki Iwashima --- net/ipv4/bpf_tcp_ops.c | 56 +++++++++++++++++++++++++++++++++++++++++- 1 file changed, 55 insertions(+), 1 deletion(-) diff --git a/net/ipv4/bpf_tcp_ops.c b/net/ipv4/bpf_tcp_ops.c index b0e14b54917e..3768b1440eb7 100644 --- a/net/ipv4/bpf_tcp_ops.c +++ b/net/ipv4/bpf_tcp_ops.c @@ -359,8 +359,62 @@ static struct bpf_struct_ops bpf_tcp_ops = { .owner = THIS_MODULE, }; +__bpf_kfunc_start_defs(); + +__bpf_kfunc int bpf_tcp_ops_set_rcvlowat(struct sock *sk, int rcvlowat, + const struct bpf_prog_aux *aux) +{ + u32 moff = aux->attach_st_ops_member_off; + bool wakeup = false; + + if (moff == offsetof(struct bpf_tcp_ops, dequeue_rcvq)) + wakeup = true; + + if (rcvlowat < 0) + rcvlowat = INT_MAX; + + return __tcp_set_rcvlowat(sk, rcvlowat, wakeup); +} + +__bpf_kfunc_end_defs(); + +BTF_KFUNCS_START(bpf_tcp_ops_rcvlowat_kfunc_set) +BTF_ID_FLAGS(func, bpf_tcp_ops_set_rcvlowat, KF_IMPLICIT_ARGS) +BTF_KFUNCS_END(bpf_tcp_ops_rcvlowat_kfunc_set) + +static int bpf_tcp_ops_rcvlowat_kfunc_filter(const struct bpf_prog *prog, + u32 kfunc_id) +{ + u32 moff; + + if (!btf_id_set8_contains(&bpf_tcp_ops_rcvlowat_kfunc_set, kfunc_id)) + return 0; + + if (prog->aux->st_ops != &bpf_tcp_ops) + return -EACCES; + + moff = prog->aux->attach_st_ops_member_off; + if (moff != offsetof(struct bpf_tcp_ops, enqueue_rcvq) && + moff != offsetof(struct bpf_tcp_ops, dequeue_rcvq)) + return -EACCES; + + return 0; +} + +static const struct btf_kfunc_id_set bpf_tcp_ops_rcvlowat_kfunc_id_set = { + .owner = THIS_MODULE, + .set = &bpf_tcp_ops_rcvlowat_kfunc_set, + .filter = bpf_tcp_ops_rcvlowat_kfunc_filter, +}; + static int __init __bpf_tcp_ops_init(void) { - return register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops); + int ret; + + ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, + &bpf_tcp_ops_rcvlowat_kfunc_id_set); + ret = ret ?: register_bpf_struct_ops(&bpf_tcp_ops, bpf_tcp_ops); + + return ret; } late_initcall(__bpf_tcp_ops_init); -- 2.55.0.1082.g2b9226bbc0-goog