From: Martin Belanger nvme_tcp_recv_skb() disables the queue and starts error recovery on any errors. But if the transport encounters an error before ->read_sock() is called (and hence before nvme_tcp_recv_skb() is called) no further action is taken. This causes nvme_tcp_io_work() to stall until a command times out. This patch checks for any error returned from ->read_sock(), and starts error recovery when the queue is still enabled (ie if nvme_tcp_recv_skb() has not detected the error). Signed-off-by: Martin Belanger Signed-off-by: Hannes Reinecke --- drivers/nvme/host/tcp.c | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/drivers/nvme/host/tcp.c b/drivers/nvme/host/tcp.c index 7eec983c0ead..2f1ae3968678 100644 --- a/drivers/nvme/host/tcp.c +++ b/drivers/nvme/host/tcp.c @@ -1422,7 +1422,23 @@ static int nvme_tcp_try_recv(struct nvme_tcp_queue *queue) queue->nr_cqe = 0; consumed = sock->ops->read_sock(sk, &rd_desc, nvme_tcp_recv_skb); release_sock(sk); - return consumed == -EAGAIN ? 0 : consumed; + if (consumed == -EAGAIN) + return 0; + + /* + * read_sock() might encounter an error before calling + * nvme_tcp_recv_skb(), so we need to check if we need + * to start error recovery here. + */ + if (unlikely(consumed < 0 && queue->rd_enabled)) { + dev_err(queue->ctrl->ctrl.device, + "queue %d: receive failed: %d\n", + nvme_tcp_queue_id(queue), consumed); + queue->rd_enabled = false; + nvme_tcp_error_recovery(&queue->ctrl->ctrl); + } + + return consumed; } static void nvme_tcp_io_work(struct work_struct *w) -- 2.51.0