Delete the fcloop remote port and then the target port while an association is being deleted. The Disconnect Association LS is completed from a work item that can run after nvmet_fc_unregister_targetport() freed the pending request, which KASAN reports as a use-after-free in fcloop_tport_lsrqst_work(). Kernel fix: "nvmet-fc: flush nvmet_wq twice on targetport unregister" Signed-off-by: Nguyen Ngoc Thang --- Changes in v2 (thanks Shin'ichiro for the review): - Note the kernel fix's commit title in the file header. - local -a ports, and use _get_fc_host_port() instead of reading ports_to_hosts directly. - Connect with --ctrl-loss-tmo 0. Without it, the host can spend up to NVMF_DEF_CTRL_LOSS_TMO (600s) retrying reconnect after we pull the remote port, which is likely the hang you saw; disconnecting it cleanly then has to wait on that reconnect state first. - Drop the swept delay before pulling the ports. I could not tell it apart from a fixed "sleep 0" here either (KASAN UAF + list_debug BUG, both with and without it), so it wasn't earning its complexity. - v1: https://lore.kernel.org/linux-block/20260921145659.17151-1-ngocthang2710.1999@gmail.com/ tests/nvme/071 | 75 ++++++++++++++++++++++++++++++++++++++++++++++ tests/nvme/071.out | 2 ++ 2 files changed, 77 insertions(+) create mode 100755 tests/nvme/071 create mode 100644 tests/nvme/071.out diff --git a/tests/nvme/071 b/tests/nvme/071 new file mode 100755 index 0000000..406a841 --- /dev/null +++ b/tests/nvme/071 @@ -0,0 +1,75 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-3.0+ +# Copyright (C) 2026 Nguyen Ngoc Thang +# +# Regression test for a use-after-free in fcloop when the target port is +# deleted right after the remote port. Deleting the association sends a +# Disconnect Association LS whose completion is queued while +# nvmet_fc_unregister_targetport() is flushing nvmet_wq. The pending LS request +# was freed before that completion ran. +# +# Kernel fix: "nvmet-fc: flush nvmet_wq twice on targetport unregister" + +. tests/nvme/rc + +DESCRIPTION="delete fcloop target port right after remote port" +QUICK=1 + +requires() { + _nvme_requires + _have_loop + _require_nvme_trtype fc +} + +set_conditions() { + _set_nvme_trtype "$@" +} + +test() { + echo "Running ${TEST_NAME}" + + _setup_nvmet + + local -a ports + local i port host_port + + _nvmet_target_setup + + _get_nvmet_ports "${def_subsysnqn}" ports + port="${ports[0]}" + host_port="$(_get_fc_host_port "${port}")" + + for ((i = 0; i < 20; i++)); do + # ctrl-loss-tmo=0 so the host drops the controller on the first + # failed reconnect instead of retrying for NVMF_DEF_CTRL_LOSS_TMO + # (600s), which would make each iteration look like it hangs. + _nvme_connect_subsys --ctrl-loss-tmo 0 + sleep 0.05 + + # Deleting the association sends a Disconnect Association LS. + _remove_nvmet_subsystem_from_port "${port}" "${def_subsysnqn}" + + # Remote port first, so the LS can't reach the host anymore. + _nvme_fcloop_del_rport "$(_host_wwnn "${host_port}")" \ + "$(_host_wwpn "${host_port}")" \ + "$(_remote_wwnn "${port}")" \ + "$(_remote_wwpn "${port}")" + _nvme_fcloop_del_tport "$(_remote_wwnn "${port}")" \ + "$(_remote_wwpn "${port}")" + + # The host keeps trying to reconnect, drop the controller. + _nvme_disconnect_subsys >> "${FULL}" 2>&1 + + _nvme_fcloop_add_tport "$(_remote_wwnn "${port}")" \ + "$(_remote_wwpn "${port}")" + _nvme_fcloop_add_rport "$(_host_wwnn "${host_port}")" \ + "$(_host_wwpn "${host_port}")" \ + "$(_remote_wwnn "${port}")" \ + "$(_remote_wwpn "${port}")" + _add_nvmet_subsys_to_port "${port}" "${def_subsysnqn}" + done + + _nvmet_target_cleanup + + echo "Test complete" +} diff --git a/tests/nvme/071.out b/tests/nvme/071.out new file mode 100644 index 0000000..146809b --- /dev/null +++ b/tests/nvme/071.out @@ -0,0 +1,2 @@ +Running nvme/071 +Test complete -- 2.43.0