Provides the functions that initialize and release the data structures used during live guest migration: * vfio_ap_init_migration_capabilities Sets the migration flags and vfio_migration_ops structure into the vfio_device object when the mdev is probed. * vfio_ap_init_migration_data Allocates and initializes the object used to maintain the state of the VFIO migration. It is called when the VFIO device is opened. * vfio_ap_release_migration_data Frees the memory of the object used to maintain the VFIO migration state. It is called when the VFIO device release callback is invoked and when the VFIO device is closed. * vfio_ap_release_mig_files This function is called from the vfio_ap_release_migration_data function (above) and releases the vfio_ap_migration_file objects contained within the vfio_ap_migation_data object used to maintain the state of the VFIO migration process. * vfio_ap_release_stop_copy_file This function is called from the vfio_ap_release_mig_files function (above) to release the vfio_ap_migration_file used during the STOP_COPY phase of migration. For now, this is a stub function that will be fully implemented in a subsequent patch after the vfio_ap_config object referenced within is allocated, as it will need to be freed according to how it is allocated * vfio_ap_release_resuming_file This function is called from the vfio_ap_release_mig_files function (above) to release the vfio_ap_migration_file used during the RESUMING phase of migration. For now, this is a stub function that will be fully implemented in a subsequent patch after the vfio_ap_config object referenced within is allocated, as it will need to be freed according to how it is allocated * vfio_ap_set_state, vfio_ap_get_state and vfio_ap_get_data_size These three functions are callback functions assigned to the vfio_ap_migration_ops (.migration_set_state, .migration_get_state and .migration_get_data_size function pointers). These are implemented as stub functions here and each will be fully implemented in a subsequent patch. Signed-off-by: Anthony Krowiak --- drivers/s390/crypto/vfio_ap_migration.c | 127 ++++++++++++++++++++++++ drivers/s390/crypto/vfio_ap_ops.c | 64 ++++++++++-- drivers/s390/crypto/vfio_ap_private.h | 4 + 3 files changed, 187 insertions(+), 8 deletions(-) diff --git a/drivers/s390/crypto/vfio_ap_migration.c b/drivers/s390/crypto/vfio_ap_migration.c index 374d3a67cb21..cf303d228a0a 100644 --- a/drivers/s390/crypto/vfio_ap_migration.c +++ b/drivers/s390/crypto/vfio_ap_migration.c @@ -4,6 +4,7 @@ * * Copyright IBM Corp. 2025 */ +#include #include "vfio_ap_private.h" /* Magic number and version for the vfio_ap_config migration blob */ @@ -111,3 +112,129 @@ struct vfio_ap_config { u64 adm[DIV_ROUND_UP(AP_DOMAINS, 64)]; struct vfio_ap_queue_info qinfo[] __counted_by(num_queues); }; + +static void +vfio_ap_release_stop_copy_file(struct vfio_ap_migration_data *mig_data) +{ + /* Stub to be implemented when the mig_data->stop_copy_mig_file.ap_config + * object is allocated. + */ +} + +static void vfio_ap_release_resuming_file(struct vfio_ap_migration_data *mig_data) +{ + /* Stub to be implemented when the mig_data->resuming_mig_file.ap_config + * object is allocated. + */ +} + +static struct file *vfio_ap_set_state(struct vfio_device *vdev, + enum vfio_device_mig_state new_state) +{ + return NULL; +} + +static int vfio_ap_get_state(struct vfio_device *vdev, + enum vfio_device_mig_state *current_state) +{ + return -EOPNOTSUPP; +} + +static int vfio_ap_get_data_size(struct vfio_device *vdev, + unsigned long *stop_copy_length) +{ + return -EOPNOTSUPP; +} + +static const struct vfio_migration_ops vfio_ap_migration_ops = { + .migration_set_state = vfio_ap_set_state, + .migration_get_state = vfio_ap_get_state, + .migration_get_data_size = vfio_ap_get_data_size, +}; + +/** + * vfio_ap_init_migrations_capabilities - initialize migration capabilities + * + * @matrix_mdev: pointer to object containing the mdev state + */ +void vfio_ap_init_migration_capabilities(struct ap_matrix_mdev *matrix_mdev) +{ + if (ap_is_se_guest()) + return; + + matrix_mdev->vdev.migration_flags = VFIO_MIGRATION_STOP_COPY; + matrix_mdev->vdev.mig_ops = &vfio_ap_migration_ops; +} + +/** + * vfio_ap_init_migration_data - initialize migration data and functions + * + * @matrix_mdev: pointer to object containing the mdev state + * + * Return: zero if initialization is successful; otherwise, returns a error. + */ +int vfio_ap_init_migration_data(struct ap_matrix_mdev *matrix_mdev) +{ + struct vfio_ap_migration_data *mig_data; + + lockdep_assert_held(&matrix_dev->mdevs_lock); + + mig_data = kzalloc_obj(struct vfio_ap_migration_data, GFP_KERNEL); + if (!mig_data) + return -ENOMEM; + + mig_data->mig_state = VFIO_DEVICE_STATE_RUNNING; + matrix_mdev->mig_data = mig_data; + + return 0; +} + +/** + * vfio_ap_release_mig_files: + * + * Free the ap_config buffers for any open migration FDs. Although a + * migration FD may still be held open by userspace, it is safe to free + * mig_data here because: + * + * 1. matrix_mdev remains valid for the lifetime of any open migration + * FD via the vfio_device registration reference taken in + * vfio_ap_open_file_stream() and dropped in + * vfio_ap_release_mig_file(). + * + * 2. mig_data is only accessed by the migration file ops + * (vfio_ap_stop_copy_read, vfio_ap_resuming_write) under + * mdevs_lock. Once mig_data is set to NULL by the caller, those + * paths will see NULL and return -ENODEV before dereferencing it. + * + * @matrix_mdev: The object used to maintain the state for a mediated device + */ +static void vfio_ap_release_mig_files(struct ap_matrix_mdev *matrix_mdev) +{ + struct vfio_ap_migration_data *mig_data; + + lockdep_assert_held(&matrix_dev->mdevs_lock); + + mig_data = matrix_mdev->mig_data; + if (!mig_data) + return; + + vfio_ap_release_stop_copy_file(mig_data); + vfio_ap_release_resuming_file(mig_data); +} + +/** + * vfio_ap_release_migration_data: reclaim private migration data + * + * @vdev: pointer to the mdev + */ +void vfio_ap_release_migration_data(struct ap_matrix_mdev *matrix_mdev) +{ + lockdep_assert_held(&matrix_dev->mdevs_lock); + + if (!matrix_mdev->mig_data) + return; + + vfio_ap_release_mig_files(matrix_mdev); + kfree(matrix_mdev->mig_data); + matrix_mdev->mig_data = NULL; +} diff --git a/drivers/s390/crypto/vfio_ap_ops.c b/drivers/s390/crypto/vfio_ap_ops.c index 36786d70a88f..90b0fce0123b 100644 --- a/drivers/s390/crypto/vfio_ap_ops.c +++ b/drivers/s390/crypto/vfio_ap_ops.c @@ -775,18 +775,30 @@ static bool vfio_ap_mdev_filter_matrix(struct ap_matrix_mdev *matrix_mdev, static int vfio_ap_mdev_init_dev(struct vfio_device *vdev) { - struct ap_matrix_mdev *matrix_mdev = - container_of(vdev, struct ap_matrix_mdev, vdev); + struct ap_matrix_mdev *matrix_mdev; + mutex_lock(&matrix_dev->mdevs_lock); + matrix_mdev = container_of(vdev, struct ap_matrix_mdev, vdev); matrix_mdev->mdev = to_mdev_device(vdev->dev); vfio_ap_matrix_init(&matrix_dev->info, &matrix_mdev->matrix); matrix_mdev->pqap_hook = handle_pqap; vfio_ap_matrix_init(&matrix_dev->info, &matrix_mdev->shadow_apcb); hash_init(matrix_mdev->qtable.queues); + mutex_unlock(&matrix_dev->mdevs_lock); return 0; } +static void vfio_ap_mdev_release_dev(struct vfio_device *vdev) +{ + struct ap_matrix_mdev *matrix_mdev; + + mutex_lock(&matrix_dev->mdevs_lock); + matrix_mdev = container_of(vdev, struct ap_matrix_mdev, vdev); + vfio_ap_release_migration_data(matrix_mdev); + mutex_unlock(&matrix_dev->mdevs_lock); +} + static int vfio_ap_mdev_probe(struct mdev_device *mdev) { struct ap_matrix_mdev *matrix_mdev; @@ -797,13 +809,28 @@ static int vfio_ap_mdev_probe(struct mdev_device *mdev) if (IS_ERR(matrix_mdev)) return PTR_ERR(matrix_mdev); + /* + * Migration capabilities must be initialized before calling + * vfio_register_emulated_iommu_dev; otherwise, the VFIO core + * will see mig_ops as NULL during the registration. This could + * prevent the VFIO core from properly setting up migration + * infrastructure like debugfs entries. + * + * This must be done before acquiring mdevs_lock to avoid an ABBA + * deadlock: vfio_register_emulated_iommu_dev() acquires dev_set->lock + * internally, while vfio_ap_mdev_open_device() is called by the VFIO + * core with dev_set->lock already held and then acquires mdevs_lock. + */ + vfio_ap_init_migration_capabilities(matrix_mdev); + ret = vfio_register_emulated_iommu_dev(&matrix_mdev->vdev); if (ret) goto err_put_vdev; + + mutex_lock(&matrix_dev->mdevs_lock); matrix_mdev->req_trigger = NULL; matrix_mdev->cfg_chg_trigger = NULL; dev_set_drvdata(&mdev->dev, matrix_mdev); - mutex_lock(&matrix_dev->mdevs_lock); list_add(&matrix_mdev->node, &matrix_dev->mdev_list); mutex_unlock(&matrix_dev->mdevs_lock); return 0; @@ -2052,19 +2079,39 @@ static int vfio_ap_mdev_reset_qlist(struct list_head *qlist) static int vfio_ap_mdev_open_device(struct vfio_device *vdev) { - struct ap_matrix_mdev *matrix_mdev = - container_of(vdev, struct ap_matrix_mdev, vdev); + struct ap_matrix_mdev *matrix_mdev; + int ret; if (!vdev->kvm) return -EINVAL; - return vfio_ap_mdev_set_kvm(matrix_mdev, vdev->kvm); + mutex_lock(&matrix_dev->mdevs_lock); + matrix_mdev = container_of(vdev, struct ap_matrix_mdev, vdev); + ret = vfio_ap_init_migration_data(matrix_mdev); + mutex_unlock(&matrix_dev->mdevs_lock); + + if (ret) + return ret; + + ret = vfio_ap_mdev_set_kvm(matrix_mdev, vdev->kvm); + if (ret) { + /* Clean up migration data on failure */ + mutex_lock(&matrix_dev->mdevs_lock); + vfio_ap_release_migration_data(matrix_mdev); + mutex_unlock(&matrix_dev->mdevs_lock); + } + + return ret; } static void vfio_ap_mdev_close_device(struct vfio_device *vdev) { - struct ap_matrix_mdev *matrix_mdev = - container_of(vdev, struct ap_matrix_mdev, vdev); + struct ap_matrix_mdev *matrix_mdev; + + mutex_lock(&matrix_dev->mdevs_lock); + matrix_mdev = container_of(vdev, struct ap_matrix_mdev, vdev); + vfio_ap_release_migration_data(matrix_mdev); + mutex_unlock(&matrix_dev->mdevs_lock); vfio_ap_mdev_unset_kvm(matrix_mdev); } @@ -2368,6 +2415,7 @@ static const struct attribute_group vfio_queue_attr_group = { static const struct vfio_device_ops vfio_ap_matrix_dev_ops = { .init = vfio_ap_mdev_init_dev, + .release = vfio_ap_mdev_release_dev, .open_device = vfio_ap_mdev_open_device, .close_device = vfio_ap_mdev_close_device, .ioctl = vfio_ap_mdev_ioctl, diff --git a/drivers/s390/crypto/vfio_ap_private.h b/drivers/s390/crypto/vfio_ap_private.h index 2b542648964b..a2a713f93674 100644 --- a/drivers/s390/crypto/vfio_ap_private.h +++ b/drivers/s390/crypto/vfio_ap_private.h @@ -172,4 +172,8 @@ void vfio_ap_on_cfg_changed(struct ap_config_info *new_config_info, void vfio_ap_on_scan_complete(struct ap_config_info *new_config_info, struct ap_config_info *old_config_info); +void vfio_ap_init_migration_capabilities(struct ap_matrix_mdev *matrix_mdev); +int vfio_ap_init_migration_data(struct ap_matrix_mdev *matrix_mdev); +void vfio_ap_release_migration_data(struct ap_matrix_mdev *matrix_mdev); + #endif /* _VFIO_AP_PRIVATE_H_ */ -- 2.53.0