The original code used DMA_ATTR_FORCE_CONTIGUOUS, which could exhaust the CMA pool when a large number of VFs were requested. Fix this by switching to the DMA streaming API. This is equivalent on Octeon platforms, which provide full I/O coherency via the SMMU. Cc: Leon Romanovsky Fixes: 73d33dbc0723 ("octeontx2-af: Use DMA_ATTR_FORCE_CONTIGUOUS attribute in DMA alloc") Signed-off-by: Ratheesh Kannoth --- v9 -> v10: Updated commit message as suggested by Leon v8 -> v9: Addressed Leon comment. - Used kzalloc instead of kmalloc. v7 -> v8: Addressed Leon comments. - Replace __get_free_pages() and __GFP_COMP with kmalloc() - Drop GFP_DMA32 retry loop and dma_capable()/phys_to_dma() mask probing - Use dma_map_single()/dma_unmap_single() instead of dma_map_page_attrs() with DMA_ATTR_REQUIRE_COHERENT - Remove defensive parameter checks and dma_max_mapping_size() from the allocator helper - Move MAX_PAGE_ORDER validation to qmem_alloc() v6 -> v7: Addressed Sashiko comments. https://lore.kernel.org/netdev/178863855246.219967.10510865726694393307@kernel.org/ v5 -> v6: Addressed review comments. https://lore.kernel.org/netdev/20260901015621.2708182-1-rkannoth@marvell.com/ v4 -> v5: Fixed compilation issues. https://lore.kernel.org/netdev/20260831024210.208447-1-rkannoth@marvell.com/ v3 -> v4: Fixed compilation issues. https://lore.kernel.org/netdev/apTpKcN_S1xIwRbZ@rkannoth-OptiPlex-7090/ v2 -> v3: Addressed sashiko comments. https://sashiko.dev/#/patchset/20260825045616.3723078-1-rkannoth%40marvell.com v1 -> v2: Rewrote patch as per sashiko comment. --- .../ethernet/marvell/octeontx2/af/common.h | 45 ++++++++++++++++--- 1 file changed, 39 insertions(+), 6 deletions(-) diff --git a/drivers/net/ethernet/marvell/octeontx2/af/common.h b/drivers/net/ethernet/marvell/octeontx2/af/common.h index 779413a383b7..78e42549d990 100644 --- a/drivers/net/ethernet/marvell/octeontx2/af/common.h +++ b/drivers/net/ethernet/marvell/octeontx2/af/common.h @@ -7,6 +7,10 @@ #ifndef COMMON_H #define COMMON_H +#include +#include +#include + #include "rvu_struct.h" #define OTX2_ALIGN 128 /* Align to cacheline */ @@ -44,6 +48,33 @@ struct qmem { u32 qsize; }; +static inline void *otx2_dma_alloc_coherent(struct device *dev, size_t size, + dma_addr_t *dma_handle) +{ + dma_addr_t dma_addr; + void *vaddr; + + vaddr = kzalloc(size, GFP_KERNEL); + if (!vaddr) + return NULL; + + dma_addr = dma_map_single(dev, vaddr, size, DMA_BIDIRECTIONAL); + if (dma_mapping_error(dev, dma_addr)) { + kfree(vaddr); + return NULL; + } + + *dma_handle = dma_addr; + return vaddr; +} + +static inline void otx2_dma_free_coherent(struct device *dev, size_t size, + void *vaddr, dma_addr_t dma_handle) +{ + dma_unmap_single(dev, dma_handle, size, DMA_BIDIRECTIONAL); + kfree(vaddr); +} + static inline int qmem_alloc(struct device *dev, struct qmem **q, int qsize, int entry_sz) { @@ -60,8 +91,11 @@ static inline int qmem_alloc(struct device *dev, struct qmem **q, qmem->entry_sz = entry_sz; qmem->alloc_sz = (qsize * entry_sz) + OTX2_ALIGN; - qmem->base = dma_alloc_attrs(dev, qmem->alloc_sz, &qmem->iova, - GFP_KERNEL, DMA_ATTR_FORCE_CONTIGUOUS); + + if (get_order(PAGE_ALIGN(qmem->alloc_sz)) > MAX_PAGE_ORDER) + return -ENOMEM; + + qmem->base = otx2_dma_alloc_coherent(dev, qmem->alloc_sz, &qmem->iova); if (!qmem->base) return -ENOMEM; @@ -80,10 +114,9 @@ static inline void qmem_free(struct device *dev, struct qmem *qmem) return; if (qmem->base) - dma_free_attrs(dev, qmem->alloc_sz, - qmem->base - qmem->align, - qmem->iova - qmem->align, - DMA_ATTR_FORCE_CONTIGUOUS); + otx2_dma_free_coherent(dev, qmem->alloc_sz, + qmem->base - qmem->align, + qmem->iova - qmem->align); devm_kfree(dev, qmem); } -- 2.43.0