Queue 0 and subordinate RX queues use different interrupt control interfaces in PHYP:
- queue 0: h_vio_signal() after h_register_logical_lan() - queue N: H_VIOCTL against the queue handle/hwirq mapping The current code is single-queue oriented and cannot safely scale to multiple RX queues in poll completion and open/close IRQ setup. Introduce queue-indexed interrupt helpers and wire them into open()/close()/poll()/interrupt in the same patch: ibmveth_toggle_irq() / enable_irq() / disable_irq() ibmveth_setup_rx_interrupts() / ibmveth_cleanup_rx_interrupts() ibmveth_schedule_rx_queue() These helpers centralize queue0-vs-subordinate dispatch. request_irq() uses &adapter->napi[i] as the per-queue cookie so the handler can resolve the queue index. Move napi_enable() into setup_rx_interrupts() (after LAN registration and buffer-pool allocation): request_irq -> napi_enable. In this single-queue tree, setup does not yet unmask PHYP; the first poll/kick still enables queue 0 via schedule_rx_queue(). Factor process-context RX kicks (open, resume, pool sysfs, netpoll) into ibmveth_schedule_rx_queue(); keep ibmveth_interrupt() as a thin IRQ-only wrapper. cleanup_rx_interrupts() masks PHYP and synchronizes IRQs before napi_disable (intentional storm-safety vs classic close ordering). Runtime remains single-queue (num_rx_queues is still 1). Signed-off-by: Mingming Cao <[email protected]> Reviewed-by: Dave Marquardt <[email protected]> Tested-by: Shaik Abdulla <[email protected]> --- Changes in v4: - Include irq.h / irqdomain.h with first irq_dispose_mapping() use. - Introduce IRQ helpers in the same patch that wires open/close/poll callers, instead of leaving unused statics. - Factor process-context RX kicks into ibmveth_schedule_rx_queue(); keep ibmveth_interrupt() as the IRQ-only wrapper. - On cleanup, mask PHYP and synchronize_irq before napi_disable (storm-safety; not fully behavior-preserving vs classic close). - Leave queue_irq[0] set after cleanup (queue 0 uses netdev->irq; next open reuses it). Only subordinate virqs are disposed. drivers/net/ethernet/ibm/ibmveth.c | 258 +++++++++++++++++++++++++---- 1 file changed, 225 insertions(+), 33 deletions(-) diff --git a/drivers/net/ethernet/ibm/ibmveth.c b/drivers/net/ethernet/ibm/ibmveth.c index 7a2ed49cad4f..664169c4d27a 100644 --- a/drivers/net/ethernet/ibm/ibmveth.c +++ b/drivers/net/ethernet/ibm/ibmveth.c @@ -21,6 +21,8 @@ #include <linux/skbuff.h> #include <linux/init.h> #include <linux/interrupt.h> +#include <linux/irq.h> +#include <linux/irqdomain.h> #include <linux/mm.h> #include <linux/pm.h> #include <linux/ethtool.h> @@ -329,6 +331,203 @@ ibmveth_cleanup_rx_resources(struct ibmveth_adapter *adapter) } } +/** + * ibmveth_toggle_irq - Common helper to enable/disable queue interrupts + * @adapter: ibmveth adapter structure + * @queue_index: Index of the queue (0 for primary, 1+ for subordinate) + * @enable: true to enable, false to disable + * + * For queue 0 (primary), uses h_vio_signal() as it's registered via + * h_register_logical_lan(). For subordinate queues (1+), uses H_VIOCTL + * with H_ENABLE/DISABLE_VIO_INTERRUPT for per-queue interrupt control. + * + * Return: 0 on success, error code otherwise + */ +static int +ibmveth_toggle_irq(struct ibmveth_adapter *adapter, int queue_index, + bool enable) +{ + unsigned long rc; + unsigned long irq = adapter->queue_irq[queue_index]; + const char *action = enable ? "enable" : "disable"; + + if (queue_index == 0) { + /* Primary queue: use h_vio_signal() */ + rc = h_vio_signal(adapter->vdev->unit_address, + enable ? VIO_IRQ_ENABLE : VIO_IRQ_DISABLE); + } else { + /* Subordinate queues: use H_VIOCTL with hardware IRQ */ + struct irq_data *irq_data = irq_get_irq_data(irq); + irq_hw_number_t hwirq; + u64 vioctl_cmd = enable ? H_ENABLE_VIO_INTERRUPT : + H_DISABLE_VIO_INTERRUPT; + + if (!irq_data) { + netdev_err(adapter->netdev, + "Failed to get IRQ data for queue %d (virq=%lu)\n", + queue_index, irq); + return -EINVAL; + } + + hwirq = irqd_to_hwirq(irq_data); + rc = plpar_hcall_norets(H_VIOCTL, + adapter->vdev->unit_address, + vioctl_cmd, + hwirq, 0, 0); + + if (rc == H_PARAMETER) { + /* H_PARAMETER is non-fatal when IRQ is already in + * the requested state. + */ + netdev_warn_once(adapter->netdev, + "H_VIOCTL %s IRQ returned H_PARAMETER for queue %d (hwirq=%lu)\n", + action, queue_index, hwirq); + return 0; + } + } + + if (rc) + netdev_err(adapter->netdev, + "Failed to %s IRQ for queue %d, rc=%ld\n", + action, queue_index, rc); + return rc; +} + +/** + * ibmveth_disable_irq - Disable interrupt for a specific queue + * @adapter: ibmveth adapter structure + * @queue_index: Index of the queue (0 for primary, 1+ for subordinate) + * + * Return: 0 on success, error code otherwise + */ +static int +ibmveth_disable_irq(struct ibmveth_adapter *adapter, int queue_index) +{ + return ibmveth_toggle_irq(adapter, queue_index, false); +} + +/** + * ibmveth_enable_irq - Enable interrupt for a specific queue + * @adapter: ibmveth adapter structure + * @queue_index: Index of the queue (0 for primary, 1+ for subordinate) + * + * Return: 0 on success, error code otherwise + */ +static int +ibmveth_enable_irq(struct ibmveth_adapter *adapter, int queue_index) +{ + return ibmveth_toggle_irq(adapter, queue_index, true); +} + +/** + * ibmveth_setup_rx_interrupts - Register IRQs and enable NAPI + * @adapter: ibmveth adapter structure + * + * Registers interrupt handlers for all RX queues and enables NAPI polling. + * On error, cleans up any successfully registered IRQs before returning. + * + * Return: 0 on success, negative error code on failure + */ +static int +ibmveth_setup_rx_interrupts(struct ibmveth_adapter *adapter) +{ + struct net_device *netdev = adapter->netdev; + int i, rc; + + for (i = 0; i < adapter->num_rx_queues; i++) { + if (!adapter->queue_irq[i]) { + netdev_err(netdev, "queue %d has invalid IRQ (0)\n", i); + rc = -EINVAL; + goto err_free_irqs; + } + + rc = request_irq(adapter->queue_irq[i], ibmveth_interrupt, + 0, netdev->name, &adapter->napi[i]); + if (rc) { + netdev_err(netdev, + "request_irq() failed for irq 0x%x queue %d: %d\n", + adapter->queue_irq[i], i, rc); + goto err_free_irqs; + } + } + + for (i = 0; i < adapter->num_rx_queues; i++) + napi_enable(&adapter->napi[i]); + + return 0; + +err_free_irqs: + while (--i >= 0) + free_irq(adapter->queue_irq[i], &adapter->napi[i]); + return rc; +} + +/** + * ibmveth_cleanup_rx_interrupts - Mask PHYP, disable NAPI, free IRQs + * @adapter: ibmveth adapter structure + * + * Tears down RX interrupt delivery for all queues. Mask PHYP before + * napi_disable so ibmveth_interrupt cannot return IRQ_HANDLED without + * masking (same storm window as scale-down). Safe for close and for + * open failure after setup_rx_interrupts() already unmasked PHYP. + */ +static void +ibmveth_cleanup_rx_interrupts(struct ibmveth_adapter *adapter) +{ + int i; + + for (i = 0; i < adapter->num_rx_queues; i++) { + if (adapter->queue_irq[i]) { + ibmveth_disable_irq(adapter, i); + synchronize_irq(adapter->queue_irq[i]); + } + } + + for (i = 0; i < adapter->num_rx_queues; i++) + napi_disable(&adapter->napi[i]); + + for (i = 0; i < adapter->num_rx_queues; i++) { + if (adapter->queue_irq[i]) + free_irq(adapter->queue_irq[i], &adapter->napi[i]); + } + + /* Dispose IRQ mappings for subordinate queues (1-15). + * Queue 0 uses netdev->irq from device tree, not irq_create_mapping(). + */ + for (i = 1; i < adapter->num_rx_queues; i++) { + if (adapter->queue_irq[i]) { + irq_dispose_mapping(adapter->queue_irq[i]); + adapter->queue_irq[i] = 0; + } + } + + /* Queue 0 uses netdev->irq; leave queue_irq[0] for next open. */ +} + +/** + * ibmveth_schedule_rx_queue - Mask PHYP IRQ and schedule NAPI for one RX queue + * @adapter: ibmveth adapter structure + * @qindex: RX queue index + * + * Shared by the IRQ handler and process-context kick paths (open, resume, + * pool sysfs, netpoll). Keep ibmveth_interrupt() as the IRQ-only wrapper. + */ +static void ibmveth_schedule_rx_queue(struct ibmveth_adapter *adapter, + int qindex) +{ + struct napi_struct *napi = &adapter->napi[qindex]; + unsigned long lpar_rc; + + if (WARN_ON(qindex < 0 || qindex >= adapter->num_rx_queues)) + return; + + if (napi_schedule_prep(napi)) { + lpar_rc = ibmveth_disable_irq(adapter, qindex); + WARN_ON(lpar_rc != H_SUCCESS); + __napi_schedule(napi); + } +} + /* setup the initial settings for a buffer pool */ static void ibmveth_init_buffer_pool(struct ibmveth_buff_pool *pool, u32 pool_index, u32 pool_size, @@ -947,8 +1146,6 @@ static int ibmveth_open(struct net_device *netdev) netdev_dbg(netdev, "open starting\n"); - napi_enable(&adapter->napi[0]); - for (i = 0; i < IBMVETH_NUM_BUFF_POOLS; i++) rxq_entries += adapter->rx_buff_pool[0][i].size; @@ -972,7 +1169,8 @@ static int ibmveth_open(struct net_device *netdev) adapter->rx_queue[0].queue_len; rxq_desc.fields.address = adapter->rx_queue[0].queue_dma; - h_vio_signal(adapter->vdev->unit_address, VIO_IRQ_DISABLE); + adapter->queue_irq[0] = netdev->irq; + ibmveth_disable_irq(adapter, 0); lpar_rc = ibmveth_register_logical_lan(adapter, rxq_desc, mac_address); @@ -993,21 +1191,16 @@ static int ibmveth_open(struct net_device *netdev) if (rc) goto out_free_tx_ltb; - netdev_dbg(netdev, "registering irq 0x%x\n", netdev->irq); - rc = request_irq(netdev->irq, ibmveth_interrupt, 0, netdev->name, - netdev); - if (rc != 0) { - netdev_err(netdev, "unable to request irq 0x%x, rc %d\n", - netdev->irq, rc); + rc = ibmveth_setup_rx_interrupts(adapter); + if (rc) { do { lpar_rc = h_free_logical_lan(adapter->vdev->unit_address); } while (H_IS_LONG_BUSY(lpar_rc) || (lpar_rc == H_BUSY)); - goto out_free_buffer_pools; } netdev_dbg(netdev, "initial replenish cycle\n"); - ibmveth_interrupt(netdev->irq, netdev); + ibmveth_schedule_rx_queue(adapter, 0); netif_tx_start_all_queues(netdev); @@ -1024,7 +1217,6 @@ static int ibmveth_open(struct net_device *netdev) out_free_filter_list: ibmveth_free_filter_list(adapter); out: - napi_disable(&adapter->napi[0]); return rc; } @@ -1036,11 +1228,10 @@ static int ibmveth_close(struct net_device *netdev) netdev_dbg(netdev, "close starting\n"); - napi_disable(&adapter->napi[0]); - netif_tx_stop_all_queues(netdev); - h_vio_signal(adapter->vdev->unit_address, VIO_IRQ_DISABLE); + /* PHYP mask + napi_disable + free_irq live in cleanup_rx_interrupts */ + ibmveth_cleanup_rx_interrupts(adapter); do { lpar_rc = h_free_logical_lan(adapter->vdev->unit_address); @@ -1051,8 +1242,6 @@ static int ibmveth_close(struct net_device *netdev) "continuing with close\n", lpar_rc); } - free_irq(netdev->irq, netdev); - ibmveth_update_rx_no_buffer(adapter); ibmveth_free_buffer_pools(adapter); @@ -1798,15 +1987,14 @@ static int ibmveth_poll(struct napi_struct *napi, int budget) /* We think we are done - reenable interrupts, * then check once more to make sure we are done. */ - lpar_rc = h_vio_signal(adapter->vdev->unit_address, VIO_IRQ_ENABLE); + lpar_rc = ibmveth_enable_irq(adapter, 0); if (WARN_ON(lpar_rc != H_SUCCESS)) { schedule_work(&adapter->work); goto out; } if (ibmveth_rxq_pending_buffer(adapter) && napi_schedule(napi)) { - lpar_rc = h_vio_signal(adapter->vdev->unit_address, - VIO_IRQ_DISABLE); + lpar_rc = ibmveth_disable_irq(adapter, 0); goto restart_poll; } @@ -1816,16 +2004,16 @@ static int ibmveth_poll(struct napi_struct *napi, int budget) static irqreturn_t ibmveth_interrupt(int irq, void *dev_instance) { - struct net_device *netdev = dev_instance; + struct napi_struct *napi = dev_instance; + struct net_device *netdev = napi->dev; struct ibmveth_adapter *adapter = netdev_priv(netdev); - unsigned long lpar_rc; + int qindex; - if (napi_schedule_prep(&adapter->napi[0])) { - lpar_rc = h_vio_signal(adapter->vdev->unit_address, - VIO_IRQ_DISABLE); - WARN_ON(lpar_rc != H_SUCCESS); - __napi_schedule(&adapter->napi[0]); - } + qindex = napi - adapter->napi; + if (WARN_ON(qindex < 0 || qindex >= adapter->num_rx_queues)) + return IRQ_NONE; + + ibmveth_schedule_rx_queue(adapter, qindex); return IRQ_HANDLED; } @@ -1930,8 +2118,10 @@ static int ibmveth_change_mtu(struct net_device *dev, int new_mtu) #ifdef CONFIG_NET_POLL_CONTROLLER static void ibmveth_poll_controller(struct net_device *dev) { - ibmveth_replenish_task(netdev_priv(dev)); - ibmveth_interrupt(dev->irq, dev); + struct ibmveth_adapter *adapter = netdev_priv(dev); + + ibmveth_replenish_task(adapter); + ibmveth_schedule_rx_queue(adapter, 0); } #endif @@ -2343,8 +2533,8 @@ static ssize_t veth_pool_store(struct kobject *kobj, struct attribute *attr, } rtnl_unlock(); - /* kick the interrupt handler to allocate/deallocate pools */ - ibmveth_interrupt(netdev->irq, netdev); + /* kick RX processing to allocate/deallocate pools */ + ibmveth_schedule_rx_queue(adapter, 0); return count; unlock_err: @@ -2384,7 +2574,9 @@ static struct kobj_type ktype_veth_pool = { static int ibmveth_resume(struct device *dev) { struct net_device *netdev = dev_get_drvdata(dev); - ibmveth_interrupt(netdev->irq, netdev); + struct ibmveth_adapter *adapter = netdev_priv(netdev); + + ibmveth_schedule_rx_queue(adapter, 0); return 0; } -- 2.50.1 (Apple Git-155)
