diff --git a/.mailmap b/.mailmap index 11750e582d09..389b94a0124e 100644 --- a/.mailmap +++ b/.mailmap @@ -977,6 +977,7 @@ Vladimir Davydov Vlastimil Babka WangYuli WangYuli +Wei Wang Weiwen Hu WeiXiong Liao Wen Gong diff --git a/Documentation/netlink/specs/netdev.yaml b/Documentation/netlink/specs/netdev.yaml index 3e3f03bd5c29..0562ba1f567a 100644 --- a/Documentation/netlink/specs/netdev.yaml +++ b/Documentation/netlink/specs/netdev.yaml @@ -851,6 +851,8 @@ operations: name: bind-tx doc: Bind dmabuf to netdev for TX attribute-set: dmabuf + # Intentionally unprivileged (no admin-perm / uns-admin-perm); see + # comment above netdev_nl_bind_tx_doit(). do: request: attributes: diff --git a/MAINTAINERS b/MAINTAINERS index 140eafcbbd78..c52ec2d7d3c1 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -5454,6 +5454,7 @@ F: include/linux/brcmphy.h BROADCOM GENET ETHERNET DRIVER M: Doug Berger M: Florian Fainelli +M: Nicolai Buchwitz R: Broadcom internal kernel review list L: netdev@vger.kernel.org S: Maintained diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index 6e6e2b19815c..2819e001797b 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -1099,6 +1099,11 @@ static void btintel_pcie_msix_tx_handle(struct btintel_pcie_data *data) txq = &data->txq; + if (cr_hia >= txq->count) { + bt_dev_err(data->hdev, "TXQ: invalid cr_hia %u", cr_hia); + return; + } + while (cr_tia != cr_hia) { data->tx_wait_done = true; wake_up(&data->tx_wait_q); @@ -1650,6 +1655,11 @@ static void btintel_pcie_msix_rx_handle(struct btintel_pcie_data *data) rxq = &data->rxq; + if (cr_hia >= rxq->count) { + bt_dev_err(hdev, "RXQ: invalid cr_hia %u", cr_hia); + return; + } + /* The firmware sends multiple CD in a single MSI-X and it needs to * process all received CDs in this interrupt. */ @@ -1657,6 +1667,12 @@ static void btintel_pcie_msix_rx_handle(struct btintel_pcie_data *data) urbd1 = &rxq->urbd1s[cr_tia]; ipc_print_urbd1(data->hdev, urbd1, cr_tia); + if (urbd1->frbd_tag >= rxq->count) { + bt_dev_err(hdev, "RXQ: invalid frbd_tag %u", + urbd1->frbd_tag); + return; + } + buf = &rxq->bufs[urbd1->frbd_tag]; if (!buf) { bt_dev_err(hdev, "RXQ: failed to get the DMA buffer for %d", diff --git a/drivers/bluetooth/btnxpuart.c b/drivers/bluetooth/btnxpuart.c index 25e7b41b349f..4e23d71d00e8 100644 --- a/drivers/bluetooth/btnxpuart.c +++ b/drivers/bluetooth/btnxpuart.c @@ -1388,9 +1388,11 @@ static int nxp_process_fw_dump(struct hci_dev *hdev, struct sk_buff *skb) msecs_to_jiffies(20000)); } - err = hci_devcd_append(hdev, skb_clone(skb, GFP_ATOMIC)); - if (err < 0) - goto free_skb; + if (IS_ENABLED(CONFIG_DEV_COREDUMP)) { + err = hci_devcd_append(hdev, skb_clone(skb, GFP_ATOMIC)); + if (err < 0) + goto free_skb; + } if (buf_len == 0) { bt_dev_warn(hdev, "==== FW dump complete ==="); diff --git a/drivers/dpll/dpll_netlink.c b/drivers/dpll/dpll_netlink.c index 45365214fbef..fb24fd53f2e1 100644 --- a/drivers/dpll/dpll_netlink.c +++ b/drivers/dpll/dpll_netlink.c @@ -1210,8 +1210,7 @@ dpll_pin_ref_sync_state_set(struct dpll_pin *pin, struct dpll_device *dpll; int ret; - ref_sync_pin = xa_find(&pin->ref_sync_pins, &ref_sync_pin_idx, - ULONG_MAX, XA_PRESENT); + ref_sync_pin = xa_load(&pin->ref_sync_pins, ref_sync_pin_idx); if (!ref_sync_pin) { NL_SET_ERR_MSG(extack, "reference sync pin not found"); return -EINVAL; diff --git a/drivers/net/bonding/bond_main.c b/drivers/net/bonding/bond_main.c index a9bff7663eec..de2489c3d9bf 100644 --- a/drivers/net/bonding/bond_main.c +++ b/drivers/net/bonding/bond_main.c @@ -490,7 +490,7 @@ static int bond_ipsec_add_sa(struct net_device *bond_dev, !real_dev->xfrmdev_ops->xdo_dev_state_add || netif_is_bond_master(real_dev)) { NL_SET_ERR_MSG_MOD(extack, "Slave does not support ipsec offload"); - err = -EINVAL; + err = -EOPNOTSUPP; goto out; } diff --git a/drivers/net/dsa/mt7530-mdio.c b/drivers/net/dsa/mt7530-mdio.c index 784dd58a7158..de42f70afcfa 100644 --- a/drivers/net/dsa/mt7530-mdio.c +++ b/drivers/net/dsa/mt7530-mdio.c @@ -227,15 +227,17 @@ mt7530_remove(struct mdio_device *mdiodev) if (!priv) return; - ret = regulator_disable(priv->core_pwr); - if (ret < 0) - dev_err(priv->dev, - "Failed to disable core power: %d\n", ret); + if (priv->id == ID_MT7530) { + ret = regulator_disable(priv->core_pwr); + if (ret < 0) + dev_err(priv->dev, + "Failed to disable core power: %d\n", ret); - ret = regulator_disable(priv->io_pwr); - if (ret < 0) - dev_err(priv->dev, "Failed to disable io pwr: %d\n", - ret); + ret = regulator_disable(priv->io_pwr); + if (ret < 0) + dev_err(priv->dev, "Failed to disable io pwr: %d\n", + ret); + } mt7530_remove_common(priv); diff --git a/drivers/net/dsa/mt7530.c b/drivers/net/dsa/mt7530.c index 3e61eb3c2b1e..96832852c65a 100644 --- a/drivers/net/dsa/mt7530.c +++ b/drivers/net/dsa/mt7530.c @@ -3593,9 +3593,6 @@ EXPORT_SYMBOL_GPL(mt7530_probe_common); void mt7530_remove_common(struct mt7530_priv *priv) { - if (priv->irq_domain) - mt7530_free_mdio_irq(priv); - dsa_unregister_switch(priv->ds); mutex_destroy(&priv->reg_mutex); diff --git a/drivers/net/dsa/mv88e6xxx/chip.c b/drivers/net/dsa/mv88e6xxx/chip.c index 7f68a0c55802..a4a8c7e11bf4 100644 --- a/drivers/net/dsa/mv88e6xxx/chip.c +++ b/drivers/net/dsa/mv88e6xxx/chip.c @@ -5639,6 +5639,68 @@ static const struct mv88e6xxx_ops mv88e6390x_ops = { .pcs_ops = &mv88e6390_pcs_ops, }; +static const struct mv88e6xxx_ops mv88e6191x_ops = { + /* MV88E6XXX_FAMILY_6393 without AVB and PTP: 6191X and 6193X */ + .irl_init_all = mv88e6390_g2_irl_init_all, + .get_eeprom = mv88e6xxx_g2_get_eeprom8, + .set_eeprom = mv88e6xxx_g2_set_eeprom8, + .set_switch_mac = mv88e6xxx_g2_set_switch_mac, + .phy_read = mv88e6xxx_g2_smi_phy_read_c22, + .phy_write = mv88e6xxx_g2_smi_phy_write_c22, + .phy_read_c45 = mv88e6xxx_g2_smi_phy_read_c45, + .phy_write_c45 = mv88e6xxx_g2_smi_phy_write_c45, + .port_set_link = mv88e6xxx_port_set_link, + .port_sync_link = mv88e6xxx_port_sync_link, + .port_set_rgmii_delay = mv88e6390_port_set_rgmii_delay, + .port_set_speed_duplex = mv88e6393x_port_set_speed_duplex, + .port_tag_remap = mv88e6390_port_tag_remap, + .port_set_policy = mv88e6393x_port_set_policy, + .port_set_frame_mode = mv88e6351_port_set_frame_mode, + .port_set_ucast_flood = mv88e6352_port_set_ucast_flood, + .port_set_mcast_flood = mv88e6352_port_set_mcast_flood, + .port_set_ether_type = mv88e6393x_port_set_ether_type, + .port_set_jumbo_size = mv88e6165_port_set_jumbo_size, + .port_egress_rate_limiting = mv88e6097_port_egress_rate_limiting, + .port_pause_limit = mv88e6390_port_pause_limit, + .port_disable_learn_limit = mv88e6xxx_port_disable_learn_limit, + .port_disable_pri_override = mv88e6xxx_port_disable_pri_override, + .port_get_cmode = mv88e6352_port_get_cmode, + .port_set_cmode = mv88e6393x_port_set_cmode, + .port_setup_message_port = mv88e6xxx_setup_message_port, + .port_set_upstream_port = mv88e6393x_port_set_upstream_port, + .port_enable_tcam = mv88e6xxx_port_enable_tcam, + .stats_snapshot = mv88e6390_g1_stats_snapshot, + .stats_set_histogram = mv88e6390_g1_stats_set_histogram, + .stats_get_sset_count = mv88e6320_stats_get_sset_count, + .stats_get_strings = mv88e6320_stats_get_strings, + .stats_get_stat = mv88e6390_stats_get_stat, + /* .set_cpu_port is missing because this family does not support a global + * CPU port, only per port CPU port which is set via + * .port_set_upstream_port method. + */ + .set_egress_port = mv88e6393x_set_egress_port, + .watchdog_ops = &mv88e6393x_watchdog_ops, + .mgmt_rsvd2cpu = mv88e6393x_port_mgmt_rsvd2cpu, + .pot_clear = mv88e6xxx_g2_pot_clear, + .hardware_reset_pre = mv88e6xxx_g2_eeprom_wait, + .hardware_reset_post = mv88e6xxx_g2_eeprom_wait, + .reset = mv88e6352_g1_reset, + .rmu_disable = mv88e6390_g1_rmu_disable, + .atu_get_hash = mv88e6165_g1_atu_get_hash, + .atu_set_hash = mv88e6165_g1_atu_set_hash, + .vtu_getnext = mv88e6390_g1_vtu_getnext, + .vtu_loadpurge = mv88e6390_g1_vtu_loadpurge, + .stu_getnext = mv88e6390_g1_stu_getnext, + .stu_loadpurge = mv88e6390_g1_stu_loadpurge, + .serdes_get_lane = mv88e6393x_serdes_get_lane, + .serdes_irq_mapping = mv88e6390_serdes_irq_mapping, + /* TODO: serdes stats */ + .gpio_ops = &mv88e6352_gpio_ops, + .phylink_get_caps = mv88e6393x_phylink_get_caps, + .pcs_ops = &mv88e6393x_pcs_ops, + .tcam_ops = &mv88e6393_tcam_ops, +}; + static const struct mv88e6xxx_ops mv88e6393x_ops = { /* MV88E6XXX_FAMILY_6393 */ .irl_init_all = mv88e6390_g2_irl_init_all, @@ -6163,8 +6225,7 @@ static const struct mv88e6xxx_info mv88e6xxx_table[] = { .atu_move_port_mask = 0x1f, .pvt = true, .multi_chip = true, - .ptp_support = true, - .ops = &mv88e6393x_ops, + .ops = &mv88e6191x_ops, }, [MV88E6193X] = { @@ -6190,8 +6251,7 @@ static const struct mv88e6xxx_info mv88e6xxx_table[] = { .atu_move_port_mask = 0x1f, .pvt = true, .multi_chip = true, - .ptp_support = true, - .ops = &mv88e6393x_ops, + .ops = &mv88e6191x_ops, }, [MV88E6220] = { diff --git a/drivers/net/ethernet/airoha/airoha_npu.c b/drivers/net/ethernet/airoha/airoha_npu.c index 5bb4817a898d..4d3195eb00f7 100644 --- a/drivers/net/ethernet/airoha/airoha_npu.c +++ b/drivers/net/ethernet/airoha/airoha_npu.c @@ -5,6 +5,7 @@ */ #include +#include #include #include #include @@ -751,12 +752,15 @@ static int airoha_npu_probe(struct platform_device *pdev) if (irq < 0) return irq; + err = devm_work_autocancel(dev, &core->wdt_work, + airoha_npu_wdt_work); + if (err) + return err; + err = devm_request_irq(dev, irq, airoha_npu_wdt_handler, IRQF_SHARED, "airoha-npu-wdt", core); if (err) return err; - - INIT_WORK(&core->wdt_work, airoha_npu_wdt_work); } /* wlan IRQ lines */ @@ -803,18 +807,8 @@ static int airoha_npu_probe(struct platform_device *pdev) return 0; } -static void airoha_npu_remove(struct platform_device *pdev) -{ - struct airoha_npu *npu = platform_get_drvdata(pdev); - int i; - - for (i = 0; i < ARRAY_SIZE(npu->cores); i++) - cancel_work_sync(&npu->cores[i].wdt_work); -} - static struct platform_driver airoha_npu_driver = { .probe = airoha_npu_probe, - .remove = airoha_npu_remove, .driver = { .name = "airoha-npu", .of_match_table = of_airoha_npu_match, diff --git a/drivers/net/ethernet/amazon/ena/ena_netdev.c b/drivers/net/ethernet/amazon/ena/ena_netdev.c index ea89619039d8..7eb6456ed0d5 100644 --- a/drivers/net/ethernet/amazon/ena/ena_netdev.c +++ b/drivers/net/ethernet/amazon/ena/ena_netdev.c @@ -4122,6 +4122,8 @@ static int ena_probe(struct pci_dev *pdev, const struct pci_device_id *ent) err_device_destroy: ena_com_delete_host_info(ena_dev); ena_com_admin_destroy(ena_dev); + ena_phc_destroy(adapter); + ena_com_mmio_reg_read_request_destroy(ena_dev); ena_devlink_destroy: ena_devlink_free(devlink); err_metrics_destroy: diff --git a/drivers/net/ethernet/atheros/atl1c/atl1c_main.c b/drivers/net/ethernet/atheros/atl1c/atl1c_main.c index 7efa3fc257b3..e58f1d2c26bd 100644 --- a/drivers/net/ethernet/atheros/atl1c/atl1c_main.c +++ b/drivers/net/ethernet/atheros/atl1c/atl1c_main.c @@ -1602,6 +1602,9 @@ static int atl1c_clean_tx(struct napi_struct *napi, int budget) AT_READ_REGW(&adapter->hw, atl1c_qregs[tpd_ring->num].tpd_cons, &hw_next_to_clean); + if (unlikely(hw_next_to_clean >= tpd_ring->count)) + hw_next_to_clean = next_to_clean; + while (next_to_clean != hw_next_to_clean) { buffer_info = &tpd_ring->buffer_info[next_to_clean]; if (buffer_info->skb) { diff --git a/drivers/net/ethernet/atheros/atl1e/atl1e_main.c b/drivers/net/ethernet/atheros/atl1e/atl1e_main.c index 40290028580b..437989edb780 100644 --- a/drivers/net/ethernet/atheros/atl1e/atl1e_main.c +++ b/drivers/net/ethernet/atheros/atl1e/atl1e_main.c @@ -1234,6 +1234,9 @@ static bool atl1e_clean_tx_irq(struct atl1e_adapter *adapter) u16 hw_next_to_clean = AT_READ_REGW(&adapter->hw, REG_TPD_CONS_IDX); u16 next_to_clean = atomic_read(&tx_ring->next_to_clean); + if (unlikely(hw_next_to_clean >= tx_ring->count)) + hw_next_to_clean = next_to_clean; + while (next_to_clean != hw_next_to_clean) { tx_buffer = &tx_ring->tx_buffer[next_to_clean]; if (tx_buffer->dma) { diff --git a/drivers/net/ethernet/atheros/atlx/atl1.c b/drivers/net/ethernet/atheros/atlx/atl1.c index 98a4d089270e..957d5598dda5 100644 --- a/drivers/net/ethernet/atheros/atlx/atl1.c +++ b/drivers/net/ethernet/atheros/atlx/atl1.c @@ -2066,6 +2066,9 @@ static int atl1_intr_tx(struct atl1_adapter *adapter) sw_tpd_next_to_clean = atomic_read(&tpd_ring->next_to_clean); cmb_tpd_next_to_clean = le16_to_cpu(adapter->cmb.cmb->tpd_cons_idx); + if (unlikely(cmb_tpd_next_to_clean >= tpd_ring->count)) + cmb_tpd_next_to_clean = sw_tpd_next_to_clean; + while (cmb_tpd_next_to_clean != sw_tpd_next_to_clean) { buffer_info = &tpd_ring->buffer_info[sw_tpd_next_to_clean]; if (buffer_info->dma) { diff --git a/drivers/net/ethernet/broadcom/bnxt/bnxt_ethtool.c b/drivers/net/ethernet/broadcom/bnxt/bnxt_ethtool.c index 62bc9cae613c..622e89587e5d 100644 --- a/drivers/net/ethernet/broadcom/bnxt/bnxt_ethtool.c +++ b/drivers/net/ethernet/broadcom/bnxt/bnxt_ethtool.c @@ -5733,7 +5733,8 @@ const struct ethtool_ops bnxt_ethtool_ops = { .op_needs_rtnl = ETHTOOL_OP_NEEDS_RTNL_SCHANNELS | ETHTOOL_OP_NEEDS_RTNL_SRINGPARAM | ETHTOOL_OP_NEEDS_RTNL_SCOALESCE | - ETHTOOL_OP_NEEDS_RTNL_RSS, + ETHTOOL_OP_NEEDS_RTNL_RSS | + ETHTOOL_OP_NEEDS_RTNL_TEST, .supported_coalesce_params = ETHTOOL_COALESCE_USECS | ETHTOOL_COALESCE_MAX_FRAMES | ETHTOOL_COALESCE_USECS_IRQ | diff --git a/drivers/net/ethernet/broadcom/genet/bcmgenet.c b/drivers/net/ethernet/broadcom/genet/bcmgenet.c index b916080f4ff1..21668e41b696 100644 --- a/drivers/net/ethernet/broadcom/genet/bcmgenet.c +++ b/drivers/net/ethernet/broadcom/genet/bcmgenet.c @@ -852,7 +852,8 @@ static int bcmgenet_get_coalesce(struct net_device *dev, ec->rx_max_coalesced_frames = bcmgenet_rdma_ring_readl(priv, 0, DMA_MBUF_DONE_THRESH); ec->rx_coalesce_usecs = - bcmgenet_rdma_readl(priv, DMA_RING0_TIMEOUT) * 8192 / 1000; + (bcmgenet_rdma_readl(priv, DMA_RING0_TIMEOUT) & + DMA_TIMEOUT_MASK) * 8192 / 1000; for (i = 0; i <= priv->hw_params->rx_queues; i++) { ring = &priv->rx_rings[i]; @@ -1346,9 +1347,8 @@ static void bcmgenet_get_ethtool_stats(struct net_device *dev, p = (char *)&stats64; p += s->stat_offset; - if (sizeof(unsigned long) != sizeof(u32) && - s->stat_sizeof == sizeof(unsigned long)) - data[i] = *(unsigned long *)p; + if (s->stat_sizeof == sizeof(u64)) + data[i] = *(u64 *)p; else data[i] = *(u32 *)p; } @@ -1763,13 +1763,12 @@ static int bcmgenet_power_up(struct bcmgenet_priv *priv, int ret = 0; u32 reg; - if (!bcmgenet_has_ext(priv)) - return ret; - - reg = bcmgenet_ext_readl(priv, EXT_EXT_PWR_MGMT); - switch (mode) { case GENET_POWER_PASSIVE: + if (!bcmgenet_has_ext(priv)) + break; + + reg = bcmgenet_ext_readl(priv, EXT_EXT_PWR_MGMT); reg &= ~(EXT_PWR_DOWN_DLL | EXT_PWR_DOWN_BIAS | EXT_ENERGY_DET_MASK); if (GENET_IS_V5(priv) && !bcmgenet_has_ephy_16nm(priv)) { @@ -1793,8 +1792,12 @@ static int bcmgenet_power_up(struct bcmgenet_priv *priv, break; case GENET_POWER_CABLE_SENSE: + if (!bcmgenet_has_ext(priv)) + break; + /* enable APD */ if (!GENET_IS_V5(priv)) { + reg = bcmgenet_ext_readl(priv, EXT_EXT_PWR_MGMT); reg |= EXT_PWR_DN_EN_LD; bcmgenet_ext_writel(priv, reg, EXT_EXT_PWR_MGMT); } @@ -3441,6 +3444,8 @@ static void bcmgenet_netif_stop(struct net_device *dev, bool stop_phy) { struct bcmgenet_priv *priv = netdev_priv(dev); + /* Stop completion polling before it can wake a stopped queue */ + bcmgenet_disable_tx_napi(priv); netif_tx_disable(dev); /* Disable MAC receive */ @@ -3455,7 +3460,6 @@ static void bcmgenet_netif_stop(struct net_device *dev, bool stop_phy) /* Disable MAC transmit. TX DMA disabled must be done before this */ umac_enable_set(priv, CMD_TX_EN, false); - bcmgenet_disable_tx_napi(priv); bcmgenet_disable_rx_napi(priv); bcmgenet_intr_disable(priv); @@ -3632,6 +3636,9 @@ static int bcmgenet_set_mac_addr(struct net_device *dev, void *p) if (netif_running(dev)) return -EBUSY; + if (!is_valid_ether_addr(addr->sa_data)) + return -EADDRNOTAVAIL; + eth_hw_addr_set(dev, addr->sa_data); return 0; @@ -4135,10 +4142,10 @@ static int bcmgenet_probe(struct platform_device *pdev) priv->rx_rings[i].rx_max_coalesced_frames = 1; /* Initialize u64 stats seq counter for 32bit machines */ - for (i = 0; i <= priv->hw_params->rx_queues; i++) + for (i = 0; i <= GENET_MAX_MQ_CNT; i++) { u64_stats_init(&priv->rx_rings[i].stats64.syncp); - for (i = 0; i <= priv->hw_params->tx_queues; i++) u64_stats_init(&priv->tx_rings[i].stats64.syncp); + } /* libphy will determine the link state */ netif_carrier_off(dev); @@ -4320,6 +4327,8 @@ static int bcmgenet_suspend(struct device *d) netif_device_detach(dev); if (device_may_wakeup(d) && priv->wolopts) { + /* Stop completion polling before it can wake a stopped queue */ + bcmgenet_disable_tx_napi(priv); netif_tx_disable(dev); /* Suspend non-wake Rx data flows */ @@ -4348,7 +4357,6 @@ static int bcmgenet_suspend(struct device *d) netdev_warn(priv->dev, "Timed out while disabling TX DMA\n"); - bcmgenet_disable_tx_napi(priv); bcmgenet_disable_rx_napi(priv); disable_irq(priv->irq1); bcmgenet_tx_reclaim_all(dev); diff --git a/drivers/net/ethernet/broadcom/tg3.c b/drivers/net/ethernet/broadcom/tg3.c index 73a4b569b03e..8b6806a79edf 100644 --- a/drivers/net/ethernet/broadcom/tg3.c +++ b/drivers/net/ethernet/broadcom/tg3.c @@ -17915,11 +17915,14 @@ static int tg3_init_one(struct pci_dev *pdev, err = tg3_get_device_address(tp, addr); if (err) { - dev_err(&pdev->dev, - "Could not obtain valid ethernet address, aborting\n"); - goto err_out_apeunmap; + dev_warn_probe(&pdev->dev, err, + "Could not obtain a valid ethernet address\n"); + if (err == -EPROBE_DEFER) + goto err_out_apeunmap; + eth_hw_addr_random(dev); + } else { + eth_hw_addr_set(dev, addr); } - eth_hw_addr_set(dev, addr); intmbx = MAILBOX_INTERRUPT_0 + TG3_64BIT_REG_LOW; rcvmbx = MAILBOX_RCVRET_CON_IDX_0 + TG3_64BIT_REG_LOW; @@ -18047,6 +18050,10 @@ static int tg3_init_one(struct pci_dev *pdev, return 0; err_out_apeunmap: + if (tg3_flag(tp, USE_PHYLIB)) + tg3_phy_fini(tp); + tg3_mdio_fini(tp); + if (tp->aperegs) { iounmap(tp->aperegs); tp->aperegs = NULL; diff --git a/drivers/net/ethernet/brocade/bna/bnad.c b/drivers/net/ethernet/brocade/bna/bnad.c index 8b75004ba7c9..55dfd4896785 100644 --- a/drivers/net/ethernet/brocade/bna/bnad.c +++ b/drivers/net/ethernet/brocade/bna/bnad.c @@ -2571,6 +2571,22 @@ bnad_ioceth_disable(struct bnad *bnad) return err; } +/* + * The IOC timers rearm one another, so deleting one cannot stop a + * sibling callback from arming it again. Shut them down so a later + * mod_timer() is ignored. + */ +static void +bnad_ioc_timers_shutdown(struct bnad *bnad) +{ + struct bfa_ioc *ioc = &bnad->bna.ioceth.ioc; + + timer_shutdown_sync(&ioc->ioc_timer); + timer_shutdown_sync(&ioc->sem_timer); + timer_shutdown_sync(&ioc->hb_timer); + timer_shutdown_sync(&ioc->iocpf_timer); +} + static int bnad_ioceth_enable(struct bnad *bnad) { @@ -3727,9 +3743,7 @@ bnad_pci_probe(struct pci_dev *pdev, bnad_res_free(bnad, &bnad->mod_res_info[0], BNA_MOD_RES_T_MAX); disable_ioceth: bnad_ioceth_disable(bnad); - timer_delete_sync(&bnad->bna.ioceth.ioc.ioc_timer); - timer_delete_sync(&bnad->bna.ioceth.ioc.sem_timer); - timer_delete_sync(&bnad->bna.ioceth.ioc.hb_timer); + bnad_ioc_timers_shutdown(bnad); spin_lock_irqsave(&bnad->bna_lock, flags); bna_uninit(bna); spin_unlock_irqrestore(&bnad->bna_lock, flags); @@ -3770,9 +3784,7 @@ bnad_pci_remove(struct pci_dev *pdev) mutex_lock(&bnad->conf_mutex); bnad_ioceth_disable(bnad); - timer_delete_sync(&bnad->bna.ioceth.ioc.ioc_timer); - timer_delete_sync(&bnad->bna.ioceth.ioc.sem_timer); - timer_delete_sync(&bnad->bna.ioceth.ioc.hb_timer); + bnad_ioc_timers_shutdown(bnad); spin_lock_irqsave(&bnad->bna_lock, flags); bna_uninit(bna); spin_unlock_irqrestore(&bnad->bna_lock, flags); diff --git a/drivers/net/ethernet/cadence/macb_main.c b/drivers/net/ethernet/cadence/macb_main.c index b8234ac4b602..8e5c034dc3a4 100644 --- a/drivers/net/ethernet/cadence/macb_main.c +++ b/drivers/net/ethernet/cadence/macb_main.c @@ -2749,14 +2749,24 @@ static int macb_alloc(struct macb *bp) size = bp->num_queues * macb_tx_ring_size_per_queue(bp); tx = dma_alloc_coherent(dev, size, &tx_dma, GFP_KERNEL); - if (!tx || upper_32_bits(tx_dma) != upper_32_bits(tx_dma + size - 1)) + if (!tx) + goto out_err; + /* Record the buffer so that the error path frees it. */ + bp->queues[0].tx_ring = tx; + bp->queues[0].tx_ring_dma = tx_dma; + if (upper_32_bits(tx_dma) != upper_32_bits(tx_dma + size - 1)) goto out_err; netdev_dbg(bp->netdev, "Allocated %zu bytes for %u TX rings at %08lx (mapped %p)\n", size, bp->num_queues, (unsigned long)tx_dma, tx); size = bp->num_queues * macb_rx_ring_size_per_queue(bp); rx = dma_alloc_coherent(dev, size, &rx_dma, GFP_KERNEL); - if (!rx || upper_32_bits(rx_dma) != upper_32_bits(rx_dma + size - 1)) + if (!rx) + goto out_err; + /* Record the buffer so that the error path frees it. */ + bp->queues[0].rx_ring = rx; + bp->queues[0].rx_ring_dma = rx_dma; + if (upper_32_bits(rx_dma) != upper_32_bits(rx_dma + size - 1)) goto out_err; netdev_dbg(bp->netdev, "Allocated %zu bytes for %u RX rings at %08lx (mapped %p)\n", size, bp->num_queues, (unsigned long)rx_dma, rx); diff --git a/drivers/net/ethernet/freescale/fman/fman.c b/drivers/net/ethernet/freescale/fman/fman.c index 299bab043175..46cc28895e56 100644 --- a/drivers/net/ethernet/freescale/fman/fman.c +++ b/drivers/net/ethernet/freescale/fman/fman.c @@ -2734,6 +2734,7 @@ static struct fman *read_dts_node(struct platform_device *of_dev) } clk_rate = clk_get_rate(clk); + clk_put(clk); if (!clk_rate) { err = -EINVAL; dev_err(&of_dev->dev, "%s: Failed to determine FM%d clock rate\n", diff --git a/drivers/net/ethernet/google/gve/gve_desc_dqo.h b/drivers/net/ethernet/google/gve/gve_desc_dqo.h index f7786b03c744..d2c86c8eeae2 100644 --- a/drivers/net/ethernet/google/gve/gve_desc_dqo.h +++ b/drivers/net/ethernet/google/gve/gve_desc_dqo.h @@ -14,6 +14,11 @@ #define GVE_TX_MAX_HDR_SIZE_DQO 255 #define GVE_TX_MIN_TSO_MSS_DQO 88 +/* HW limit. This also has to fit in the 14 bits of the mss field of + * struct gve_tx_tso_context_desc_dqo. + */ +#define GVE_TX_MAX_TSO_MSS_DQO 9728 + #ifndef __LITTLE_ENDIAN_BITFIELD #error "Only little endian supported" #endif diff --git a/drivers/net/ethernet/google/gve/gve_tx_dqo.c b/drivers/net/ethernet/google/gve/gve_tx_dqo.c index 80ab0a449ff5..616c1921aebe 100644 --- a/drivers/net/ethernet/google/gve/gve_tx_dqo.c +++ b/drivers/net/ethernet/google/gve/gve_tx_dqo.c @@ -577,15 +577,18 @@ static int gve_prep_tso(struct sk_buff *skb) int header_len; int err; - /* Note: HW requires MSS (gso_size) to be <= 9728 and the total length - * of the TSO to be <= 262143. + /* Note: HW requires the total length of the TSO to be <= 262143, + * this is enforced by netif_set_tso_max_size(). * - * However, we don't validate these because: - * - Hypervisor enforces a limit of 9K MTU - * - Kernel will not produce a TSO larger than 64k + * MSS (gso_size) can not be trusted: packets forwarded from a tap or + * injected by a packet socket can carry an arbitrary value, while the + * mss field of the TSO context descriptor is only 14 bits wide. + * + * A too big MSS is dropped here instead of being rejected from + * gve_features_check_dqo(), because software segmentation would + * produce packets larger than the device can send. */ - - if (unlikely(shinfo->gso_size < GVE_TX_MIN_TSO_MSS_DQO)) + if (unlikely(shinfo->gso_size > GVE_TX_MAX_TSO_MSS_DQO)) return -1; /* Needed because we will modify header. */ @@ -918,13 +921,22 @@ static bool gve_can_send_tso(const struct sk_buff *skb) { const int max_bufs_per_seg = GVE_TX_MAX_DATA_DESCS - 1; const struct skb_shared_info *shinfo = skb_shinfo(skb); - const int header_len = skb_tcp_all_headers(skb); const int gso_size = shinfo->gso_size; int cur_seg_num_bufs; int prev_frag_size; int cur_seg_size; + int header_len; int i; + if (unlikely(gso_size < GVE_TX_MIN_TSO_MSS_DQO)) + return false; + + /* Must match the header length programmed by gve_prep_tso(). */ + if (skb_is_gso_tcp(skb)) + header_len = skb_tcp_all_headers(skb); + else + header_len = skb_transport_offset(skb) + sizeof(struct udphdr); + cur_seg_size = skb_headlen(skb) - header_len; prev_frag_size = skb_headlen(skb); cur_seg_num_bufs = cur_seg_size > 0; @@ -966,7 +978,17 @@ netdev_features_t gve_features_check_dqo(struct sk_buff *skb, struct net_device *dev, netdev_features_t features) { - if (skb_is_gso(skb) && !gve_can_send_tso(skb)) + if (!skb_is_gso(skb)) + return features; + + /* Keep the GSO bits for a too big MSS, so that gve_prep_tso() drops + * the packet: software segmentation would give packets larger than + * the device can send. + */ + if (skb_shinfo(skb)->gso_size > GVE_TX_MAX_TSO_MSS_DQO) + return features; + + if (!gve_can_send_tso(skb)) return features & ~NETIF_F_GSO_MASK; return features; diff --git a/drivers/net/ethernet/hisilicon/hns/hns_dsaf_mac.c b/drivers/net/ethernet/hisilicon/hns/hns_dsaf_mac.c index bc6b269be299..f1cb6d56e4b9 100644 --- a/drivers/net/ethernet/hisilicon/hns/hns_dsaf_mac.c +++ b/drivers/net/ethernet/hisilicon/hns/hns_dsaf_mac.c @@ -793,6 +793,7 @@ static int hns_mac_register_phy(struct hns_mac_cb *mac_cb) dev_err(mac_cb->dev, "mac%d mdio is NULL, dsaf will probe again later\n", mac_cb->mac_id); + put_device(&pdev->dev); return -EPROBE_DEFER; } @@ -801,6 +802,8 @@ static int hns_mac_register_phy(struct hns_mac_cb *mac_cb) dev_dbg(mac_cb->dev, "mac%d register phy addr:%d\n", mac_cb->mac_id, addr); + put_device(&pdev->dev); + return rc; } diff --git a/drivers/net/ethernet/ibm/emac/core.c b/drivers/net/ethernet/ibm/emac/core.c index 1d46cf6c2c12..e7043523457c 100644 --- a/drivers/net/ethernet/ibm/emac/core.c +++ b/drivers/net/ethernet/ibm/emac/core.c @@ -3044,6 +3044,15 @@ static int emac_probe(struct platform_device *ofdev) if (err) goto err_gone; + if (emac_phy_supports_gige(dev->phy_mode)) { + ndev->netdev_ops = &emac_gige_netdev_ops; + dev->commac.ops = &emac_commac_sg_ops; + } else { + ndev->netdev_ops = &emac_netdev_ops; + dev->commac.ops = &emac_commac_ops; + } + ndev->ethtool_ops = &emac_ethtool_ops; + dev->emacp = devm_platform_ioremap_resource(ofdev, 0); if (IS_ERR(dev->emacp)) { err = PTR_ERR(dev->emacp); @@ -3076,7 +3085,6 @@ static int emac_probe(struct platform_device *ofdev) dev->mdio_instance = platform_get_drvdata(dev->mdio_dev); /* Register with MAL */ - dev->commac.ops = &emac_commac_ops; dev->commac.dev = dev; dev->commac.tx_chan_mask = MAL_CHAN_MASK(dev->mal_tx_chan); dev->commac.rx_chan_mask = MAL_CHAN_MASK(dev->mal_rx_chan); @@ -3144,12 +3152,6 @@ static int emac_probe(struct platform_device *ofdev) ndev->features |= ndev->hw_features | NETIF_F_RXCSUM; } ndev->watchdog_timeo = 5 * HZ; - if (emac_phy_supports_gige(dev->phy_mode)) { - ndev->netdev_ops = &emac_gige_netdev_ops; - dev->commac.ops = &emac_commac_sg_ops; - } else - ndev->netdev_ops = &emac_netdev_ops; - ndev->ethtool_ops = &emac_ethtool_ops; /* MTU range: 46 - 1500 or whatever is in OF */ ndev->min_mtu = EMAC_MIN_MTU; diff --git a/drivers/net/ethernet/marvell/octeontx2/af/common.h b/drivers/net/ethernet/marvell/octeontx2/af/common.h index 779413a383b7..78e42549d990 100644 --- a/drivers/net/ethernet/marvell/octeontx2/af/common.h +++ b/drivers/net/ethernet/marvell/octeontx2/af/common.h @@ -7,6 +7,10 @@ #ifndef COMMON_H #define COMMON_H +#include +#include +#include + #include "rvu_struct.h" #define OTX2_ALIGN 128 /* Align to cacheline */ @@ -44,6 +48,33 @@ struct qmem { u32 qsize; }; +static inline void *otx2_dma_alloc_coherent(struct device *dev, size_t size, + dma_addr_t *dma_handle) +{ + dma_addr_t dma_addr; + void *vaddr; + + vaddr = kzalloc(size, GFP_KERNEL); + if (!vaddr) + return NULL; + + dma_addr = dma_map_single(dev, vaddr, size, DMA_BIDIRECTIONAL); + if (dma_mapping_error(dev, dma_addr)) { + kfree(vaddr); + return NULL; + } + + *dma_handle = dma_addr; + return vaddr; +} + +static inline void otx2_dma_free_coherent(struct device *dev, size_t size, + void *vaddr, dma_addr_t dma_handle) +{ + dma_unmap_single(dev, dma_handle, size, DMA_BIDIRECTIONAL); + kfree(vaddr); +} + static inline int qmem_alloc(struct device *dev, struct qmem **q, int qsize, int entry_sz) { @@ -60,8 +91,11 @@ static inline int qmem_alloc(struct device *dev, struct qmem **q, qmem->entry_sz = entry_sz; qmem->alloc_sz = (qsize * entry_sz) + OTX2_ALIGN; - qmem->base = dma_alloc_attrs(dev, qmem->alloc_sz, &qmem->iova, - GFP_KERNEL, DMA_ATTR_FORCE_CONTIGUOUS); + + if (get_order(PAGE_ALIGN(qmem->alloc_sz)) > MAX_PAGE_ORDER) + return -ENOMEM; + + qmem->base = otx2_dma_alloc_coherent(dev, qmem->alloc_sz, &qmem->iova); if (!qmem->base) return -ENOMEM; @@ -80,10 +114,9 @@ static inline void qmem_free(struct device *dev, struct qmem *qmem) return; if (qmem->base) - dma_free_attrs(dev, qmem->alloc_sz, - qmem->base - qmem->align, - qmem->iova - qmem->align, - DMA_ATTR_FORCE_CONTIGUOUS); + otx2_dma_free_coherent(dev, qmem->alloc_sz, + qmem->base - qmem->align, + qmem->iova - qmem->align); devm_kfree(dev, qmem); } diff --git a/drivers/net/ethernet/marvell/octeontx2/af/rvu_debugfs.c b/drivers/net/ethernet/marvell/octeontx2/af/rvu_debugfs.c index 904374baae6f..2927633465d9 100644 --- a/drivers/net/ethernet/marvell/octeontx2/af/rvu_debugfs.c +++ b/drivers/net/ethernet/marvell/octeontx2/af/rvu_debugfs.c @@ -714,110 +714,72 @@ static int get_max_column_width(struct rvu *rvu) } /* Dumps current provisioning status of all RVU block LFs */ -static ssize_t rvu_dbg_rsrc_attach_status(struct file *filp, - char __user *buffer, - size_t count, loff_t *ppos) +static int rvu_dbg_rsrc_attach_status(struct seq_file *filp, void *unused) { - int index, off = 0, flag = 0, len = 0, i = 0; - struct rvu *rvu = filp->private_data; - int bytes_not_copied = 0; + struct rvu *rvu = filp->private; + int index, pf, vf, pcifunc; struct rvu_block block; - int pf, vf, pcifunc; - int buf_size = 2048; int lf_str_size; char *lfs; - char *buf; - /* don't allow partial reads */ - if (*ppos != 0) - return 0; - - buf = kzalloc(buf_size, GFP_KERNEL); - if (!buf) - return -ENOMEM; - - /* Get the maximum width of a column */ lf_str_size = get_max_column_width(rvu); + if (lf_str_size < 0) + return lf_str_size; lfs = kzalloc(lf_str_size, GFP_KERNEL); - if (!lfs) { - kfree(buf); + if (!lfs) return -ENOMEM; - } - off += scnprintf(&buf[off], buf_size - 1 - off, "%-*s", lf_str_size, - "pcifunc"); + + seq_printf(filp, "%-*s", lf_str_size, "pcifunc"); for (index = 0; index < BLK_COUNT; index++) - if (strlen(rvu->hw->block[index].name)) { - off += scnprintf(&buf[off], buf_size - 1 - off, - "%-*s", lf_str_size, - rvu->hw->block[index].name); - } + if (strlen(rvu->hw->block[index].name)) + seq_printf(filp, "%-*s", lf_str_size, + rvu->hw->block[index].name); - off += scnprintf(&buf[off], buf_size - 1 - off, "\n"); - bytes_not_copied = copy_to_user(buffer + (i * off), buf, off); - if (bytes_not_copied) - goto out; - - i++; - *ppos += off; + seq_putc(filp, '\n'); for (pf = 0; pf < rvu->hw->total_pfs; pf++) { for (vf = 0; vf <= rvu->hw->total_vfs; vf++) { - off = 0; - flag = 0; pcifunc = rvu_make_pcifunc(rvu->pdev, pf, vf); if (!pcifunc) continue; - if (vf) { - sprintf(lfs, "PF%d:VF%d", pf, vf - 1); - off = scnprintf(&buf[off], - buf_size - 1 - off, - "%-*s", lf_str_size, lfs); - } else { - sprintf(lfs, "PF%d", pf); - off = scnprintf(&buf[off], - buf_size - 1 - off, - "%-*s", lf_str_size, lfs); - } - for (index = 0; index < BLK_COUNT; index++) { block = rvu->hw->block[index]; if (!strlen(block.name)) continue; - len = 0; - lfs[len] = '\0'; + lfs[0] = '\0'; get_lf_str_list(&block, pcifunc, lfs); if (strlen(lfs)) - flag = 1; - - off += scnprintf(&buf[off], buf_size - 1 - off, - "%-*s", lf_str_size, lfs); + break; } - if (flag) { - off += scnprintf(&buf[off], - buf_size - 1 - off, "\n"); - bytes_not_copied = copy_to_user(buffer + - (i * off), - buf, off); - if (bytes_not_copied) - goto out; + if (index == BLK_COUNT) + continue; - i++; - *ppos += off; + if (vf) + sprintf(lfs, "PF%d:VF%d", pf, vf - 1); + else + sprintf(lfs, "PF%d", pf); + seq_printf(filp, "%-*s", lf_str_size, lfs); + + for (index = 0; index < BLK_COUNT; index++) { + block = rvu->hw->block[index]; + if (!strlen(block.name)) + continue; + + lfs[0] = '\0'; + get_lf_str_list(&block, pcifunc, lfs); + seq_printf(filp, "%-*s", lf_str_size, lfs); } + seq_putc(filp, '\n'); } } -out: kfree(lfs); - kfree(buf); - if (bytes_not_copied) - return -EFAULT; - return *ppos; + return 0; } -RVU_DEBUG_FOPS(rsrc_status, rsrc_attach_status, NULL); +RVU_DEBUG_SEQ_FOPS(rsrc_status, rsrc_attach_status, NULL); static int rvu_dbg_rvu_pf_cgx_map_display(struct seq_file *filp, void *unused) { diff --git a/drivers/net/ethernet/mediatek/mtk_eth_soc.c b/drivers/net/ethernet/mediatek/mtk_eth_soc.c index fd7a49ae88d0..2ea5dfe85539 100644 --- a/drivers/net/ethernet/mediatek/mtk_eth_soc.c +++ b/drivers/net/ethernet/mediatek/mtk_eth_soc.c @@ -4509,6 +4509,10 @@ static int mtk_unreg_dev(struct mtk_eth *eth) mac = netdev_priv(eth->netdev[i]); if (MTK_HAS_CAPS(eth->soc->caps, MTK_QDMA)) unregister_netdevice_notifier(&mac->device_notifier); + + if (eth->netdev[i]->reg_state != NETREG_REGISTERED) + continue; + unregister_netdev(eth->netdev[i]); } @@ -5344,7 +5348,7 @@ static int mtk_probe(struct platform_device *pdev) err = register_netdev(eth->netdev[i]); if (err) { dev_err(eth->dev, "error bringing up device\n"); - goto err_deinit_ppe; + goto err_unreg_netdev; } else netif_info(eth, probe, eth->netdev[i], "mediatek frame engine at 0x%08lx, irq %d\n", diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en/rep/bridge.c b/drivers/net/ethernet/mellanox/mlx5/core/en/rep/bridge.c index baac38bece14..4b7b0a0fc2b2 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/en/rep/bridge.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/en/rep/bridge.c @@ -85,9 +85,16 @@ mlx5_esw_bridge_lower_rep_vport_num_vhca_id_get(struct net_device *dev, struct m struct net_device *lower_dev; struct list_head *iter; - if (netif_is_lag_master(dev) || mlx5e_eswitch_rep(dev)) - return mlx5_esw_bridge_rep_vport_num_vhca_id_get(dev, esw, vport_num, - esw_owner_vhca_id); + if (netif_is_lag_master(dev) || mlx5e_eswitch_rep(dev)) { + struct net_device *rep; + + rep = mlx5_esw_bridge_rep_vport_num_vhca_id_get(dev, esw, vport_num, + esw_owner_vhca_id); + if (rep && !mlx5_esw_bridge_port_exists(*vport_num, *esw_owner_vhca_id, + esw->br_offloads)) + return NULL; + return rep; + } netdev_for_each_lower_dev(dev, lower_dev, iter) { struct net_device *rep; @@ -104,6 +111,28 @@ mlx5_esw_bridge_lower_rep_vport_num_vhca_id_get(struct net_device *dev, struct m return NULL; } +static bool mlx5_esw_bridge_rep_port_lookup(struct net_device *dev, + struct mlx5_esw_bridge_offloads *br_offloads, + u16 *vport_num, u16 *esw_owner_vhca_id) +{ + if (!mlx5_esw_bridge_rep_vport_num_vhca_id_get(dev, br_offloads->esw, vport_num, + esw_owner_vhca_id)) + return false; + + return mlx5_esw_bridge_port_exists(*vport_num, *esw_owner_vhca_id, br_offloads); +} + +static bool mlx5_esw_bridge_lower_rep_port_lookup(struct net_device *dev, + struct mlx5_esw_bridge_offloads *br_offloads, + u16 *vport_num, u16 *esw_owner_vhca_id) +{ + if (!mlx5_esw_bridge_lower_rep_vport_num_vhca_id_get(dev, br_offloads->esw, vport_num, + esw_owner_vhca_id)) + return false; + + return mlx5_esw_bridge_port_exists(*vport_num, *esw_owner_vhca_id, br_offloads); +} + static bool mlx5_esw_bridge_is_local(struct net_device *dev, struct net_device *rep, struct mlx5_eswitch *esw) { @@ -218,8 +247,7 @@ mlx5_esw_bridge_port_obj_add(struct net_device *dev, u16 vport_num, esw_owner_vhca_id; int err; - if (!mlx5_esw_bridge_rep_vport_num_vhca_id_get(dev, br_offloads->esw, &vport_num, - &esw_owner_vhca_id)) + if (!mlx5_esw_bridge_rep_port_lookup(dev, br_offloads, &vport_num, &esw_owner_vhca_id)) return 0; port_obj_info->handled = true; @@ -251,8 +279,7 @@ mlx5_esw_bridge_port_obj_del(struct net_device *dev, const struct switchdev_obj_port_mdb *mdb; u16 vport_num, esw_owner_vhca_id; - if (!mlx5_esw_bridge_rep_vport_num_vhca_id_get(dev, br_offloads->esw, &vport_num, - &esw_owner_vhca_id)) + if (!mlx5_esw_bridge_rep_port_lookup(dev, br_offloads, &vport_num, &esw_owner_vhca_id)) return 0; port_obj_info->handled = true; @@ -283,8 +310,8 @@ mlx5_esw_bridge_port_obj_attr_set(struct net_device *dev, u16 vport_num, esw_owner_vhca_id; int err = 0; - if (!mlx5_esw_bridge_lower_rep_vport_num_vhca_id_get(dev, br_offloads->esw, &vport_num, - &esw_owner_vhca_id)) + if (!mlx5_esw_bridge_lower_rep_port_lookup(dev, br_offloads, &vport_num, + &esw_owner_vhca_id)) return 0; port_attr_info->handled = true; diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en/tc_ct.c b/drivers/net/ethernet/mellanox/mlx5/core/en/tc_ct.c index 6c87a1c7db09..c1841ce74d9a 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/en/tc_ct.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/en/tc_ct.c @@ -1082,6 +1082,9 @@ mlx5_tc_ct_shared_counter_get(struct mlx5_tc_ct_priv *ct_priv, spin_unlock_bh(&ct_priv->ht_lock); + if (rev_entry) + mlx5_tc_ct_entry_put(rev_entry); + create_counter: shared_counter = mlx5_tc_ct_counter_create(ct_priv); diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec_fs.c b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec_fs.c index 329608c59313..8ffa8068e90a 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec_fs.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec_fs.c @@ -1564,14 +1564,14 @@ static void setup_fte_addr6(struct mlx5_flow_spec *spec, memcpy(MLX5_ADDR_OF(fte_match_param, spec->match_value, outer_headers.src_ipv4_src_ipv6.ipv6_layout.ipv6), saddr, 16); memcpy(MLX5_ADDR_OF(fte_match_param, spec->match_criteria, - outer_headers.src_ipv4_src_ipv6.ipv6_layout.ipv6), dmask, 16); + outer_headers.src_ipv4_src_ipv6.ipv6_layout.ipv6), smask, 16); } if (!addr6_all_zero(daddr)) { memcpy(MLX5_ADDR_OF(fte_match_param, spec->match_value, outer_headers.dst_ipv4_dst_ipv6.ipv6_layout.ipv6), daddr, 16); memcpy(MLX5_ADDR_OF(fte_match_param, spec->match_criteria, - outer_headers.dst_ipv4_dst_ipv6.ipv6_layout.ipv6), smask, 16); + outer_headers.dst_ipv4_dst_ipv6.ipv6_layout.ipv6), dmask, 16); } } diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/macsec.c b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/macsec.c index daff53ba7d09..38a3415acf7a 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/macsec.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/macsec.c @@ -1724,6 +1724,8 @@ void mlx5e_macsec_build_netdev(struct mlx5e_priv *priv) mlx5_core_dbg(priv->mdev, "mlx5e: MACsec acceleration enabled\n"); netdev->macsec_ops = &macsec_offload_ops; netdev->features |= NETIF_F_HW_MACSEC; + netdev->hw_features |= NETIF_F_HW_MACSEC; + netdev->vlan_features |= NETIF_F_HW_MACSEC; netif_keep_dst(netdev); } diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_main.c b/drivers/net/ethernet/mellanox/mlx5/core/en_main.c index fc110a7d16e8..4c6060d54fbe 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/en_main.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/en_main.c @@ -5851,7 +5851,6 @@ static void mlx5e_build_nic_netdev(struct net_device *netdev) netdev->vlan_features |= NETIF_F_SG; netdev->vlan_features |= NETIF_F_HW_CSUM; - netdev->vlan_features |= NETIF_F_HW_MACSEC; netdev->vlan_features |= NETIF_F_GRO; netdev->vlan_features |= NETIF_F_TSO; netdev->vlan_features |= NETIF_F_TSO6; diff --git a/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.c b/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.c index 87b5fd349594..b4cf3c5ac0dd 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.c @@ -1649,10 +1649,8 @@ int mlx5_esw_bridge_vport_unlink(struct net_device *br_netdev, u16 vport_num, int err; port = mlx5_esw_bridge_port_lookup(vport_num, esw_owner_vhca_id, br_offloads); - if (!port) { - NL_SET_ERR_MSG_MOD(extack, "Port is not attached to any bridge"); - return -EINVAL; - } + if (!port) + return 0; if (port->bridge->ifindex != br_netdev->ifindex) { NL_SET_ERR_MSG_MOD(extack, "Port is attached to another bridge"); return -EINVAL; @@ -1682,10 +1680,19 @@ int mlx5_esw_bridge_vport_peer_unlink(struct net_device *br_netdev, u16 vport_nu struct mlx5_esw_bridge_offloads *br_offloads, struct netlink_ext_ack *extack) { + if (!MLX5_CAP_ESW(br_offloads->esw->dev, merged_eswitch)) + return 0; + return mlx5_esw_bridge_vport_unlink(br_netdev, vport_num, esw_owner_vhca_id, br_offloads, extack); } +bool mlx5_esw_bridge_port_exists(u16 vport_num, u16 esw_owner_vhca_id, + struct mlx5_esw_bridge_offloads *br_offloads) +{ + return mlx5_esw_bridge_port_lookup(vport_num, esw_owner_vhca_id, br_offloads); +} + int mlx5_esw_bridge_port_vlan_add(u16 vport_num, u16 esw_owner_vhca_id, u16 vid, u16 flags, struct mlx5_esw_bridge_offloads *br_offloads, struct netlink_ext_ack *extack) diff --git a/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.h b/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.h index d6f539161993..a4e59cc21089 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.h +++ b/drivers/net/ethernet/mellanox/mlx5/core/esw/bridge.h @@ -80,6 +80,8 @@ int mlx5_esw_bridge_vlan_proto_set(u16 vport_num, u16 esw_owner_vhca_id, u16 pro struct mlx5_esw_bridge_offloads *br_offloads); int mlx5_esw_bridge_mcast_set(u16 vport_num, u16 esw_owner_vhca_id, bool enable, struct mlx5_esw_bridge_offloads *br_offloads); +bool mlx5_esw_bridge_port_exists(u16 vport_num, u16 esw_owner_vhca_id, + struct mlx5_esw_bridge_offloads *br_offloads); int mlx5_esw_bridge_port_vlan_add(u16 vport_num, u16 esw_owner_vhca_id, u16 vid, u16 flags, struct mlx5_esw_bridge_offloads *br_offloads, struct netlink_ext_ack *extack); diff --git a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c index c655f6e32e9b..dd14cdc378de 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c @@ -1266,25 +1266,45 @@ void mlx5_lag_remove_devices(struct mlx5_lag *ldev) mlx5_lag_remove_devices_filter(ldev, MLX5_LAG_FILTER_PORTS); } +static int mlx5_lag_reload_ib_reps_idx(struct mlx5_lag *ldev, int idx, + u32 flags) +{ + struct lag_func *pf = mlx5_lag_pf(ldev, idx); + struct mlx5_eswitch *esw; + int ret; + + if (pf->dev->priv.flags & flags) + return 0; + + esw = pf->dev->priv.eswitch; + mlx5_esw_reps_block(esw); + ret = mlx5_eswitch_reload_ib_reps(esw); + mlx5_esw_reps_unblock(esw); + + return ret; +} + static int mlx5_lag_reload_ib_reps_unlocked(struct mlx5_lag *ldev, u32 flags, u32 filter, bool cont_on_fail) { - struct lag_func *pf; + int master_idx = mlx5_lag_get_dev_index_by_seq_filter(ldev, MLX5_LAG_P1, + filter); int ret; int i; - mlx5_lag_for_each(i, 0, ldev, filter) { - pf = mlx5_lag_pf(ldev, i); - if (!(pf->dev->priv.flags & flags)) { - struct mlx5_eswitch *esw; + if (master_idx < 0) + return -EINVAL; - esw = pf->dev->priv.eswitch; - mlx5_esw_reps_block(esw); - ret = mlx5_eswitch_reload_ib_reps(esw); - mlx5_esw_reps_unblock(esw); - if (ret && !cont_on_fail) - return ret; - } + ret = mlx5_lag_reload_ib_reps_idx(ldev, master_idx, flags); + if (ret && !cont_on_fail) + return ret; + + mlx5_lag_for_each(i, 0, ldev, filter) { + if (i == master_idx) + continue; + ret = mlx5_lag_reload_ib_reps_idx(ldev, i, flags); + if (ret && !cont_on_fail) + return ret; } return 0; diff --git a/drivers/net/ethernet/mellanox/mlx5/core/lag/shared_fdb.c b/drivers/net/ethernet/mellanox/mlx5/core/lag/shared_fdb.c index 6b4ad3c53f2f..424040918fa3 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/lag/shared_fdb.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/lag/shared_fdb.c @@ -270,6 +270,7 @@ int mlx5_lag_shared_fdb_create(struct mlx5_lag *ldev, pf->sd_fdb_active = false; } mlx5_lag_destroy_single_fdb_filter(ldev, group_id); + mlx5_lag_unload_reps_from_locked(ldev, filter); } err_add_devices: mlx5_lag_add_devices_filter(ldev, filter); diff --git a/drivers/net/ethernet/mellanox/mlx5/core/lib/devcom.c b/drivers/net/ethernet/mellanox/mlx5/core/lib/devcom.c index 64f92427602d..75855481522b 100644 --- a/drivers/net/ethernet/mellanox/mlx5/core/lib/devcom.c +++ b/drivers/net/ethernet/mellanox/mlx5/core/lib/devcom.c @@ -37,6 +37,7 @@ struct mlx5_devcom_comp { struct mlx5_devcom_key key; mlx5_devcom_event_handler_t handler; struct kref ref; + int nr_devs; bool ready; struct rw_semaphore sem; struct lock_class_key lock_key; @@ -170,6 +171,7 @@ devcom_alloc_comp_dev(struct mlx5_devcom_dev *devc, down_write(&comp->sem); list_add_tail(&devcom->list, &comp->comp_dev_list_head); + WRITE_ONCE(comp->nr_devs, comp->nr_devs + 1); up_write(&comp->sem); return devcom; @@ -182,6 +184,7 @@ devcom_free_comp_dev(struct mlx5_devcom_comp_dev *devcom) down_write(&comp->sem); list_del(&devcom->list); + WRITE_ONCE(comp->nr_devs, comp->nr_devs - 1); up_write(&comp->sem); kref_put(&devcom->devc->ref, mlx5_devcom_dev_release); @@ -284,7 +287,7 @@ int mlx5_devcom_comp_get_size(struct mlx5_devcom_comp_dev *devcom) { struct mlx5_devcom_comp *comp = devcom->comp; - return kref_read(&comp->ref); + return READ_ONCE(comp->nr_devs); } int mlx5_devcom_locked_send_event(struct mlx5_devcom_comp_dev *devcom, diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_csr.h b/drivers/net/ethernet/meta/fbnic/fbnic_csr.h index 64b958df7774..baba3471bf5a 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_csr.h +++ b/drivers/net/ethernet/meta/fbnic/fbnic_csr.h @@ -974,6 +974,7 @@ enum { /* PUL User Registers */ #define FBNIC_CSR_START_PUL_USER 0x31000 /* CSR section delimiter */ #define FBNIC_PUL_OB_TLP_HDR_AW_CFG 0x3103d /* 0xc40f4 */ +#define FBNIC_PUL_OB_TLP_HDR_AW_CFG_FLUSH_MODE CSR_BIT(20) #define FBNIC_PUL_OB_TLP_HDR_AW_CFG_FLUSH CSR_BIT(19) #define FBNIC_PUL_OB_TLP_HDR_AW_CFG_BME CSR_BIT(18) #define FBNIC_PUL_OB_TLP_HDR_AW_CFG_RDE_ATTR CSR_GENMASK(17, 15) @@ -1215,6 +1216,10 @@ enum { #define FBNIC_IPC_MBX_DESC_LEN_MASK DESC_GENMASK(63, 48) #define FBNIC_IPC_MBX_DESC_EOM DESC_BIT(46) #define FBNIC_IPC_MBX_DESC_ADDR_MASK DESC_GENMASK(45, 3) +/* Set with FW_CMPL when the FW completed a descriptor without successfully + * processing it (e.g. a mailbox DMA error); the completion has no valid data. + */ +#define FBNIC_IPC_MBX_DESC_FW_ERR DESC_BIT(2) #define FBNIC_IPC_MBX_DESC_FW_CMPL DESC_BIT(1) #define FBNIC_IPC_MBX_DESC_HOST_CMPL DESC_BIT(0) diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_debugfs.c b/drivers/net/ethernet/meta/fbnic/fbnic_debugfs.c index 3c4563c8f403..6edfa0aa69f1 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_debugfs.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_debugfs.c @@ -539,8 +539,8 @@ static void fbnic_dbg_fw_mbx_display(struct seq_file *s, /* Generate header */ seq_puts(s, mbx_idx == FBNIC_IPC_MBX_RX_IDX ? "Rx\n" : "Tx\n"); - seq_printf(s, "Rdy: %d Head: %d Tail: %d\n", - mbx->ready, mbx->head, mbx->tail); + seq_printf(s, "Rdy: %d Head: %d Tail: %d resp_error: %llu\n", + mbx->ready, mbx->head, mbx->tail, mbx->resp_error); snprintf(hdr, sizeof(hdr), "%3s %-4s %s %-12s %s %-3s %-16s\n", "Idx", "Len", "E", "Addr", "F", "H", "Raw"); diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c b/drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c index 0e47088ec44b..76e9a545bb16 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c @@ -313,6 +313,11 @@ fbnic_get_ringparam(struct net_device *netdev, struct ethtool_ringparam *ring, kernel_ring->hds_thresh = fbn->hds_thresh; } +static u32 fbnic_ring_size_pow2(u32 size) +{ + return size ? roundup_pow_of_two(size) : 0; +} + static void fbnic_set_rings(struct fbnic_net *fbn, struct ethtool_ringparam *ring, struct kernel_ethtool_ringparam *kernel_ring) @@ -334,10 +339,10 @@ fbnic_set_ringparam(struct net_device *netdev, struct ethtool_ringparam *ring, struct fbnic_net *clone; int err; - ring->rx_pending = roundup_pow_of_two(ring->rx_pending); - ring->rx_mini_pending = roundup_pow_of_two(ring->rx_mini_pending); - ring->rx_jumbo_pending = roundup_pow_of_two(ring->rx_jumbo_pending); - ring->tx_pending = roundup_pow_of_two(ring->tx_pending); + ring->rx_pending = fbnic_ring_size_pow2(ring->rx_pending); + ring->rx_mini_pending = fbnic_ring_size_pow2(ring->rx_mini_pending); + ring->rx_jumbo_pending = fbnic_ring_size_pow2(ring->rx_jumbo_pending); + ring->tx_pending = fbnic_ring_size_pow2(ring->tx_pending); /* These are absolute minimums allowing the device and driver to operate * but not necessarily guarantee reasonable performance. Settings below @@ -2025,7 +2030,8 @@ static const struct ethtool_ops fbnic_ethtool_ops = { ETHTOOL_OP_NEEDS_RTNL_SPAUSEPARAM | ETHTOOL_OP_NEEDS_RTNL_SCHANNELS | ETHTOOL_OP_NEEDS_RTNL_SRINGPARAM | - ETHTOOL_OP_NEEDS_RTNL_GLINK, + ETHTOOL_OP_NEEDS_RTNL_GLINK | + ETHTOOL_OP_NEEDS_RTNL_TEST, .get_drvinfo = fbnic_get_drvinfo, .get_regs_len = fbnic_get_regs_len, .get_regs = fbnic_get_regs, diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_fw.c b/drivers/net/ethernet/meta/fbnic/fbnic_fw.c index 283d25fae79e..6d7eb8479edf 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_fw.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_fw.c @@ -60,8 +60,15 @@ static void fbnic_mbx_reset_desc_ring(struct fbnic_dev *fbd, int mbx_idx) */ switch (mbx_idx) { case FBNIC_IPC_MBX_RX_IDX: + /* Clearing BME blocks the device from writing to the host + * but leaves the requests parked in the write pipeline. The + * write path only clears outstanding requests when both FLUSH + * and FLUSH_MODE are set; FLUSH_MODE lets them drain without + * landing on the host. + */ wr32(fbd, FBNIC_PUL_OB_TLP_HDR_AW_CFG, - FBNIC_PUL_OB_TLP_HDR_AW_CFG_FLUSH); + FBNIC_PUL_OB_TLP_HDR_AW_CFG_FLUSH | + FBNIC_PUL_OB_TLP_HDR_AW_CFG_FLUSH_MODE); break; case FBNIC_IPC_MBX_TX_IDX: wr32(fbd, FBNIC_PUL_OB_TLP_HDR_AR_CFG, @@ -285,6 +292,12 @@ static void fbnic_mbx_process_tx_msgs(struct fbnic_dev *fbd) if (!(desc & FBNIC_IPC_MBX_DESC_FW_CMPL)) break; + if (desc & FBNIC_IPC_MBX_DESC_FW_ERR) { + tx_mbx->resp_error++; + dev_warn_ratelimited(fbd->dev, + "FW completed a Tx mailbox request with an error\n"); + } + fbnic_mbx_unmap_and_free_msg(fbd, FBNIC_IPC_MBX_TX_IDX, head); head++; @@ -1666,6 +1679,13 @@ static void fbnic_mbx_process_rx_msgs(struct fbnic_dev *fbd) if (!(desc & FBNIC_IPC_MBX_DESC_FW_CMPL)) break; + if (desc & FBNIC_IPC_MBX_DESC_FW_ERR) { + rx_mbx->resp_error++; + dev_warn_ratelimited(fbd->dev, + "FW reported an error on an Rx mailbox message; dropping\n"); + goto next_page; + } + dma_sync_single_for_cpu(fbd->dev, rx_mbx->buf_info[head].addr, FBNIC_RX_PAGE_SIZE, DMA_FROM_DEVICE); @@ -1733,7 +1753,9 @@ void fbnic_mbx_poll(struct fbnic_dev *fbd) int fbnic_mbx_poll_tx_ready(struct fbnic_dev *fbd) { struct fbnic_fw_mbx *tx_mbx = &fbd->mbx[FBNIC_IPC_MBX_TX_IDX]; + struct fbnic_fw_mbx *rx_mbx = &fbd->mbx[FBNIC_IPC_MBX_RX_IDX]; unsigned long timeout = jiffies + 10 * HZ + 1; + u64 tx_resp_error, rx_resp_error; int err, i; do { @@ -1764,6 +1786,9 @@ int fbnic_mbx_poll_tx_ready(struct fbnic_dev *fbd) * mgmt.version once we get the actual version from the firmware * in the capabilities request message. */ +send_cap_req: + tx_resp_error = tx_mbx->resp_error; + rx_resp_error = rx_mbx->resp_error; err = fbnic_fw_xmit_simple_msg(fbd, FBNIC_TLV_MSG_ID_HOST_CAP_REQ); if (err) goto clean_mbx; @@ -1781,9 +1806,27 @@ int fbnic_mbx_poll_tx_ready(struct fbnic_dev *fbd) msleep(20); fbnic_mbx_poll(fbd); + /* A valid capabilities response ends the poll. Check it + * before the FW_ERR retry below so a response parsed in the + * same poll as an unrelated FW_ERR is not discarded. + */ + if (fbd->fw_cap.running.mgmt.version >= MIN_FW_VER_CODE) + break; + /* set err, but wait till mgmt.version check to report it */ - if (!time_is_after_jiffies(timeout)) + if (!time_is_after_jiffies(timeout)) { err = -ETIMEDOUT; + continue; + } + + /* The FW can flag our capabilities request (Tx) or its + * response (Rx) with FW_ERR, in which case it produced no + * usable response. The ring is not wedged, so re-issue the + * request instead of spinning until the timeout. + */ + if (tx_mbx->resp_error != tx_resp_error || + rx_mbx->resp_error != rx_resp_error) + goto send_cap_req; } return 0; diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_fw.h b/drivers/net/ethernet/meta/fbnic/fbnic_fw.h index d84723e4cfa3..5f9969247e30 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_fw.h +++ b/drivers/net/ethernet/meta/fbnic/fbnic_fw.h @@ -13,6 +13,7 @@ struct fbnic_tlv_msg; struct fbnic_fw_mbx { u8 ready, head, tail; + u64 resp_error; struct { struct fbnic_tlv_msg *msg; dma_addr_t addr; diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_pci.c b/drivers/net/ethernet/meta/fbnic/fbnic_pci.c index 8b9bc9e8ea56..c6698e3002a1 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_pci.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_pci.c @@ -434,6 +434,7 @@ static int fbnic_pm_suspend(struct device *dev) { struct fbnic_dev *fbd = dev_get_drvdata(dev); struct net_device *netdev = fbd->netdev; + struct fbnic_net *fbn; if (fbnic_init_failure(fbd)) goto null_uc_addr; @@ -441,11 +442,16 @@ static int fbnic_pm_suspend(struct device *dev) rtnl_lock(); netdev_lock(netdev); + fbn = netdev_priv(netdev); + netif_device_detach(netdev); if (netif_running(netdev)) netdev->netdev_ops->ndo_stop(netdev); + /* The IRQs are about to be freed, so drop the napi vector count */ + fbn->num_napi = 0; + netdev_unlock(netdev); rtnl_unlock(); @@ -508,16 +514,20 @@ static int __fbnic_pm_resume(struct device *dev) if (fbnic_init_failure(fbd)) return 0; + rtnl_lock(); + netdev_lock(netdev); + fbn = netdev_priv(netdev); /* Reset the queues if needed */ fbnic_reset_queues(fbn, fbn->num_tx_queues, fbn->num_rx_queues); - rtnl_lock(); - netdev_lock(netdev); - - if (netif_running(netdev)) + if (netif_running(netdev)) { err = __fbnic_open(fbn); + /* On failure the vectors are freed, so drop the count */ + if (err) + fbn->num_napi = 0; + } netdev_unlock(netdev); rtnl_unlock(); diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c index e7918d3f6aba..10caacffee0f 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -1622,7 +1623,7 @@ fbnic_alloc_qt_page_pools(struct fbnic_net *fbn, struct fbnic_q_triad *qt, return 0; err_destroy_sub0: - page_pool_destroy(pp); + page_pool_destroy(qt->sub0.page_pool); return PTR_ERR(pp); } @@ -1791,7 +1792,7 @@ int fbnic_alloc_napi_vectors(struct fbnic_net *fbn) int err; /* Allocate 1 Tx queue per napi vector */ - if (num_napi < FBNIC_MAX_TXQS && num_napi == num_tx + num_rx) { + if (num_napi <= FBNIC_MAX_TXQS && num_napi == num_tx + num_rx) { while (num_tx) { err = fbnic_alloc_napi_vector(fbd, fbn, num_napi, v_idx, @@ -2853,6 +2854,17 @@ void fbnic_napi_depletion_check(struct net_device *netdev) fbnic_wrfl(fbd); } +/* Returns the napi vector servicing an Rx queue, or NULL if the datapath + * is torn down. The association is published by fbnic_set_netif_napi() + * and cleared by fbnic_reset_netif_napi(), both under the instance lock. + */ +static struct fbnic_napi_vector *fbnic_rxq_nv(struct net_device *dev, int idx) +{ + struct napi_struct *napi = __netif_get_rx_queue(dev, idx)->napi; + + return napi ? container_of(napi, struct fbnic_napi_vector, napi) : NULL; +} + static int fbnic_queue_mem_alloc(struct net_device *dev, struct netdev_queue_config *qcfg, void *qmem, int idx) @@ -2865,8 +2877,16 @@ static int fbnic_queue_mem_alloc(struct net_device *dev, if (!netif_running(dev)) return fbnic_alloc_qt_page_pools(fbn, qt, idx); + /* A failed PCIe recovery or resume can leave the datapath torn down + * while netif_running() is still true. This ndo runs before + * netdev_rx_queue_restart() checks netif_running(), so bail out + * rather than touching rings and vectors that are already freed. + */ + nv = fbnic_rxq_nv(dev, idx); + if (!nv) + return -ENETDOWN; + real = container_of(fbn->rx[idx], struct fbnic_q_triad, cmpl); - nv = fbn->napi[idx % fbn->num_napi]; fbnic_ring_init(&qt->sub0, real->sub0.doorbell, real->sub0.q_idx, real->sub0.flags); @@ -2917,7 +2937,7 @@ static int fbnic_queue_start(struct net_device *dev, struct fbnic_q_triad *real; real = container_of(fbn->rx[idx], struct fbnic_q_triad, cmpl); - nv = fbn->napi[idx % fbn->num_napi]; + nv = fbnic_rxq_nv(dev, idx); fbnic_aggregate_ring_bdq_counters(fbn, &real->sub0); fbnic_aggregate_ring_bdq_counters(fbn, &real->sub1); @@ -2939,7 +2959,7 @@ static int fbnic_queue_stop(struct net_device *dev, void *qmem, int idx) int err; real = container_of(fbn->rx[idx], struct fbnic_q_triad, cmpl); - nv = fbn->napi[idx % fbn->num_napi]; + nv = fbnic_rxq_nv(dev, idx); fbnic_dbg_nv_exit(nv); napi_disable_locked(&nv->napi); diff --git a/drivers/net/ethernet/netronome/nfp/crypto/ipsec.c b/drivers/net/ethernet/netronome/nfp/crypto/ipsec.c index 9e7c285eaa6b..960d7513aa8d 100644 --- a/drivers/net/ethernet/netronome/nfp/crypto/ipsec.c +++ b/drivers/net/ethernet/netronome/nfp/crypto/ipsec.c @@ -625,11 +625,12 @@ int nfp_net_ipsec_rx(struct nfp_meta_parsed *meta, struct sk_buff *skb) xa_lock(&nn->xa_ipsec); x = xa_load(&nn->xa_ipsec, saidx); + if (x) + xfrm_state_hold(x); xa_unlock(&nn->xa_ipsec); if (!x) return -EINVAL; - xfrm_state_hold(x); sp->xvec[sp->len++] = x; sp->olen++; xo = xfrm_offload(skb); diff --git a/drivers/net/ethernet/sfc/efx.c b/drivers/net/ethernet/sfc/efx.c index 3806cd3dd7f4..953f1d90b9b1 100644 --- a/drivers/net/ethernet/sfc/efx.c +++ b/drivers/net/ethernet/sfc/efx.c @@ -904,6 +904,10 @@ static const struct pci_device_id efx_pci_table[] = { .driver_data = (unsigned long)&efx_x4_nic_type}, {PCI_DEVICE(PCI_VENDOR_ID_SOLARFLARE, 0x2c03), /* X4 PF (FF only) */ .driver_data = (unsigned long)&efx_x4_nic_type}, + {PCI_DEVICE(PCI_VENDOR_ID_SOLARFLARE, 0x8c03), /* X4D PF (FF/LL) */ + .driver_data = (unsigned long)&efx_x4_nic_type}, + {PCI_DEVICE(PCI_VENDOR_ID_SOLARFLARE, 0xac03), /* X4D PF (FF only) */ + .driver_data = (unsigned long)&efx_x4_nic_type}, {0} /* end of list */ }; diff --git a/drivers/net/ethernet/spacemit/k1_emac.c b/drivers/net/ethernet/spacemit/k1_emac.c index f7f16397a2c2..d641ac26a1e8 100644 --- a/drivers/net/ethernet/spacemit/k1_emac.c +++ b/drivers/net/ethernet/spacemit/k1_emac.c @@ -803,6 +803,9 @@ static void emac_tx_mem_map(struct emac_priv *priv, struct sk_buff *skb) while (i != head) { emac_free_tx_buf(priv, i); + tx_desc_addr = &((struct emac_desc *)tx_ring->desc_addr)[i]; + memset(tx_desc_addr, 0, sizeof(*tx_desc_addr)); + if (++i == tx_ring->total_cnt) i = 0; } diff --git a/drivers/net/ethernet/stmicro/stmmac/dwmac-rk.c b/drivers/net/ethernet/stmicro/stmmac/dwmac-rk.c index 8d7042e68926..72bdbcb5e863 100644 --- a/drivers/net/ethernet/stmicro/stmmac/dwmac-rk.c +++ b/drivers/net/ethernet/stmicro/stmmac/dwmac-rk.c @@ -1162,8 +1162,11 @@ static int gmac_clk_enable(struct rk_priv_data *bsp_priv, bool enable) return ret; ret = clk_prepare_enable(bsp_priv->clk_phy); - if (ret) + if (ret) { + clk_bulk_disable_unprepare(bsp_priv->num_clks, + bsp_priv->clks); return ret; + } rk_configure_io_clksel(bsp_priv); rk_ungate_rmii_clock(bsp_priv); diff --git a/drivers/net/ethernet/stmicro/stmmac/dwmac4_descs.c b/drivers/net/ethernet/stmicro/stmmac/dwmac4_descs.c index 2994df41ec2c..c6a8f8d73501 100644 --- a/drivers/net/ethernet/stmicro/stmmac/dwmac4_descs.c +++ b/drivers/net/ethernet/stmicro/stmmac/dwmac4_descs.c @@ -474,11 +474,11 @@ static void dwmac4_set_sarc(struct dma_desc *p, u32 sarc_type) sarc_type)); } -static int set_16kib_bfsize(int mtu) +static int set_16kib_bfsize(int len) { int ret = 0; - if (unlikely(mtu >= BUF_SIZE_8KiB)) + if (unlikely(len > BUF_SIZE_8KiB)) ret = BUF_SIZE_16KiB; return ret; } diff --git a/drivers/net/ethernet/stmicro/stmmac/hwif.h b/drivers/net/ethernet/stmicro/stmmac/hwif.h index 9314bcb85c22..857f7562c6c6 100644 --- a/drivers/net/ethernet/stmicro/stmmac/hwif.h +++ b/drivers/net/ethernet/stmicro/stmmac/hwif.h @@ -540,7 +540,7 @@ struct stmmac_mode_ops { bool (*is_jumbo_frm)(unsigned int len, bool enh_desc); int (*jumbo_frm)(struct stmmac_tx_queue *tx_q, struct sk_buff *skb, int csum); - int (*set_16kib_bfsize)(int mtu); + int (*set_16kib_bfsize)(int len); void (*init_desc3)(struct dma_desc *p); void (*refill_desc3)(struct stmmac_rx_queue *rx_q, struct dma_desc *p); void (*clean_desc3)(struct stmmac_tx_queue *tx_q, struct dma_desc *p); diff --git a/drivers/net/ethernet/stmicro/stmmac/ring_mode.c b/drivers/net/ethernet/stmicro/stmmac/ring_mode.c index f7949419eb9f..d2f0c321661d 100644 --- a/drivers/net/ethernet/stmicro/stmmac/ring_mode.c +++ b/drivers/net/ethernet/stmicro/stmmac/ring_mode.c @@ -124,10 +124,10 @@ static void clean_desc3(struct stmmac_tx_queue *tx_q, struct dma_desc *p) p->des3 = 0; } -static int set_16kib_bfsize(int mtu) +static int set_16kib_bfsize(int len) { int ret = 0; - if (unlikely(mtu > BUF_SIZE_8KiB)) + if (unlikely(len > BUF_SIZE_8KiB)) ret = BUF_SIZE_16KiB; return ret; } diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c index 1fb5f804ea23..3f34d491c959 100644 --- a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c @@ -1536,17 +1536,17 @@ static unsigned int stmmac_rx_offset(struct stmmac_priv *priv) return NET_SKB_PAD + NET_IP_ALIGN; } -static int stmmac_set_bfsize(int mtu) +static int stmmac_set_bfsize(int len) { int ret; - if (mtu >= BUF_SIZE_8KiB) + if (len > BUF_SIZE_8KiB) ret = BUF_SIZE_16KiB; - else if (mtu >= BUF_SIZE_4KiB) + else if (len > BUF_SIZE_4KiB) ret = BUF_SIZE_8KiB; - else if (mtu >= BUF_SIZE_2KiB) + else if (len > BUF_SIZE_2KiB) ret = BUF_SIZE_4KiB; - else if (mtu > DEFAULT_BUFSIZE) + else if (len > DEFAULT_BUFSIZE) ret = BUF_SIZE_2KiB; else ret = DEFAULT_BUFSIZE; @@ -4063,7 +4063,7 @@ static struct stmmac_dma_conf * stmmac_setup_dma_desc(struct stmmac_priv *priv, unsigned int mtu) { struct stmmac_dma_conf *dma_conf; - int bfsize, ret; + int bfsize, len, ret; u8 chan; dma_conf = kzalloc_obj(*dma_conf); @@ -4073,13 +4073,15 @@ stmmac_setup_dma_desc(struct stmmac_priv *priv, unsigned int mtu) return ERR_PTR(-ENOMEM); } - /* Returns 0 or BUF_SIZE_16KiB if mtu > 8KiB and dwmac4 or ring mode */ - bfsize = stmmac_set_16kib_bfsize(priv, mtu); + len = mtu + ETH_HLEN + 2 * VLAN_HLEN + ETH_FCS_LEN; + + /* Returns 0 or BUF_SIZE_16KiB if len > 8KiB and dwmac4 or ring mode */ + bfsize = stmmac_set_16kib_bfsize(priv, len); if (bfsize < 0) bfsize = 0; if (bfsize < BUF_SIZE_16KiB) - bfsize = stmmac_set_bfsize(mtu); + bfsize = stmmac_set_bfsize(len); dma_conf->dma_buf_sz = bfsize; /* Chose the tx/rx size from the already defined one in the @@ -5887,6 +5889,7 @@ static int stmmac_rx(struct stmmac_priv *priv, int limit, u32 queue) if (!skb) { page_pool_recycle_direct(rx_q->page_pool, buf->page); + buf->page = NULL; rx_dropped++; count++; goto drain_data; diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_selftests.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_selftests.c index 6372ec7c3f31..c25dc9f89270 100644 --- a/drivers/net/ethernet/stmicro/stmmac/stmmac_selftests.c +++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_selftests.c @@ -12,6 +12,7 @@ #include #include #include +#include #include #include #include @@ -29,6 +30,7 @@ struct stmmachdr { sizeof(struct stmmachdr)) #define STMMAC_TEST_PKT_MAGIC 0xdeadcafecafedeadULL #define STMMAC_LB_TIMEOUT msecs_to_jiffies(200) +#define STMMAC_SFT_MAX_LPI (5 * USEC_PER_SEC) struct stmmac_packet_attrs { int vlan; @@ -237,6 +239,10 @@ struct stmmac_test_priv { struct stmmac_packet_attrs *packet; struct packet_type pt; struct completion comp; + __be16 packet_type; + int (*func)(struct sk_buff *skb, struct net_device *ndev, + struct packet_type *pt, struct net_device *orig_ndev); + bool capture_all; int double_vlan; int vlan_id; int ok; @@ -316,6 +322,52 @@ static int stmmac_test_loopback_validate(struct sk_buff *skb, return 0; } +static int stmmac_sft_filter(struct sk_buff *skb, struct net_device *ndev, + struct packet_type *pt, + struct net_device *orig_ndev) +{ + struct stmmac_test_priv *tpriv = pt->af_packet_priv; + struct ethhdr *hdr = eth_hdr(skb); + int ret = 0; + + if (hdr->h_proto == tpriv->packet_type) { + struct sk_buff *nskb = skb_clone(skb, GFP_ATOMIC); + + if (nskb) + ret = tpriv->func(nskb, ndev, pt, orig_ndev); + } + + kfree_skb(skb); + return ret; +} + +static void stmmac_sft_add_pack(struct packet_type *pt) +{ + struct stmmac_test_priv *tpriv = pt->af_packet_priv; + + if (netdev_uses_dsa(tpriv->pt.dev) || tpriv->capture_all) { + tpriv->packet_type = tpriv->pt.type; + tpriv->func = tpriv->pt.func; + + /* DSA conduit will report ETH_P_XDSA, so our packet handler + * won't match. Let's register a ETH_P_ALL match and filter + * manually in stmmac_sft_filter. This is also useful for + * VLAN tests, to capture packets otherwise marked as + * OTHERHOST. + */ + tpriv->pt.type = htons(ETH_P_ALL); + tpriv->pt.func = stmmac_sft_filter; + tpriv->pt.ignore_outgoing = true; + } + + dev_add_pack(pt); +} + +static void stmmac_sft_remove_pack(struct packet_type *pt) +{ + dev_remove_pack(pt); +} + static int __stmmac_test_loopback(struct stmmac_priv *priv, struct stmmac_packet_attrs *attr) { @@ -337,7 +389,7 @@ static int __stmmac_test_loopback(struct stmmac_priv *priv, tpriv->packet = attr; if (!attr->dont_wait) - dev_add_pack(&tpriv->pt); + stmmac_sft_add_pack(&tpriv->pt); skb = stmmac_test_get_udp_skb(priv, attr); if (!skb) { @@ -360,7 +412,7 @@ static int __stmmac_test_loopback(struct stmmac_priv *priv, cleanup: if (!attr->dont_wait) - dev_remove_pack(&tpriv->pt); + stmmac_sft_remove_pack(&tpriv->pt); kfree(tpriv); return ret; } @@ -414,12 +466,16 @@ static int stmmac_test_mmc(struct stmmac_priv *priv) static int stmmac_test_eee(struct stmmac_priv *priv) { struct stmmac_extra_stats *initial, *final; - int retries = 10; + unsigned long timeout, max_duration; int ret; if (!priv->dma_cap.eee || !priv->eee_active) return -EOPNOTSUPP; + /* Bail out if the configured LPI timer is too long */ + if (priv->tx_lpi_timer > STMMAC_SFT_MAX_LPI) + return -EOPNOTSUPP; + initial = kzalloc_obj(*initial); if (!initial) return -ENOMEM; @@ -430,14 +486,21 @@ static int stmmac_test_eee(struct stmmac_priv *priv) goto out_free_initial; } + /* Snapshot stats, we want to count the in_lpi events. We may enter + * LPI just after the packet was sent. + */ memcpy(initial, &priv->xstats, sizeof(*initial)); + /* Send a frame, then wait to enter LPI */ ret = stmmac_test_mac_loopback(priv); if (ret) goto out_free_final; + max_duration = usecs_to_jiffies(2 * priv->tx_lpi_timer); + /* We have no traffic in the line so, sooner or later it will go LPI */ - while (--retries) { + timeout = jiffies + max_duration; + while (!time_after(jiffies, timeout)) { memcpy(final, &priv->xstats, sizeof(*final)); if (final->irq_tx_path_in_lpi_mode_n > @@ -446,20 +509,38 @@ static int stmmac_test_eee(struct stmmac_priv *priv) msleep(100); } - if (!retries) { + memcpy(final, &priv->xstats, sizeof(*final)); + if (final->irq_tx_path_in_lpi_mode_n <= + initial->irq_tx_path_in_lpi_mode_n) { ret = -ETIMEDOUT; goto out_free_final; } - if (final->irq_tx_path_in_lpi_mode_n <= - initial->irq_tx_path_in_lpi_mode_n) { - ret = -EINVAL; + /* Re-snapshot, as we want to measure exit_lpi events. We should be + * in LPI right now. + */ + memcpy(initial, &priv->xstats, sizeof(*initial)); + + /* TX something so we go out of LPI */ + ret = stmmac_test_mac_loopback(priv); + if (ret) goto out_free_final; + + /* Wait for the exit LPI interrupt */ + timeout = jiffies + max_duration; + while (!time_after(jiffies, timeout)) { + memcpy(final, &priv->xstats, sizeof(*final)); + + if (final->irq_tx_path_exit_lpi_mode_n > + initial->irq_tx_path_exit_lpi_mode_n) + break; + msleep(100); } + memcpy(final, &priv->xstats, sizeof(*final)); if (final->irq_tx_path_exit_lpi_mode_n <= initial->irq_tx_path_exit_lpi_mode_n) { - ret = -EINVAL; + ret = -ETIMEDOUT; goto out_free_final; } @@ -767,7 +848,7 @@ static int stmmac_test_flowctrl(struct stmmac_priv *priv) tpriv->pt.func = stmmac_test_flowctrl_validate; tpriv->pt.dev = priv->dev; tpriv->pt.af_packet_priv = tpriv; - dev_add_pack(&tpriv->pt); + stmmac_sft_add_pack(&tpriv->pt); /* Compute minimum number of packets to make FIFO full */ pkt_count = rx_fifo_size; @@ -823,7 +904,7 @@ static int stmmac_test_flowctrl(struct stmmac_priv *priv) cleanup: dev_mc_del(priv->dev, paddr); dev_set_promiscuity(priv->dev, -1); - dev_remove_pack(&tpriv->pt); + stmmac_sft_remove_pack(&tpriv->pt); kfree(tpriv); return ret; } @@ -865,6 +946,11 @@ static int stmmac_test_vlan_validate(struct sk_buff *skb, goto out; if (skb_headlen(skb) < (STMMAC_TEST_PKT_SIZE - ETH_HLEN)) goto out; + + ehdr = (struct ethhdr *)skb_mac_header(skb); + if (!ether_addr_equal_unaligned(ehdr->h_dest, tpriv->packet->dst)) + goto out; + if (tpriv->vlan_id) { if (skb->vlan_proto != htons(proto)) goto out; @@ -876,10 +962,6 @@ static int stmmac_test_vlan_validate(struct sk_buff *skb, } } - ehdr = (struct ethhdr *)skb_mac_header(skb); - if (!ether_addr_equal_unaligned(ehdr->h_dest, tpriv->packet->dst)) - goto out; - ihdr = ip_hdr(skb); if (tpriv->double_vlan) ihdr = (struct iphdr *)(skb_network_header(skb) + 4); @@ -921,6 +1003,7 @@ static int __stmmac_test_vlanfilt(struct stmmac_priv *priv) tpriv->pt.dev = priv->dev; tpriv->pt.af_packet_priv = tpriv; tpriv->packet = &attr; + tpriv->capture_all = true; /* * As we use HASH filtering, false positives may appear. This is a @@ -928,18 +1011,20 @@ static int __stmmac_test_vlanfilt(struct stmmac_priv *priv) * HASH values. */ tpriv->vlan_id = 0x123; - dev_add_pack(&tpriv->pt); ret = vlan_vid_add(priv->dev, htons(ETH_P_8021Q), tpriv->vlan_id); if (ret) goto cleanup; + attr.vlan = 1; + attr.dst = priv->dev->dev_addr; + attr.sport = 9; + attr.dport = 9; + + stmmac_sft_add_pack(&tpriv->pt); + for (i = 0; i < 4; i++) { - attr.vlan = 1; attr.vlan_id_out = tpriv->vlan_id + i; - attr.dst = priv->dev->dev_addr; - attr.sport = 9; - attr.dport = 9; skb = stmmac_test_get_udp_skb(priv, &attr); if (!skb) { @@ -966,9 +1051,9 @@ static int __stmmac_test_vlanfilt(struct stmmac_priv *priv) } vlan_del: + stmmac_sft_remove_pack(&tpriv->pt); vlan_vid_del(priv->dev, htons(ETH_P_8021Q), tpriv->vlan_id); cleanup: - dev_remove_pack(&tpriv->pt); kfree(tpriv); return ret; } @@ -1015,6 +1100,7 @@ static int __stmmac_test_dvlanfilt(struct stmmac_priv *priv) tpriv->pt.dev = priv->dev; tpriv->pt.af_packet_priv = tpriv; tpriv->packet = &attr; + tpriv->capture_all = true; /* * As we use HASH filtering, false positives may appear. This is a @@ -1022,18 +1108,20 @@ static int __stmmac_test_dvlanfilt(struct stmmac_priv *priv) * HASH values. */ tpriv->vlan_id = 0x123; - dev_add_pack(&tpriv->pt); ret = vlan_vid_add(priv->dev, htons(ETH_P_8021AD), tpriv->vlan_id); if (ret) goto cleanup; + attr.vlan = 2; + attr.dst = priv->dev->dev_addr; + attr.sport = 9; + attr.dport = 9; + + stmmac_sft_add_pack(&tpriv->pt); + for (i = 0; i < 4; i++) { - attr.vlan = 2; attr.vlan_id_out = tpriv->vlan_id + i; - attr.dst = priv->dev->dev_addr; - attr.sport = 9; - attr.dport = 9; skb = stmmac_test_get_udp_skb(priv, &attr); if (!skb) { @@ -1060,9 +1148,9 @@ static int __stmmac_test_dvlanfilt(struct stmmac_priv *priv) } vlan_del: + stmmac_sft_remove_pack(&tpriv->pt); vlan_vid_del(priv->dev, htons(ETH_P_8021AD), tpriv->vlan_id); cleanup: - dev_remove_pack(&tpriv->pt); kfree(tpriv); return ret; } @@ -1293,7 +1381,7 @@ static int stmmac_test_vlanoff_common(struct stmmac_priv *priv, bool svlan) tpriv->pt.af_packet_priv = tpriv; tpriv->packet = &attr; tpriv->vlan_id = 0x123; - dev_add_pack(&tpriv->pt); + tpriv->capture_all = true; ret = vlan_vid_add(priv->dev, htons(proto), tpriv->vlan_id); if (ret) @@ -1301,6 +1389,8 @@ static int stmmac_test_vlanoff_common(struct stmmac_priv *priv, bool svlan) attr.dst = priv->dev->dev_addr; + stmmac_sft_add_pack(&tpriv->pt); + skb = stmmac_test_get_udp_skb(priv, &attr); if (!skb) { ret = -ENOMEM; @@ -1318,9 +1408,9 @@ static int stmmac_test_vlanoff_common(struct stmmac_priv *priv, bool svlan) ret = tpriv->ok ? 0 : -ETIMEDOUT; vlan_del: + stmmac_sft_remove_pack(&tpriv->pt); vlan_vid_del(priv->dev, htons(proto), tpriv->vlan_id); cleanup: - dev_remove_pack(&tpriv->pt); kfree(tpriv); return ret; } @@ -1332,7 +1422,7 @@ static int stmmac_test_vlanoff(struct stmmac_priv *priv) static int stmmac_test_svlanoff(struct stmmac_priv *priv) { - if (!priv->dma_cap.dvlan) + if (!(priv->dev->features & NETIF_F_HW_VLAN_STAG_TX)) return -EOPNOTSUPP; return stmmac_test_vlanoff_common(priv, true); } @@ -1699,6 +1789,9 @@ static int __stmmac_test_jumbo(struct stmmac_priv *priv, u16 queue) struct stmmac_packet_attrs attr = { }; int size = priv->dma_conf.dma_buf_sz; + if (!dwmac_is_xmac(priv->plat->core_type)) + size -= NET_IP_ALIGN; + attr.dst = priv->dev->dev_addr; attr.max_size = size - ETH_FCS_LEN; attr.queue_mapping = queue; diff --git a/drivers/net/ethernet/ti/netcp_core.c b/drivers/net/ethernet/ti/netcp_core.c index eb8fc2ed05f4..4f9e20468bbb 100644 --- a/drivers/net/ethernet/ti/netcp_core.c +++ b/drivers/net/ethernet/ti/netcp_core.c @@ -2225,7 +2225,7 @@ static int netcp_probe(struct platform_device *pdev) return -ENOMEM; pm_runtime_enable(&pdev->dev); - ret = pm_runtime_get_sync(&pdev->dev); + ret = pm_runtime_resume_and_get(&pdev->dev); if (ret < 0) { dev_err(dev, "Failed to enable NETCP power-domain\n"); pm_runtime_disable(&pdev->dev); diff --git a/drivers/net/ethernet/wangxun/libwx/wx_hw.c b/drivers/net/ethernet/wangxun/libwx/wx_hw.c index 122c4952d203..113552586be7 100644 --- a/drivers/net/ethernet/wangxun/libwx/wx_hw.c +++ b/drivers/net/ethernet/wangxun/libwx/wx_hw.c @@ -2518,6 +2518,7 @@ int wx_sw_init(struct wx *wx) } spin_lock_init(&wx->hw_stats_lock); + spin_lock_init(&wx->ptp_tx_lock); mutex_init(&wx->reset_lock); bitmap_zero(wx->state, WX_STATE_NBITS); bitmap_zero(wx->flags, WX_PF_FLAGS_NBITS); diff --git a/drivers/net/ethernet/wangxun/libwx/wx_lib.c b/drivers/net/ethernet/wangxun/libwx/wx_lib.c index ed5aad7857bd..dcbf5811046e 100644 --- a/drivers/net/ethernet/wangxun/libwx/wx_lib.c +++ b/drivers/net/ethernet/wangxun/libwx/wx_lib.c @@ -1200,9 +1200,11 @@ static int wx_tx_map(struct wx_ring *tx_ring, i--; } - dev_kfree_skb_any(first->skb); - first->skb = NULL; - + /* first->skb is released by the caller, which keeps a reference on it + * until the PTP cleanup has compared it against wx->ptp_tx_skb. That + * prevents the address from being reused by a newer request while the + * comparison is pending. + */ tx_ring->next_to_use = i; return -ENOMEM; @@ -1649,9 +1651,11 @@ static netdev_tx_t wx_xmit_frame_ring(struct sk_buff *skb, if (unlikely(skb_shinfo(skb)->tx_flags & SKBTX_HW_TSTAMP) && wx->ptp_clock) { + unsigned long flags; + + spin_lock_irqsave(&wx->ptp_tx_lock, flags); if (wx->tstamp_config.tx_type == HWTSTAMP_TX_ON && - !test_and_set_bit_lock(WX_STATE_PTP_TX_IN_PROGRESS, - wx->state)) { + !test_and_set_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state)) { skb_shinfo(skb)->tx_flags |= SKBTX_IN_PROGRESS; tx_flags |= WX_TX_FLAGS_TSTAMP; wx->ptp_tx_skb = skb_get(skb); @@ -1659,6 +1663,7 @@ static netdev_tx_t wx_xmit_frame_ring(struct sk_buff *skb, } else { wx->tx_hwtstamp_skipped++; } + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); } /* record initial flags and protocol */ @@ -1677,19 +1682,35 @@ static netdev_tx_t wx_xmit_frame_ring(struct sk_buff *skb, wx->atr(tx_ring, first, ptype); if (wx_tx_map(tx_ring, first, hdr_len)) - goto cleanup_tx_tstamp; + goto out_drop; return NETDEV_TX_OK; out_drop: + /* The frame never reached the hardware, so no timestamp will ever be + * reported for it and the request has to be cancelled. The slot is + * shared, though: wx_ptp_clear_tx_timestamp() or wx_ptp_tx_hang() may + * have dropped our request already, and a transmit on another queue + * can have claimed the slot since. Only cancel it while it is still + * ours, otherwise we would free somebody else's skb and release their + * in-progress bit. + */ + if (unlikely(tx_flags & WX_TX_FLAGS_TSTAMP)) { + struct sk_buff *ptp_tx_skb = NULL; + unsigned long flags; + + spin_lock_irqsave(&wx->ptp_tx_lock, flags); + if (wx->ptp_tx_skb == skb) { + ptp_tx_skb = wx->ptp_tx_skb; + wx->ptp_tx_skb = NULL; + clear_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); + wx->tx_hwtstamp_errors++; + } + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + + dev_kfree_skb_any(ptp_tx_skb); + } dev_kfree_skb_any(first->skb); first->skb = NULL; -cleanup_tx_tstamp: - if (unlikely(tx_flags & WX_TX_FLAGS_TSTAMP)) { - dev_kfree_skb_any(wx->ptp_tx_skb); - wx->ptp_tx_skb = NULL; - wx->tx_hwtstamp_errors++; - clear_bit_unlock(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); - } return NETDEV_TX_OK; } diff --git a/drivers/net/ethernet/wangxun/libwx/wx_ptp.c b/drivers/net/ethernet/wangxun/libwx/wx_ptp.c index 4708e7f3958f..65b8937f6e94 100644 --- a/drivers/net/ethernet/wangxun/libwx/wx_ptp.c +++ b/drivers/net/ethernet/wangxun/libwx/wx_ptp.c @@ -129,6 +129,34 @@ static int wx_ptp_settime64(struct ptp_clock_info *ptp, return 0; } +/** + * __wx_ptp_detach_tx_skb - detach the skb tracking the Tx timestamp request + * @wx: the private board structure + * + * Detach the skb of the outstanding request and release the in-progress bit, + * so that a new request can be submitted. + * + * This performs no register access. Callers that need a timestamp the hardware + * may have left latched must unlatch it themselves, while the device is known + * to be alive. wx_ptp_quiesce() runs during PCIe error recovery, where MMIO is + * not reliable, and therefore deliberately skips the unlatch. + * + * Context: Expects wx->ptp_tx_lock to be held by the caller. + * Return: the detached skb, or NULL if no request was outstanding. The caller + * owns the returned reference and must release it once the lock is dropped. + */ +static struct sk_buff *__wx_ptp_detach_tx_skb(struct wx *wx) +{ + struct sk_buff *skb = wx->ptp_tx_skb; + + lockdep_assert_held(&wx->ptp_tx_lock); + + wx->ptp_tx_skb = NULL; + clear_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); + + return skb; +} + /** * wx_ptp_clear_tx_timestamp - utility function to clear Tx timestamp state * @wx: the private board structure @@ -139,12 +167,16 @@ static int wx_ptp_settime64(struct ptp_clock_info *ptp, */ static void wx_ptp_clear_tx_timestamp(struct wx *wx) { + struct sk_buff *skb; + unsigned long flags; + + spin_lock_irqsave(&wx->ptp_tx_lock, flags); + /* Unlatch a timestamp the hardware may have left pending. */ rd32ptp(wx, WX_TSC_1588_STMPH); - if (wx->ptp_tx_skb) { - dev_kfree_skb_any(wx->ptp_tx_skb); - wx->ptp_tx_skb = NULL; - } - clear_bit_unlock(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); + skb = __wx_ptp_detach_tx_skb(wx); + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + + dev_kfree_skb_any(skb); } /** @@ -175,49 +207,54 @@ static void wx_ptp_convert_to_hwtstamp(struct wx *wx, } /** - * wx_ptp_tx_hwtstamp - utility function which checks for TX time stamp + * wx_ptp_tx_hwtstamp_work - check for a pending Tx time stamp * @wx: the private board struct * - * if the timestamp is valid, we convert it into the timecounter ns - * value, then store that result into the shhwtstamps structure which - * is passed up the network stack + * If a Tx timestamp request is outstanding and the hardware has latched a + * valid value, we convert it into the timecounter ns value, then store that + * result into the shhwtstamps structure which is passed up the network stack. + * + * Return: 0 when there is nothing left to poll for, -1 when the timestamp is + * not available yet and the caller should poll again. */ -static void wx_ptp_tx_hwtstamp(struct wx *wx) -{ - struct skb_shared_hwtstamps shhwtstamps; - struct sk_buff *skb = wx->ptp_tx_skb; - u64 regval = 0; - - regval |= (u64)rd32ptp(wx, WX_TSC_1588_STMPL); - regval |= (u64)rd32ptp(wx, WX_TSC_1588_STMPH) << 32; - - wx_ptp_convert_to_hwtstamp(wx, &shhwtstamps, regval); - - wx->ptp_tx_skb = NULL; - clear_bit_unlock(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); - skb_tstamp_tx(skb, &shhwtstamps); - dev_kfree_skb_any(skb); - wx->tx_hwtstamp_pkts++; -} - static int wx_ptp_tx_hwtstamp_work(struct wx *wx) { + struct skb_shared_hwtstamps shhwtstamps; + unsigned long flags; + struct sk_buff *skb; u32 tsynctxctl; + u64 regval = 0; + + spin_lock_irqsave(&wx->ptp_tx_lock, flags); /* we have to have a valid skb to poll for a timestamp */ if (!wx->ptp_tx_skb) { - wx_ptp_clear_tx_timestamp(wx); + rd32ptp(wx, WX_TSC_1588_STMPH); + __wx_ptp_detach_tx_skb(wx); + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); return 0; } /* stop polling once we have a valid timestamp */ tsynctxctl = rd32ptp(wx, WX_TSC_1588_CTL); - if (tsynctxctl & WX_TSC_1588_CTL_VALID) { - wx_ptp_tx_hwtstamp(wx); - return 0; + if (!(tsynctxctl & WX_TSC_1588_CTL_VALID)) { + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + return -1; } - return -1; + regval |= (u64)rd32ptp(wx, WX_TSC_1588_STMPL); + regval |= (u64)rd32ptp(wx, WX_TSC_1588_STMPH) << 32; + skb = wx->ptp_tx_skb; + wx->ptp_tx_skb = NULL; + clear_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + + wx_ptp_convert_to_hwtstamp(wx, &shhwtstamps, regval); + skb_tstamp_tx(skb, &shhwtstamps); + dev_kfree_skb_any(skb); + wx->tx_hwtstamp_pkts++; + + return 0; } /** @@ -296,24 +333,29 @@ static void wx_ptp_rx_hang(struct wx *wx) */ static void wx_ptp_tx_hang(struct wx *wx) { - bool timeout = time_is_before_jiffies(wx->ptp_tx_start + - WX_PTP_TX_TIMEOUT); + struct sk_buff *skb = NULL; + unsigned long flags; - if (!wx->ptp_tx_skb) - return; - - if (!test_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state)) - return; + spin_lock_irqsave(&wx->ptp_tx_lock, flags); /* If we haven't received a timestamp within the timeout, it is * reasonable to assume that it will never occur, so we can unlock the * timestamp bit when this occurs. */ - if (timeout) { - wx_ptp_clear_tx_timestamp(wx); - wx->tx_hwtstamp_timeouts++; - dev_warn(&wx->pdev->dev, "clearing Tx timestamp hang\n"); + if (wx->ptp_tx_skb && + test_bit(WX_STATE_PTP_TX_IN_PROGRESS, wx->state) && + time_is_before_jiffies(wx->ptp_tx_start + WX_PTP_TX_TIMEOUT)) { + rd32ptp(wx, WX_TSC_1588_STMPH); + skb = __wx_ptp_detach_tx_skb(wx); } + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + + if (!skb) + return; + + dev_kfree_skb_any(skb); + wx->tx_hwtstamp_timeouts++; + dev_warn(&wx->pdev->dev, "clearing Tx timestamp hang\n"); } static long wx_ptp_do_aux_work(struct ptp_clock_info *ptp) @@ -841,6 +883,9 @@ EXPORT_SYMBOL(wx_ptp_stop); void wx_ptp_quiesce(struct wx *wx) { + struct sk_buff *skb; + unsigned long flags; + if (!test_and_clear_bit(WX_STATE_PTP_RUNNING, wx->state)) return; @@ -849,11 +894,14 @@ void wx_ptp_quiesce(struct wx *wx) if (wx->ptp_clock) ptp_cancel_worker_sync(wx->ptp_clock); - if (wx->ptp_tx_skb) { - dev_kfree_skb_any(wx->ptp_tx_skb); - wx->ptp_tx_skb = NULL; - } - clear_bit_unlock(WX_STATE_PTP_TX_IN_PROGRESS, wx->state); + /* Drop a pending Tx timestamp request. Do not touch the registers + * here: quiesce runs during PCIe error recovery, where the device may + * already be gone and MMIO is not reliable. + */ + spin_lock_irqsave(&wx->ptp_tx_lock, flags); + skb = __wx_ptp_detach_tx_skb(wx); + spin_unlock_irqrestore(&wx->ptp_tx_lock, flags); + dev_kfree_skb_any(skb); if (wx->ptp_clock) { ptp_clock_unregister(wx->ptp_clock); diff --git a/drivers/net/ethernet/wangxun/libwx/wx_type.h b/drivers/net/ethernet/wangxun/libwx/wx_type.h index 9454e90258d8..afd980dbb793 100644 --- a/drivers/net/ethernet/wangxun/libwx/wx_type.h +++ b/drivers/net/ethernet/wangxun/libwx/wx_type.h @@ -1430,6 +1430,8 @@ struct wx { unsigned long last_overflow_check; unsigned long last_rx_ptp_check; unsigned long ptp_tx_start; + /* protects ptp_tx_skb, ptp_tx_start and the in-progress state bit */ + spinlock_t ptp_tx_lock; seqlock_t hw_tc_lock; /* seqlock for ptp */ struct cyclecounter hw_cc; struct timecounter hw_tc; diff --git a/drivers/net/ethernet/wangxun/txgbe/txgbe_fdir.c b/drivers/net/ethernet/wangxun/txgbe/txgbe_fdir.c index a84010828551..59a47532618c 100644 --- a/drivers/net/ethernet/wangxun/txgbe/txgbe_fdir.c +++ b/drivers/net/ethernet/wangxun/txgbe/txgbe_fdir.c @@ -591,15 +591,24 @@ static void txgbe_fdir_filter_restore(struct wx *wx) queue = TXGBE_RDB_FDIR_DROP_QUEUE; } else { u32 ring = ethtool_get_flow_spec_ring(filter->action); + u8 vf = ethtool_get_flow_spec_ring_vf(filter->action); - if (ring >= wx->num_rx_queues) { + if (!vf && ring >= wx->num_rx_queues) { wx_err(wx, "FDIR restore failed, ring:%u\n", ring); continue; + } else if (vf && (vf > wx->num_vfs || + ring >= wx->num_rx_queues_per_pool)) { + wx_err(wx, "FDIR restore failed, vf:%u, ring:%u\n", + vf, ring); + continue; } /* Map the ring onto the absolute queue index */ - queue = wx->rx_ring[ring]->reg_idx; + if (!vf) + queue = wx->rx_ring[ring]->reg_idx; + else + queue = ((vf - 1) * wx->num_rx_queues_per_pool) + ring; } ret = txgbe_fdir_write_perfect_filter(wx, diff --git a/drivers/net/macsec.c b/drivers/net/macsec.c index 6f9f3aceffaa..78a19b134632 100644 --- a/drivers/net/macsec.c +++ b/drivers/net/macsec.c @@ -3539,6 +3539,22 @@ static int macsec_dev_init(struct net_device *dev) if (err) return err; + err = -ENOMEM; + macsec->stats = netdev_alloc_pcpu_stats(struct pcpu_secy_stats); + if (!macsec->stats) + goto destroy_gro_cells; + + macsec->secy.tx_sc.stats = + netdev_alloc_pcpu_stats(struct pcpu_tx_sc_stats); + if (!macsec->secy.tx_sc.stats) + goto free_secy_stats; + + macsec->secy.tx_sc.md_dst = metadata_dst_alloc(0, METADATA_MACSEC, + GFP_KERNEL); + if (!macsec->secy.tx_sc.md_dst) + goto free_tx_sc_stats; + macsec->secy.tx_sc.md_dst->u.macsec_info.sci = macsec->secy.sci; + macsec_inherit_tso_max(dev); dev->hw_features = real_dev->hw_features & MACSEC_OFFLOAD_FEATURES; @@ -3551,8 +3567,6 @@ static int macsec_dev_init(struct net_device *dev) macsec_set_head_tail_room(dev); - if (is_zero_ether_addr(dev->dev_addr)) - eth_hw_addr_inherit(dev, real_dev); if (is_zero_ether_addr(dev->broadcast)) memcpy(dev->broadcast, real_dev->broadcast, dev->addr_len); @@ -3560,6 +3574,14 @@ static int macsec_dev_init(struct net_device *dev) netdev_hold(real_dev, &macsec->dev_tracker, GFP_KERNEL); return 0; + +free_tx_sc_stats: + free_percpu(macsec->secy.tx_sc.stats); +free_secy_stats: + free_percpu(macsec->stats); +destroy_gro_cells: + gro_cells_destroy(&macsec->gro_cells); + return err; } static void macsec_dev_uninit(struct net_device *dev) @@ -4116,26 +4138,11 @@ static sci_t dev_to_sci(struct net_device *dev, __be16 port) return make_sci(dev->dev_addr, port); } -static int macsec_add_dev(struct net_device *dev, sci_t sci, u8 icv_len) +static void macsec_init_secy(struct net_device *dev, sci_t sci, u8 icv_len) { struct macsec_dev *macsec = macsec_priv(dev); struct macsec_secy *secy = &macsec->secy; - macsec->stats = netdev_alloc_pcpu_stats(struct pcpu_secy_stats); - if (!macsec->stats) - return -ENOMEM; - - secy->tx_sc.stats = netdev_alloc_pcpu_stats(struct pcpu_tx_sc_stats); - if (!secy->tx_sc.stats) - return -ENOMEM; - - secy->tx_sc.md_dst = metadata_dst_alloc(0, METADATA_MACSEC, GFP_KERNEL); - if (!secy->tx_sc.md_dst) - /* macsec and secy percpu stats will be freed when unregistering - * net_device in macsec_free_netdev() - */ - return -ENOMEM; - if (sci == MACSEC_UNDEF_SCI) sci = dev_to_sci(dev, MACSEC_PORT_ES); @@ -4149,15 +4156,12 @@ static int macsec_add_dev(struct net_device *dev, sci_t sci, u8 icv_len) secy->xpn = DEFAULT_XPN; secy->sci = sci; - secy->tx_sc.md_dst->u.macsec_info.sci = sci; secy->tx_sc.active = true; secy->tx_sc.encoding_sa = DEFAULT_ENCODING_SA; secy->tx_sc.encrypt = DEFAULT_ENCRYPT; secy->tx_sc.send_sci = DEFAULT_SEND_SCI; secy->tx_sc.end_station = false; secy->tx_sc.scb = false; - - return 0; } static struct lock_class_key macsec_netdev_addr_lock_key; @@ -4220,6 +4224,24 @@ static int macsec_newlink(struct net_device *dev, if (rx_handler && rx_handler != macsec_handle_frame) return -EBUSY; + if (is_zero_ether_addr(dev->dev_addr)) + eth_hw_addr_inherit(dev, real_dev); + + if (data && data[IFLA_MACSEC_SCI]) + sci = nla_get_sci(data[IFLA_MACSEC_SCI]); + else if (data && data[IFLA_MACSEC_PORT]) + sci = dev_to_sci(dev, nla_get_be16(data[IFLA_MACSEC_PORT])); + else + sci = dev_to_sci(dev, MACSEC_PORT_ES); + + /* Registration can notify listeners before returning. */ + macsec_init_secy(dev, sci, icv_len); + if (data) { + err = macsec_changelink_common(dev, data); + if (err) + return err; + } + err = register_netdevice(dev); if (err < 0) return err; @@ -4232,31 +4254,11 @@ static int macsec_newlink(struct net_device *dev, if (err < 0) goto unregister; - /* need to be already registered so that ->init has run and - * the MAC addr is set - */ - if (data && data[IFLA_MACSEC_SCI]) - sci = nla_get_sci(data[IFLA_MACSEC_SCI]); - else if (data && data[IFLA_MACSEC_PORT]) - sci = dev_to_sci(dev, nla_get_be16(data[IFLA_MACSEC_PORT])); - else - sci = dev_to_sci(dev, MACSEC_PORT_ES); - if (rx_handler && sci_exists(real_dev, sci)) { err = -EBUSY; goto unlink; } - err = macsec_add_dev(dev, sci, icv_len); - if (err) - goto unlink; - - if (data) { - err = macsec_changelink_common(dev, data); - if (err) - goto del_dev; - } - /* If h/w offloading is available, propagate to the device */ if (macsec_is_offloaded(macsec)) { const struct macsec_ops *ops; diff --git a/drivers/net/mdio/mdio-realtek-rtl9300.c b/drivers/net/mdio/mdio-realtek-rtl9300.c index afd52a1cd7f8..9ce2b7807532 100644 --- a/drivers/net/mdio/mdio-realtek-rtl9300.c +++ b/drivers/net/mdio/mdio-realtek-rtl9300.c @@ -88,6 +88,8 @@ #define RTL9310_SMI_INDRT_ACCESS_BC_PHYID_CTRL 0x0c14 #define RTL9310_BC_PORT_ID GENMASK(10, 5) #define RTL9310_SMI_INDRT_ACCESS_CTRL_1 0x0c04 +#define RTL9310_SMI_INDRT_EXT_PAGE GENMASK(8, 0) +#define RTL9310_SMI_INDRT_EXT_PAGE_NO_CHANGE 0x1ff #define RTL9310_SMI_INDRT_ACCESS_CTRL_2_LOW 0x0c08 #define RTL9310_SMI_INDRT_ACCESS_CTRL_2_HIGH 0x0c0c #define RTL9310_SMI_INDRT_ACCESS_CTRL_3 0x0c10 /* I/O fields flipped */ @@ -325,6 +327,8 @@ static int otto_emdio_9310_read_c22(struct mii_bus *bus, int port, int regnum, u .broadcast = FIELD_PREP(RTL9310_BC_PORT_ID, port), .c22_data = FIELD_PREP(RTL9310_PHY_CTRL_REG_ADDR, regnum) | FIELD_PREP(RTL9310_PHY_CTRL_MAIN_PAGE, RAW_PAGE(priv)), + .ext_page = FIELD_PREP(RTL9310_SMI_INDRT_EXT_PAGE, + RTL9310_SMI_INDRT_EXT_PAGE_NO_CHANGE), }; return otto_emdio_read_cmd(bus, RTL9310_PHY_CTRL_TYPE_C22, &cmd_data, @@ -337,6 +341,8 @@ static int otto_emdio_9310_write_c22(struct mii_bus *bus, int port, int regnum, struct otto_emdio_cmd_regs cmd_data = { .c22_data = FIELD_PREP(RTL9310_PHY_CTRL_REG_ADDR, regnum) | FIELD_PREP(RTL9310_PHY_CTRL_MAIN_PAGE, RAW_PAGE(priv)), + .ext_page = FIELD_PREP(RTL9310_SMI_INDRT_EXT_PAGE, + RTL9310_SMI_INDRT_EXT_PAGE_NO_CHANGE), .io_data = FIELD_PREP(RTL9310_PHY_CTRL_INDATA, value), .port_mask_high = (u32)(BIT_ULL(port) >> 32), .port_mask_low = (u32)(BIT_ULL(port)), diff --git a/drivers/net/ovpn/netlink.c b/drivers/net/ovpn/netlink.c index 4dad85294198..5432bc2eb8e8 100644 --- a/drivers/net/ovpn/netlink.c +++ b/drivers/net/ovpn/netlink.c @@ -100,6 +100,8 @@ static bool ovpn_nl_attr_sockaddr_remote(struct nlattr **attrs, struct sockaddr_in6 *sin6; struct sockaddr_in *sin; struct in6_addr *in6; + struct nlattr *scope; + u32 scope_id = 0; __be16 port = 0; __be32 *in; @@ -114,6 +116,9 @@ static bool ovpn_nl_attr_sockaddr_remote(struct nlattr **attrs, } else if (attrs[OVPN_A_PEER_REMOTE_IPV6]) { ss->ss_family = AF_INET6; in6 = nla_data(attrs[OVPN_A_PEER_REMOTE_IPV6]); + scope = attrs[OVPN_A_PEER_REMOTE_IPV6_SCOPE_ID]; + if (scope) + scope_id = nla_get_u32(scope); } else { return false; } @@ -126,6 +131,7 @@ static bool ovpn_nl_attr_sockaddr_remote(struct nlattr **attrs, if (!ipv6_addr_v4mapped(in6)) { sin6 = (struct sockaddr_in6 *)ss; sin6->sin6_port = port; + sin6->sin6_scope_id = scope_id; memcpy(&sin6->sin6_addr, in6, sizeof(*in6)); break; } @@ -179,6 +185,39 @@ static sa_family_t ovpn_nl_family_get(struct nlattr *addr4, return AF_UNSPEC; } +static int ovpn_nl_peer_check_vpn_addrs(const struct in_addr *addr4, + const struct in6_addr *addr6, + struct genl_info *info) +{ + int addr6_type; + + if (addr4->s_addr == htonl(INADDR_ANY) && ipv6_addr_any(addr6)) { + NL_SET_ERR_MSG_MOD(info->extack, + "at least one VPN IP must be configured in MP mode"); + return -EINVAL; + } + + if (ipv4_is_multicast(addr4->s_addr) || ipv4_is_lbcast(addr4->s_addr) || + ipv4_is_loopback(addr4->s_addr)) { + NL_SET_ERR_MSG_MOD(info->extack, + "VPN IPv4 address must be valid unicast or any"); + return -EADDRNOTAVAIL; + } + + if (!ipv6_addr_any(addr6)) { + addr6_type = ipv6_addr_type(addr6); + + if (!(addr6_type & IPV6_ADDR_UNICAST) || + (addr6_type & (IPV6_ADDR_LOOPBACK | IPV6_ADDR_COMPATv4))) { + NL_SET_ERR_MSG_MOD(info->extack, + "VPN IPv6 address must be valid unicast or any"); + return -EADDRNOTAVAIL; + } + } + + return 0; +} + static int ovpn_nl_peer_precheck(struct ovpn_priv *ovpn, struct genl_info *info, struct nlattr **attrs) @@ -346,8 +385,10 @@ static int ovpn_nl_peer_modify(struct ovpn_peer *peer, struct genl_info *info, int ovpn_nl_peer_new_doit(struct sk_buff *skb, struct genl_info *info) { - struct nlattr *attrs[OVPN_A_PEER_MAX + 1]; + struct in_addr vpn_addr4 = { .s_addr = htonl(INADDR_ANY) }; + struct in6_addr vpn_addr6 = IN6ADDR_ANY_INIT; struct ovpn_priv *ovpn = info->user_ptr[0]; + struct nlattr *attrs[OVPN_A_PEER_MAX + 1]; struct ovpn_socket *ovpn_sock; struct socket *sock = NULL; struct ovpn_peer *peer; @@ -371,11 +412,18 @@ int ovpn_nl_peer_new_doit(struct sk_buff *skb, struct genl_info *info) return -EINVAL; /* in MP mode VPN IPs are required for selecting the right peer */ - if (ovpn->mode == OVPN_MODE_MP && !attrs[OVPN_A_PEER_VPN_IPV4] && - !attrs[OVPN_A_PEER_VPN_IPV6]) { - NL_SET_ERR_MSG_FMT_MOD(info->extack, - "VPN IP must be provided in MP mode"); - return -EINVAL; + if (ovpn->mode == OVPN_MODE_MP) { + if (attrs[OVPN_A_PEER_VPN_IPV4]) + vpn_addr4.s_addr = + nla_get_in_addr(attrs[OVPN_A_PEER_VPN_IPV4]); + if (attrs[OVPN_A_PEER_VPN_IPV6]) + vpn_addr6 = + nla_get_in6_addr(attrs[OVPN_A_PEER_VPN_IPV6]); + + ret = ovpn_nl_peer_check_vpn_addrs(&vpn_addr4, &vpn_addr6, + info); + if (ret < 0) + return ret; } peer_id = nla_get_u32(attrs[OVPN_A_PEER_ID]); @@ -474,8 +522,10 @@ int ovpn_nl_peer_new_doit(struct sk_buff *skb, struct genl_info *info) int ovpn_nl_peer_set_doit(struct sk_buff *skb, struct genl_info *info) { - struct nlattr *attrs[OVPN_A_PEER_MAX + 1]; struct ovpn_priv *ovpn = info->user_ptr[0]; + struct nlattr *attrs[OVPN_A_PEER_MAX + 1]; + struct in6_addr vpn_addr6; + struct in_addr vpn_addr4; struct ovpn_socket *sock; struct ovpn_peer *peer; u32 peer_id; @@ -522,28 +572,58 @@ int ovpn_nl_peer_set_doit(struct sk_buff *skb, struct genl_info *info) rcu_read_unlock(); spin_lock_bh(&ovpn->lock); - ret = ovpn_nl_peer_modify(peer, info, attrs); - if (ret < 0) { - spin_unlock_bh(&ovpn->lock); - ovpn_peer_put(peer); - return ret; + + vpn_addr4 = peer->vpn_addrs.ipv4; + vpn_addr6 = peer->vpn_addrs.ipv6; + + /* reject peer with conflicting VPN address */ + if (attrs[OVPN_A_PEER_VPN_IPV4]) { + vpn_addr4.s_addr = nla_get_in_addr(attrs[OVPN_A_PEER_VPN_IPV4]); + if (ovpn_peer_vpn_addr_conflict4(ovpn, peer, &vpn_addr4)) + goto addr_conflict; } + if (attrs[OVPN_A_PEER_VPN_IPV6]) { + vpn_addr6 = nla_get_in6_addr(attrs[OVPN_A_PEER_VPN_IPV6]); + if (ovpn_peer_vpn_addr_conflict6(ovpn, peer, &vpn_addr6)) + goto addr_conflict; + } + + /* in MP mode VPN IPs are required for selecting the right peer */ + if (ovpn->mode == OVPN_MODE_MP) { + ret = ovpn_nl_peer_check_vpn_addrs(&vpn_addr4, &vpn_addr6, + info); + if (ret < 0) + goto unlock; + } + + ret = ovpn_nl_peer_modify(peer, info, attrs); + if (ret < 0) + goto unlock; /* ret == 1 means that VPN IPv4/6 has been modified and rehashing * is required */ - if (ret > 0) + if (ret > 0) { ovpn_peer_hash_vpn_ip(peer); + ret = 0; + } /* if the remote endpoint was updated, the by_transp_addr hash bucket * also needs to be refreshed, otherwise incoming packets from the new * remote address would fail the lockless lookup */ if (attrs[OVPN_A_PEER_REMOTE_IPV4] || attrs[OVPN_A_PEER_REMOTE_IPV6]) ovpn_peer_hash_transp_addr(peer); + +unlock: spin_unlock_bh(&ovpn->lock); ovpn_peer_put(peer); - return 0; + return ret; +addr_conflict: + NL_SET_ERR_MSG_FMT_MOD(info->extack, + "VPN IP is already assigned to another peer"); + ret = -EADDRINUSE; + goto unlock; } static int ovpn_nl_send_peer(struct sk_buff *skb, const struct genl_info *info, diff --git a/drivers/net/ovpn/peer.c b/drivers/net/ovpn/peer.c index c95656ca7c35..2067825bb5b6 100644 --- a/drivers/net/ovpn/peer.c +++ b/drivers/net/ovpn/peer.c @@ -113,6 +113,7 @@ struct ovpn_peer *ovpn_peer_new(struct ovpn_priv *ovpn, u32 id) RCU_INIT_POINTER(peer->bind, NULL); ovpn_crypto_state_init(&peer->crypto); spin_lock_init(&peer->lock); + seqcount_spinlock_init(&peer->route_key_seq, &peer->lock); kref_init(&peer->refcount); ovpn_peer_stats_init(&peer->vpn_stats); ovpn_peer_stats_init(&peer->link_stats); @@ -199,13 +200,12 @@ static void __ovpn_peer_hash_transp_addr(struct ovpn_peer *peer, */ void ovpn_peer_endpoints_update(struct ovpn_peer *peer, struct sk_buff *skb) { + const void *local_ip = NULL; struct sockaddr_storage ss; struct sockaddr_in6 *sa6; - bool reset_cache = false; struct sockaddr_in *sa; struct ovpn_bind *bind; - const void *local_ip; - size_t salen = 0; + bool floated = false; spin_lock_bh(&peer->lock); bind = rcu_dereference_protected(peer->bind, @@ -232,8 +232,7 @@ void ovpn_peer_endpoints_update(struct ovpn_peer *peer, struct sk_buff *skb) .sin_addr.s_addr = ip_hdr(skb)->saddr, .sin_port = udp_hdr(skb)->source, }; - salen = sizeof(*sa); - reset_cache = true; + floated = true; break; } @@ -245,10 +244,12 @@ void ovpn_peer_endpoints_update(struct ovpn_peer *peer, struct sk_buff *skb) netdev_name(peer->ovpn->dev), peer->id, &bind->local.ipv4.s_addr, &ip_hdr(skb)->daddr); - bind->local.ipv4.s_addr = ip_hdr(skb)->daddr; - reset_cache = true; + local_ip = &ip_hdr(skb)->daddr; + memcpy(&ss, &bind->remote, sizeof(struct sockaddr_in)); + break; } - break; + /* nothing changed */ + goto unlock; case htons(ETH_P_IPV6): /* float check */ if (unlikely(!ovpn_bind_skb_src_match(bind, skb))) { @@ -270,8 +271,7 @@ void ovpn_peer_endpoints_update(struct ovpn_peer *peer, struct sk_buff *skb) ipv6_iface_scope_id(&ipv6_hdr(skb)->saddr, skb->skb_iif), }; - salen = sizeof(*sa6); - reset_cache = true; + floated = true; break; } @@ -284,26 +284,30 @@ void ovpn_peer_endpoints_update(struct ovpn_peer *peer, struct sk_buff *skb) netdev_name(peer->ovpn->dev), peer->id, &bind->local.ipv6, &ipv6_hdr(skb)->daddr); - bind->local.ipv6 = ipv6_hdr(skb)->daddr; - reset_cache = true; + local_ip = &ipv6_hdr(skb)->daddr; + memcpy(&ss, &bind->remote, sizeof(struct sockaddr_in6)); + break; } - break; + /* nothing changed */ + goto unlock; default: goto unlock; } - if (unlikely(reset_cache)) - dst_cache_reset(&peer->dst_cache); - - /* if the peer did not float, we can bail out now */ - if (likely(!salen)) - goto unlock; - if (unlikely(ovpn_peer_reset_sockaddr(peer, (struct sockaddr_storage *)&ss, local_ip) < 0)) goto unlock; + /* reset the cache only after a successful bind update to avoid useless + * cache misses on concurrent TX + */ + dst_cache_reset(&peer->dst_cache); + + /* if only the local address changed, bail out now */ + if (!floated) + goto unlock; + net_dbg_ratelimited("%s: peer %d floated to %pIScp", netdev_name(peer->ovpn->dev), peer->id, &ss); @@ -484,7 +488,7 @@ static struct ovpn_peer *ovpn_peer_get_by_vpn_addr4(struct ovpn_priv *ovpn, * Return: the peer if found or NULL otherwise */ static struct ovpn_peer *ovpn_peer_get_by_vpn_addr6(struct ovpn_priv *ovpn, - struct in6_addr *addr) + const struct in6_addr *addr) { struct hlist_nulls_head *nhead; struct hlist_nulls_node *ntmp; @@ -509,6 +513,64 @@ static struct ovpn_peer *ovpn_peer_get_by_vpn_addr6(struct ovpn_priv *ovpn, return NULL; } +/** + * ovpn_peer_vpn_addr_conflict4 - check if the VPN v4 address is already in use + * @ovpn: the openvpn instance to search + * @peer: peer being added or updated, or NULL + * @addr: VPN IPv4 address to check + * + * Check whether @addr is already assigned to another peer. @peer is ignored + * when found, allowing peer updates that keep an existing address. + * Unspecified addresses are ignored. + * + * Note: the caller must hold @ovpn->lock. + * + * Return: true on conflict, false otherwise. + */ +bool ovpn_peer_vpn_addr_conflict4(struct ovpn_priv *ovpn, + const struct ovpn_peer *peer, + const struct in_addr *addr) +{ + struct ovpn_peer *tmp = NULL; + + lockdep_assert_held(&ovpn->lock); + + /* we don't hash INADDR_ANY, no conflict in that case */ + if (addr->s_addr != htonl(INADDR_ANY)) + tmp = ovpn_peer_get_by_vpn_addr4(ovpn, addr->s_addr); + + return tmp && tmp != peer; +} + +/** + * ovpn_peer_vpn_addr_conflict6 - check if the VPN v6 address is already in use + * @ovpn: the openvpn instance to search + * @peer: peer being added or updated, or NULL + * @addr: VPN IPv6 address to check + * + * Check whether @addr is already assigned to another peer. @peer is ignored + * when found, allowing peer updates that keep an existing address. + * Unspecified addresses are ignored. + * + * Note: the caller must hold @ovpn->lock. + * + * Return: true on conflict, false otherwise. + */ +bool ovpn_peer_vpn_addr_conflict6(struct ovpn_priv *ovpn, + const struct ovpn_peer *peer, + const struct in6_addr *addr) +{ + struct ovpn_peer *tmp = NULL; + + lockdep_assert_held(&ovpn->lock); + + /* we don't hash ::, no conflict in that case */ + if (!ipv6_addr_any(addr)) + tmp = ovpn_peer_get_by_vpn_addr6(ovpn, addr); + + return tmp && tmp != peer; +} + /** * ovpn_peer_transp_match - check if sockaddr and peer binding match * @peer: the peer to get the binding from @@ -990,10 +1052,11 @@ void ovpn_peer_hash_vpn_ip(struct ovpn_peer *peer) if (hlist_unhashed(&peer->hash_entry_id)) return; - if (peer->vpn_addrs.ipv4.s_addr != htonl(INADDR_ANY)) { - /* remove potential old hashing */ - hlist_nulls_del_init_rcu(&peer->hash_entry_addr4); + /* remove potential old hashing */ + hlist_nulls_del_init_rcu(&peer->hash_entry_addr4); + hlist_nulls_del_init_rcu(&peer->hash_entry_addr6); + if (peer->vpn_addrs.ipv4.s_addr != htonl(INADDR_ANY)) { nhead = ovpn_get_hash_head(peer->ovpn->peers->by_vpn_addr4, &peer->vpn_addrs.ipv4, sizeof(peer->vpn_addrs.ipv4)); @@ -1001,9 +1064,6 @@ void ovpn_peer_hash_vpn_ip(struct ovpn_peer *peer) } if (!ipv6_addr_any(&peer->vpn_addrs.ipv6)) { - /* remove potential old hashing */ - hlist_nulls_del_init_rcu(&peer->hash_entry_addr6); - nhead = ovpn_get_hash_head(peer->ovpn->peers->by_vpn_addr6, &peer->vpn_addrs.ipv6, sizeof(peer->vpn_addrs.ipv6)); @@ -1038,6 +1098,13 @@ static int ovpn_peer_add_mp(struct ovpn_priv *ovpn, struct ovpn_peer *peer) goto out; } + /* reject peer with conflicting VPN address */ + if (ovpn_peer_vpn_addr_conflict4(ovpn, NULL, &peer->vpn_addrs.ipv4) || + ovpn_peer_vpn_addr_conflict6(ovpn, NULL, &peer->vpn_addrs.ipv6)) { + ret = -EADDRINUSE; + goto out; + } + bind = rcu_dereference_protected(peer->bind, true); /* peers connected via TCP have bind == NULL */ if (bind) { diff --git a/drivers/net/ovpn/peer.h b/drivers/net/ovpn/peer.h index dfa5c0037e02..1879bfb76992 100644 --- a/drivers/net/ovpn/peer.h +++ b/drivers/net/ovpn/peer.h @@ -10,6 +10,7 @@ #ifndef _NET_OVPN_OVPNPEER_H_ #define _NET_OVPN_OVPNPEER_H_ +#include #include #include @@ -17,6 +18,16 @@ #include "socket.h" #include "stats.h" +/** + * struct ovpn_route_key - route key used for the peer dst cache + * @mark: fwmark used for route lookup + * @sport: UDP source port used for route lookup + */ +struct ovpn_route_key { + u32 mark; + __be16 sport; +}; + /** * struct ovpn_peer - the main remote peer object * @ovpn: main openvpn instance this peer belongs to @@ -45,6 +56,8 @@ * @tcp.sk_cb.ops: pointer to the original prot_ops object (TCP only) * @crypto: the crypto configuration (ciphers, keys, etc..) * @dst_cache: cache for dst_entry used to send to peer + * @route_key: route key matching the current dst cache contents + * @route_key_seq: seqcount protecting lockless route_key reads * @bind: remote peer binding * @keepalive_interval: seconds after which a new keepalive should be sent * @keepalive_xmit_exp: future timestamp when next keepalive should be sent @@ -55,7 +68,7 @@ * @vpn_stats: per-peer in-VPN TX/RX stats * @link_stats: per-peer link/transport TX/RX stats * @delete_reason: why peer was deleted (i.e. timeout, transport error, ..) - * @lock: protects binding to peer (bind) and keepalive* fields + * @lock: protects binding to peer (bind), route_key and keepalive* fields * @refcount: reference counter * @rcu: used to free peer in an RCU safe way * @release_entry: entry for the socket release list @@ -99,6 +112,8 @@ struct ovpn_peer { } tcp; struct ovpn_crypto_state crypto; struct dst_cache dst_cache; + struct ovpn_route_key route_key; + seqcount_spinlock_t route_key_seq; struct ovpn_bind __rcu *bind; unsigned long keepalive_interval; unsigned long keepalive_xmit_exp; @@ -109,7 +124,7 @@ struct ovpn_peer { struct ovpn_peer_stats vpn_stats; struct ovpn_peer_stats link_stats; enum ovpn_del_peer_reason delete_reason; - spinlock_t lock; /* protects bind and keepalive* */ + spinlock_t lock; /* protects bind, route_key and keepalive* */ struct kref refcount; struct rcu_head rcu; struct llist_node release_entry; @@ -149,6 +164,12 @@ struct ovpn_peer *ovpn_peer_get_by_transp_addr(struct ovpn_priv *ovpn, struct ovpn_peer *ovpn_peer_get_by_id(struct ovpn_priv *ovpn, u32 peer_id); struct ovpn_peer *ovpn_peer_get_by_dst(struct ovpn_priv *ovpn, struct sk_buff *skb); +bool ovpn_peer_vpn_addr_conflict4(struct ovpn_priv *ovpn, + const struct ovpn_peer *peer, + const struct in_addr *addr); +bool ovpn_peer_vpn_addr_conflict6(struct ovpn_priv *ovpn, + const struct ovpn_peer *peer, + const struct in6_addr *addr); void ovpn_peer_hash_vpn_ip(struct ovpn_peer *peer); void ovpn_peer_hash_transp_addr(struct ovpn_peer *peer); bool ovpn_peer_check_by_src(struct ovpn_priv *ovpn, struct sk_buff *skb, diff --git a/drivers/net/ovpn/udp.c b/drivers/net/ovpn/udp.c index 7f69e8890b5b..055cdb1bee13 100644 --- a/drivers/net/ovpn/udp.c +++ b/drivers/net/ovpn/udp.c @@ -131,6 +131,77 @@ static int ovpn_udp_encap_recv(struct sock *sk, struct sk_buff *skb) return 0; } +static bool ovpn_route_key_equal(const struct ovpn_route_key *a, + const struct ovpn_route_key *b) +{ + return a->mark == b->mark && a->sport == b->sport; +} + +/** + * ovpn_dst_cache_check_key - reset peer dst cache after key changes + * @peer: the peer owning the dst cache + * @cache: the cache that might need to be reset + * @key: the route key for the packet being transmitted + * + * Reset the peer dst cache if it was populated for a different route key. + */ +static void ovpn_dst_cache_check_key(struct ovpn_peer *peer, + struct dst_cache *cache, + const struct ovpn_route_key *key) +{ + struct ovpn_route_key old_key; + unsigned int seq; + + /* snapshot the saved key before deciding whether the cache matches */ + do { + seq = read_seqcount_begin(&peer->route_key_seq); + old_key = peer->route_key; + } while (read_seqcount_retry(&peer->route_key_seq, seq)); + + /* nothing changed: the current cache can be reused */ + if (likely(ovpn_route_key_equal(&old_key, key))) + return; + + /* recheck under lock because another path may have updated the key */ + spin_lock_bh(&peer->lock); + if (!ovpn_route_key_equal(&peer->route_key, key)) { + write_seqcount_begin(&peer->route_key_seq); + peer->route_key = *key; + dst_cache_reset(cache); + write_seqcount_end(&peer->route_key_seq); + } + spin_unlock_bh(&peer->lock); +} + +/** + * ovpn_dst_cache_current - check whether a route lookup matches peer state + * @peer: the peer owning the bind and dst cache + * @bind: the RCU bind used for the route lookup + * @key: the route key used for the route lookup + * + * Check that @bind is still the current peer bind and that @key still matches + * the peer route key. The caller must hold @peer->lock. The TX path keeps + * @bind inside an RCU read-side critical section, so pointer identity is enough + * to detect whether the bind was replaced while the route lookup was running. + * + * Return: true if the lookup result still matches the current peer state and + * may update the dst cache or replace the bind. + */ +static bool ovpn_dst_cache_current(const struct ovpn_peer *peer, + const struct ovpn_bind *bind, + const struct ovpn_route_key *key) +{ + const struct ovpn_bind *curr_bind; + + lockdep_assert_held(&peer->lock); + + curr_bind = rcu_dereference_protected(peer->bind, + lockdep_is_held(&peer->lock)); + + return curr_bind == bind && + ovpn_route_key_equal(key, &peer->route_key); +} + /** * ovpn_udp4_output - send IPv4 packet over udp socket * @peer: the destination peer @@ -138,21 +209,26 @@ static int ovpn_udp_encap_recv(struct sock *sk, struct sk_buff *skb) * @cache: dst cache * @sk: the socket to send the packet over * @skb: the packet to send + * @key: the route key snapshot used for cache validation and flow lookup * * Return: 0 on success or a negative error code otherwise */ static int ovpn_udp4_output(struct ovpn_peer *peer, struct ovpn_bind *bind, struct dst_cache *cache, struct sock *sk, - struct sk_buff *skb) + struct sk_buff *skb, + const struct ovpn_route_key *key) { + struct sockaddr_storage remote; + struct in_addr local = {}; + bool reset_local = false; struct rtable *rt; struct flowi4 fl = { .saddr = bind->local.ipv4.s_addr, .daddr = bind->remote.in4.sin_addr.s_addr, - .fl4_sport = inet_sk(sk)->inet_sport, + .fl4_sport = key->sport, .fl4_dport = bind->remote.in4.sin_port, .flowi4_proto = sk->sk_protocol, - .flowi4_mark = sk->sk_mark, + .flowi4_mark = key->mark, }; int ret; @@ -161,26 +237,19 @@ static int ovpn_udp4_output(struct ovpn_peer *peer, struct ovpn_bind *bind, if (rt) goto transmit; - if (unlikely(!inet_confirm_addr(sock_net(sk), NULL, 0, fl.saddr, - RT_SCOPE_HOST))) { - /* we may end up here when the cached address is not usable - * anymore. In this case we reset address/cache and perform a - * new look up + if (fl.saddr && unlikely(!inet_confirm_addr(sock_net(sk), NULL, 0, + fl.saddr, RT_SCOPE_HOST))) { + /* The learned local address is not usable anymore. + * Retry with source address autoselection. */ fl.saddr = 0; - spin_lock_bh(&peer->lock); - bind->local.ipv4.s_addr = 0; - spin_unlock_bh(&peer->lock); - dst_cache_reset(cache); + reset_local = true; } rt = ip_route_output_flow(sock_net(sk), &fl, sk); if (IS_ERR(rt) && PTR_ERR(rt) == -EINVAL) { fl.saddr = 0; - spin_lock_bh(&peer->lock); - bind->local.ipv4.s_addr = 0; - spin_unlock_bh(&peer->lock); - dst_cache_reset(cache); + reset_local = true; rt = ip_route_output_flow(sock_net(sk), &fl, sk); } @@ -193,7 +262,30 @@ static int ovpn_udp4_output(struct ovpn_peer *peer, struct ovpn_bind *bind, ret); goto err; } - dst_cache_set_ip4(cache, &rt->dst, fl.saddr); + + /* avoid storing a stale cache or local address */ + spin_lock_bh(&peer->lock); + if (likely(ovpn_dst_cache_current(peer, bind, key))) { + if (!reset_local) { + dst_cache_set_ip4(cache, &rt->dst, fl.saddr); + spin_unlock_bh(&peer->lock); + goto transmit; + } + + /* invalidate per-CPU dst entries that may still carry + * the stale source + */ + dst_cache_reset(cache); + + /* preserve the current remote */ + memcpy(&remote, &bind->remote, sizeof(struct sockaddr_in)); + /* The current packet already has a valid wildcard-source route. + * If replacing the bind fails, leave the stale local in place; + * a later cache miss will retry the repair. + */ + ovpn_peer_reset_sockaddr(peer, &remote, &local); + } + spin_unlock_bh(&peer->lock); transmit: udp_tunnel_xmit_skb(rt, sk, skb, fl.saddr, fl.daddr, 0, @@ -213,23 +305,28 @@ static int ovpn_udp4_output(struct ovpn_peer *peer, struct ovpn_bind *bind, * @cache: dst cache * @sk: the socket to send the packet over * @skb: the packet to send + * @key: the route key snapshot used for cache validation and flow lookup * * Return: 0 on success or a negative error code otherwise */ static int ovpn_udp6_output(struct ovpn_peer *peer, struct ovpn_bind *bind, struct dst_cache *cache, struct sock *sk, - struct sk_buff *skb) + struct sk_buff *skb, + const struct ovpn_route_key *key) { + struct in6_addr local = in6addr_any; + struct sockaddr_storage remote; + bool reset_local = false; struct dst_entry *dst; int ret; struct flowi6 fl = { .saddr = bind->local.ipv6, .daddr = bind->remote.in6.sin6_addr, - .fl6_sport = inet_sk(sk)->inet_sport, + .fl6_sport = key->sport, .fl6_dport = bind->remote.in6.sin6_port, .flowi6_proto = sk->sk_protocol, - .flowi6_mark = sk->sk_mark, + .flowi6_mark = key->mark, .flowi6_oif = bind->remote.in6.sin6_scope_id, }; @@ -238,16 +335,13 @@ static int ovpn_udp6_output(struct ovpn_peer *peer, struct ovpn_bind *bind, if (dst) goto transmit; - if (unlikely(!ipv6_chk_addr(sock_net(sk), &fl.saddr, NULL, 0))) { - /* we may end up here when the cached address is not usable - * anymore. In this case we reset address/cache and perform a - * new look up + if (!ipv6_addr_any(&fl.saddr) && + unlikely(!ipv6_chk_addr(sock_net(sk), &fl.saddr, NULL, 0))) { + /* The learned local address is not usable anymore. + * Retry with source address autoselection. */ fl.saddr = in6addr_any; - spin_lock_bh(&peer->lock); - bind->local.ipv6 = in6addr_any; - spin_unlock_bh(&peer->lock); - dst_cache_reset(cache); + reset_local = true; } dst = ip6_dst_lookup_flow(sock_net(sk), sk, &fl, NULL); @@ -258,7 +352,30 @@ static int ovpn_udp6_output(struct ovpn_peer *peer, struct ovpn_bind *bind, &bind->remote.in6, ret); goto err; } - dst_cache_set_ip6(cache, dst, &fl.saddr); + + /* avoid storing a stale cache or local address */ + spin_lock_bh(&peer->lock); + if (likely(ovpn_dst_cache_current(peer, bind, key))) { + if (!reset_local) { + dst_cache_set_ip6(cache, dst, &fl.saddr); + spin_unlock_bh(&peer->lock); + goto transmit; + } + + /* invalidate per-CPU dst entries that may still carry + * the stale source + */ + dst_cache_reset(cache); + + /* preserve the current remote */ + memcpy(&remote, &bind->remote, sizeof(struct sockaddr_in6)); + /* The current packet already has a valid wildcard-source route. + * If replacing the bind fails, leave the stale local in place; + * a later cache miss will retry the repair. + */ + ovpn_peer_reset_sockaddr(peer, &remote, &local); + } + spin_unlock_bh(&peer->lock); transmit: /* user IPv6 packets may be larger than the transport interface @@ -287,6 +404,7 @@ static int ovpn_udp6_output(struct ovpn_peer *peer, struct ovpn_bind *bind, * @cache: dst cache * @sk: the socket to send the packet over * @skb: the packet to send + * @key: route key snapshot used for cache validation and flow lookup * * rcu_read_lock should be held on entry. * On return, the skb is consumed. @@ -294,7 +412,8 @@ static int ovpn_udp6_output(struct ovpn_peer *peer, struct ovpn_bind *bind, * Return: 0 on success or a negative error code otherwise */ static int ovpn_udp_output(struct ovpn_peer *peer, struct dst_cache *cache, - struct sock *sk, struct sk_buff *skb) + struct sock *sk, struct sk_buff *skb, + struct ovpn_route_key *key) { struct ovpn_bind *bind; int ret; @@ -314,11 +433,11 @@ static int ovpn_udp_output(struct ovpn_peer *peer, struct dst_cache *cache, switch (bind->remote.in4.sin_family) { case AF_INET: - ret = ovpn_udp4_output(peer, bind, cache, sk, skb); + ret = ovpn_udp4_output(peer, bind, cache, sk, skb, key); break; #if IS_ENABLED(CONFIG_IPV6) case AF_INET6: - ret = ovpn_udp6_output(peer, bind, cache, sk, skb); + ret = ovpn_udp6_output(peer, bind, cache, sk, skb, key); break; #endif default: @@ -340,15 +459,21 @@ static int ovpn_udp_output(struct ovpn_peer *peer, struct dst_cache *cache, void ovpn_udp_send_skb(struct ovpn_peer *peer, struct sock *sk, struct sk_buff *skb) { + struct ovpn_route_key key = { + .mark = READ_ONCE(sk->sk_mark), + .sport = READ_ONCE(inet_sk(sk)->inet_sport), + }; int ret; skb->dev = peer->ovpn->dev; - skb->mark = READ_ONCE(sk->sk_mark); + skb->mark = key.mark; /* no checksum performed at this layer */ skb->ip_summed = CHECKSUM_NONE; + ovpn_dst_cache_check_key(peer, &peer->dst_cache, &key); + /* crypto layer -> transport (UDP) */ - ret = ovpn_udp_output(peer, &peer->dst_cache, sk, skb); + ret = ovpn_udp_output(peer, &peer->dst_cache, sk, skb, &key); if (unlikely(ret < 0)) kfree_skb(skb); } diff --git a/drivers/net/pcs/pcs-rzn1-miic.c b/drivers/net/pcs/pcs-rzn1-miic.c index 2b72fa98ddf1..cb74861e823c 100644 --- a/drivers/net/pcs/pcs-rzn1-miic.c +++ b/drivers/net/pcs/pcs-rzn1-miic.c @@ -683,7 +683,8 @@ static int miic_parse_dt(struct miic *miic, u32 *mode_cfg) if (!dt_val) return -ENOMEM; - memset(dt_val, MIIC_MODCTRL_CONF_NONE, sizeof(*dt_val)); + memset(dt_val, MIIC_MODCTRL_CONF_NONE, + sizeof(*dt_val) * miic->of_data->conf_conv_count); if (of_property_read_u32(np, "renesas,miic-switch-portin", &conf) == 0) dt_val[0] = conf; diff --git a/drivers/net/pcs/pcs-xpcs.c b/drivers/net/pcs/pcs-xpcs.c index 0337e2bcc012..b415b93d77c1 100644 --- a/drivers/net/pcs/pcs-xpcs.c +++ b/drivers/net/pcs/pcs-xpcs.c @@ -1545,8 +1545,10 @@ static int xpcs_init_clks(struct dw_xpcs *xpcs) return dev_err_probe(dev, ret, "Failed to get clocks\n"); ret = clk_bulk_prepare_enable(DW_XPCS_NUM_CLKS, xpcs->clks); - if (ret) + if (ret) { + clk_bulk_put(DW_XPCS_NUM_CLKS, xpcs->clks); return dev_err_probe(dev, ret, "Failed to enable clocks\n"); + } return 0; } diff --git a/drivers/net/phy/intel-xway.c b/drivers/net/phy/intel-xway.c index afbcec711744..3cee31bb931f 100644 --- a/drivers/net/phy/intel-xway.c +++ b/drivers/net/phy/intel-xway.c @@ -16,6 +16,11 @@ #define XWAY_MDIO_ISTAT 0x1A /* interrupt status */ #define XWAY_MDIO_LED 0x1B /* led control */ +#define XWAY_MDIO_GCTRL_TM_MASK GENMASK(15, 13) +#define XWAY_MDIO_GCTRL_TM(mode) FIELD_PREP(XWAY_MDIO_GCTRL_TM_MASK, (mode)) +#define XWAY_MDIO_GCTRL_TM_NOP XWAY_MDIO_GCTRL_TM(0) /* Normal operation */ +#define XWAY_MDIO_GCTRL_TM_CDIAG XWAY_MDIO_GCTRL_TM(6) /* Cable diagnostics */ + #define XWAY_MDIO_ERRCNT_SEL GENMASK(11, 8) #define XWAY_MDIO_ERRCNT_COUNT GENMASK(7, 0) #define XWAY_MDIO_ERRCNT_SEL_RXERR 0 @@ -326,6 +331,28 @@ static int xway_gphy_probe(struct phy_device *phydev) return 0; } +static int xway_11g_int_config_init(struct phy_device *phydev) +{ + int err; + + /* An issue has been sporadically observed after device power-on on the + * first link-up attempt in 100BASE-TX mode resulting in either the + * link-up taking a long time, or failing to link-up altogether. + * + * Workaround: + * After power-on, enable Cable Diagnostic Mode for all ports and + * disable it. + */ + err = phy_modify(phydev, MII_CTRL1000, XWAY_MDIO_GCTRL_TM_MASK, XWAY_MDIO_GCTRL_TM_CDIAG); + if (err) + return err; + err = phy_modify(phydev, MII_CTRL1000, XWAY_MDIO_GCTRL_TM_MASK, XWAY_MDIO_GCTRL_TM_NOP); + if (err) + return err; + + return xway_gphy_config_init(phydev); +} + static int xway_gphy14_config_aneg(struct phy_device *phydev) { int reg, err; @@ -735,7 +762,7 @@ static struct phy_driver xway_gphy[] = { .phy_id_mask = 0xffffffff, .name = "Intel XWAY PHY11G (xRX v1.2 integrated)", /* PHY_GBIT_FEATURES */ - .config_init = xway_gphy_config_init, + .config_init = xway_11g_int_config_init, .probe = xway_gphy_probe, .handle_interrupt = xway_gphy_handle_interrupt, .config_intr = xway_gphy_config_intr, diff --git a/drivers/net/phy/micrel.c b/drivers/net/phy/micrel.c index ae830781824b..5c8461db7b4b 100644 --- a/drivers/net/phy/micrel.c +++ b/drivers/net/phy/micrel.c @@ -6344,6 +6344,7 @@ static int lanphy_write_reg_data(struct phy_device *phydev, data->val); if (ret) break; + data++; } return ret; diff --git a/drivers/net/phy/phylink.c b/drivers/net/phy/phylink.c index a1458da8111b..1bbcf46c8356 100644 --- a/drivers/net/phy/phylink.c +++ b/drivers/net/phy/phylink.c @@ -2129,7 +2129,6 @@ static int phylink_bringup_phy(struct phylink *pl, struct phy_device *phy, mutex_lock(&pl->phydev_mutex); mutex_lock(&phy->lock); mutex_lock(&pl->state_mutex); - pl->phydev = phy; pl->phy_state.interface = interface; pl->phy_state.pause = MLO_PAUSE_NONE; pl->phy_state.speed = SPEED_UNKNOWN; @@ -2196,10 +2195,25 @@ static int phylink_bringup_phy(struct phylink *pl, struct phy_device *phy, ret = 0; } - if (ret == 0 && phy_interrupt_is_valid(phy)) + if (ret) + return ret; + + /* Nothing below can fail, so the PHY can be recorded now. Doing it + * here rather than above keeps a failed bringup from leaving + * pl->phydev pointing at a PHY the caller is about to detach. + */ + mutex_lock(&pl->phydev_mutex); + mutex_lock(&phy->lock); + mutex_lock(&pl->state_mutex); + pl->phydev = phy; + mutex_unlock(&pl->state_mutex); + mutex_unlock(&phy->lock); + mutex_unlock(&pl->phydev_mutex); + + if (phy_interrupt_is_valid(phy)) phy_request_interrupt(phy); - return ret; + return 0; } static int phylink_attach_phy(struct phylink *pl, struct phy_device *phy, diff --git a/drivers/net/usb/catc.c b/drivers/net/usb/catc.c index 96e82f94edcf..39b678f175dd 100644 --- a/drivers/net/usb/catc.c +++ b/drivers/net/usb/catc.c @@ -233,17 +233,26 @@ static void catc_rx_done(struct urb *urb) } do { - if(!catc->is_f5u011) { - pkt_len = le16_to_cpup((__le16*)pkt_start); - if (pkt_len > urb->actual_length) { + int remaining = urb->actual_length - + (pkt_start - (u8 *)urb->transfer_buffer); + + if (!catc->is_f5u011) { + if (remaining < pkt_offset) { catc->netdev->stats.rx_length_errors++; catc->netdev->stats.rx_errors++; break; } + pkt_len = le16_to_cpup((__le16 *)pkt_start); } else { pkt_len = urb->actual_length; } + if (pkt_len < ETH_HLEN || pkt_len + pkt_offset > remaining) { + catc->netdev->stats.rx_length_errors++; + catc->netdev->stats.rx_errors++; + break; + } + if (!(skb = dev_alloc_skb(pkt_len))) return; diff --git a/drivers/net/usb/cdc_mbim.c b/drivers/net/usb/cdc_mbim.c index 877fb0ed7d3d..a7010a0664c7 100644 --- a/drivers/net/usb/cdc_mbim.c +++ b/drivers/net/usb/cdc_mbim.c @@ -635,6 +635,11 @@ static const struct usb_device_id mbim_devs[] = { .driver_info = (unsigned long)&cdc_mbim_info, }, + /* MeiG Smart SRM821 ZLP conformance */ + { USB_DEVICE_AND_INTERFACE_INFO(0x2dee, 0x4d53, USB_CLASS_COMM, USB_CDC_SUBCLASS_MBIM, USB_CDC_PROTO_NONE), + .driver_info = (unsigned long)&cdc_mbim_info, + }, + /* Some Huawei devices, ME906s-158 (12d1:15c1) and E3372 * (12d1:157d), are known to fail unless the NDP is placed * after the IP packets. Applying the quirk to all Huawei diff --git a/drivers/net/usb/lan78xx.c b/drivers/net/usb/lan78xx.c index cb782d81d84f..5655941f1478 100644 --- a/drivers/net/usb/lan78xx.c +++ b/drivers/net/usb/lan78xx.c @@ -5239,10 +5239,12 @@ static bool lan78xx_submit_deferred_urbs(struct lan78xx_net *dev) !netif_carrier_ok(dev->net) || pipe_halted) { lan78xx_release_tx_buf(dev, skb); + usb_put_urb(urb); continue; } ret = usb_submit_urb(urb, GFP_ATOMIC); + usb_put_urb(urb); if (ret == 0) { netif_trans_update(dev->net); diff --git a/drivers/net/usb/qmi_wwan.c b/drivers/net/usb/qmi_wwan.c index f51cf9cb9421..0e2ab567dd40 100644 --- a/drivers/net/usb/qmi_wwan.c +++ b/drivers/net/usb/qmi_wwan.c @@ -1086,6 +1086,7 @@ static const struct usb_device_id products[] = { {QMI_MATCH_FF_FF_FF(0x2c7c, 0x0125)}, /* Quectel EC25, EC20 R2.0 Mini PCIe */ {QMI_MATCH_FF_FF_FF(0x2c7c, 0x013d)}, /* Quectel RG660QB */ {QMI_MATCH_FF_FF_FF(0x2c7c, 0x0306)}, /* Quectel EP06/EG06/EM06 */ + {QMI_MATCH_FF_FF_FF(0x2c7c, 0x030b)}, /* Quectel EM060K/EG120K-EA */ {QMI_MATCH_FF_FF_FF(0x2c7c, 0x0512)}, /* Quectel EG12/EM12 */ {QMI_MATCH_FF_FF_FF(0x2c7c, 0x0620)}, /* Quectel EM160R-GL */ {QMI_MATCH_FF_FF_FF(0x2c7c, 0x0800)}, /* Quectel RM500Q-GL */ diff --git a/drivers/net/usb/sr9700.c b/drivers/net/usb/sr9700.c index 937e6fef3ac6..50981a28376a 100644 --- a/drivers/net/usb/sr9700.c +++ b/drivers/net/usb/sr9700.c @@ -355,7 +355,8 @@ static int sr9700_rx_fixup(struct usbnet *dev, struct sk_buff *skb) /* ignore the CRC length */ len = (skb->data[1] | (skb->data[2] << 8)) - 4; - if (len > ETH_FRAME_LEN || len > skb->len || len < 0) + if (len > ETH_FRAME_LEN || len < 0 || + len > skb->len - SR_RX_OVERHEAD) return 0; /* the last packet of current skb */ diff --git a/drivers/net/veth.c b/drivers/net/veth.c index 6ed3ee81153f..71227d0389aa 100644 --- a/drivers/net/veth.c +++ b/drivers/net/veth.c @@ -1054,6 +1054,7 @@ static int __veth_napi_enable_range(struct net_device *dev, int start, int end) for (i = start; i < end; i++) { struct veth_rq *rq = &priv->rq[i]; + rcu_assign_pointer(rq->xdp_prog, priv->_xdp_prog); napi_enable(&rq->xdp_napi); rcu_assign_pointer(priv->rq[i].napi, &priv->rq[i].xdp_napi); } @@ -1088,6 +1089,7 @@ static void veth_napi_del_range(struct net_device *dev, int start, int end) rcu_assign_pointer(priv->rq[i].napi, NULL); napi_disable(&rq->xdp_napi); + rcu_assign_pointer(rq->xdp_prog, NULL); __netif_napi_del(&rq->xdp_napi); } synchronize_net(); diff --git a/drivers/net/virtio_net.c b/drivers/net/virtio_net.c index e34c52d059d3..bf82ef9874ab 100644 --- a/drivers/net/virtio_net.c +++ b/drivers/net/virtio_net.c @@ -3349,6 +3349,14 @@ static netdev_tx_t start_xmit(struct sk_buff *skb, struct net_device *dev) else virtqueue_disable_cb(sq->vq); + if (!use_napi && + unlikely(skb_orphan_frags(skb, GFP_ATOMIC))) { + DEV_STATS_INC(dev, tx_dropped); + dev_kfree_skb_any(skb); + kick = !xmit_more || netif_xmit_stopped(txq); + goto kick_vq; + } + /* timestamp packet in software */ skb_tx_timestamp(skb); @@ -3381,6 +3389,7 @@ static netdev_tx_t start_xmit(struct sk_buff *skb, struct net_device *dev) kick = use_napi ? __netdev_tx_sent_queue(txq, skb->len, xmit_more) : !xmit_more || netif_xmit_stopped(txq); +kick_vq: if (kick) { if (virtqueue_kick_prepare(sq->vq) && virtqueue_notify(sq->vq)) { u64_stats_update_begin(&sq->stats.syncp); diff --git a/drivers/net/vrf.c b/drivers/net/vrf.c index a0557a3a7026..d4dc6d690a75 100644 --- a/drivers/net/vrf.c +++ b/drivers/net/vrf.c @@ -1175,8 +1175,6 @@ static int vrf_prepare_mac_header(struct sk_buff *skb, skb->protocol = eth->h_proto; skb->pkt_type = PACKET_HOST; - skb_postpush_rcsum(skb, skb->data, ETH_HLEN); - skb_pull_inline(skb, ETH_HLEN); return 0; diff --git a/drivers/net/vxlan/vxlan_core.c b/drivers/net/vxlan/vxlan_core.c index c1d54339fa2b..3390e341d1e5 100644 --- a/drivers/net/vxlan/vxlan_core.c +++ b/drivers/net/vxlan/vxlan_core.c @@ -1958,13 +1958,15 @@ static struct sk_buff *vxlan_na_create(struct sk_buff *request, struct ipv6hdr *pip6; u8 *daddr; int na_olen = 8; /* opt hdr + ETH_ALEN for target */ + int headroom; int ns_olen; int i, len; if (dev == NULL || !pskb_may_pull(request, request->len)) return NULL; - len = LL_RESERVED_SPACE(dev) + sizeof(struct ipv6hdr) + + headroom = LL_RESERVED_SPACE(dev); + len = headroom + sizeof(struct ipv6hdr) + sizeof(*na) + na_olen + dev->needed_tailroom; reply = alloc_skb(len, GFP_ATOMIC); if (reply == NULL) @@ -1972,7 +1974,7 @@ static struct sk_buff *vxlan_na_create(struct sk_buff *request, reply->protocol = htons(ETH_P_IPV6); reply->dev = dev; - skb_reserve(reply, LL_RESERVED_SPACE(request->dev)); + skb_reserve(reply, headroom); skb_push(reply, sizeof(struct ethhdr)); skb_reset_mac_header(reply); diff --git a/drivers/nfc/microread/microread.c b/drivers/nfc/microread/microread.c index dfa2490db545..2bfafa94e83d 100644 --- a/drivers/nfc/microread/microread.c +++ b/drivers/nfc/microread/microread.c @@ -251,9 +251,9 @@ static int microread_start_poll(struct nfc_hci_dev *hdev, param[1] |= (1 << 1); if ((im_protocols | tm_protocols) & NFC_PROTO_NFC_DEP_MASK) { - hdev->gb = nfc_get_local_general_bytes(hdev->ndev, - &hdev->gb_len); - if (hdev->gb == NULL || hdev->gb_len == 0) { + nfc_get_local_general_bytes(hdev->ndev, hdev->gb, + sizeof(hdev->gb), &hdev->gb_len); + if (hdev->gb_len == 0) { im_protocols &= ~NFC_PROTO_NFC_DEP_MASK; tm_protocols &= ~NFC_PROTO_NFC_DEP_MASK; } diff --git a/drivers/nfc/nfcmrvl/fw_dnld.c b/drivers/nfc/nfcmrvl/fw_dnld.c index 2b8f401d8fd7..8b9d5257320d 100644 --- a/drivers/nfc/nfcmrvl/fw_dnld.c +++ b/drivers/nfc/nfcmrvl/fw_dnld.c @@ -263,9 +263,14 @@ static int process_state_fw_dnld(struct nfcmrvl_private *priv, * B8..N: payload */ - /* Remove NCI HDR */ - skb_pull(skb, 3); - if (skb->data[0] != HELPER_CMD_PACKET_FORMAT || skb->len != 5) { + if (skb->len != NCI_DATA_HDR_SIZE + 5) { + nfc_err(priv->dev, "bad command"); + return -EINVAL; + } + + /* Remove NCI header */ + skb_pull(skb, NCI_DATA_HDR_SIZE); + if (skb->data[0] != HELPER_CMD_PACKET_FORMAT) { nfc_err(priv->dev, "bad command"); return -EINVAL; } diff --git a/drivers/nfc/pn533/pn533.c b/drivers/nfc/pn533/pn533.c index f5a6a7c20d5a..b0133e51dce9 100644 --- a/drivers/nfc/pn533/pn533.c +++ b/drivers/nfc/pn533/pn533.c @@ -1357,10 +1357,11 @@ static int pn533_poll_dep(struct nfc_dev *nfc_dev) u8 *next, nfcid3[NFC_NFCID3_MAXSIZE]; u8 passive_data[PASSIVE_DATA_LEN] = {0x00, 0xff, 0xff, 0x00, 0x3}; - if (!dev->gb) { - dev->gb = nfc_get_local_general_bytes(nfc_dev, &dev->gb_len); - - if (!dev->gb || !dev->gb_len) { + if (!dev->gb_len) { + nfc_get_local_general_bytes(nfc_dev, dev->gb, + sizeof(dev->gb), + &dev->gb_len); + if (!dev->gb_len) { dev->poll_dep = 0; queue_work(dev->wq, &dev->rf_work); } @@ -1658,8 +1659,9 @@ static int pn533_start_poll(struct nfc_dev *nfc_dev, } if (tm_protocols) { - dev->gb = nfc_get_local_general_bytes(nfc_dev, &dev->gb_len); - if (dev->gb == NULL) + nfc_get_local_general_bytes(nfc_dev, dev->gb, + sizeof(dev->gb), &dev->gb_len); + if (dev->gb_len == 0) tm_protocols = 0; } diff --git a/drivers/nfc/pn533/pn533.h b/drivers/nfc/pn533/pn533.h index 09e35b8693f5..5ab668e05121 100644 --- a/drivers/nfc/pn533/pn533.h +++ b/drivers/nfc/pn533/pn533.h @@ -6,6 +6,8 @@ * Copyright (C) 2012-2013 Tieto Poland */ +#include + #define PN533_DEVICE_STD 0x1 #define PN533_DEVICE_PASORI 0x2 #define PN533_DEVICE_ACR122U 0x3 @@ -166,7 +168,7 @@ struct pn533 { struct timer_list listen_timer; int cancel_listen; - u8 *gb; + u8 gb[NFC_MAX_GT_LEN]; size_t gb_len; u8 tgt_available_prots; diff --git a/drivers/nfc/pn533/usb.c b/drivers/nfc/pn533/usb.c index efb07f944fce..972eaac09e59 100644 --- a/drivers/nfc/pn533/usb.c +++ b/drivers/nfc/pn533/usb.c @@ -319,7 +319,9 @@ static bool pn533_acr122_is_rx_frame_valid(void *_frame, struct pn533 *dev) if (frame->ccid.type != 0x83) return false; - if (!frame->ccid.datalen) + if (frame->ccid.datalen < 2 || + frame->ccid.datalen > PN533_ACR122_FRAME_MAX_PAYLOAD_LEN + + PN533_ACR122_RX_FRAME_TAIL_LEN) return false; if (frame->data[frame->ccid.datalen - 2] == 0x63) diff --git a/drivers/nfc/pn544/pn544.c b/drivers/nfc/pn544/pn544.c index 9d0a16ac465e..c4fa70e45c14 100644 --- a/drivers/nfc/pn544/pn544.c +++ b/drivers/nfc/pn544/pn544.c @@ -377,10 +377,9 @@ static int pn544_hci_start_poll(struct nfc_hci_dev *hdev, return r; if ((im_protocols | tm_protocols) & NFC_PROTO_NFC_DEP_MASK) { - hdev->gb = nfc_get_local_general_bytes(hdev->ndev, - &hdev->gb_len); - pr_debug("generate local bytes %p\n", hdev->gb); - if (hdev->gb == NULL || hdev->gb_len == 0) { + nfc_get_local_general_bytes(hdev->ndev, hdev->gb, + sizeof(hdev->gb), &hdev->gb_len); + if (hdev->gb_len == 0) { im_protocols &= ~NFC_PROTO_NFC_DEP_MASK; tm_protocols &= ~NFC_PROTO_NFC_DEP_MASK; } diff --git a/drivers/nfc/port100.c b/drivers/nfc/port100.c index b613f5e2fd57..e769a30b8b5c 100644 --- a/drivers/nfc/port100.c +++ b/drivers/nfc/port100.c @@ -636,6 +636,13 @@ static void port100_recv_response(struct urb *urb) in_frame = dev->in_urb->transfer_buffer; + if (urb->actual_length < PORT100_FRAME_HEADER_LEN || + urb->actual_length < port100_rx_frame_size(in_frame)) { + nfc_err(&dev->interface->dev, "Received a truncated frame\n"); + cmd->status = -EIO; + goto sched_wq; + } + if (!port100_rx_frame_is_valid(in_frame)) { nfc_err(&dev->interface->dev, "Received an invalid frame\n"); cmd->status = -EIO; diff --git a/drivers/nfc/st21nfca/core.c b/drivers/nfc/st21nfca/core.c index fd39a05c9622..b5c1ca3acfbe 100644 --- a/drivers/nfc/st21nfca/core.c +++ b/drivers/nfc/st21nfca/core.c @@ -351,10 +351,10 @@ static int st21nfca_hci_start_poll(struct nfc_hci_dev *hdev, if (r < 0) return r; } else { - hdev->gb = nfc_get_local_general_bytes(hdev->ndev, - &hdev->gb_len); - - if (hdev->gb == NULL || hdev->gb_len == 0) { + nfc_get_local_general_bytes(hdev->ndev, hdev->gb, + sizeof(hdev->gb), + &hdev->gb_len); + if (hdev->gb_len == 0) { im_protocols &= ~NFC_PROTO_NFC_DEP_MASK; tm_protocols &= ~NFC_PROTO_NFC_DEP_MASK; } @@ -577,9 +577,7 @@ static int st21nfca_get_iso15693_inventory(struct nfc_hci_dev *hdev, if (r < 0) goto exit; - skb_pull(inventory_skb, 2); - - if (inventory_skb->len == 0 || + if (!skb_pull(inventory_skb, 2) || inventory_skb->len < 2 || inventory_skb->len > NFC_ISO15693_UID_MAXSIZE) { r = -EPROTO; goto exit; diff --git a/drivers/nfc/st21nfca/i2c.c b/drivers/nfc/st21nfca/i2c.c index a4c93ff7c5b0..0f44c783bd04 100644 --- a/drivers/nfc/st21nfca/i2c.c +++ b/drivers/nfc/st21nfca/i2c.c @@ -289,27 +289,36 @@ static int check_crc(u8 *buf, int buflen) */ static int st21nfca_hci_i2c_repack(struct sk_buff *skb) { - int i, j, r, size; + int read, write, r, size; - if (skb->len < 1 || (skb->len > 1 && skb->data[1] != 0)) + if (skb->len < ST21NFCA_FRAME_HEADROOM || + !IS_START_OF_FRAME(skb->data)) return -EBADMSG; size = get_frame_size(skb->data, skb->len); if (size > 0) { + if (size < ST21NFCA_FRAME_HEADROOM + 2) + return -EBADMSG; + skb_trim(skb, size); /* remove ST21NFCA byte stuffing for upper layer */ - for (i = 1, j = 0; i < skb->len; i++) { - if (skb->data[i + j] == + for (read = 1, write = 1; read < skb->len;) { + if (skb->data[read] == (u8) ST21NFCA_ESCAPE_BYTE_STUFFING) { - skb->data[i] = skb->data[i + j + 1] - | ST21NFCA_BYTE_STUFFING_MASK; - i++; - j++; + if (read + 1 == skb->len) + return -EBADMSG; + + skb->data[write++] = skb->data[read + 1] + | ST21NFCA_BYTE_STUFFING_MASK; + read += 2; + } else { + skb->data[write++] = skb->data[read++]; } - skb->data[i] = skb->data[i + j]; } /* remove byte stuffing useless byte */ - skb_trim(skb, i - j); + skb_trim(skb, write); + if (skb->len < ST21NFCA_FRAME_HEADROOM + 2) + return -EBADMSG; /* remove ST21NFCA_SOF_EOF from head */ skb_pull(skb, 1); diff --git a/drivers/nfc/trf7970a.c b/drivers/nfc/trf7970a.c index 60883001fa5d..ddfc58c29964 100644 --- a/drivers/nfc/trf7970a.c +++ b/drivers/nfc/trf7970a.c @@ -1997,8 +1997,10 @@ static int trf7970a_startup(struct trf7970a *trf) return ret; ret = trf7970a_update_rx_gain_reduction(trf); - if (ret) + if (ret) { + trf7970a_power_down(trf); return ret; + } pm_runtime_set_active(trf->dev); pm_runtime_enable(trf->dev); diff --git a/drivers/nfc/virtual_ncidev.c b/drivers/nfc/virtual_ncidev.c index 8eeb447ac96e..e51c647b27eb 100644 --- a/drivers/nfc/virtual_ncidev.c +++ b/drivers/nfc/virtual_ncidev.c @@ -195,7 +195,8 @@ static const struct file_operations virtual_ncidev_fops = { .write = virtual_ncidev_write, .open = virtual_ncidev_open, .release = virtual_ncidev_close, - .unlocked_ioctl = virtual_ncidev_ioctl + .unlocked_ioctl = virtual_ncidev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static struct miscdevice miscdev = { diff --git a/include/linux/ethtool.h b/include/linux/ethtool.h index 253600c0eccd..c4c9ce038611 100644 --- a/include/linux/ethtool.h +++ b/include/linux/ethtool.h @@ -944,6 +944,7 @@ struct kernel_ethtool_ts_info { #define ETHTOOL_OP_NEEDS_RTNL_SPAUSEPARAM BIT(6) #define ETHTOOL_OP_NEEDS_RTNL_RSS BIT(7) #define ETHTOOL_OP_NEEDS_RTNL_GLINK BIT(8) +#define ETHTOOL_OP_NEEDS_RTNL_TEST BIT(9) /** * struct ethtool_ops - optional netdev operations @@ -981,6 +982,7 @@ struct kernel_ethtool_ts_info { * - netdev_update_features() * - netif_set_real_num_tx_queues() * - ethtool_op_get_link() (syncs link watch under rtnl_lock) + * - netif_open() / netif_close() (used by @self_test) * * @get_drvinfo: Report driver/device information. Modern drivers no * longer have to implement this callback. Most fields are diff --git a/include/linux/if_vlan.h b/include/linux/if_vlan.h index 20cc16ea4e5a..4846032bf4ff 100644 --- a/include/linux/if_vlan.h +++ b/include/linux/if_vlan.h @@ -365,6 +365,9 @@ static inline int __vlan_insert_inner_tag(struct sk_buff *skb, const u8 meta_len = mac_len > ETH_TLEN ? skb_metadata_len(skb) : 0; struct vlan_ethhdr *veth; + if (unlikely(!pskb_may_pull(skb, mac_len))) + return -EINVAL; + if (skb_cow_head(skb, meta_len + VLAN_HLEN) < 0) return -ENOMEM; diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h index c8e21903074c..84308498a3a8 100644 --- a/include/linux/skbuff.h +++ b/include/linux/skbuff.h @@ -1834,22 +1834,6 @@ static inline void skb_zcopy_set(struct sk_buff *skb, struct ubuf_info *uarg, } } -static inline void skb_zcopy_set_nouarg(struct sk_buff *skb, void *val) -{ - skb_shinfo(skb)->destructor_arg = (void *)((uintptr_t) val | 0x1UL); - skb_shinfo(skb)->flags |= SKBFL_ZEROCOPY_FRAG; -} - -static inline bool skb_zcopy_is_nouarg(struct sk_buff *skb) -{ - return (uintptr_t) skb_shinfo(skb)->destructor_arg & 0x1UL; -} - -static inline void *skb_zcopy_get_nouarg(struct sk_buff *skb) -{ - return (void *)((uintptr_t) skb_shinfo(skb)->destructor_arg & ~0x1UL); -} - static inline void net_zcopy_put(struct ubuf_info *uarg) { if (uarg) @@ -1872,8 +1856,7 @@ static inline void skb_zcopy_clear(struct sk_buff *skb, bool zerocopy_success) struct ubuf_info *uarg = skb_zcopy(skb); if (uarg) { - if (!skb_zcopy_is_nouarg(skb)) - uarg->ops->complete(skb, uarg, zerocopy_success); + uarg->ops->complete(skb, uarg, zerocopy_success); skb_shinfo(skb)->flags &= ~SKBFL_ALL_ZEROCOPY; } diff --git a/include/net/dst.h b/include/net/dst.h index 307073eae7f8..dbedfe72e1fd 100644 --- a/include/net/dst.h +++ b/include/net/dst.h @@ -455,7 +455,8 @@ static inline unsigned int dst_dev_overhead(struct dst_entry *dst, struct sk_buff *skb) { if (likely(dst)) - return LL_RESERVED_SPACE(dst->dev); + return max_t(unsigned int, skb->mac_len, + LL_RESERVED_SPACE(dst->dev)); return skb->mac_len; } diff --git a/include/net/gue.h b/include/net/gue.h index caefd6da8693..d377155fd0b3 100644 --- a/include/net/gue.h +++ b/include/net/gue.h @@ -84,8 +84,9 @@ static inline size_t guehdr_priv_flags_len(__be32 flags) } /* Validate standard and private flags. Returns non-zero (meaning invalid) - * if there is an unknown standard or private flags, or the options length for - * the flags exceeds the options length specific in hlen of the GUE header. + * if there is an unknown standard or private flags, if the options length for + * the flags exceeds the options length specified in hlen of the GUE header, or + * if a private option contains invalid data. */ static inline int validate_gue_flags(struct guehdr *guehdr, size_t optlen) { @@ -103,8 +104,8 @@ static inline int validate_gue_flags(struct guehdr *guehdr, size_t optlen) /* Private flags are last four bytes accounted in * guehdr_flags_len */ - __be32 pflags = *(__be32 *)((void *)&guehdr[1] + - len - GUE_LEN_PRIV); + void *data = (void *)&guehdr[1] + len; + __be32 pflags = *(__be32 *)(data - GUE_LEN_PRIV); if (pflags & ~GUE_PFLAGS_ALL) return 1; @@ -112,6 +113,16 @@ static inline int validate_gue_flags(struct guehdr *guehdr, size_t optlen) len += guehdr_priv_flags_len(pflags); if (len > optlen) return 1; + + if (pflags & GUE_PFLAG_REMCSUM) { + __be16 *pd = data; + + /* The field offset pd[1] must not be less + * than the start pd[0]. + */ + if (ntohs(pd[1]) < ntohs(pd[0])) + return 1; + } } return 0; diff --git a/include/net/ip6_route.h b/include/net/ip6_route.h index b9e8d2b759e9..0f9b7a260d25 100644 --- a/include/net/ip6_route.h +++ b/include/net/ip6_route.h @@ -101,12 +101,12 @@ static inline struct dst_entry *ip6_route_output(struct net *net, } /* Only conditionally release dst if flags indicates - * !RT6_LOOKUP_F_DST_NOREF or dst is in uncached_list. + * !RT6_LOOKUP_F_DST_NOREF or dst is uncached. */ static inline void ip6_rt_put_flags(struct rt6_info *rt, int flags) { if (!(flags & RT6_LOOKUP_F_DST_NOREF) || - !list_empty(&rt->dst.rt_uncached)) + rt->dst.rt_uncached_list) ip6_rt_put(rt); } diff --git a/include/net/netfilter/nf_conntrack.h b/include/net/netfilter/nf_conntrack.h index bc42dd0e10e6..c39425e54d87 100644 --- a/include/net/netfilter/nf_conntrack.h +++ b/include/net/netfilter/nf_conntrack.h @@ -185,6 +185,11 @@ static inline void nf_ct_put(struct nf_conn *ct) nf_ct_destroy(&ct->ct_general); } +static inline bool nf_ct_shared(const struct nf_conn *ct) +{ + return refcount_read(&ct->ct_general.use) > 1; +} + /* load module; enable/disable conntrack in this namespace */ int nf_ct_netns_get(struct net *net, u8 nfproto); void nf_ct_netns_put(struct net *net, u8 nfproto); diff --git a/include/net/nfc/hci.h b/include/net/nfc/hci.h index 756c11084f65..86ed63e5d533 100644 --- a/include/net/nfc/hci.h +++ b/include/net/nfc/hci.h @@ -144,7 +144,7 @@ struct nfc_hci_dev { data_exchange_cb_t async_cb; void *async_cb_context; - u8 *gb; + u8 gb[NFC_MAX_GT_LEN]; size_t gb_len; unsigned long quirks; diff --git a/include/net/nfc/nfc.h b/include/net/nfc/nfc.h index c54df042db6b..bcafab5c53e5 100644 --- a/include/net/nfc/nfc.h +++ b/include/net/nfc/nfc.h @@ -273,7 +273,8 @@ struct sk_buff *nfc_alloc_recv_skb(unsigned int size, gfp_t gfp); int nfc_set_remote_general_bytes(struct nfc_dev *dev, const u8 *gt, u8 gt_len); -u8 *nfc_get_local_general_bytes(struct nfc_dev *dev, size_t *gb_len); +u8 *nfc_get_local_general_bytes(struct nfc_dev *dev, u8 *out_gb, + size_t gb_max_len, size_t *gb_len); int nfc_fw_download_done(struct nfc_dev *dev, const char *firmware_name, u32 result); diff --git a/include/net/tcp.h b/include/net/tcp.h index 436495ff2271..4416cdf9bf30 100644 --- a/include/net/tcp.h +++ b/include/net/tcp.h @@ -1232,9 +1232,9 @@ static inline bool tcp_skb_can_collapse_to(const struct sk_buff *skb) static inline bool tcp_skb_can_collapse(const struct sk_buff *to, const struct sk_buff *from) { - /* skb_cmp_decrypted() not needed, use tcp_write_collapse_fence() */ return likely(tcp_skb_can_collapse_to(to) && mptcp_skb_can_collapse(to, from) && + !skb_cmp_decrypted(to, from) && skb_pure_zcopy_same(to, from) && skb_frags_readable(to) == skb_frags_readable(from)); } @@ -2327,7 +2327,7 @@ static inline void tcp_rtx_queue_unlink_and_free(struct sk_buff *skb, struct soc static inline void tcp_write_collapse_fence(struct sock *sk) { - struct sk_buff *skb = tcp_write_queue_tail(sk); + struct sk_buff *skb = tcp_write_queue_tail(sk) ?: tcp_rtx_queue_tail(sk); if (skb) TCP_SKB_CB(skb)->eor = 1; diff --git a/net/8021q/vlan_dev.c b/net/8021q/vlan_dev.c index 2859cbac3f26..c949c6a82945 100644 --- a/net/8021q/vlan_dev.c +++ b/net/8021q/vlan_dev.c @@ -55,6 +55,11 @@ static int vlan_dev_hard_header(struct sk_buff *skb, struct net_device *dev, int rc; if (!(vlan->flags & VLAN_FLAG_REORDER_HDR)) { + unsigned int hlen = READ_ONCE(dev->hard_header_len) + + READ_ONCE(dev->needed_headroom); + + if (skb_cow_head(skb, hlen) < 0) + return -ENOMEM; vhdr = skb_push(skb, VLAN_HLEN); vlan_tci = vlan->vlan_id; diff --git a/net/bluetooth/bnep/core.c b/net/bluetooth/bnep/core.c index f7d88c33e23e..ad24d2486665 100644 --- a/net/bluetooth/bnep/core.c +++ b/net/bluetooth/bnep/core.c @@ -270,9 +270,14 @@ static int bnep_rx_extension(struct bnep_session *s, struct sk_buff *skb) BT_DBG("type 0x%x len %u", h->type, h->len); + if (skb->len < h->len) { + err = -EILSEQ; + break; + } + switch (h->type & BNEP_TYPE_MASK) { case BNEP_EXT_CONTROL: - bnep_rx_control(s, skb->data, skb->len); + bnep_rx_control(s, skb->data, h->len); break; default: @@ -373,6 +378,11 @@ static int bnep_rx_frame(struct bnep_session *s, struct sk_buff *skb) goto badframe; } + if ((type & BNEP_TYPE_MASK) == BNEP_CONTROL) { + kfree_skb(skb); + return 0; + } + /* Strip 802.1p header */ if (ntohs(s->eh.h_proto) == ETH_P_8021Q) { if (!skb_pull(skb, 4)) @@ -451,6 +461,11 @@ static int bnep_tx_frame(struct bnep_session *s, struct sk_buff *skb) goto send; } + if (skb->len < ETH_HLEN) { + kfree_skb(skb); + return 0; + } + iv[il++] = (struct kvec) { &type, 1 }; len++; diff --git a/net/bluetooth/bnep/netdev.c b/net/bluetooth/bnep/netdev.c index ee1e39a3daff..b451ef457741 100644 --- a/net/bluetooth/bnep/netdev.c +++ b/net/bluetooth/bnep/netdev.c @@ -166,6 +166,12 @@ static netdev_tx_t bnep_net_xmit(struct sk_buff *skb, BT_DBG("skb %p, dev %p", skb, dev); + if (!pskb_may_pull(skb, ETH_HLEN)) { + dev->stats.tx_dropped++; + kfree_skb(skb); + return NETDEV_TX_OK; + } + #ifdef CONFIG_BT_BNEP_MC_FILTER if (bnep_net_mc_filter(skb, s)) { kfree_skb(skb); @@ -218,7 +224,7 @@ void bnep_net_setup(struct net_device *dev) dev->addr_len = ETH_ALEN; ether_setup(dev); - dev->min_mtu = 0; + dev->min_mtu = ETH_MIN_MTU; dev->max_mtu = ETH_MAX_MTU; dev->priv_flags &= ~IFF_TX_SKB_SHARING; dev->netdev_ops = &bnep_netdev_ops; diff --git a/net/bluetooth/hci_conn.c b/net/bluetooth/hci_conn.c index fa72cf8aaa7a..96195d2fd10f 100644 --- a/net/bluetooth/hci_conn.c +++ b/net/bluetooth/hci_conn.c @@ -2079,6 +2079,8 @@ struct hci_conn *hci_bind_cis(struct hci_dev *hdev, bdaddr_t *dst, cis->conn_timeout = timeout; } + hci_conn_hold(cis); + if (cis->state == BT_CONNECTED) return cis; @@ -2120,7 +2122,6 @@ struct hci_conn *hci_bind_cis(struct hci_dev *hdev, bdaddr_t *dst, return ERR_PTR(-EINVAL); } - hci_conn_hold(cis); cis->state = BT_BOUND; return cis; @@ -2374,10 +2375,13 @@ struct hci_conn *hci_bind_bis(struct hci_dev *hdev, bdaddr_t *dst, __u8 sid, parent = hci_conn_hash_lookup_big(hdev, conn->iso_qos.bcast.big); if (parent && parent != conn) { + hci_conn_hold(parent); link = hci_conn_link(parent, conn); hci_conn_drop(conn); - if (!link) + if (!link) { + hci_conn_drop(parent); return ERR_PTR(-ENOLINK); + } } return conn; @@ -2497,6 +2501,12 @@ struct hci_conn *hci_connect_cis(struct hci_dev *hdev, bdaddr_t *dst, return cis; } + /* The existing link already owns the hold on its parent. */ + if (cis->link) { + hci_conn_drop(le); + return cis; + } + link = hci_conn_link(le, cis); hci_conn_drop(cis); if (!link) { diff --git a/net/bluetooth/hci_sock.c b/net/bluetooth/hci_sock.c index 070ca388f9ac..6d56c77741e1 100644 --- a/net/bluetooth/hci_sock.c +++ b/net/bluetooth/hci_sock.c @@ -164,6 +164,7 @@ static bool is_filtered_packet(struct sock *sk, struct sk_buff *skb) { struct hci_filter *flt; int flt_type, flt_event; + u8 event; /* Apply filter */ flt = &hci_pi(sk)->filter; @@ -177,7 +178,11 @@ static bool is_filtered_packet(struct sock *sk, struct sk_buff *skb) if (hci_skb_pkt_type(skb) != HCI_EVENT_PKT) return false; - flt_event = (*(__u8 *)skb->data & HCI_FLT_EVENT_BITS); + if (skb->len < 1) + return true; + + event = *(__u8 *)skb->data; + flt_event = event & HCI_FLT_EVENT_BITS; if (!hci_test_bit(flt_event, &flt->event_mask)) return true; @@ -186,11 +191,17 @@ static bool is_filtered_packet(struct sock *sk, struct sk_buff *skb) if (!flt->opcode) return false; - if (flt_event == HCI_EV_CMD_COMPLETE && + if (event == HCI_EV_CMD_COMPLETE && skb->len < 5) + return true; + + if (event == HCI_EV_CMD_COMPLETE && flt->opcode != get_unaligned((__le16 *)(skb->data + 3))) return true; - if (flt_event == HCI_EV_CMD_STATUS && + if (event == HCI_EV_CMD_STATUS && skb->len < 6) + return true; + + if (event == HCI_EV_CMD_STATUS && flt->opcode != get_unaligned((__le16 *)(skb->data + 4))) return true; @@ -1881,7 +1892,8 @@ static int hci_sock_sendmsg(struct socket *sock, struct msghdr *msg, u16 ocf = hci_opcode_ocf(opcode); if (((ogf > HCI_SFLT_MAX_OGF) || - !hci_test_bit(ocf & HCI_FLT_OCF_BITS, + (ocf > HCI_FLT_OCF_BITS) || + !hci_test_bit(ocf, &hci_sec_filter.ocf_mask[ogf])) && !capable(CAP_NET_RAW)) { err = -EPERM; diff --git a/net/bluetooth/iso.c b/net/bluetooth/iso.c index eb99653f33f9..7657c2a0abbf 100644 --- a/net/bluetooth/iso.c +++ b/net/bluetooth/iso.c @@ -496,6 +496,7 @@ static int iso_connect_cis(struct sock *sk) struct hci_dev *hdev; bdaddr_t src, dst; u8 src_type; + bool already_attached; int err; lock_sock(sk); @@ -568,8 +569,14 @@ static int iso_connect_cis(struct sock *sk) goto unlock; } + iso_conn_lock(conn); + already_attached = iso_pi(sk)->conn == conn && conn->sk == sk; + iso_conn_unlock(conn); + err = iso_chan_add(conn, sk, NULL); iso_conn_put(conn); + if (already_attached || err == -EBUSY) + hci_conn_drop(hcon); if (err) goto unlock; diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 644e31160d55..aaa2a1cd489a 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -6702,9 +6702,17 @@ static int l2cap_stream_rx(struct l2cap_chan *chan, struct l2cap_ctrl *control, static int l2cap_data_rcv(struct l2cap_chan *chan, struct sk_buff *skb) { struct l2cap_ctrl *control = &bt_cb(skb)->l2cap; - u16 len; + u16 len, min_len; u8 event; + min_len = test_bit(FLAG_EXT_CTRL, &chan->flags) ? + L2CAP_EXT_CTRL_SIZE : L2CAP_ENH_CTRL_SIZE; + if (chan->fcs == L2CAP_FCS_CRC16) + min_len += L2CAP_FCS_SIZE; + + if (skb->len < min_len) + goto drop; + __unpack_control(chan, skb); len = skb->len; diff --git a/net/bluetooth/mgmt.c b/net/bluetooth/mgmt.c index ac4864e56ec7..41956cdde982 100644 --- a/net/bluetooth/mgmt.c +++ b/net/bluetooth/mgmt.c @@ -496,13 +496,9 @@ static int read_unconf_index_list(struct sock *sk, struct hci_dev *hdev, read_lock(&hci_dev_list_lock); - count = 0; - list_for_each_entry(d, &hci_dev_list, list) { - if (hci_dev_test_flag(d, HCI_UNCONFIGURED)) - count++; - } + count = list_count_nodes(&hci_dev_list); - rp_len = sizeof(*rp) + (2 * count); + rp_len = sizeof(*rp) + (sizeof(__le16) * count); rp = kmalloc(rp_len, GFP_ATOMIC); if (!rp) { read_unlock(&hci_dev_list_lock); @@ -2316,6 +2312,8 @@ static void mesh_send_start_complete(struct hci_dev *hdev, void *data, int err) hci_dev_clear_flag(hdev, HCI_MESH_SENDING); /* Send Complete Error Code for handle */ mesh_send_complete(hdev, mesh_tx, false); + if (err != -ECANCELED) + mesh_next(hdev, NULL, 0); return; } @@ -2425,19 +2423,28 @@ static int send_cancel(struct hci_dev *hdev, void *data) do { mesh_tx = mgmt_mesh_next(hdev, cmd->sk); - if (mesh_tx) - mesh_send_complete(hdev, mesh_tx, false); + if (mesh_tx) { + if (!hci_cmd_sync_dequeue(hdev, mesh_send_sync, + mesh_tx, NULL)) + mesh_send_complete(hdev, mesh_tx, false); + } } while (mesh_tx); } else { mesh_tx = mgmt_mesh_find(hdev, cancel->handle); - if (mesh_tx && mesh_tx->sk == cmd->sk) - mesh_send_complete(hdev, mesh_tx, false); + if (mesh_tx && mesh_tx->sk == cmd->sk) { + if (!hci_cmd_sync_dequeue(hdev, mesh_send_sync, + mesh_tx, NULL)) + mesh_send_complete(hdev, mesh_tx, false); + } } mgmt_cmd_complete(cmd->sk, hdev->id, MGMT_OP_MESH_SEND_CANCEL, 0, NULL, 0); + if (!hci_dev_test_flag(hdev, HCI_MESH_SENDING)) + mesh_next(hdev, NULL, 0); + return 0; } diff --git a/net/bluetooth/rfcomm/core.c b/net/bluetooth/rfcomm/core.c index f7463f092283..d91e2a6ee26c 100644 --- a/net/bluetooth/rfcomm/core.c +++ b/net/bluetooth/rfcomm/core.c @@ -1817,7 +1817,8 @@ static struct rfcomm_session *rfcomm_recv_frame(struct rfcomm_session *s, return s; } - if (skb->len < sizeof(*hdr) + 1) { + if (skb->len < sizeof(*hdr) + 1 || + (!__test_ea(hdr->len) && skb->len < sizeof(*hdr) + 2)) { kfree_skb(skb); return s; } diff --git a/net/bluetooth/rfcomm/sock.c b/net/bluetooth/rfcomm/sock.c index e2486bc11cbc..fb924d0e34ec 100644 --- a/net/bluetooth/rfcomm/sock.c +++ b/net/bluetooth/rfcomm/sock.c @@ -786,8 +786,10 @@ static int rfcomm_sock_getsockopt_old(struct socket *sock, int optname, break; case RFCOMM_CONNINFO: - if (sk->sk_state != BT_CONNECTED && - !rfcomm_pi(sk)->dlc->defer_setup) { + if ((sk->sk_state != BT_CONNECTED && + !(sk->sk_state == BT_CONNECT2 && + rfcomm_pi(sk)->dlc->defer_setup)) || + !rfcomm_pi(sk)->dlc->session) { err = -ENOTCONN; break; } diff --git a/net/bluetooth/smp.c b/net/bluetooth/smp.c index 6091c47cb002..d23f9d0729c4 100644 --- a/net/bluetooth/smp.c +++ b/net/bluetooth/smp.c @@ -2269,6 +2269,23 @@ static u8 smp_cmd_security_req(struct l2cap_conn *conn, struct sk_buff *skb) bt_dev_dbg(hdev, "conn %p", conn); + /* SMP over BR/EDR only covers cross-transport key derivation; the + * Security Request procedure has no BR/EDR counterpart. Reject it + * here, otherwise smp_ltk_encrypt() finds the peer's LE LTK + * (ADDR_LE_DEV_PUBLIC and BDADDR_BREDR are both 0) and issues + * HCI_OP_LE_START_ENC on the ACL handle, which the controller + * rejects and hci_cs_le_start_enc() turns into a disconnect. Reply + * without smp_failure(): this is not an authentication failure, and + * MGMT_EV_AUTH_FAILED would make bluetoothd drop the device. + */ + if (hcon->type != LE_LINK) { + u8 reason = SMP_CMD_NOTSUPP; + + smp_send_cmd(conn, SMP_CMD_PAIRING_FAIL, sizeof(reason), + &reason); + return 0; + } + if (skb->len < sizeof(*rp)) return SMP_INVALID_PARAMS; diff --git a/net/bridge/br_mdb.c b/net/bridge/br_mdb.c index e0c7020b12f5..a01bd280c722 100644 --- a/net/bridge/br_mdb.c +++ b/net/bridge/br_mdb.c @@ -1523,6 +1523,8 @@ static void br_mdb_flush_pgs(struct net_bridge *br, } br_multicast_del_pg(mp, p, pp); + /* br_multicast_del_pg() can remove other groups from this list. */ + pp = &mp->ports; } } diff --git a/net/bridge/br_stp_bpdu.c b/net/bridge/br_stp_bpdu.c index 74ec42ba1e7d..21d092f5acbb 100644 --- a/net/bridge/br_stp_bpdu.c +++ b/net/bridge/br_stp_bpdu.c @@ -52,7 +52,10 @@ static void br_send_bpdu(struct net_bridge_port *p, LLC_SAP_BSPAN, LLC_PDU_CMD); llc_pdu_init_as_ui_cmd(skb); - llc_mac_hdr_init(skb, p->dev->dev_addr, p->br->group_addr); + if (llc_mac_hdr_init(skb, p->dev->dev_addr, p->br->group_addr)) { + kfree_skb(skb); + return; + } skb_reset_mac_header(skb); diff --git a/net/core/dev.c b/net/core/dev.c index c67900354fa6..f660fccfc0db 100644 --- a/net/core/dev.c +++ b/net/core/dev.c @@ -2901,7 +2901,7 @@ int __netif_set_xps_queue(struct net_device *dev, const unsigned long *mask, dev = netdev_get_tx_queue(dev, index)->sb_dev ? : dev; tc = netdev_txq_to_tc(dev, index); - if (tc < 0) + if (tc < 0 || tc >= num_tc) return -EINVAL; } @@ -5376,7 +5376,8 @@ void kick_defer_list_purge(unsigned int cpu) backlog_unlock_irq_restore(sd, flags); } else if (!cmpxchg(&sd->defer_ipi_scheduled, 0, 1)) { - smp_call_function_single_async(cpu, &sd->defer_csd); + if (smp_call_function_single_async(cpu, &sd->defer_csd)) + WRITE_ONCE(sd->defer_ipi_scheduled, 0); } } @@ -6900,25 +6901,35 @@ bool napi_complete_done(struct napi_struct *n, int work_done) } EXPORT_SYMBOL(napi_complete_done); -static void skb_defer_free_flush(void) +static void __skb_defer_free_flush(struct skb_defer_node *sdn, int budget) { struct llist_node *free_list; struct sk_buff *skb, *next; + + if (llist_empty(&sdn->defer_list)) + return; + atomic_long_set(&sdn->defer_count, 0); + free_list = llist_del_all(&sdn->defer_list); + + llist_for_each_entry_safe(skb, next, free_list, ll_node) { + prefetch(next); + napi_consume_skb(skb, budget); + } +} + +void skb_defer_node_flush(struct skb_defer_node *sdn) +{ + __skb_defer_free_flush(sdn, 0); +} + +static void skb_defer_free_flush(void) +{ struct skb_defer_node *sdn; int node; for_each_node(node) { sdn = this_cpu_ptr(net_hotdata.skb_defer_nodes) + node; - - if (llist_empty(&sdn->defer_list)) - continue; - atomic_long_set(&sdn->defer_count, 0); - free_list = llist_del_all(&sdn->defer_list); - - llist_for_each_entry_safe(skb, next, free_list, ll_node) { - prefetch(next); - napi_consume_skb(skb, 1); - } + __skb_defer_free_flush(sdn, 1); } } @@ -12897,6 +12908,7 @@ static int dev_cpu_dead(unsigned int oldcpu) struct sk_buff **list_skb; struct sk_buff *skb; unsigned int cpu; + int node; struct softnet_data *sd, *oldsd, *remsd = NULL; local_irq_disable(); @@ -12957,6 +12969,17 @@ static int dev_cpu_dead(unsigned int oldcpu) rps_input_queue_head_incr(oldsd); } + for_each_node(node) + skb_defer_node_flush(per_cpu_ptr(net_hotdata.skb_defer_nodes, + oldcpu) + node); + node = cpu_to_node(oldcpu); + if (node_possible(node) && + !cpumask_intersects(cpumask_of_node(node), cpu_online_mask)) { + for_each_possible_cpu(cpu) + skb_defer_node_flush(per_cpu_ptr(net_hotdata.skb_defer_nodes, + cpu) + node); + } + return 0; } diff --git a/net/core/dev.h b/net/core/dev.h index b757faead4d1..04fb0e9a571e 100644 --- a/net/core/dev.h +++ b/net/core/dev.h @@ -399,6 +399,8 @@ static inline void napi_assert_will_not_race(const struct napi_struct *napi) WARN_ON(READ_ONCE(napi->list_owner) != -1); } +struct skb_defer_node; +void skb_defer_node_flush(struct skb_defer_node *sdn); void kick_defer_list_purge(unsigned int cpu); int dev_set_hwtstamp_phylib(struct net_device *dev, diff --git a/net/core/dev_ioctl.c b/net/core/dev_ioctl.c index a320e264eaaf..164643140a52 100644 --- a/net/core/dev_ioctl.c +++ b/net/core/dev_ioctl.c @@ -276,19 +276,18 @@ int dev_get_hwtstamp_phylib(struct net_device *dev, if (phy_is_default_hwtstamp(dev->phydev)) return phy_hwtstamp_get(dev->phydev, cfg); + if (!dev->netdev_ops->ndo_hwtstamp_get) + return -EOPNOTSUPP; + return dev->netdev_ops->ndo_hwtstamp_get(dev, cfg); } static int dev_get_hwtstamp(struct net_device *dev, struct ifreq *ifr) { - const struct net_device_ops *ops = dev->netdev_ops; struct kernel_hwtstamp_config kernel_cfg = {}; struct hwtstamp_config cfg; int err; - if (!ops->ndo_hwtstamp_get) - return -EOPNOTSUPP; - if (!netif_device_present(dev)) return -ENODEV; @@ -359,12 +358,18 @@ int dev_set_hwtstamp_phylib(struct net_device *dev, cfg->source = phy_ts ? HWTSTAMP_SOURCE_PHYLIB : HWTSTAMP_SOURCE_NETDEV; if (phy_ts && dev->see_all_hwtstamp_requests) { + if (!ops->ndo_hwtstamp_get) + return -EOPNOTSUPP; + err = ops->ndo_hwtstamp_get(dev, &old_cfg); if (err) return err; } if (!phy_ts || dev->see_all_hwtstamp_requests) { + if (!ops->ndo_hwtstamp_set) + return -EOPNOTSUPP; + err = ops->ndo_hwtstamp_set(dev, cfg, extack); if (err) { if (extack->_msg) @@ -390,7 +395,6 @@ int dev_set_hwtstamp_phylib(struct net_device *dev, static int dev_set_hwtstamp(struct net_device *dev, struct ifreq *ifr) { - const struct net_device_ops *ops = dev->netdev_ops; struct kernel_hwtstamp_config kernel_cfg = {}; struct netlink_ext_ack extack = {}; struct hwtstamp_config cfg; @@ -413,9 +417,6 @@ static int dev_set_hwtstamp(struct net_device *dev, struct ifreq *ifr) return err; } - if (!ops->ndo_hwtstamp_set) - return -EOPNOTSUPP; - if (!netif_device_present(dev)) return -ENODEV; @@ -441,15 +442,11 @@ static int dev_set_hwtstamp(struct net_device *dev, struct ifreq *ifr) int generic_hwtstamp_get_lower(struct net_device *dev, struct kernel_hwtstamp_config *kernel_cfg) { - const struct net_device_ops *ops = dev->netdev_ops; int err; if (!netif_device_present(dev)) return -ENODEV; - if (!ops->ndo_hwtstamp_get) - return -EOPNOTSUPP; - netdev_lock_ops(dev); err = dev_get_hwtstamp_phylib(dev, kernel_cfg); netdev_unlock_ops(dev); @@ -462,15 +459,11 @@ int generic_hwtstamp_set_lower(struct net_device *dev, struct kernel_hwtstamp_config *kernel_cfg, struct netlink_ext_ack *extack) { - const struct net_device_ops *ops = dev->netdev_ops; int err; if (!netif_device_present(dev)) return -ENODEV; - if (!ops->ndo_hwtstamp_set) - return -EOPNOTSUPP; - netdev_lock_ops(dev); err = dev_set_hwtstamp_phylib(dev, kernel_cfg, extack); netdev_unlock_ops(dev); diff --git a/net/core/netdev-genl.c b/net/core/netdev-genl.c index cb18db681640..33b9f4eb9565 100644 --- a/net/core/netdev-genl.c +++ b/net/core/netdev-genl.c @@ -1168,6 +1168,12 @@ netdev_find_netmem_tx_dev(struct net_device *dev) return NULL; } +/* Note: NETDEV_CMD_BIND_TX is intentionally unprivileged (no + * GENL_ADMIN_PERM / GENL_UNS_ADMIN_PERM). Unlike bind-rx, which configures + * shared NIC RX queues, bind-tx only DMA-maps the caller's dmabuf so they can + * transmit from it on their own sockets without affecting other traffic or + * device state. + */ int netdev_nl_bind_tx_doit(struct sk_buff *skb, struct genl_info *info) { struct net_devmem_dmabuf_binding *binding; diff --git a/net/core/skbuff.c b/net/core/skbuff.c index 609f2c7f4a47..c3042d822afa 100644 --- a/net/core/skbuff.c +++ b/net/core/skbuff.c @@ -5977,7 +5977,8 @@ static int skb_checksum_setup_ipv6(struct sk_buff *skb, bool recalculate) err = skb_maybe_pull_tail(skb, off + sizeof(struct ipv6_opt_hdr), - MAX_IPV6_HDR_LEN); + off + + sizeof(struct ipv6_opt_hdr)); if (err < 0) goto out; @@ -5992,7 +5993,8 @@ static int skb_checksum_setup_ipv6(struct sk_buff *skb, bool recalculate) err = skb_maybe_pull_tail(skb, off + sizeof(struct ip_auth_hdr), - MAX_IPV6_HDR_LEN); + off + + sizeof(struct ip_auth_hdr)); if (err < 0) goto out; @@ -6007,7 +6009,8 @@ static int skb_checksum_setup_ipv6(struct sk_buff *skb, bool recalculate) err = skb_maybe_pull_tail(skb, off + sizeof(struct frag_hdr), - MAX_IPV6_HDR_LEN); + off + + sizeof(struct frag_hdr)); if (err < 0) goto out; @@ -7357,8 +7360,8 @@ void skb_attempt_defer_free(struct sk_buff *skb) struct skb_defer_node *sdn; unsigned long defer_count; unsigned int defer_max; + int cpu, my_cpu; bool kick; - int cpu; if (static_branch_unlikely(&skb_defer_disable_key)) goto nodefer; @@ -7368,7 +7371,8 @@ void skb_attempt_defer_free(struct sk_buff *skb) goto nodefer; cpu = skb->alloc_cpu; - if (cpu == raw_smp_processor_id() || + my_cpu = raw_smp_processor_id(); + if (cpu == my_cpu || WARN_ON_ONCE(cpu >= nr_cpu_ids) || !cpu_online(cpu)) { nodefer: kfree_skb_napi_cache(skb); @@ -7379,7 +7383,7 @@ nodefer: kfree_skb_napi_cache(skb); DEBUG_NET_WARN_ON_ONCE(skb->destructor); DEBUG_NET_WARN_ON_ONCE(skb_nfct(skb)); - sdn = per_cpu_ptr(net_hotdata.skb_defer_nodes, cpu) + numa_node_id(); + sdn = per_cpu_ptr(net_hotdata.skb_defer_nodes, cpu) + cpu_to_node(my_cpu); defer_max = READ_ONCE(net_hotdata.sysctl_skb_defer_max); defer_count = atomic_long_inc_return(&sdn->defer_count); @@ -7389,6 +7393,11 @@ nodefer: kfree_skb_napi_cache(skb); llist_add(&skb->ll_node, &sdn->defer_list); + if (unlikely(!cpu_online(cpu) || my_cpu != raw_smp_processor_id())) { + skb_defer_node_flush(sdn); + return; + } + /* Send an IPI every time queue reaches half capacity. */ kick = (defer_count - 1) == (defer_max >> 1); diff --git a/net/ethtool/common.h b/net/ethtool/common.h index 4e5356e26f40..ae32e7fdb563 100644 --- a/net/ethtool/common.h +++ b/net/ethtool/common.h @@ -163,6 +163,8 @@ ethtool_ioctl_needs_rtnl(const struct net_device *dev, u32 ethcmd) return ops->op_needs_rtnl & ETHTOOL_OP_NEEDS_RTNL_RSS; case ETHTOOL_GLINK: return ops->op_needs_rtnl & ETHTOOL_OP_NEEDS_RTNL_GLINK; + case ETHTOOL_TEST: + return ops->op_needs_rtnl & ETHTOOL_OP_NEEDS_RTNL_TEST; } return false; } diff --git a/net/ipv4/arp.c b/net/ipv4/arp.c index d409f606aec0..60009d92e071 100644 --- a/net/ipv4/arp.c +++ b/net/ipv4/arp.c @@ -1278,6 +1278,7 @@ int arp_ioctl(struct net *net, unsigned int cmd, void __user *arg) err = copy_from_user(&r, arg, sizeof(struct arpreq)); if (err) return -EFAULT; + r.arp_dev[IFNAMSIZ - 1] = '\0'; break; default: return -EINVAL; diff --git a/net/ipv4/devinet.c b/net/ipv4/devinet.c index a90be57c63be..5b6b11c943e4 100644 --- a/net/ipv4/devinet.c +++ b/net/ipv4/devinet.c @@ -2117,9 +2117,10 @@ static int inet_validate_link_af(const struct net_device *dev, return err; if (tb[IFLA_INET_CONF]) { - err = nla_parse_nested(nested_tb, IPV4_DEVCONF_MAX, - tb[IFLA_INET_CONF], inet_devconf_policy, - extack); + err = nla_parse(nested_tb, IPV4_DEVCONF_MAX, + nla_data(tb[IFLA_INET_CONF]), + nla_len(tb[IFLA_INET_CONF]), + inet_devconf_policy, extack); if (err < 0) return err; diff --git a/net/ipv4/fib_semantics.c b/net/ipv4/fib_semantics.c index 7a362f2e2c2b..5c9021ea3a79 100644 --- a/net/ipv4/fib_semantics.c +++ b/net/ipv4/fib_semantics.c @@ -2176,6 +2176,15 @@ static bool fib_good_nh(const struct fib_nh *nh) return !!(state & NUD_VALID); } +static __be32 fib_nh_saddr(struct net *net, const struct fib_info *fi, + struct fib_nh *nh, int genid) +{ + if (READ_ONCE(nh->nh_saddr_genid) == genid) + return READ_ONCE(nh->nh_saddr); + + return fib_info_update_nhc_saddr(net, &nh->nh_common, fi->fib_scope); +} + void fib_select_multipath(struct fib_result *res, int hash, const struct flowi4 *fl4) { @@ -2184,6 +2193,7 @@ void fib_select_multipath(struct fib_result *res, int hash, bool use_neigh; int score = -1; __be32 saddr; + int genid; if (unlikely(res->fi->nh)) { nexthop_path_fib_result(res, hash); @@ -2192,6 +2202,7 @@ void fib_select_multipath(struct fib_result *res, int hash, use_neigh = READ_ONCE(net->ipv4.sysctl_fib_multipath_use_neigh); saddr = fl4 ? fl4->saddr : 0; + genid = saddr ? atomic_read(&net->ipv4.dev_addr_genid) : 0; change_nexthops(fi) { int nh_upper_bound, nh_score = 0; @@ -2204,7 +2215,7 @@ void fib_select_multipath(struct fib_result *res, int hash, (use_neigh && !fib_good_nh(nexthop_nh))) continue; - if (saddr && nexthop_nh->nh_saddr == saddr) + if (saddr && fib_nh_saddr(net, fi, nexthop_nh, genid) == saddr) nh_score += 2; if (hash <= nh_upper_bound) nh_score++; diff --git a/net/ipv4/fou_core.c b/net/ipv4/fou_core.c index 5e867f1b5c1d..076fdca44f51 100644 --- a/net/ipv4/fou_core.c +++ b/net/ipv4/fou_core.c @@ -600,6 +600,10 @@ static int fou_create(struct net *net, struct fou_cfg *cfg, /* Initial for fou type */ switch (cfg->type) { case FOU_ENCAP_DIRECT: + if (!cfg->protocol) { + err = -EINVAL; + goto error; + } tunnel_cfg.encap_rcv = fou_udp_recv; tunnel_cfg.gro_receive = fou_gro_receive; tunnel_cfg.gro_complete = fou_gro_complete; diff --git a/net/ipv4/ip_gre.c b/net/ipv4/ip_gre.c index 82309efd417e..e4878e9aa636 100644 --- a/net/ipv4/ip_gre.c +++ b/net/ipv4/ip_gre.c @@ -1464,6 +1464,12 @@ static int ipgre_changelink(struct net_device *dev, struct nlattr *tb[], if (!rtnl_dev_link_net_capable(dev, t->net)) return -EPERM; + if (data && data[IFLA_GRE_COLLECT_METADATA] && !t->collect_md) { + NL_SET_ERR_MSG(extack, + "Enabling collect_md on an existing device is not supported"); + return -EOPNOTSUPP; + } + err = ipgre_newlink_encap_setup(dev, data); if (err) return err; @@ -1496,6 +1502,12 @@ static int erspan_changelink(struct net_device *dev, struct nlattr *tb[], if (!rtnl_dev_link_net_capable(dev, t->net)) return -EPERM; + if (data && data[IFLA_GRE_COLLECT_METADATA] && !t->collect_md) { + NL_SET_ERR_MSG(extack, + "Enabling collect_md on an existing device is not supported"); + return -EOPNOTSUPP; + } + err = ipgre_newlink_encap_setup(dev, data); if (err) return err; diff --git a/net/ipv4/ipconfig.c b/net/ipv4/ipconfig.c index 155db067eaec..1b8585404a41 100644 --- a/net/ipv4/ipconfig.c +++ b/net/ipv4/ipconfig.c @@ -676,6 +676,24 @@ static const u8 ic_bootp_cookie[4] = { 99, 130, 83, 99 }; #ifdef IPCONFIG_DHCP +static bool __init +ic_dhcp_add_option(u8 **options, const u8 *end, u8 type, const void *value, + int len) +{ + u8 *e = *options; + + /* leave room for the option header and the END marker */ + if (len > U8_MAX || end - e < len + 3) + return false; + + *e++ = type; + *e++ = len; + memcpy(e, value, len); + *options = e + len; + + return true; +} + static void __init ic_dhcp_init_options(u8 *options, struct ic_device *d) { @@ -691,6 +709,7 @@ ic_dhcp_init_options(u8 *options, struct ic_device *d) 42, /* NTP servers */ }; u8 mt = (ic_servaddr == NONE) ? DHCPDISCOVER : DHCPREQUEST; + u8 *end = options + sizeof(((struct bootp_pkt *)0)->exten); u8 *e = options; int len; @@ -721,31 +740,19 @@ ic_dhcp_init_options(u8 *options, struct ic_device *d) e += sizeof(ic_req_params); if (ic_host_name_set) { - *e++ = 12; /* host-name */ len = strlen(utsname()->nodename); - *e++ = len; - memcpy(e, utsname()->nodename, len); - e += len; + ic_dhcp_add_option(&e, end, 12, utsname()->nodename, len); } if (*vendor_class_identifier) { - pr_info("DHCP: sending class identifier \"%s\"\n", - vendor_class_identifier); - *e++ = 60; /* Class-identifier */ len = strlen(vendor_class_identifier); - *e++ = len; - memcpy(e, vendor_class_identifier, len); - e += len; + if (ic_dhcp_add_option(&e, end, 60, vendor_class_identifier, len)) + pr_info("DHCP: sending class identifier \"%s\"\n", + vendor_class_identifier); } len = strlen(dhcp_client_identifier + 1); - /* the minimum length of identifier is 2, include 1 byte type, - * and can not be larger than the length of options - */ - if (len >= 1 && len < 312 - (e - options) - 1) { - *e++ = 61; - *e++ = len + 1; - memcpy(e, dhcp_client_identifier, len + 1); - e += len + 1; - } + /* the minimum length of identifier is 2, include 1 byte type */ + if (len >= 1) + ic_dhcp_add_option(&e, end, 61, dhcp_client_identifier, len + 1); *e++ = 255; /* End of the list */ } diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c index d960e3de7d50..e0c392e29de5 100644 --- a/net/ipv4/tcp_output.c +++ b/net/ipv4/tcp_output.c @@ -3886,6 +3886,7 @@ void tcp_send_active_reset(struct sock *sk, enum sk_rst_reason reason) */ int tcp_send_synack(struct sock *sk) { + struct tcp_sock *tp = tcp_sk(sk); struct sk_buff *skb; skb = tcp_rtx_queue_head(sk); @@ -3903,6 +3904,8 @@ int tcp_send_synack(struct sock *sk) if (!nskb) return -ENOMEM; INIT_LIST_HEAD(&nskb->tcp_tsorted_anchor); + if (skb == tp->retransmit_skb_hint) + tp->retransmit_skb_hint = nskb; tcp_highest_sack_replace(sk, skb, nskb); tcp_rtx_queue_unlink_and_free(skb, sk); __skb_header_release(nskb); diff --git a/net/ipv4/udp.c b/net/ipv4/udp.c index bb8cfc62cb00..b090bd1f59e8 100644 --- a/net/ipv4/udp.c +++ b/net/ipv4/udp.c @@ -617,14 +617,23 @@ void udp_lib_hash4(struct sock *sk, u16 hash) struct net *net = sock_net(sk); struct udp_table *udptable; - /* Connected udp socket can re-connect to another remote address, which - * will be handled by rehash. Thus no need to redo hash4 here. - */ - if (udp_hashed4(sk)) - return; - udptable = net->ipv4.udp_table; hslot = udp_hashslot(udptable, net, udp_sk(sk)->udp_port_hash); + + /* A connected socket can re-connect to another address. rehash() + * relocates it, but only runs when the local address changes, so a + * socket bound to a specific address would stay filed under the + * previous peer's hash. Move it here. + */ + if (udp_hashed4(sk)) { + if (udp_sk(sk)->udp_lrpa_hash != hash) { + spin_lock_bh(&hslot->lock); + udp_rehash4(udptable, sk, hash); + spin_unlock_bh(&hslot->lock); + } + return; + } + hslot2 = udp_hashslot2(udptable, udp_sk(sk)->udp_portaddr_hash); hslot4 = udp_hashslot4(udptable, hash); udp_sk(sk)->udp_lrpa_hash = hash; @@ -2197,9 +2206,31 @@ int __udp_disconnect(struct sock *sk, int flags) } EXPORT_SYMBOL(__udp_disconnect); +/* __udp_disconnect() takes a socket out of the 4-tuple hash table only via + * ->rehash() or ->unhash(), and neither runs for a socket bound to a + * specific address and port. Remove it here, before its peer is cleared. + */ +static void udp_unhash4_on_disconnect(struct sock *sk) +{ + struct net *net = sock_net(sk); + struct udp_table *udptable; + struct udp_hslot *hslot; + + if (!udp_hashed4(sk)) + return; + + udptable = net->ipv4.udp_table; + hslot = udp_hashslot(udptable, net, udp_sk(sk)->udp_port_hash); + + spin_lock_bh(&hslot->lock); + udp_unhash4(udptable, sk); + spin_unlock_bh(&hslot->lock); +} + int udp_disconnect(struct sock *sk, int flags) { lock_sock(sk); + udp_unhash4_on_disconnect(sk); __udp_disconnect(sk, flags); release_sock(sk); return 0; @@ -3131,6 +3162,7 @@ int udp_abort(struct sock *sk, int err) sk->sk_err = err; sk_error_report(sk); + udp_unhash4_on_disconnect(sk); __udp_disconnect(sk, 0); out: diff --git a/net/ipv6/exthdrs_core.c b/net/ipv6/exthdrs_core.c index 9d06d487e8b1..4a9748338cf4 100644 --- a/net/ipv6/exthdrs_core.c +++ b/net/ipv6/exthdrs_core.c @@ -278,6 +278,9 @@ int ipv6_find_hdr(const struct sk_buff *skb, unsigned int *offset, hdrlen = ipv6_optlen(hp); if (!found) { + if (skb->len - start < hdrlen) + return -EBADMSG; + nexthdr = hp->nexthdr; start += hdrlen; } diff --git a/net/ipv6/ip6_fib.c b/net/ipv6/ip6_fib.c index 9ea75703b38d..9ff761962b45 100644 --- a/net/ipv6/ip6_fib.c +++ b/net/ipv6/ip6_fib.c @@ -1043,8 +1043,8 @@ static void fib6_purge_rt(struct fib6_info *rt, struct fib6_node *fn, struct fib6_table *table = rt->fib6_table; /* Flush all cached dst in exception table */ - rt6_flush_exceptions(rt); fib6_drop_pcpu_from(rt); + rt6_flush_exceptions(rt); if (rt->nh) { spin_lock(&rt->nh->lock); diff --git a/net/ipv6/ip6_gre.c b/net/ipv6/ip6_gre.c index 8ebda0b6a78b..e61cb10b50dc 100644 --- a/net/ipv6/ip6_gre.c +++ b/net/ipv6/ip6_gre.c @@ -2279,7 +2279,7 @@ static int ip6erspan_changelink(struct net_device *dev, struct nlattr *tb[], return PTR_ERR(t); ip6erspan_set_version(data, &p); - ip6gre_tunnel_unlink_md(ign, t); + ip6erspan_tunnel_unlink_md(ign, t); ip6gre_tunnel_unlink(ign, t); ip6erspan_tnl_change(t, &p, !tb[IFLA_MTU]); ip6erspan_tunnel_link_md(ign, t); diff --git a/net/ipv6/netfilter/ip6t_rpfilter.c b/net/ipv6/netfilter/ip6t_rpfilter.c index 67c87a88cde4..b5def30c3127 100644 --- a/net/ipv6/netfilter/ip6t_rpfilter.c +++ b/net/ipv6/netfilter/ip6t_rpfilter.c @@ -61,7 +61,7 @@ static bool rpfilter_lookup_reverse6(struct net *net, const struct sk_buff *skb, fl6.flowi6_oif = dev->ifindex; rt = (void *)ip6_route_lookup(net, &fl6, skb, lookup_flags); - if (rt->dst.error) + if (rt->dst.error || !rt->rt6i_idev) goto out; if (rt->rt6i_flags & (RTF_REJECT|RTF_ANYCAST)) diff --git a/net/ipv6/netfilter/ip6t_rt.c b/net/ipv6/netfilter/ip6t_rt.c index 8051425213dd..9880faf3cc7d 100644 --- a/net/ipv6/netfilter/ip6t_rt.c +++ b/net/ipv6/netfilter/ip6t_rt.c @@ -96,7 +96,8 @@ static bool rt_mt6(const struct sk_buff *skb, struct xt_action_param *par) unsigned int i = 0; for (temp = 0; - temp < (unsigned int)((hdrlen - 8) / 16); + temp < (unsigned int)((hdrlen - 8) / 16) && + i < rtinfo->addrnr; temp++) { ap = skb_header_pointer(skb, ptr @@ -112,8 +113,6 @@ static bool rt_mt6(const struct sk_buff *skb, struct xt_action_param *par) if (ipv6_addr_equal(ap, &rtinfo->addrs[i])) i++; - if (i == rtinfo->addrnr) - break; } if (i == rtinfo->addrnr) return ret; @@ -162,6 +161,12 @@ static int rt_mt6_check(const struct xt_mtchk_param *par) pr_info_ratelimited("too many addresses specified\n"); return -EINVAL; } + + if ((rtinfo->flags & IP6T_RT_FST_MASK) && !rtinfo->addrnr) { + pr_info_ratelimited("address list match requested but addrnr is 0\n"); + return -EINVAL; + } + if ((rtinfo->flags & (IP6T_RT_RES | IP6T_RT_FST_MASK)) && (!(rtinfo->flags & IP6T_RT_TYP) || (rtinfo->rt_type != 0) || diff --git a/net/ipv6/route.c b/net/ipv6/route.c index 08bd68f1b5bb..153ce16628c1 100644 --- a/net/ipv6/route.c +++ b/net/ipv6/route.c @@ -139,6 +139,7 @@ void rt6_uncached_list_add(struct rt6_info *rt) { struct uncached_list *ul = raw_cpu_ptr(&rt6_uncached_list); + /* Set once and never cleared: non-NULL marks an uncached route. */ rt->dst.rt_uncached_list = ul; spin_lock_bh(&ul->lock); @@ -1729,6 +1730,11 @@ static int rt6_insert_exception(struct rt6_info *nrt, spin_lock_bh(&rt6_exception_lock); + if (f6i->fib6_destroying) { + err = -ENOENT; + goto out; + } + bucket = rcu_dereference_protected(nh->rt6i_exception_bucket, lockdep_is_held(&rt6_exception_lock)); if (!bucket) { @@ -2721,8 +2727,8 @@ struct dst_entry *ip6_route_output_flags(struct net *net, rcu_read_lock(); dst = ip6_route_output_flags_noref(net, sk, fl6, flags); rt6 = dst_rt6_info(dst); - /* For dst cached in uncached_list, refcnt is already taken. */ - if (list_empty(&rt6->dst.rt_uncached) && !dst_hold_safe(dst)) { + /* For an uncached dst, refcnt is already taken. */ + if (!rt6->dst.rt_uncached_list && !dst_hold_safe(dst)) { dst = &net->ipv6.ip6_null_entry->dst; dst_hold(dst); } @@ -2831,7 +2837,7 @@ INDIRECT_CALLABLE_SCOPE struct dst_entry *ip6_dst_check(struct dst_entry *dst, from = rcu_dereference(rt->from); if (from && (rt->rt6i_flags & RTF_PCPU || - unlikely(!list_empty(&rt->dst.rt_uncached)))) + unlikely(rt->dst.rt_uncached_list))) dst_ret = rt6_dst_from_check(rt, from, cookie); else dst_ret = rt6_check(rt, from, cookie); diff --git a/net/ipv6/seg6.c b/net/ipv6/seg6.c index 62a7eb779202..8c2b156c227a 100644 --- a/net/ipv6/seg6.c +++ b/net/ipv6/seg6.c @@ -138,8 +138,8 @@ void seg6_icmp_srh(struct sk_buff *skb, struct inet6_skb_parm *opt) static struct genl_family seg6_genl_family; static const struct nla_policy seg6_genl_policy[SEG6_ATTR_MAX + 1] = { - [SEG6_ATTR_DST] = { .type = NLA_BINARY, - .len = sizeof(struct in6_addr) }, + [SEG6_ATTR_DST] = + NLA_POLICY_EXACT_LEN(sizeof(struct in6_addr)), [SEG6_ATTR_DSTLEN] = { .type = NLA_S32, }, [SEG6_ATTR_HMACKEYID] = { .type = NLA_U32, }, [SEG6_ATTR_SECRET] = { .type = NLA_BINARY, }, diff --git a/net/llc/llc_c_ac.c b/net/llc/llc_c_ac.c index 724ecd741d4c..1aa7fe28acdd 100644 --- a/net/llc/llc_c_ac.c +++ b/net/llc/llc_c_ac.c @@ -437,7 +437,7 @@ int llc_conn_ac_resend_i_xxx_x_set_0_or_send_rr(struct sock *sk, if (likely(!rc)) llc_conn_send_pdu(sk, nskb); else - kfree_skb(skb); + kfree_skb(nskb); } if (rc) { nr = LLC_I_GET_NR(pdu); diff --git a/net/llc/llc_s_ac.c b/net/llc/llc_s_ac.c index 98deee560373..831998211b52 100644 --- a/net/llc/llc_s_ac.c +++ b/net/llc/llc_s_ac.c @@ -121,6 +121,8 @@ int llc_sap_action_send_xid_r(struct llc_sap *sap, struct sk_buff *skb) rc = llc_mac_hdr_init(nskb, mac_sa, mac_da); if (likely(!rc)) rc = dev_queue_xmit(nskb); + else + kfree_skb(nskb); out: return rc; } @@ -170,6 +172,8 @@ int llc_sap_action_send_test_r(struct llc_sap *sap, struct sk_buff *skb) rc = llc_mac_hdr_init(nskb, mac_sa, mac_da); if (likely(!rc)) rc = dev_queue_xmit(nskb); + else + kfree_skb(nskb); out: return rc; } diff --git a/net/llc/llc_sap.c b/net/llc/llc_sap.c index 1bd446a21092..3904a1b4ba84 100644 --- a/net/llc/llc_sap.c +++ b/net/llc/llc_sap.c @@ -19,12 +19,12 @@ #include #include -static int llc_mac_header_len(unsigned short devtype) +static int llc_mac_header_len(struct net_device *dev) { - switch (devtype) { + switch (dev->type) { case ARPHRD_ETHER: case ARPHRD_LOOPBACK: - return sizeof(struct ethhdr); + return LL_RESERVED_SPACE(dev); } return 0; } @@ -45,7 +45,7 @@ struct sk_buff *llc_alloc_frame(struct sock *sk, struct net_device *dev, int hlen = type == LLC_PDU_TYPE_U ? 3 : 4; struct sk_buff *skb; - hlen += llc_mac_header_len(dev->type); + hlen += llc_mac_header_len(dev); skb = alloc_skb(hlen + data_size, GFP_ATOMIC); if (skb) { diff --git a/net/mctp/route.c b/net/mctp/route.c index b19c63a5691a..e7c95eeacb48 100644 --- a/net/mctp/route.c +++ b/net/mctp/route.c @@ -825,7 +825,7 @@ static struct mctp_sk_key *mctp_lookup_prealloc_tag(struct mctp_sock *msk, spin_lock_irqsave(&mns->keys_lock, flags); - hlist_for_each_entry(tmp, &mns->keys, hlist) { + hlist_for_each_entry(tmp, &msk->keys, sklist) { if (tmp->net != netid) continue; diff --git a/net/netfilter/ipvs/ip_vs_core.c b/net/netfilter/ipvs/ip_vs_core.c index ba0957798bad..fd503f0efb57 100644 --- a/net/netfilter/ipvs/ip_vs_core.c +++ b/net/netfilter/ipvs/ip_vs_core.c @@ -1960,6 +1960,12 @@ ip_vs_in_icmp(struct netns_ipvs *ipvs, struct sk_buff *skb, int *related, /* Ensure the IP header is present in headroom */ if (!pskb_may_pull(skb, hlen_orig)) goto ignore_tunnel; + skb_set_transport_header(skb, hlen_orig); + /* Before now we may used ihl from skb frag, revalidate it after + * copying it into skb head to prevent out-of-bounds access + */ + if (ip_hdr(skb)->ihl * 4 != hlen_orig) + goto ignore_tunnel; IP_VS_DBG(12, "Sending ICMP for %pI4->%pI4: t=%u, c=%u, i=%u\n", &ip_hdr(skb)->saddr, &ip_hdr(skb)->daddr, type, code, ntohl(info)); diff --git a/net/netfilter/nf_conntrack_netlink.c b/net/netfilter/nf_conntrack_netlink.c index 579ada063b1b..4e5d7c701436 100644 --- a/net/netfilter/nf_conntrack_netlink.c +++ b/net/netfilter/nf_conntrack_netlink.c @@ -3392,7 +3392,8 @@ static bool expect_iter_name(struct nf_conntrack_expect *exp, void *data) struct nf_conntrack_helper *helper; const char *name = data; - helper = rcu_dereference(exp->helper); + helper = rcu_dereference_protected(exp->helper, + lockdep_is_held(&nf_conntrack_expect_lock)); if (!helper) return false; diff --git a/net/netfilter/nf_flow_table_offload.c b/net/netfilter/nf_flow_table_offload.c index 801a3dd9ceea..6757fd89c1f1 100644 --- a/net/netfilter/nf_flow_table_offload.c +++ b/net/netfilter/nf_flow_table_offload.c @@ -995,7 +995,6 @@ static void flow_offload_work_del(struct flow_offload_work *offload) flow_offload_tuple_del(offload, FLOW_OFFLOAD_DIR_ORIGINAL); if (test_bit(NF_FLOW_HW_BIDIRECTIONAL, &offload->flow->flags)) flow_offload_tuple_del(offload, FLOW_OFFLOAD_DIR_REPLY); - set_bit(NF_FLOW_HW_DEAD, &offload->flow->flags); } static void flow_offload_tuple_stats(struct flow_offload_work *offload, @@ -1059,6 +1058,12 @@ static void flow_offload_work_handler(struct work_struct *work) } clear_bit(NF_FLOW_HW_PENDING, &offload->flow->flags); + if (offload->cmd == FLOW_CLS_DESTROY) { + /* Publish after the worker's last flow access. */ + smp_mb__before_atomic(); + set_bit(NF_FLOW_HW_DEAD, &offload->flow->flags); + } + kfree(offload); } diff --git a/net/netfilter/nf_tables_api.c b/net/netfilter/nf_tables_api.c index c0b754a2d45b..b59628e6240c 100644 --- a/net/netfilter/nf_tables_api.c +++ b/net/netfilter/nf_tables_api.c @@ -6995,11 +6995,14 @@ static int nft_setelem_catchall_insert(const struct net *net, { struct nft_set_elem_catchall *catchall; u8 genmask = nft_genmask_next(net); + u64 tstamp = nft_net_tstamp(net); struct nft_set_ext *ext; list_for_each_entry(catchall, &set->catchall_list, list) { ext = nft_set_elem_ext(set, catchall->elem); - if (nft_set_elem_active(ext, genmask)) { + if (nft_set_elem_active(ext, genmask) && + !__nft_set_elem_expired(ext, tstamp) && + !nft_set_elem_is_dead(ext)) { *priv = catchall->elem; return -EEXIST; } @@ -7092,11 +7095,14 @@ static int nft_setelem_catchall_deactivate(const struct net *net, struct nft_set_elem *elem) { struct nft_set_elem_catchall *catchall; + u64 tstamp = nft_net_tstamp(net); struct nft_set_ext *ext; list_for_each_entry(catchall, &set->catchall_list, list) { ext = nft_set_elem_ext(set, catchall->elem); - if (!nft_is_active_next(net, ext)) + if (!nft_is_active_next(net, ext) || + __nft_set_elem_expired(ext, tstamp) || + nft_set_elem_is_dead(ext)) continue; kfree(elem->priv); diff --git a/net/netfilter/nfnetlink_queue.c b/net/netfilter/nfnetlink_queue.c index c727668b0c5b..a3bc00280051 100644 --- a/net/netfilter/nfnetlink_queue.c +++ b/net/netfilter/nfnetlink_queue.c @@ -1593,6 +1593,7 @@ nfqnl_rcv_nl_event(struct notifier_block *this, if (event == NETLINK_URELEASE && n->protocol == NETLINK_NETFILTER) { int i; + nfnl_lock(NFNL_SUBSYS_QUEUE); /* destroy all instances for this portid */ spin_lock(&q->instances_lock); for (i = 0; i < INSTANCE_BUCKETS; i++) { @@ -1606,6 +1607,7 @@ nfqnl_rcv_nl_event(struct notifier_block *this, } } spin_unlock(&q->instances_lock); + nfnl_unlock(NFNL_SUBSYS_QUEUE); } return NOTIFY_DONE; } @@ -1925,9 +1927,9 @@ static int nfqnl_recv_config(struct sk_buff *skb, const struct nfnl_info *info, /* Lookup queue under RCU. After peer_portid check (or for new queue * in BIND case), the queue is owned by the socket sending this message. - * A socket cannot simultaneously send a message and close, so while - * processing this CONFIG message, nfqnl_rcv_nl_event() (triggered by - * socket close) cannot destroy this queue. Safe to use without RCU. + * nfqnl_rcv_nl_event() will block on the nfnl subsys mutex that is + * held by the caller, so the queue cannot be destroyed in parallel, + * even after we drop the RCU read lock. */ rcu_read_lock(); queue = instance_lookup(q, queue_num); diff --git a/net/netfilter/nft_synproxy.c b/net/netfilter/nft_synproxy.c index 9ed288c9d168..554a96a000f4 100644 --- a/net/netfilter/nft_synproxy.c +++ b/net/netfilter/nft_synproxy.c @@ -118,7 +118,8 @@ static void nft_synproxy_do_eval(const struct nft_synproxy *priv, return; } - if (nf_ip_checksum(skb, nft_hook(pkt), thoff, IPPROTO_TCP)) { + if (nf_checksum(skb, nft_hook(pkt), thoff, IPPROTO_TCP, + nft_pf(pkt))) { regs->verdict.code = NF_DROP; return; } diff --git a/net/netlink/genetlink.c b/net/netlink/genetlink.c index 41d37442f186..5cc1037d4917 100644 --- a/net/netlink/genetlink.c +++ b/net/netlink/genetlink.c @@ -1656,7 +1656,7 @@ static void *ctrl_dumppolicy_prep(struct sk_buff *skb, } static int ctrl_dumppolicy_put_op(struct sk_buff *skb, - struct netlink_callback *cb, + struct netlink_callback *cb, u32 cmd, struct genl_split_ops *doit, struct genl_split_ops *dumpit) { @@ -1677,7 +1677,7 @@ static int ctrl_dumppolicy_put_op(struct sk_buff *skb, if (!nest_pol) goto err; - nest_op = nla_nest_start(skb, doit->cmd); + nest_op = nla_nest_start(skb, cmd); if (!nest_op) goto err; @@ -1721,7 +1721,8 @@ static int ctrl_dumppolicy(struct sk_buff *skb, struct netlink_callback *cb) &doit, &dumpit))) return -ENOENT; - if (ctrl_dumppolicy_put_op(skb, cb, &doit, &dumpit)) + if (ctrl_dumppolicy_put_op(skb, cb, ctx->op, + &doit, &dumpit)) return skb->len; /* done with the per-op policy index list */ @@ -1730,6 +1731,7 @@ static int ctrl_dumppolicy(struct sk_buff *skb, struct netlink_callback *cb) while (ctx->dump_map) { if (ctrl_dumppolicy_put_op(skb, cb, + ctx->op_iter->cmd, &ctx->op_iter->doit, &ctx->op_iter->dumpit)) return skb->len; diff --git a/net/nfc/core.c b/net/nfc/core.c index a92a6566e6a0..f521669293f0 100644 --- a/net/nfc/core.c +++ b/net/nfc/core.c @@ -279,10 +279,10 @@ static struct nfc_target *nfc_find_target(struct nfc_dev *dev, u32 target_idx) int nfc_dep_link_up(struct nfc_dev *dev, int target_index, u8 comm_mode) { - int rc = 0; - u8 *gb; - size_t gb_len; struct nfc_target *target; + u8 gb[NFC_MAX_GT_LEN]; + size_t gb_len = 0; + int rc = 0; pr_debug("dev_name=%s comm %d\n", dev_name(&dev->dev), comm_mode); @@ -301,7 +301,7 @@ int nfc_dep_link_up(struct nfc_dev *dev, int target_index, u8 comm_mode) goto error; } - gb = nfc_llcp_general_bytes(dev, &gb_len); + nfc_get_local_general_bytes(dev, gb, sizeof(gb), &gb_len); if (gb_len > NFC_MAX_GT_LEN) { rc = -EINVAL; goto error; @@ -644,11 +644,10 @@ int nfc_set_remote_general_bytes(struct nfc_dev *dev, const u8 *gb, u8 gb_len) } EXPORT_SYMBOL(nfc_set_remote_general_bytes); -u8 *nfc_get_local_general_bytes(struct nfc_dev *dev, size_t *gb_len) +u8 *nfc_get_local_general_bytes(struct nfc_dev *dev, u8 *out_gb, + size_t gb_max_len, size_t *gb_len) { - pr_debug("dev_name=%s\n", dev_name(&dev->dev)); - - return nfc_llcp_general_bytes(dev, gb_len); + return nfc_llcp_general_bytes(dev, out_gb, gb_max_len, gb_len); } EXPORT_SYMBOL(nfc_get_local_general_bytes); diff --git a/net/nfc/digital_dep.c b/net/nfc/digital_dep.c index 3982fa084737..968547c306a5 100644 --- a/net/nfc/digital_dep.c +++ b/net/nfc/digital_dep.c @@ -1490,14 +1490,14 @@ static int digital_tg_send_atr_res(struct nfc_digital_dev *ddev, struct digital_atr_req *atr_req) { struct digital_atr_res *atr_res; + u8 gb[NFC_MAX_GT_LEN]; struct sk_buff *skb; - u8 *gb, payload_bits; + u8 payload_bits; size_t gb_len; int rc; - gb = nfc_get_local_general_bytes(ddev->nfc_dev, &gb_len); - if (!gb) - gb_len = 0; + nfc_get_local_general_bytes(ddev->nfc_dev, gb, sizeof(gb), + &gb_len); skb = digital_skb_alloc(ddev, sizeof(struct digital_atr_res) + gb_len); if (!skb) diff --git a/net/nfc/llcp.h b/net/nfc/llcp.h index d8345ed57c95..23ae7a0112d3 100644 --- a/net/nfc/llcp.h +++ b/net/nfc/llcp.h @@ -91,6 +91,7 @@ struct nfc_llcp_local { struct hlist_head pending_sdreqs; struct timer_list sdreq_timer; struct work_struct sdreq_timeout_work; + struct work_struct release_work; u8 sdreq_next_tid; /* sockets array */ diff --git a/net/nfc/llcp_commands.c b/net/nfc/llcp_commands.c index ca89fe967d6a..80a00938c869 100644 --- a/net/nfc/llcp_commands.c +++ b/net/nfc/llcp_commands.c @@ -135,7 +135,7 @@ struct nfc_llcp_sdp_tlv *nfc_llcp_build_sdreq_tlv(u8 tid, const char *uri, { struct nfc_llcp_sdp_tlv *sdreq; - pr_debug("uri: %s, len: %zu\n", uri, uri_len); + pr_debug("uri: %.*s, len: %zu\n", (int)uri_len, uri, uri_len); /* sdreq->tlv_len is u8, takes uri_len, + 3 for header, + 1 for NULL */ if (WARN_ON_ONCE(uri_len > U8_MAX - 4)) diff --git a/net/nfc/llcp_core.c b/net/nfc/llcp_core.c index cac1b5487064..74bf817007cf 100644 --- a/net/nfc/llcp_core.c +++ b/net/nfc/llcp_core.c @@ -20,6 +20,8 @@ static LIST_HEAD(llcp_devices); /* Protects llcp_devices list */ static DEFINE_SPINLOCK(llcp_devices_lock); +static struct workqueue_struct *llcp_wq; + static void nfc_llcp_rx_skb(struct nfc_llcp_local *local, struct sk_buff *skb); void nfc_llcp_sock_link(struct llcp_sock_list *l, struct sock *sk) @@ -63,21 +65,33 @@ static void nfc_llcp_socket_purge(struct nfc_llcp_sock *sock) } } +static struct sock *nfc_llcp_sock_list_pop(struct llcp_sock_list *l) +{ + struct sock *sk; + + write_lock(&l->lock); + sk = sk_head(&l->head); + if (sk) { + sock_hold(sk); + sk_del_node_init(sk); + } + write_unlock(&l->lock); + + return sk; +} + static void nfc_llcp_socket_release(struct nfc_llcp_local *local, bool device, int err) { struct sock *sk; - struct hlist_node *tmp; struct nfc_llcp_sock *llcp_sock; skb_queue_purge(&local->tx_queue); - write_lock(&local->sockets.lock); - - sk_for_each_safe(sk, tmp, &local->sockets.head) { + while ((sk = nfc_llcp_sock_list_pop(&local->sockets))) { llcp_sock = nfc_llcp_sock(sk); - bh_lock_sock(sk); + lock_sock(sk); nfc_llcp_socket_purge(llcp_sock); @@ -91,17 +105,27 @@ static void nfc_llcp_socket_release(struct nfc_llcp_local *local, bool device, list_for_each_entry_safe(lsk, n, &llcp_sock->accept_queue, accept_queue) { + bool put_creation = false; + accept_sk = &lsk->sk; - bh_lock_sock(accept_sk); + lock_sock_nested(accept_sk, + SINGLE_DEPTH_NESTING); - nfc_llcp_accept_unlink(accept_sk); + if (nfc_llcp_sock(accept_sk)->parent == sk) { + nfc_llcp_accept_unlink(accept_sk); + nfc_llcp_sock_unlink(&local->sockets, accept_sk); - if (err) - accept_sk->sk_err = err; - accept_sk->sk_state = LLCP_CLOSED; - accept_sk->sk_state_change(sk); + if (err) + accept_sk->sk_err = err; + accept_sk->sk_state = LLCP_CLOSED; + accept_sk->sk_state_change(accept_sk); + sock_orphan(accept_sk); + put_creation = true; + } - bh_unlock_sock(accept_sk); + release_sock(accept_sk); + if (put_creation) + sock_put(accept_sk); /* creation ref */ } } @@ -110,23 +134,18 @@ static void nfc_llcp_socket_release(struct nfc_llcp_local *local, bool device, sk->sk_state = LLCP_CLOSED; sk->sk_state_change(sk); - bh_unlock_sock(sk); - - sk_del_node_init(sk); + release_sock(sk); + sock_put(sk); } - write_unlock(&local->sockets.lock); - /* If we still have a device, we keep the RAW sockets alive */ if (device == true) return; - write_lock(&local->raw_sockets.lock); - - sk_for_each_safe(sk, tmp, &local->raw_sockets.head) { + while ((sk = nfc_llcp_sock_list_pop(&local->raw_sockets))) { llcp_sock = nfc_llcp_sock(sk); - bh_lock_sock(sk); + lock_sock(sk); nfc_llcp_socket_purge(llcp_sock); @@ -135,26 +154,20 @@ static void nfc_llcp_socket_release(struct nfc_llcp_local *local, bool device, sk->sk_state = LLCP_CLOSED; sk->sk_state_change(sk); - bh_unlock_sock(sk); - - sk_del_node_init(sk); + release_sock(sk); + sock_put(sk); } - - write_unlock(&local->raw_sockets.lock); } static struct nfc_llcp_local *nfc_llcp_local_get(struct nfc_llcp_local *local) { - /* Since using nfc_llcp_local may result in usage of nfc_dev, whenever - * we hold a reference to local, we also need to hold a reference to - * the device to avoid UAF. - */ - if (!nfc_get_device(local->dev->idx)) + if (!local) return NULL; - kref_get(&local->ref); + if (kref_get_unless_zero(&local->ref)) + return local; - return local; + return NULL; } static void local_cleanup(struct nfc_llcp_local *local) @@ -172,30 +185,34 @@ static void local_cleanup(struct nfc_llcp_local *local) nfc_llcp_free_sdp_tlv_list(&local->pending_sdreqs); } +static void local_release_work(struct work_struct *work) +{ + struct nfc_llcp_local *local; + struct nfc_dev *dev; + + local = container_of(work, struct nfc_llcp_local, release_work); + dev = local->dev; + + local_cleanup(local); + kfree(local); + nfc_put_device(dev); +} + static void local_release(struct kref *ref) { struct nfc_llcp_local *local; local = container_of(ref, struct nfc_llcp_local, ref); - local_cleanup(local); - kfree(local); + queue_work(llcp_wq, &local->release_work); } int nfc_llcp_local_put(struct nfc_llcp_local *local) { - struct nfc_dev *dev; - int ret; - - if (local == NULL) + if (!local) return 0; - dev = local->dev; - - ret = kref_put(&local->ref, local_release); - nfc_put_device(dev); - - return ret; + return kref_put(&local->ref, local_release); } static struct nfc_llcp_sock *nfc_llcp_sock_get(struct nfc_llcp_local *local, @@ -341,7 +358,7 @@ static int nfc_llcp_wks_sap(const char *service_name, size_t service_name_len) { int sap, num_wks; - pr_debug("%s\n", service_name); + pr_debug("%.*s\n", (int)service_name_len, service_name); if (service_name == NULL) return -EINVAL; @@ -352,7 +369,8 @@ static int nfc_llcp_wks_sap(const char *service_name, size_t service_name_len) if (wks[sap] == NULL) continue; - if (strncmp(wks[sap], service_name, service_name_len) == 0) + if (strlen(wks[sap]) == service_name_len && + !strncmp(wks[sap], service_name, service_name_len)) return sap; } @@ -635,23 +653,32 @@ static int nfc_llcp_build_gb(struct nfc_llcp_local *local) return ret; } -u8 *nfc_llcp_general_bytes(struct nfc_dev *dev, size_t *general_bytes_len) +u8 *nfc_llcp_general_bytes(struct nfc_dev *dev, u8 *out_gb, size_t gb_max_len, + size_t *general_bytes_len) { struct nfc_llcp_local *local; + if (!out_gb || !general_bytes_len) + return NULL; + local = nfc_llcp_find_local(dev); - if (local == NULL) { + if (!local) { *general_bytes_len = 0; return NULL; } nfc_llcp_build_gb(local); - *general_bytes_len = local->gb_len; + if (local->gb_len) { + *general_bytes_len = min_t(size_t, local->gb_len, gb_max_len); + memcpy(out_gb, local->gb, *general_bytes_len); + } else { + *general_bytes_len = 0; + } nfc_llcp_local_put(local); - return local->gb; + return out_gb; } int nfc_llcp_set_remote_gb(struct nfc_dev *dev, const u8 *gb, u8 gb_len) @@ -1074,6 +1101,9 @@ static void nfc_llcp_recv_hdlc(struct nfc_llcp_local *local, struct sock *sk; u8 dsap, ssap, ptype, ns, nr; + if (!pskb_may_pull(skb, LLCP_HEADER_SIZE + LLCP_SEQUENCE_SIZE)) + return; + ptype = nfc_llcp_ptype(skb); dsap = nfc_llcp_dsap(skb); ssap = nfc_llcp_ssap(skb); @@ -1251,6 +1281,7 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, struct nfc_llcp_sock *llcp_sock; struct sock *sk; u8 dsap, ssap, reason; + bool connecting = false; dsap = nfc_llcp_dsap(skb); ssap = nfc_llcp_ssap(skb); @@ -1262,6 +1293,7 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, case LLCP_DM_NOBOUND: case LLCP_DM_REJ: llcp_sock = nfc_llcp_connecting_sock_get(local, dsap); + connecting = true; break; default: @@ -1276,10 +1308,33 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, sk = &llcp_sock->sk; + lock_sock(sk); + + /* Check if socket was destroyed whilst waiting for the lock */ + if (!sk_hashed(sk)) { + release_sock(sk); + nfc_llcp_sock_put(llcp_sock); + return; + } + + /* + * For DM(NOBOUND)/DM(REJ) the socket is still linked on the + * connecting_sockets list. Unlink it here, under the socket lock, + * before moving it to LLCP_CLOSED: llcp_sock_release() selects the + * list to unlink from by sk_state, so leaving a connecting socket + * in the CLOSED state would make it unlink from the wrong list and + * corrupt the connecting_sockets list / desync the socket refcount. + * This mirrors nfc_llcp_recv_cc(). + */ + if (connecting) + nfc_llcp_sock_unlink(&local->connecting_sockets, sk); + sk->sk_err = ENXIO; sk->sk_state = LLCP_CLOSED; sk->sk_state_change(sk); + release_sock(sk); + nfc_llcp_sock_put(llcp_sock); } @@ -1680,6 +1735,7 @@ int nfc_llcp_register_device(struct nfc_dev *ndev) INIT_WORK(&local->rx_work, nfc_llcp_rx_work); INIT_WORK(&local->timeout_work, nfc_llcp_timeout_work); + INIT_WORK(&local->release_work, local_release_work); rwlock_init(&local->sockets.lock); rwlock_init(&local->connecting_sockets.lock); @@ -1723,10 +1779,23 @@ void nfc_llcp_unregister_device(struct nfc_dev *dev) int __init nfc_llcp_init(void) { - return nfc_llcp_sock_init(); + int ret; + + llcp_wq = alloc_workqueue("nfc_llcp_wq", WQ_UNBOUND, 0); + if (!llcp_wq) + return -ENOMEM; + + ret = nfc_llcp_sock_init(); + if (ret) { + destroy_workqueue(llcp_wq); + return ret; + } + + return 0; } void nfc_llcp_exit(void) { nfc_llcp_sock_exit(); + destroy_workqueue(llcp_wq); } diff --git a/net/nfc/llcp_sock.c b/net/nfc/llcp_sock.c index 5558d8a4d48b..1e5ee4bcde68 100644 --- a/net/nfc/llcp_sock.c +++ b/net/nfc/llcp_sock.c @@ -392,11 +392,12 @@ void nfc_llcp_accept_unlink(struct sock *sk) pr_debug("state %d\n", sk->sk_state); - list_del_init(&llcp_sock->accept_queue); - sk_acceptq_removed(llcp_sock->parent); - llcp_sock->parent = NULL; - - sock_put(sk); + if (llcp_sock->parent) { + list_del_init(&llcp_sock->accept_queue); + sk_acceptq_removed(llcp_sock->parent); + llcp_sock->parent = NULL; + sock_put(sk); + } } void nfc_llcp_accept_enqueue(struct sock *parent, struct sock *sk) @@ -423,12 +424,20 @@ struct sock *nfc_llcp_accept_dequeue(struct sock *parent, list_for_each_entry_safe(lsk, n, &llcp_parent->accept_queue, accept_queue) { + struct nfc_llcp_local *local; + sk = &lsk->sk; - lock_sock(sk); + lock_sock_nested(sk, SINGLE_DEPTH_NESTING); if (sk->sk_state == LLCP_CLOSED) { - release_sock(sk); + local = nfc_llcp_sock(sk)->local; + nfc_llcp_accept_unlink(sk); + if (local) + nfc_llcp_sock_unlink(&local->sockets, sk); + sock_orphan(sk); + release_sock(sk); + sock_put(sk); continue; } @@ -464,7 +473,7 @@ static int llcp_sock_accept(struct socket *sock, struct socket *newsock, pr_debug("parent %p\n", sk); - lock_sock_nested(sk, SINGLE_DEPTH_NESTING); + lock_sock(sk); if (sk->sk_state != LLCP_LISTEN) { ret = -EBADFD; @@ -490,7 +499,12 @@ static int llcp_sock_accept(struct socket *sock, struct socket *newsock, release_sock(sk); timeo = schedule_timeout(timeo); - lock_sock_nested(sk, SINGLE_DEPTH_NESTING); + lock_sock(sk); + + if (sk->sk_state != LLCP_LISTEN) { + ret = -EBADFD; + break; + } } __set_current_state(TASK_RUNNING); remove_wait_queue(sk_sleep(sk), &wait); @@ -629,13 +643,24 @@ static int llcp_sock_release(struct socket *sock) list_for_each_entry_safe(lsk, n, &llcp_sock->accept_queue, accept_queue) { - accept_sk = &lsk->sk; - lock_sock(accept_sk); + bool put_creation = false; - nfc_llcp_send_disconnect(lsk); - nfc_llcp_accept_unlink(accept_sk); + accept_sk = &lsk->sk; + lock_sock_nested(accept_sk, SINGLE_DEPTH_NESTING); + + if (nfc_llcp_sock(accept_sk)->parent == sk) { + nfc_llcp_send_disconnect(lsk); + nfc_llcp_accept_unlink(accept_sk); + nfc_llcp_sock_unlink(&local->sockets, accept_sk); + + accept_sk->sk_state = LLCP_CLOSED; + sock_orphan(accept_sk); + put_creation = true; + } release_sock(accept_sk); + if (put_creation) + sock_put(accept_sk); /* creation ref */ } } @@ -734,12 +759,16 @@ static int llcp_sock_connect(struct socket *sock, struct sockaddr_unsized *_addr llcp_sock->service_name_len = min_t(unsigned int, addr->service_name_len, NFC_LLCP_MAX_SERVICE_NAME); - llcp_sock->service_name = kmemdup(addr->service_name, - llcp_sock->service_name_len, - GFP_KERNEL); - if (!llcp_sock->service_name) { - ret = -ENOMEM; - goto sock_llcp_release; + if (llcp_sock->service_name_len == 0) { + llcp_sock->service_name = NULL; + } else { + llcp_sock->service_name = kmemdup(addr->service_name, + llcp_sock->service_name_len, + GFP_KERNEL); + if (!llcp_sock->service_name) { + ret = -ENOMEM; + goto sock_llcp_release; + } } nfc_llcp_sock_link(&local->connecting_sockets, sk); diff --git a/net/nfc/nci/core.c b/net/nfc/nci/core.c index 5f46c4b5720f..73e3a96470ac 100644 --- a/net/nfc/nci/core.c +++ b/net/nfc/nci/core.c @@ -780,15 +780,15 @@ static int nci_set_local_general_bytes(struct nfc_dev *nfc_dev) { struct nci_dev *ndev = nfc_get_drvdata(nfc_dev); struct nci_set_config_param param; + u8 gb[NFC_MAX_GT_LEN]; int rc; - param.val = nfc_get_local_general_bytes(nfc_dev, ¶m.len); - if ((param.val == NULL) || (param.len == 0)) + nfc_get_local_general_bytes(nfc_dev, gb, sizeof(gb), + ¶m.len); + if (param.len == 0) return 0; - if (param.len > NFC_MAX_GT_LEN) - return -EINVAL; - + param.val = gb; param.id = NCI_PN_ATR_REQ_GEN_BYTES; rc = nci_request(ndev, nci_set_config_req, ¶m, diff --git a/net/nfc/netlink.c b/net/nfc/netlink.c index 0c58824cb150..224bdfa2dd0d 100644 --- a/net/nfc/netlink.c +++ b/net/nfc/netlink.c @@ -1181,7 +1181,7 @@ static int nfc_genl_llc_sdreq(struct sk_buff *skb, struct genl_info *info) if (rc != 0) { rc = -EINVAL; - goto put_local; + goto free_list; } if (!sdp_attrs[NFC_SDP_ATTR_URI]) @@ -1200,7 +1200,7 @@ static int nfc_genl_llc_sdreq(struct sk_buff *skb, struct genl_info *info) sdreq = nfc_llcp_build_sdreq_tlv(tid, uri, uri_len); if (sdreq == NULL) { rc = -ENOMEM; - goto put_local; + goto free_list; } tlvs_len += sdreq->tlv_len; @@ -1215,6 +1215,9 @@ static int nfc_genl_llc_sdreq(struct sk_buff *skb, struct genl_info *info) rc = nfc_llcp_send_snl_sdreq(local, &sdreq_list, tlvs_len); +free_list: + nfc_llcp_free_sdp_tlv_list(&sdreq_list); + put_local: nfc_llcp_local_put(local); diff --git a/net/nfc/nfc.h b/net/nfc/nfc.h index 0b1e6466f4fb..82c5dfdad10e 100644 --- a/net/nfc/nfc.h +++ b/net/nfc/nfc.h @@ -49,7 +49,8 @@ void nfc_llcp_mac_is_up(struct nfc_dev *dev, u32 target_idx, int nfc_llcp_register_device(struct nfc_dev *dev); void nfc_llcp_unregister_device(struct nfc_dev *dev); int nfc_llcp_set_remote_gb(struct nfc_dev *dev, const u8 *gb, u8 gb_len); -u8 *nfc_llcp_general_bytes(struct nfc_dev *dev, size_t *general_bytes_len); +u8 *nfc_llcp_general_bytes(struct nfc_dev *dev, u8 *out_gb, size_t gb_max_len, + size_t *general_bytes_len); int nfc_llcp_data_received(struct nfc_dev *dev, struct sk_buff *skb); struct nfc_llcp_local *nfc_llcp_find_local(struct nfc_dev *dev); int nfc_llcp_local_put(struct nfc_llcp_local *local); diff --git a/net/openvswitch/conntrack.c b/net/openvswitch/conntrack.c index 0f433688e17b..d3326edcabf7 100644 --- a/net/openvswitch/conntrack.c +++ b/net/openvswitch/conntrack.c @@ -734,6 +734,18 @@ static int __ovs_ct_lookup(struct net *net, struct sw_flow_key *key, enum ip_conntrack_info ctinfo; struct nf_conn *ct; + /* If the ct entry is not confirmed and shared with some other skb, + * e.g., a cloned one, we can't just modify it with the commit as we + * must not modify the extension set. Reset. + */ + if (cached && info->commit) { + ct = nf_ct_get(skb, &ctinfo); + if (ct && !nf_ct_is_confirmed(ct) && nf_ct_shared(ct)) { + nf_reset_ct(skb); + cached = false; + } + } + if (!cached) { struct nf_hook_state state = { .hook = NF_INET_PRE_ROUTING, @@ -766,8 +778,6 @@ static int __ovs_ct_lookup(struct net *net, struct sw_flow_key *key, ct = nf_ct_get(skb, &ctinfo); if (ct) { - bool add_helper = false; - /* Packets starting a new connection must be NATted before the * helper, so that the helper knows about the NAT. We enforce * this by delaying both NAT and helper calls for unconfirmed @@ -799,7 +809,6 @@ static int __ovs_ct_lookup(struct net *net, struct sw_flow_key *key, GFP_ATOMIC); if (err) return err; - add_helper = true; /* helper installed, add seqadj if NAT is required */ if (info->nat && !nfct_seqadj(ct)) { @@ -808,14 +817,14 @@ static int __ovs_ct_lookup(struct net *net, struct sw_flow_key *key, } } - /* Call the helper only if: - * - nf_conntrack_in() was executed above ("!cached") or a - * helper was just attached ("add_helper") for a confirmed - * connection, or - * - When committing an unconfirmed connection. + /* Call the helper only if nf_conntrack_in() was executed + * above ("!cached"). + * + * For unconfirmed connections it will be called later during + * commit as we need to have all the other extensions allocated + * before the call. */ - if ((nf_ct_is_confirmed(ct) ? !cached || add_helper : - info->commit)) { + if (nf_ct_is_confirmed(ct) && !cached) { int err = nf_ct_helper(skb, ct, ctinfo, info->family); err = verdict_to_errno(err); @@ -1019,6 +1028,14 @@ static int ovs_ct_commit(struct net *net, struct sw_flow_key *key, return err; nf_conn_act_ct_ext_add(skb, ct, ctinfo); + + /* Call the helpers now. We couldn't do this before as + * all the extensions must be allocated before the call. + */ + err = nf_ct_helper(skb, ct, ctinfo, info->family); + err = verdict_to_errno(err); + if (err) + return err; } else if (IS_ENABLED(CONFIG_NF_CONNTRACK_LABELS) && labels_nonzero(&info->labels.mask)) { err = ovs_ct_set_labels(ct, key, &info->labels.value, diff --git a/net/packet/af_packet.c b/net/packet/af_packet.c index 50cae32ae269..7c83e01526ed 100644 --- a/net/packet/af_packet.c +++ b/net/packet/af_packet.c @@ -617,7 +617,7 @@ static int prb_calc_retire_blk_tmo(struct packet_sock *po, return DEFAULT_PRB_RETIRE_TOV; div = ecmd.base.speed / 1000; - mbits = (blk_size_in_bytes * 8) / (1024 * 1024); + mbits = (u64)blk_size_in_bytes * 8 / (1024 * 1024); if (div) mbits /= div; @@ -2530,26 +2530,6 @@ static int tpacket_rcv(struct sk_buff *skb, struct net_device *dev, goto drop_n_restore; } -static void tpacket_destruct_skb(struct sk_buff *skb) -{ - struct packet_sock *po = pkt_sk(skb->sk); - - if (likely(po->tx_ring.pg_vec)) { - void *ph; - __u32 ts; - - ph = skb_zcopy_get_nouarg(skb); - - ts = __packet_set_timestamp(po, ph, skb); - __packet_set_status(po, ph, TP_STATUS_AVAILABLE | ts); - - packet_dec_pending(&po->tx_ring); - complete(&po->skb_completion); - } - - sock_wfree(skb); -} - static int __packet_snd_vnet_parse(struct virtio_net_hdr *vnet_hdr, size_t len) { if ((vnet_hdr->flags & VIRTIO_NET_HDR_F_NEEDS_CSUM) && @@ -2589,27 +2569,56 @@ static int packet_snd_vnet_parse(struct msghdr *msg, size_t *len, return 0; } +struct tpacket_uarg { + struct ubuf_info ubuf; + struct packet_sock *po; + void *ph; +}; + +static void tpacket_ubuf_complete(struct sk_buff *skb, struct ubuf_info *uarg, + bool success) +{ + struct tpacket_uarg *tu = container_of(uarg, struct tpacket_uarg, ubuf); + struct packet_sock *po = tu->po; + void *ph = tu->ph; + __u32 ts; + + DEBUG_NET_WARN_ON_ONCE(!skb); + + if (!refcount_dec_and_test(&uarg->refcnt)) + return; + + ts = __packet_set_timestamp(po, ph, skb); + __packet_set_status(po, ph, TP_STATUS_AVAILABLE | ts); + + packet_dec_pending(&po->tx_ring); + complete(&po->skb_completion); + + kfree(tu); + sk_free(&po->sk); +} + +static const struct ubuf_info_ops tpacket_ubuf_ops = { + .complete = tpacket_ubuf_complete, +}; + static int tpacket_fill_skb(struct packet_sock *po, struct sk_buff *skb, - void *frame, struct net_device *dev, void *data, int tp_len, + struct net_device *dev, void *data, int tp_len, __be16 proto, unsigned char *addr, int hlen, int copylen, int hard_header_len, const struct sockcm_cookie *sockc) { - union tpacket_uhdr ph; int to_write, offset, len, nr_frags, len_max; struct socket *sock = po->sk.sk_socket; struct page *page; int err; - ph.raw = frame; - skb->protocol = proto; skb->dev = dev; skb->priority = sockc->priority; skb->mark = sockc->mark; skb_set_delivery_type_by_clockid(skb, sockc->transmit_time, po->sk.sk_clockid); skb_setup_tx_timestamp(skb, sockc); - skb_zcopy_set_nouarg(skb, ph.raw); skb_reserve(skb, hlen); skb_reset_network_header(skb); @@ -2749,6 +2758,7 @@ static int tpacket_snd(struct packet_sock *po, struct msghdr *msg) struct virtio_net_hdr vnet_hdr; bool has_vnet_hdr = false; struct sockcm_cookie sockc; + struct tpacket_uarg *uarg; __be16 proto; int err, reserve = 0; void *ph; @@ -2876,7 +2886,7 @@ static int tpacket_snd(struct packet_sock *po, struct msghdr *msg) err = len_sum; goto out_status; } - tp_len = tpacket_fill_skb(po, skb, ph, dev, data, tp_len, proto, + tp_len = tpacket_fill_skb(po, skb, dev, data, tp_len, proto, addr, hlen, copylen, hard_header_len, &sockc); if (likely(tp_len >= 0) && @@ -2908,7 +2918,24 @@ static int tpacket_snd(struct packet_sock *po, struct msghdr *msg) virtio_net_hdr_set_proto(skb, &vnet_hdr); } - skb->destructor = tpacket_destruct_skb; + uarg = kmalloc(sizeof(*uarg), GFP_KERNEL); + if (unlikely(!uarg)) { + if (likely(len_sum > 0)) + err = len_sum; + else + err = -ENOMEM; + goto out_status; + } + uarg->po = po; + uarg->ph = ph; + uarg->ubuf.ops = &tpacket_ubuf_ops; + uarg->ubuf.flags = SKBFL_ZEROCOPY_FRAG; + refcount_set(&uarg->ubuf.refcnt, 1); + + /* Hold a sk_wmem_alloc reference until completion */ + refcount_inc(&po->sk.sk_wmem_alloc); + skb_zcopy_init(skb, &uarg->ubuf); + __packet_set_status(po, ph, TP_STATUS_SENDING); packet_inc_pending(&po->tx_ring); @@ -4486,21 +4513,20 @@ static struct pgv *alloc_pg_vec(struct tpacket_req *req, int order, bool tx_ring vec->len = block_nr; pg_vec = vec->pg_vec; + if (tx_ring) { + vec->deferred = kzalloc_obj(*vec->deferred, + GFP_KERNEL | __GFP_NOWARN); + if (!vec->deferred) + goto out_free_pgvec; + vec->deferred->vec = vec; + INIT_DELAYED_WORK(&vec->deferred->work, + packet_free_pg_vec_work); + } + for (i = 0; i < block_nr; i++) { pg_vec[i].buffer = alloc_one_pg_vec_page(order); if (unlikely(!pg_vec[i].buffer)) goto out_free_pgvec; - - if (tx_ring && !vec->deferred && - is_vmalloc_addr(pg_vec[i].buffer)) { - vec->deferred = kzalloc_obj(*vec->deferred, - GFP_KERNEL | __GFP_NOWARN); - if (!vec->deferred) - goto out_free_pgvec; - vec->deferred->vec = vec; - INIT_DELAYED_WORK(&vec->deferred->work, - packet_free_pg_vec_work); - } } out: diff --git a/net/rds/connection.c b/net/rds/connection.c index b6c4beb50eaf..c752a8623cfc 100644 --- a/net/rds/connection.c +++ b/net/rds/connection.c @@ -276,6 +276,12 @@ static struct rds_connection *__rds_conn_create(struct net *net, conn->c_trans = trans; + /* The transport may just have been swapped for loopback; size the + * set of paths - which is also what rds_conn_destroy() tears down + * again - by the transport the connection actually uses. + */ + npaths = (trans->t_mp_capable ? RDS_MPATH_WORKERS : 1); + init_waitqueue_head(&conn->c_hs_waitq); for (i = 0; i < npaths; i++) { __rds_conn_path_init(conn, &conn->c_path[i], diff --git a/net/rds/ib_frmr.c b/net/rds/ib_frmr.c index bd861191157b..8397aa4a17ac 100644 --- a/net/rds/ib_frmr.c +++ b/net/rds/ib_frmr.c @@ -204,19 +204,16 @@ static int rds_ib_map_frmr(struct rds_ib_device *rds_ibdev, */ rds_ib_teardown_mr(ibmr); - ibmr->sg = sg; - ibmr->sg_len = sg_len; - ibmr->sg_dma_len = 0; frmr->sg_byte_len = 0; - WARN_ON(ibmr->sg_dma_len); - ibmr->sg_dma_len = ib_dma_map_sg(dev, ibmr->sg, ibmr->sg_len, + ibmr->sg_dma_len = ib_dma_map_sg(dev, sg, sg_len, DMA_BIDIRECTIONAL); if (unlikely(!ibmr->sg_dma_len)) { pr_warn("RDS/IB: %s failed!\n", __func__); return -EBUSY; } - frmr->sg_byte_len = 0; + ibmr->sg = sg; + ibmr->sg_len = sg_len; frmr->dma_npages = 0; len = 0; @@ -264,6 +261,8 @@ static int rds_ib_map_frmr(struct rds_ib_device *rds_ibdev, ib_dma_unmap_sg(rds_ibdev->dev, ibmr->sg, ibmr->sg_len, DMA_BIDIRECTIONAL); ibmr->sg_dma_len = 0; + ibmr->sg = NULL; + ibmr->sg_len = 0; return ret; } diff --git a/net/sched/act_api.c b/net/sched/act_api.c index 3f653721c45f..e45a63be397c 100644 --- a/net/sched/act_api.c +++ b/net/sched/act_api.c @@ -758,7 +758,7 @@ static int tcf_idr_delete_index(struct tcf_idrinfo *idrinfo, u32 index) mutex_lock(&idrinfo->lock); p = idr_find(&idrinfo->action_idr, index); - if (!p) { + if (IS_ERR_OR_NULL(p)) { mutex_unlock(&idrinfo->lock); return -ENOENT; } diff --git a/net/sched/act_ct.c b/net/sched/act_ct.c index 9080cb386c16..411e3dd92d07 100644 --- a/net/sched/act_ct.c +++ b/net/sched/act_ct.c @@ -432,11 +432,10 @@ static void tcf_ct_flow_table_add(struct tcf_ct_flow_table *ct_ft, if (test_and_set_bit(IPS_OFFLOAD_BIT, &ct->status)) return; + /* NULL if ct is dying (raced flush) or the atomic alloc failed. */ entry = flow_offload_alloc(ct); - if (!entry) { - WARN_ON_ONCE(1); + if (!entry) goto err_alloc; - } if (tcp) { ct->proto.tcp.seen[0].flags |= IP_CT_TCP_FLAG_BE_LIBERAL; @@ -980,14 +979,13 @@ TC_INDIRECT_SCOPE int tcf_ct_act(struct sk_buff *skb, const struct tc_action *a, struct tcf_result *res) { struct net *net = dev_net(skb->dev); + bool cached, commit, clear, nat; enum ip_conntrack_info ctinfo; struct tcf_ct *c = to_ct(a); struct nf_conn *tmpl = NULL; struct nf_hook_state state; - bool cached, commit, clear; int nh_ofs, err, retval; struct tcf_ct_params *p; - bool add_helper = false; bool skb_is_ours = false; bool skip_add = false; bool defrag = false; @@ -999,6 +997,7 @@ TC_INDIRECT_SCOPE int tcf_ct_act(struct sk_buff *skb, const struct tc_action *a, retval = p->action; commit = p->ct_action & TCA_CT_ACT_COMMIT; clear = p->ct_action & TCA_CT_ACT_CLEAR; + nat = p->ct_action & TCA_CT_ACT_NAT; tmpl = p->tmpl; tcf_lastuse_update(&c->tcf_tm); @@ -1047,6 +1046,19 @@ TC_INDIRECT_SCOPE int tcf_ct_act(struct sk_buff *skb, const struct tc_action *a, * different zone. */ cached = tcf_ct_skb_nfct_cached(net, skb, p); + + /* If the ct entry is not confirmed and shared with some other skb, + * e.g., a cloned one, we can't just modify it with a commit or nat + * as we must not modify the extension set. Reset. + */ + if (cached && (commit || nat)) { + ct = nf_ct_get(skb, &ctinfo); + if (ct && !nf_ct_is_confirmed(ct) && nf_ct_shared(ct)) { + nf_reset_ct(skb); + cached = false; + } + } + if (!cached) { if (tcf_ct_flow_table_lookup(p, skb, family)) { skip_add = true; @@ -1083,26 +1095,32 @@ TC_INDIRECT_SCOPE int tcf_ct_act(struct sk_buff *skb, const struct tc_action *a, err = __nf_ct_try_assign_helper(ct, p->tmpl, GFP_ATOMIC); if (err) goto drop; - add_helper = true; - if (p->ct_action & TCA_CT_ACT_NAT && !nfct_seqadj(ct)) { + + if (nat && !nfct_seqadj(ct)) { if (!nfct_seqadj_ext_add(ct)) goto drop; } } - if (nf_ct_is_confirmed(ct) ? ((!cached && !skip_add) || add_helper) : commit) { - err = nf_ct_helper(skb, ct, ctinfo, family); - if (err != NF_ACCEPT) - goto nf_error; - } - if (commit) { tcf_ct_act_set_mark(ct, p->mark, p->mark_mask); tcf_ct_act_set_labels(ct, p->labels, p->labels_mask); if (!nf_ct_is_confirmed(ct)) nf_conn_act_ct_ext_add(skb, ct, ctinfo); + } + /* Run helpers for the connection if nf_conntrack_in() was executed + * or if we're about to commit. This has to be done after all the + * extensions are already added. + */ + if (nf_ct_is_confirmed(ct) ? (!cached && !skip_add) : commit) { + err = nf_ct_helper(skb, ct, ctinfo, family); + if (err != NF_ACCEPT) + goto nf_error; + } + + if (commit) { /* This will take care of sending queued events * even if the connection is already confirmed. */ diff --git a/net/sched/act_gate.c b/net/sched/act_gate.c index 5d228a402204..6d6d45e03c07 100644 --- a/net/sched/act_gate.c +++ b/net/sched/act_gate.c @@ -681,7 +681,35 @@ static void tcf_gate_stats_update(struct tc_action *a, u64 bytes, u64 packets, static size_t tcf_gate_get_fill_size(const struct tc_action *act) { - return nla_total_size(sizeof(struct tc_gate)); + struct tcf_gate *gact = to_gate(act); + const struct tcf_gate_params *p; + struct tcfg_gate_entry *entry; + size_t size = nla_total_size(sizeof(struct tc_gate)) /* TCA_GATE_PARMS */ + + 3 * nla_total_size_64bit(sizeof(u64)) /* TCA_GATE_BASE_TIME + * TCA_GATE_CYCLE_TIME + * TCA_GATE_CYCLE_TIME_EXT + */ + + nla_total_size(sizeof(s32)) /* TCA_GATE_CLOCKID */ + + nla_total_size(sizeof(u32)) /* TCA_GATE_FLAGS */ + + nla_total_size(sizeof(s32)) /* TCA_GATE_PRIORITY */ + + nla_total_size(0); /* TCA_GATE_ENTRY_LIST */ + /* TCA_GATE_TM is budgeted by tcf_action_shared_attrs_size() */ + + rcu_read_lock(); + p = rcu_dereference(gact->param); + if (p) { + list_for_each_entry_rcu(entry, &p->entries, list) + /* TCA_GATE_ONE_ENTRY nest and its attributes */ + size += nla_total_size(0) + + nla_total_size(sizeof(u32)) /* TCA_GATE_ENTRY_INDEX */ + + nla_total_size(0) /* TCA_GATE_ENTRY_GATE */ + + nla_total_size(sizeof(u32)) /* TCA_GATE_ENTRY_INTERVAL */ + + nla_total_size(sizeof(s32)) /* TCA_GATE_ENTRY_MAX_OCTETS */ + + nla_total_size(sizeof(s32)); /* TCA_GATE_ENTRY_IPV */ + } + rcu_read_unlock(); + + return size; } static void tcf_gate_entry_destructor(void *priv) diff --git a/net/sched/act_ife.c b/net/sched/act_ife.c index 9cea71fc1db3..2afd68983ece 100644 --- a/net/sched/act_ife.c +++ b/net/sched/act_ife.c @@ -737,6 +737,7 @@ static int tcf_ife_decode(struct sk_buff *skb, const struct tc_action *a, u8 *curr_data; u16 mtype; u16 dlen; + int ret; curr_data = ife_tlv_meta_decode(tlv_data, ifehdr_end, &mtype, &dlen, NULL); @@ -745,13 +746,19 @@ static int tcf_ife_decode(struct sk_buff *skb, const struct tc_action *a, return TC_ACT_SHOT; } - if (find_decode_metaid(skb, p, mtype, dlen, curr_data)) { - /* abuse overlimits to count when we receive metadata - * but dont have an ops for it + ret = find_decode_metaid(skb, p, mtype, dlen, curr_data); + if (ret < 0) { + /* abuse overlimits to count metadata we cannot + * decode: no ops for it, or the decoder rejected it */ - pr_info_ratelimited("Unknown metaid %d dlen %d\n", - mtype, dlen); qstats_cpu_overlimit_inc(ife->common.cpu_qstats); + + if (ret == -ENOENT) + pr_info_ratelimited("Unknown metaid %d dlen %d\n", + mtype, dlen); + else + pr_info_ratelimited("Failed to decode metaid %d dlen %d err %d\n", + mtype, dlen, ret); } } diff --git a/net/sched/act_meta_mark.c b/net/sched/act_meta_mark.c index ea0573cb8b2d..e2f61b22bf0f 100644 --- a/net/sched/act_meta_mark.c +++ b/net/sched/act_meta_mark.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -28,9 +29,10 @@ static int skbmark_encode(struct sk_buff *skb, void *skbdata, static int skbmark_decode(struct sk_buff *skb, void *data, u16 len) { - u32 ifemark = *(u32 *)data; + if (len != sizeof(u32)) + return -EINVAL; - skb->mark = ntohl(ifemark); + skb->mark = get_unaligned_be32(data); return 0; } diff --git a/net/sched/act_meta_skbprio.c b/net/sched/act_meta_skbprio.c index 2df3133ce5ad..5cdb57931eab 100644 --- a/net/sched/act_meta_skbprio.c +++ b/net/sched/act_meta_skbprio.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -33,9 +34,10 @@ static int skbprio_encode(struct sk_buff *skb, void *skbdata, static int skbprio_decode(struct sk_buff *skb, void *data, u16 len) { - u32 ifeprio = *(u32 *)data; + if (len != sizeof(u32)) + return -EINVAL; - skb->priority = ntohl(ifeprio); + skb->priority = get_unaligned_be32(data); return 0; } diff --git a/net/sched/act_meta_skbtcindex.c b/net/sched/act_meta_skbtcindex.c index 44547caead46..8803710c0905 100644 --- a/net/sched/act_meta_skbtcindex.c +++ b/net/sched/act_meta_skbtcindex.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -28,9 +29,10 @@ static int skbtcindex_encode(struct sk_buff *skb, void *skbdata, static int skbtcindex_decode(struct sk_buff *skb, void *data, u16 len) { - u16 ifetc_index = *(u16 *)data; + if (len != sizeof(u16)) + return -EINVAL; - skb->tc_index = ntohs(ifetc_index); + skb->tc_index = get_unaligned_be16(data); return 0; } diff --git a/net/sched/cls_u32.c b/net/sched/cls_u32.c index a3e65c8cf29e..76ce2d124079 100644 --- a/net/sched/cls_u32.c +++ b/net/sched/cls_u32.c @@ -1003,8 +1003,16 @@ static int u32_change(struct net *net, struct sk_buff *in_skb, return -ENOMEM; } } else { - err = idr_alloc_u32(&tp_c->handle_idr, ht, &handle, - handle, GFP_KERNEL); + /* The IDR is keyed on the mapped id, and that is + * what the destroy paths remove. Ask for it here, + * so a manual handle colliding with the + * auto-allocated id space is rejected (-ENOSPC) + * instead of aliasing a future auto id. + */ + u32 id = handle2id(handle); + + err = idr_alloc_u32(&tp_c->handle_idr, ht, &id, id, + GFP_KERNEL); if (err) { kfree(ht); return err; diff --git a/net/sched/em_text.c b/net/sched/em_text.c index 343f1aebeec2..4132f8c3c5fc 100644 --- a/net/sched/em_text.c +++ b/net/sched/em_text.c @@ -113,7 +113,7 @@ static void em_text_destroy(struct tcf_ematch *m) static int em_text_dump(struct sk_buff *skb, struct tcf_ematch *m) { struct text_match *tm = EM_TEXT_PRIV(m); - struct tcf_em_text conf; + struct tcf_em_text conf = {}; strscpy(conf.algo, tm->config->ops->name); conf.from_offset = tm->from_offset; diff --git a/net/sched/sch_hfsc.c b/net/sched/sch_hfsc.c index e87f5021a199..284490fd6ca9 100644 --- a/net/sched/sch_hfsc.c +++ b/net/sched/sch_hfsc.c @@ -386,6 +386,15 @@ cftree_update(struct hfsc_class *cl) #define SM_MASK ((1ULL << SM_SHIFT) - 1) #define ISM_MASK ((1ULL << ISM_SHIFT) - 1) +/* + * Cap on the non-descending hops a classify walk may take before its + * filter chain is treated as misconfigured. A flowid binding that was + * legal at bind time can become lateral once hfsc_adjust_levels() + * raises a class level; a few such hops are legitimate, an unbounded + * run means the chain cycles. + */ +#define HFSC_CLASSIFY_MAX_DRIFT 8 + static inline u64 seg_x2y(u64 x, u64 sm) { @@ -1133,6 +1142,7 @@ hfsc_classify(struct sk_buff *skb, struct Qdisc *sch, int *qerr) struct hfsc_class *head, *cl; struct tcf_result res; struct tcf_proto *tcf; + unsigned int drift; int result; if (TC_H_MAJ(skb->priority ^ sch->handle) == 0 && @@ -1142,6 +1152,7 @@ hfsc_classify(struct sk_buff *skb, struct Qdisc *sch, int *qerr) *qerr = NET_XMIT_SUCCESS | __NET_XMIT_BYPASS; head = &q->root; + drift = HFSC_CLASSIFY_MAX_DRIFT; tcf = rcu_dereference_bh(q->root.filter_list); while (tcf && (result = tcf_classify_qdisc(skb, tcf, &res, false)) >= 0) { #ifdef CONFIG_NET_CLS_ACT @@ -1167,6 +1178,17 @@ hfsc_classify(struct sk_buff *skb, struct Qdisc *sch, int *qerr) if (cl->level == 0) return cl; /* hit leaf class */ + /* + * flowid binds skip the level check above (res.class is set + * at bind time and levels drift after), so a walk can follow + * lateral hops without descending; a bounded number of them + * is legal, more means the chain cycles. + */ + if (cl->level >= head->level && drift-- == 0) { + pr_warn_ratelimited("hfsc: classify hop budget exhausted, dropping packet\n"); + return NULL; + } + /* apply inner filter chain */ tcf = rcu_dereference_bh(cl->filter_list); head = cl; diff --git a/net/sched/sch_teql.c b/net/sched/sch_teql.c index 9e52afc2d980..409ce50cc0db 100644 --- a/net/sched/sch_teql.c +++ b/net/sched/sch_teql.c @@ -265,14 +265,11 @@ __teql_resolve(struct sk_buff *skb, struct sk_buff *skb_res, } if (neigh_event_send(n, skb_res) == 0) { - int err; char haddr[MAX_ADDR_LEN]; neigh_ha_snapshot(haddr, n, dev); - err = dev_hard_header(skb, dev, ntohs(skb_protocol(skb, false)), - haddr, NULL, skb->len); - - if (err < 0) + if (dev_hard_header(skb, dev, ntohs(skb_protocol(skb, false)), + haddr, NULL, skb->len) < 0) err = -EINVAL; } else { err = (skb_res == NULL) ? -EAGAIN : 1; diff --git a/net/sctp/associola.c b/net/sctp/associola.c index c0512c827d0f..4521be3bd85a 100644 --- a/net/sctp/associola.c +++ b/net/sctp/associola.c @@ -1289,18 +1289,19 @@ void sctp_assoc_update_retran_path(struct sctp_association *asoc) /* Manually skip the head element. */ if (&trans->transports == &asoc->peer.transport_addr_list) continue; - if (trans->state == SCTP_UNCONFIRMED) - continue; - trans_next = sctp_trans_elect_best(trans, trans_next); - /* Active is good enough for immediate return. */ - if (trans_next->state == SCTP_ACTIVE) - break; + if (trans->state != SCTP_UNCONFIRMED) { + trans_next = sctp_trans_elect_best(trans, trans_next); + /* Active is good enough for immediate return. */ + if (trans_next->state == SCTP_ACTIVE) + break; + } /* We've reached the end, time to update path. */ if (trans == asoc->peer.retran_path) break; } - asoc->peer.retran_path = trans_next; + if (trans_next) + asoc->peer.retran_path = trans_next; pr_debug("%s: association:%p updated new path to addr:%pISpc\n", __func__, asoc, &asoc->peer.retran_path->ipaddr.sa); diff --git a/net/sctp/input.c b/net/sctp/input.c index 864741fae418..9494cfa51106 100644 --- a/net/sctp/input.c +++ b/net/sctp/input.c @@ -436,9 +436,10 @@ void sctp_icmp_proto_unreachable(struct sock *sk, if (timer_pending(&t->proto_unreach_timer)) return; else { - if (!mod_timer(&t->proto_unreach_timer, - jiffies + (HZ/20))) - sctp_transport_hold(t); + sctp_transport_hold(t); + if (mod_timer(&t->proto_unreach_timer, + jiffies + (HZ / 20))) + sctp_transport_put(t); } } else { struct net *net = sock_net(sk); diff --git a/net/sctp/sm_sideeffect.c b/net/sctp/sm_sideeffect.c index 0d99b7e8c082..35f540fb15fc 100644 --- a/net/sctp/sm_sideeffect.c +++ b/net/sctp/sm_sideeffect.c @@ -244,8 +244,9 @@ void sctp_generate_t3_rtx_event(struct timer_list *t) pr_debug("%s: sock is busy\n", __func__); /* Try again later. */ - if (!mod_timer(&transport->T3_rtx_timer, jiffies + (HZ/20))) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->T3_rtx_timer, jiffies + (HZ / 20))) + sctp_transport_put(transport); goto out_unlock; } @@ -280,8 +281,9 @@ static void sctp_generate_timeout_event(struct sctp_association *asoc, timeout_type); /* Try again later. */ - if (!mod_timer(&asoc->timers[timeout_type], jiffies + (HZ/20))) - sctp_association_hold(asoc); + sctp_association_hold(asoc); + if (mod_timer(&asoc->timers[timeout_type], jiffies + (HZ / 20))) + sctp_association_put(asoc); goto out_unlock; } @@ -378,8 +380,9 @@ void sctp_generate_heartbeat_event(struct timer_list *t) pr_debug("%s: sock is busy\n", __func__); /* Try again later. */ - if (!mod_timer(&transport->hb_timer, jiffies + (HZ/20))) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->hb_timer, jiffies + (HZ / 20))) + sctp_transport_put(transport); goto out_unlock; } @@ -388,8 +391,9 @@ void sctp_generate_heartbeat_event(struct timer_list *t) timeout = sctp_transport_timeout(transport); if (elapsed < timeout) { elapsed = timeout - elapsed; - if (!mod_timer(&transport->hb_timer, jiffies + elapsed)) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->hb_timer, jiffies + elapsed)) + sctp_transport_put(transport); goto out_unlock; } @@ -422,9 +426,10 @@ void sctp_generate_proto_unreach_event(struct timer_list *t) pr_debug("%s: sock is busy\n", __func__); /* Try again later. */ - if (!mod_timer(&transport->proto_unreach_timer, - jiffies + (HZ/20))) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->proto_unreach_timer, + jiffies + (HZ / 20))) + sctp_transport_put(transport); goto out_unlock; } @@ -458,8 +463,9 @@ void sctp_generate_reconf_event(struct timer_list *t) pr_debug("%s: sock is busy\n", __func__); /* Try again later. */ - if (!mod_timer(&transport->reconf_timer, jiffies + (HZ / 20))) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->reconf_timer, jiffies + (HZ / 20))) + sctp_transport_put(transport); goto out_unlock; } @@ -495,8 +501,9 @@ void sctp_generate_probe_event(struct timer_list *t) pr_debug("%s: sock is busy\n", __func__); /* Try again later. */ - if (!mod_timer(&transport->probe_timer, jiffies + (HZ / 20))) - sctp_transport_hold(transport); + sctp_transport_hold(transport); + if (mod_timer(&transport->probe_timer, jiffies + (HZ / 20))) + sctp_transport_put(transport); goto out_unlock; } diff --git a/net/sctp/sm_statefuns.c b/net/sctp/sm_statefuns.c index c701cff6ea67..8dc65498763c 100644 --- a/net/sctp/sm_statefuns.c +++ b/net/sctp/sm_statefuns.c @@ -2654,6 +2654,8 @@ static enum sctp_disposition sctp_sf_do_5_2_6_stale( sctp_add_cmd_sf(commands, SCTP_CMD_REPLY, SCTP_CHUNK(reply)); + sctp_add_cmd_sf(commands, SCTP_CMD_DISCARD_PACKET, SCTP_NULL()); + return SCTP_DISPOSITION_CONSUME; nomem: diff --git a/net/smc/smc_core.c b/net/smc/smc_core.c index 04aedd957543..9974149659c2 100644 --- a/net/smc/smc_core.c +++ b/net/smc/smc_core.c @@ -1849,6 +1849,7 @@ void smcr_port_err(struct smc_ib_device *smcibdev, u8 ibport) struct smc_link_group *lgr, *n; int i; + spin_lock_bh(&smc_lgr_list.lock); list_for_each_entry_safe(lgr, n, &smc_lgr_list.list, list) { if (strncmp(smcibdev->pnetid[ibport - 1], lgr->pnet_id, SMC_MAX_PNETID_LEN)) @@ -1863,6 +1864,7 @@ void smcr_port_err(struct smc_ib_device *smcibdev, u8 ibport) smcr_link_down_cond_sched(lnk); } } + spin_unlock_bh(&smc_lgr_list.lock); } static void smc_link_down_work(struct work_struct *work) diff --git a/net/smc/smc_ib.c b/net/smc/smc_ib.c index 9bb495707445..daaa8a72da90 100644 --- a/net/smc/smc_ib.c +++ b/net/smc/smc_ib.c @@ -333,6 +333,7 @@ static bool smc_ib_check_link_gid(u8 gid[SMC_GID_SIZE], bool smcrv2, static void smc_ib_gid_check(struct smc_ib_device *smcibdev, u8 ibport) { struct smc_link_group *lgr; + bool stale_gid = false; int i; spin_lock_bh(&smc_lgr_list.lock); @@ -348,11 +349,16 @@ static void smc_ib_gid_check(struct smc_ib_device *smcibdev, u8 ibport) continue; if (!smc_ib_check_link_gid(lgr->lnk[i].gid, lgr->smc_version == SMC_V2, - smcibdev, ibport)) - smcr_port_err(smcibdev, ibport); + smcibdev, ibport)) { + stale_gid = true; + goto out; + } } } +out: spin_unlock_bh(&smc_lgr_list.lock); + if (stale_gid) + smcr_port_err(smcibdev, ibport); } static int smc_ib_remember_port_attr(struct smc_ib_device *smcibdev, u8 ibport) diff --git a/net/tipc/group.c b/net/tipc/group.c index 14e6732624e2..74f6d3dac078 100644 --- a/net/tipc/group.c +++ b/net/tipc/group.c @@ -797,10 +797,10 @@ void tipc_group_proto_rcv(struct tipc_group *grp, bool *usr_wakeup, tipc_group_open(m, usr_wakeup); return; case GRP_ACK_MSG: - if (!m) + if (!m || !grp->bc_ackers) return; acked = msg_grp_bc_acked(hdr); - if (less_eq(acked, m->bc_acked)) + if (acked != grp->bc_snd_nxt || m->bc_acked == acked) return; m->bc_acked = acked; if (--grp->bc_ackers) diff --git a/net/tipc/monitor.c b/net/tipc/monitor.c index a94b9b36a700..1a438e312d66 100644 --- a/net/tipc/monitor.c +++ b/net/tipc/monitor.c @@ -632,9 +632,10 @@ static void mon_timeout(struct timer_list *t) { struct tipc_monitor *mon = timer_container_of(mon, t, timer); struct tipc_peer *self; - int best_member_cnt = dom_size(mon->peer_cnt) - 1; + int best_member_cnt; write_lock_bh(&mon->lock); + best_member_cnt = dom_size(mon->peer_cnt) - 1; self = mon->self; if (self && (best_member_cnt != self->applied)) { mon_update_local_domain(mon); diff --git a/net/vmw_vsock/af_vsock.c b/net/vmw_vsock/af_vsock.c index f840498b58af..9b71479a2b29 100644 --- a/net/vmw_vsock/af_vsock.c +++ b/net/vmw_vsock/af_vsock.c @@ -2889,6 +2889,9 @@ static int vsock_net_child_mode_string(const struct ctl_table *table, int write, net = container_of(table->data, struct net, vsock.child_ns_mode); + if (!*lenp) + return 0; + ret = __vsock_net_mode_string(table, write, buffer, lenp, ppos, vsock_net_child_mode(net), &new_mode); if (ret) diff --git a/tools/testing/selftests/nci/nci_dev.c b/tools/testing/selftests/nci/nci_dev.c index 312f84ee0444..07427fa42888 100644 --- a/tools/testing/selftests/nci/nci_dev.c +++ b/tools/testing/selftests/nci/nci_dev.c @@ -8,6 +8,7 @@ #include #include +#include #include #include #include @@ -87,6 +88,16 @@ struct msgtemplate { char buf[MAX_MSG_SIZE]; }; +static int join_thread_status(pthread_t thread) +{ + void *thread_ret = NULL; + + if (pthread_join(thread, &thread_ret)) + return -1; + + return (int)(intptr_t)thread_ret; +} + static int create_nl_socket(void) { int fd; @@ -182,7 +193,7 @@ static int get_family_id(int sd, __u32 pid, __u32 *event_group) } ans; struct nlattr *na; int resp_len; - __u16 id; + __u16 id = 0; int len; int rc; @@ -438,13 +449,13 @@ FIXTURE_SETUP(NCI) else rc = pthread_create(&thread_t, NULL, virtual_dev_open, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_UP, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); ASSERT_EQ(status, 0); self->open_state = true; } @@ -509,12 +520,12 @@ FIXTURE_TEARDOWN(NCI) rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); ASSERT_EQ(status, 0); } @@ -585,12 +596,11 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p void *nla_start_poll_data[2] = {&dev_idx, &proto}; int nla_start_poll_len[2] = {4, 4}; pthread_t thread_t; - int status; int rc; rc = pthread_create(&thread_t, NULL, virtual_poll_start, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_mt_nla(sd, fid, pid, NFC_CMD_START_POLL, 2, nla_start_poll_type, @@ -598,19 +608,17 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p if (rc != 0) return rc; - pthread_join(thread_t, (void **)&status); - return status; + return join_thread_status(thread_t); } int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid) { pthread_t thread_t; - int status; int rc; rc = pthread_create(&thread_t, NULL, virtual_poll_stop, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_with_idx(sd, fid, pid, @@ -618,8 +626,7 @@ int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid) if (rc != 0) return rc; - pthread_join(thread_t, (void **)&status); - return status; + return join_thread_status(thread_t); } TEST_F(NCI, start_poll) @@ -830,10 +837,14 @@ int disconnect_tag(int nfc_sock, int virtual_fd) status = pthread_create(&thread_t, NULL, virtual_deactivate_proc, (void *)&virtual_fd); + if (status) + return status; close(nfc_sock); - pthread_join(thread_t, (void **)&status); - return status; + if (status) + return -1; + + return join_thread_status(thread_t); } TEST_F(NCI, t4t_tag_read) @@ -874,13 +885,13 @@ TEST_F(NCI, deinit) else rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); EXPECT_EQ(rc, 0); - pthread_join(thread_t, (void **)&status); + status = join_thread_status(thread_t); self->open_state = 0; ASSERT_EQ(status, 0); diff --git a/tools/testing/selftests/net/config b/tools/testing/selftests/net/config index 30d5fcb09a83..737e7e6327b3 100644 --- a/tools/testing/selftests/net/config +++ b/tools/testing/selftests/net/config @@ -118,10 +118,10 @@ CONFIG_NFT_NAT=m CONFIG_NUMA=y CONFIG_OPENVSWITCH=m CONFIG_PAGE_POOL_STATS=y -CONFIG_SYSCTL=y CONFIG_PSAMPLE=m CONFIG_RPS=y CONFIG_SYN_COOKIES=y +CONFIG_SYSCTL=y CONFIG_SYSFS=y CONFIG_TAP=m CONFIG_TCP_CONG_DCTCP=y diff --git a/tools/testing/selftests/net/nl_nlctrl.py b/tools/testing/selftests/net/nl_nlctrl.py index fe1f66dc9435..237b3d273260 100755 --- a/tools/testing/selftests/net/nl_nlctrl.py +++ b/tools/testing/selftests/net/nl_nlctrl.py @@ -9,40 +9,86 @@ from lib.py import ksft_run, ksft_exit from lib.py import ksft_eq, ksft_ge, ksft_true, ksft_in, ksft_not_in from lib.py import NetdevFamily, EthtoolFamily, NlctrlFamily +# Families we can expect to always be around, and which between them +# cover ops with a do, with a dump, and with both. +FAMILIES = ('nlctrl', 'netdev') -def getfamily_do(ctrl) -> None: - """Query a single family by name and validate its ops.""" - fam = ctrl.getfamily({'family-name': 'netdev'}) - ksft_eq(fam['family-name'], 'netdev') + +def _get_ops(ctrl, name): + """Get the ops of a family, keyed by command id.""" + fam = ctrl.getfamily({'family-name': name}) + ksft_eq(fam['family-name'], name) ksft_true(fam['family-id'] > 0) # The format of ops is quite odd, [{$idx: {"id"...}}, {$idx: {"id"...}}] # Discard the indices and re-key by command id. ops_by_id = {v['id']: v for op in fam['ops'] for v in op.values()} - ksft_eq(len(ops_by_id), len(fam['ops'])) + ksft_eq(len(ops_by_id), len(fam['ops']), + comment=f"{name} lists a command twice") + return ops_by_id - # All ops should have a policy (either do or dump has one) - for op in ops_by_id.values(): - ksft_in('cmd-cap-haspol', op['flags'], - comment=f"op {op['id']} missing haspol") + +def _get_policy_map(ctrl, req): + """ + The policy map in the Netlink replies looks like this: + + [{'family-id': 16, 'op-policy': {'do': 0, 'dump': 0, 'op-id': 3}}, + {'family-id': 16, 'op-policy': {'dump': 1, 'op-id': 4}}, ...] + + Return the mapping: + + {3:{'do','dump'}, 4:{'dump'}} + + The policy itself is discarded here, only return which command has policy. + """ + pol_map = {} + for msg in ctrl.getpolicy(req, dump=True): + if 'op-policy' not in msg: + continue + modes = dict(msg['op-policy']) + cmd = modes.pop('op-id') + ksft_not_in(cmd, pol_map, comment=f"command {cmd} reported twice") + pol_map[cmd] = set(modes.keys()) + return pol_map + + +def getfamily_do(ctrl) -> None: + """Query single families by name and validate their ops.""" + ops = {name: _get_ops(ctrl, name) for name in FAMILIES} + + for name, ops_by_id in ops.items(): + for op in ops_by_id.values(): + # All ops in nlctrl and netdev have a policy + ksft_in('cmd-cap-haspol', op['flags'], + comment=f"{name} op {op['id']} missing haspol") + ksft_true(op['flags'] & {'cmd-cap-do', 'cmd-cap-dump'}, + comment=f"{name} op {op['id']} has no handler") + + # nlctrl getfamily (id 3) does both, getpolicy (id 10) is dump-only + ksft_in('cmd-cap-do', ops['nlctrl'][3]['flags']) + ksft_in('cmd-cap-dump', ops['nlctrl'][3]['flags']) + ksft_not_in('cmd-cap-do', ops['nlctrl'][10]['flags']) + ksft_in('cmd-cap-dump', ops['nlctrl'][10]['flags']) + + netdev = ops['netdev'] # dev-get (id 1) should support both do and dump - ksft_in('cmd-cap-do', ops_by_id[1]['flags']) - ksft_in('cmd-cap-dump', ops_by_id[1]['flags']) + ksft_in('cmd-cap-do', netdev[1]['flags']) + ksft_in('cmd-cap-dump', netdev[1]['flags']) # qstats-get (id 12) is dump-only - ksft_not_in('cmd-cap-do', ops_by_id[12]['flags']) - ksft_in('cmd-cap-dump', ops_by_id[12]['flags']) + ksft_not_in('cmd-cap-do', netdev[12]['flags']) + ksft_in('cmd-cap-dump', netdev[12]['flags']) # napi-set (id 14) is do-only and requires admin - ksft_in('cmd-cap-do', ops_by_id[14]['flags']) - ksft_not_in('cmd-cap-dump', ops_by_id[14]['flags']) - ksft_in('admin-perm', ops_by_id[14]['flags']) + ksft_in('cmd-cap-do', netdev[14]['flags']) + ksft_not_in('cmd-cap-dump', netdev[14]['flags']) + ksft_in('admin-perm', netdev[14]['flags']) # Notification-only commands (dev-add/del/change-ntf etc.) must # not appear in the ops list since they have no do/dump handlers. for ntf_id in [2, 3, 4, 6, 7, 8]: - ksft_not_in(ntf_id, ops_by_id, + ksft_not_in(ntf_id, netdev, comment=f"ntf-only cmd {ntf_id} should not be in ops") @@ -103,6 +149,41 @@ def getpolicy_dump(_ctrl) -> None: comment="linkinfo-set should not have a dump policy") +def getpolicy_op_map(ctrl) -> None: + """Check the op-to-policy map consistency. Each op with 'haspol' flag + has to have a policy. The policy back-references must name only + real ops that exist, have given modes (do vs dump) and have 'haspol'. + """ + for name in FAMILIES: + ops_by_id = _get_ops(ctrl, name) + haspol = {cmd for cmd, op in ops_by_id.items() + if 'cmd-cap-haspol' in op['flags']} + + pol_map = _get_policy_map(ctrl, {'family-name': name}) + ksft_eq(set(pol_map), haspol, + comment=f"{name} policy map does not match the op list") + + # Walk the op list rather than the map, the map may be missing + # the very op we are after. Asking for a command the family does + # not have is an error, so it must not come from the map either. + for cmd in sorted(haspol): + modes = pol_map.get(cmd, set()) + + # The kernel only reports a mode the op actually has. + if 'do' in modes: + ksft_in('cmd-cap-do', ops_by_id[cmd]['flags'], + comment=f"{name} cmd {cmd} has no do") + if 'dump' in modes: + ksft_in('cmd-cap-dump', ops_by_id[cmd]['flags'], + comment=f"{name} cmd {cmd} has no dump") + + # Asking for one op builds the map in a different place in + # the kernel, it has to report what the full dump did. + single = _get_policy_map(ctrl, {'family-name': name, 'op': cmd}) + ksft_eq(single, {cmd: modes}, + comment=f"{name} cmd {cmd} policy differs from the dump") + + def getpolicy_by_op(_ctrl) -> None: """Query policy for specific ops, check attr names are resolved.""" ndev = NetdevFamily() @@ -122,6 +203,7 @@ def main() -> None: ksft_run([getfamily_do, getfamily_dump, getpolicy_dump, + getpolicy_op_map, getpolicy_by_op], args=(ctrl, )) ksft_exit() diff --git a/tools/testing/selftests/net/ovpn/common.sh b/tools/testing/selftests/net/ovpn/common.sh index 2d844eb3aa6e..5e9c81e885e6 100644 --- a/tools/testing/selftests/net/ovpn/common.sh +++ b/tools/testing/selftests/net/ovpn/common.sh @@ -136,6 +136,19 @@ ovpn_create_ns() { ip netns add "ovpn_peer${1}" } +ovpn_peer_vpn_addr() { + local peer="$1" + local file + + if [ "${OVPN_PROTO}" == "UDP" ]; then + file="${OVPN_UDP_PEERS_FILE}" + else + file="${OVPN_TCP_PEERS_FILE}" + fi + + awk -v peer="${peer}" '$1 == peer {print $NF; exit}' "${file}" +} + ovpn_setup_ns() { local peer="ovpn_peer${1}" local server_ns="ovpn_peer0" diff --git a/tools/testing/selftests/net/ovpn/ovpn-cli.c b/tools/testing/selftests/net/ovpn/ovpn-cli.c index f4effa7580c0..3b612a8a18fe 100644 --- a/tools/testing/selftests/net/ovpn/ovpn-cli.c +++ b/tools/testing/selftests/net/ovpn/ovpn-cli.c @@ -650,6 +650,26 @@ static int ovpn_connect(struct ovpn_ctx *ovpn) return ret; } +static int ovpn_nl_put_vpn_addr(struct nl_msg *msg, + const struct ovpn_ctx *ovpn) +{ + if (!ovpn->peer_ip_set) + return 0; + + switch (ovpn->peer_ip.in4.sin_family) { + case AF_INET: + return nla_put_u32(msg, OVPN_A_PEER_VPN_IPV4, + ovpn->peer_ip.in4.sin_addr.s_addr); + case AF_INET6: + return nla_put(msg, OVPN_A_PEER_VPN_IPV6, + sizeof(struct in6_addr), + &ovpn->peer_ip.in6.sin6_addr); + default: + fprintf(stderr, "Invalid family for peer address\n"); + return -EAFNOSUPPORT; + } +} + static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp) { struct nlattr *attr; @@ -691,22 +711,9 @@ static int ovpn_new_peer(struct ovpn_ctx *ovpn, bool is_tcp) } } - if (ovpn->peer_ip_set) { - switch (ovpn->peer_ip.in4.sin_family) { - case AF_INET: - NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_VPN_IPV4, - ovpn->peer_ip.in4.sin_addr.s_addr); - break; - case AF_INET6: - NLA_PUT(ctx->nl_msg, OVPN_A_PEER_VPN_IPV6, - sizeof(struct in6_addr), - &ovpn->peer_ip.in6.sin6_addr); - break; - default: - fprintf(stderr, "Invalid family for peer address\n"); - goto nla_put_failure; - } - } + ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn); + if (ret) + goto nla_put_failure; nla_nest_end(ctx->nl_msg, attr); @@ -732,6 +739,10 @@ static int ovpn_set_peer(struct ovpn_ctx *ovpn) ovpn->keepalive_interval); NLA_PUT_U32(ctx->nl_msg, OVPN_A_PEER_KEEPALIVE_TIMEOUT, ovpn->keepalive_timeout); + + ret = ovpn_nl_put_vpn_addr(ctx->nl_msg, ovpn); + if (ret) + goto nla_put_failure; nla_nest_end(ctx->nl_msg, attr); ret = ovpn_nl_msg_send(ctx, NULL); @@ -1730,13 +1741,14 @@ static void usage(const char *cmd) fprintf(stderr, "\tmark: socket FW mark value\n"); fprintf(stderr, - "* set_peer : set peer attributes\n"); + "* set_peer [vpnaddr]: set peer attributes\n"); fprintf(stderr, "\tiface: ovpn interface name\n"); fprintf(stderr, "\tpeer_id: peer ID of the peer to modify\n"); fprintf(stderr, "\tkeepalive_interval: interval for sending ping messages\n"); fprintf(stderr, "\tkeepalive_timeout: time after which a peer is timed out\n"); + fprintf(stderr, "\tvpnaddr: peer VPN IP\n"); fprintf(stderr, "* del_peer : delete peer\n"); fprintf(stderr, "\tiface: ovpn interface name\n"); @@ -2090,6 +2102,8 @@ static int ovpn_run_cmd(struct ovpn_ctx *ovpn) return ret; ret = ovpn_new_peer(ovpn, false); + if (ret < 0) + return ret; ovpn_waitbg(); break; case CMD_NEW_MULTI_PEER: @@ -2331,6 +2345,12 @@ static int ovpn_parse_cmd_args(struct ovpn_ctx *ovpn, int argc, char *argv[]) "keepalive interval value out of range\n"); return -1; } + + if (argc > 6) { + ret = ovpn_parse_remote(ovpn, NULL, NULL, argv[6]); + if (ret < 0) + return -1; + } break; case CMD_DEL_PEER: if (argc < 4) diff --git a/tools/testing/selftests/net/ovpn/test.sh b/tools/testing/selftests/net/ovpn/test.sh index 9b5610837032..392109d5e14e 100755 --- a/tools/testing/selftests/net/ovpn/test.sh +++ b/tools/testing/selftests/net/ovpn/test.sh @@ -56,6 +56,76 @@ ovpn_prepare_network() { done } +ovpn_new_test_peer() { + local peer_id="$1" + + shift + ip netns exec ovpn_peer0 "${OVPN_CLI}" new_peer tun0 \ + "${peer_id}" none 65000 10.10.1.2 1 "$@" +} + +ovpn_set_peer_vpn_addr() { + ip netns exec ovpn_peer0 "${OVPN_CLI}" set_peer tun0 \ + "$1" 60 120 "$2" +} + +ovpn_run_vpn_addr_validation() { + local addr + local peer1_addr4 + local test_peer_id=$((OVPN_NUM_PEERS + 1)) + local test_peer_addr6="2001:db8::2" + # Do not include 0.0.0.0 or :: here. They are invalid on creation, but + # clear one address family on update and are valid if the other remains. + local -a invalid_addrs=( + "127.0.0.1" + "224.0.0.1" + "255.255.255.255" + "::1" + "::192.0.2.1" + "::ffff:192.0.2.1" + "ff02::1" + ) + + peer1_addr4=$(ovpn_peer_vpn_addr 1) + + ovpn_cmd_fail "reject peer without VPN address" \ + ovpn_new_test_peer "${test_peer_id}" + + for addr in "0.0.0.0" "::" "${invalid_addrs[@]}"; do + ovpn_cmd_fail "reject new peer VPN address ${addr}" \ + ovpn_new_test_peer "${test_peer_id}" "${addr}" + done + + ovpn_cmd_fail "reject duplicate IPv4 address on peer creation" \ + ovpn_new_test_peer "${test_peer_id}" "${peer1_addr4}" + ovpn_cmd_fail "reject clearing the last peer VPN address" \ + ovpn_set_peer_vpn_addr 1 0.0.0.0 + + for addr in "${invalid_addrs[@]}"; do + ovpn_cmd_fail "reject updated peer VPN address ${addr}" \ + ovpn_set_peer_vpn_addr 1 "${addr}" + done + + ovpn_cmd_fail "reject duplicate IPv4 address on peer update" \ + ovpn_set_peer_vpn_addr 2 "${peer1_addr4}" + + ovpn_cmd_ok "add peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 "${test_peer_addr6}" + ovpn_cmd_fail "reject duplicate IPv6 address on peer creation" \ + ovpn_new_test_peer "${test_peer_id}" "${test_peer_addr6}" + ovpn_cmd_fail "reject duplicate IPv6 address on peer update" \ + ovpn_set_peer_vpn_addr 2 "${test_peer_addr6}" + + ovpn_cmd_ok "clear peer IPv4 address" \ + ovpn_set_peer_vpn_addr 1 0.0.0.0 + ovpn_cmd_fail "reject clearing the remaining peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 :: + ovpn_cmd_ok "restore peer IPv4 address" \ + ovpn_set_peer_vpn_addr 1 "${peer1_addr4}" + ovpn_cmd_ok "clear peer IPv6 address" \ + ovpn_set_peer_vpn_addr 1 :: +} + ovpn_run_basic_traffic() { local p local header1 @@ -293,15 +363,16 @@ trap ovpn_stage_err ERR ktap_print_header if [ "${OVPN_FLOAT}" == "1" ]; then - ktap_set_plan 13 + ktap_set_plan 14 else - ktap_set_plan 12 + ktap_set_plan 13 fi ovpn_cleanup modprobe -q ovpn || true ovpn_run_stage "setup network topology" ovpn_prepare_network +ovpn_run_stage "validate peer VPN addresses" ovpn_run_vpn_addr_validation ovpn_run_stage "run baseline data traffic" ovpn_run_basic_traffic ovpn_run_stage "run LAN traffic behind peer1" ovpn_run_lan_traffic [ "${OVPN_FLOAT}" == "1" ] && ovpn_run_stage "run floating peer checks" \ diff --git a/tools/testing/selftests/net/packetdrill/config b/tools/testing/selftests/net/packetdrill/config index 83dde525c53c..b7df8bf30920 100644 --- a/tools/testing/selftests/net/packetdrill/config +++ b/tools/testing/selftests/net/packetdrill/config @@ -4,8 +4,8 @@ CONFIG_IPV6=y CONFIG_NET_NS=y CONFIG_NET_SCH_FIFO=y CONFIG_NET_SCH_FQ=y -CONFIG_SYSCTL=y CONFIG_SYN_COOKIES=y +CONFIG_SYSCTL=y CONFIG_TCP_CONG_CUBIC=y CONFIG_TCP_MD5SIG=y CONFIG_TUN=y diff --git a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json index e2b03f2b5e89..edc5148a8d97 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json +++ b/tools/testing/selftests/tc-testing/tc-tests/filters/u32.json @@ -376,5 +376,53 @@ "teardown": [ "$TC qdisc del dev $DUMMY clsact" ] + }, + { + "id": "35fc", + "name": "u32 manual table then auto table: auto allocation must not alias a live manual handle", + "category": [ + "filter", + "u32" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DEV1 ingress", + "$TC filter add dev $DEV1 ingress protocol ip pref 1 handle 801: u32 divisor 16" + ], + "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 2 u32 divisor 16", + "expExitCode": "0", + "verifyCmd": "$TC -d filter show dev $DEV1 ingress", + "matchPattern": "fh 801:", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress" + ] + }, + { + "id": "a6e8", + "name": "u32 manual table add/del does not leak its idr entry (re-adding the same handle succeeds)", + "category": [ + "filter", + "u32" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DEV1 ingress", + "$TC filter add dev $DEV1 ingress protocol ip pref 1 u32 divisor 16", + "$TC filter add dev $DEV1 ingress protocol ip pref 5 handle 901: u32 divisor 1", + "$TC filter del dev $DEV1 ingress protocol ip pref 5 handle 901: u32" + ], + "cmdUnderTest": "$TC filter add dev $DEV1 ingress protocol ip pref 6 handle 901: u32 divisor 1", + "expExitCode": "0", + "verifyCmd": "$TC -d filter show dev $DEV1 ingress", + "matchPattern": "fh 901: ht divisor 1", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress" + ] } ] diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json index c98c339424d4..4f6bbb8b57f9 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json +++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hfsc.json @@ -169,5 +169,39 @@ "teardown": [ "$TC qdisc del dev $DUMMY handle 1: root" ] + }, + { + "id": "8c39", + "name": "HFSC classify walk still reaches leaf after lateral drift", + "category": [ + "qdisc", + "hfsc" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "ip link set lo up", + "$TC qdisc add dev lo handle 1: root hfsc default 30", + "$TC class add dev lo parent 1: classid 1:1 hfsc rt m2 100kbit", + "$TC class add dev lo parent 1:1 classid 1:10 hfsc rt m2 50kbit", + "$TC class add dev lo parent 1: classid 1:2 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1: protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:1", + "$TC filter add dev lo parent 1:1 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:2", + "$TC class add dev lo parent 1:2 classid 1:20 hfsc rt m2 10kbit", + "$TC class add dev lo parent 1: classid 1:3 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1:2 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:3", + "$TC class add dev lo parent 1:3 classid 1:30 hfsc rt m2 10kbit", + "$TC class add dev lo parent 1:3 classid 1:31 hfsc rt m2 100kbit", + "$TC filter add dev lo parent 1:3 protocol ip prio 1 u32 match u8 0 0 at 0 flowid 1:31" + ], + "cmdUnderTest": "ping -n -c 10 -W 1 127.0.0.1", + "expExitCode": "0", + "verifyCmd": "$TC -s class show dev lo", + "matchPattern": "class hfsc 1:31 parent 1:3 rt[^\\n]*\\n Sent [0-9]+ bytes [1-9][0-9]* pkt", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev lo handle 1: root" + ] } ]