The mempool related code has no strong ties with the netdev_dpdk object. Move this code to a new netdev-dpdk-common.c file for reuse.
Signed-off-by: David Marchand <[email protected]> --- lib/netdev-dpdk-common.c | 447 +++++++++++++++++++++++++++++++++++ lib/netdev-dpdk-common.h | 60 +++++ lib/netdev-dpdk.c | 492 +-------------------------------------- tests/system-dpdk.at | 1 + 4 files changed, 520 insertions(+), 480 deletions(-) diff --git a/lib/netdev-dpdk-common.c b/lib/netdev-dpdk-common.c index e58f4049ab..772830eeb5 100644 --- a/lib/netdev-dpdk-common.c +++ b/lib/netdev-dpdk-common.c @@ -16,12 +16,459 @@ #include <config.h> +#include <rte_malloc.h> +#include <rte_mbuf.h> +#include <rte_mempool.h> + +#include "dp-packet.h" +#include "hash.h" #include "netdev-dpdk-common.h" #include "netdev-provider.h" +#include "openvswitch/list.h" +#include "openvswitch/ofp-parse.h" #include "openvswitch/vlog.h" +#include "ovs-thread.h" +#include "smap.h" +#include "util.h" VLOG_DEFINE_THIS_MODULE(netdev_dpdk_common); +static bool per_port_memory = false; /* Status of per port memory support */ + +/* MAX_NB_MBUF can be divided by 2 many times, until MIN_NB_MBUF */ +BUILD_ASSERT_DECL(MAX_NB_MBUF % ROUND_DOWN_POW2(MAX_NB_MBUF / MIN_NB_MBUF) + == 0); + +/* The smallest possible NB_MBUF that we're going to try should be a multiple + * of MP_CACHE_SZ. This is advised by DPDK documentation. */ +BUILD_ASSERT_DECL((MAX_NB_MBUF / ROUND_DOWN_POW2(MAX_NB_MBUF / MIN_NB_MBUF)) + % MP_CACHE_SZ == 0); + +struct ovs_mutex dpdk_mp_mutex OVS_ACQ_AFTER(dpdk_mutex) + = OVS_MUTEX_INITIALIZER; + +/* Contains all 'struct dpdk_mp's. */ +static struct ovs_list dpdk_mp_list OVS_GUARDED_BY(dpdk_mp_mutex) + = OVS_LIST_INITIALIZER(&dpdk_mp_list); + +struct user_mempool_config { + int adj_mtu; + int socket_id; +}; + +static struct user_mempool_config *user_mempools = NULL; +static int n_user_mempools; + +bool +dpdk_mp_per_port_memory(void) +{ + return per_port_memory; +} + +/* DPDK NIC drivers allocate RX buffers at a particular granularity, typically + * aligned at 1k or less. If a declared mbuf size is not a multiple of this + * value, insufficient buffers are allocated to accomodate the packet in its + * entirety. Furthermore, certain drivers need to ensure that there is also + * sufficient space in the Rx buffer to accommodate two VLAN tags (for QinQ + * frames). If the RX buffer is too small, then the driver enables scatter RX + * behaviour, which reduces performance. To prevent this, use a buffer size + * that is closest to 'mtu', but which satisfies the aforementioned criteria. + */ +static uint32_t +dpdk_buf_size(int mtu) +{ + return ROUND_UP(MTU_TO_MAX_FRAME_LEN(mtu), NETDEV_DPDK_MBUF_ALIGN) + + RTE_PKTMBUF_HEADROOM; +} + +static int +dpdk_get_user_adjusted_mtu(int port_adj_mtu, int port_mtu, int port_socket_id) +{ + int best_adj_user_mtu = INT_MAX; + + for (unsigned i = 0; i < n_user_mempools; i++) { + int user_adj_mtu, user_socket_id; + + user_adj_mtu = user_mempools[i].adj_mtu; + user_socket_id = user_mempools[i].socket_id; + if (port_adj_mtu > user_adj_mtu + || (user_socket_id != INT_MAX + && user_socket_id != port_socket_id)) { + continue; + } + if (user_adj_mtu < best_adj_user_mtu) { + /* This is the is the lowest valid user MTU. */ + best_adj_user_mtu = user_adj_mtu; + if (best_adj_user_mtu == port_adj_mtu) { + /* Found an exact fit, no need to keep searching. */ + break; + } + } + } + if (best_adj_user_mtu == INT_MAX) { + VLOG_DBG("No user configured shared mempool mbuf sizes found " + "suitable for port with MTU %d, NUMA %d.", port_mtu, + port_socket_id); + best_adj_user_mtu = port_adj_mtu; + } else { + VLOG_DBG("Found user configured shared mempool with mbufs " + "of size %d, suitable for port with MTU %d, NUMA %d.", + MTU_TO_FRAME_LEN(best_adj_user_mtu), port_mtu, + port_socket_id); + } + return best_adj_user_mtu; +} + +static void +ovs_rte_pktmbuf_init(struct rte_mempool *mp OVS_UNUSED, + void *opaque_arg OVS_UNUSED, + void *_p, + unsigned i OVS_UNUSED) +{ + struct rte_mbuf *pkt = _p; + + dp_packet_init_dpdk((struct dp_packet *) pkt); +} + +static int +dpdk_mp_full(const struct rte_mempool *mp) OVS_REQUIRES(dpdk_mp_mutex) +{ + /* At this point we want to know if all the mbufs are back + * in the mempool. rte_mempool_full() is not atomic but it's + * the best available and as we are no longer requesting mbufs + * from the mempool, it means mbufs will not move from + * 'mempool ring' --> 'mempool cache'. In rte_mempool_full() + * the ring is counted before caches, so we won't get false + * positives in this use case and we handle false negatives. + * + * If future implementations of rte_mempool_full() were to change + * it could be possible for a false positive. Even that would + * likely be ok, as there are additional checks during mempool + * freeing but it would make things racey. + */ + return rte_mempool_full(mp); +} + +/* Free unused mempools. */ +static void +dpdk_mp_sweep(void) OVS_REQUIRES(dpdk_mp_mutex) +{ + struct dpdk_mp *dmp; + + LIST_FOR_EACH_SAFE (dmp, list_node, &dpdk_mp_list) { + if (!dmp->refcount && dpdk_mp_full(dmp->mp)) { + VLOG_DBG("Freeing mempool \"%s\"", dmp->mp->name); + ovs_list_remove(&dmp->list_node); + rte_mempool_free(dmp->mp); + rte_free(dmp); + } + } +} + +/* Calculating the required number of mbufs differs depending on the + * mempool model being used. Check if per port memory is in use before + * calculating. + */ +static uint32_t +dpdk_calculate_mbufs(struct dpdk_mp_config *cfg, int mtu) +{ + uint32_t n_mbufs; + + if (!per_port_memory) { + /* Shared memory are being used. + * XXX: this is a really rough method of provisioning memory. + * It's impossible to determine what the exact memory requirements are + * when the number of ports and rxqs that utilize a particular mempool + * can change dynamically at runtime. For now, use this rough + * heurisitic. + */ + if (mtu >= RTE_ETHER_MTU) { + n_mbufs = MAX_NB_MBUF; + } else { + n_mbufs = MIN_NB_MBUF; + } + } else { + /* Per port memory is being used. + * XXX: rough estimation of number of mbufs required for this port: + * <packets required to fill the device rxqs> + * + <packets that could be stuck on other ports txqs> + * + <packets in the pmd threads> + * + <additional memory for corner cases> + */ + n_mbufs = cfg->n_rxq * cfg->rxq_size + + cfg->n_txq * cfg->txq_size + + MIN(RTE_MAX_LCORE, cfg->n_rxq) * NETDEV_MAX_BURST + + MIN_NB_MBUF; + } + + return n_mbufs; +} + +static struct dpdk_mp * +dpdk_mp_create(struct dpdk_mp_config *cfg, int mtu) +{ + char mp_name[RTE_MEMPOOL_NAMESIZE]; + const char *netdev_name = cfg->name; + int socket_id = cfg->socket_id; + uint32_t n_mbufs = 0; + uint32_t mbuf_size = 0; + uint32_t aligned_mbuf_size = 0; + uint32_t mbuf_priv_data_len = 0; + uint32_t pkt_size = 0; + uint32_t hash = hash_string(netdev_name, 0); + struct dpdk_mp *dmp = NULL; + int ret; + + dmp = dpdk_zmalloc(sizeof *dmp); + if (!dmp) { + return NULL; + } + dmp->socket_id = socket_id; + dmp->mtu = mtu; + dmp->refcount = 1; + + /* Get the size of each mbuf, based on the MTU */ + mbuf_size = MTU_TO_FRAME_LEN(mtu); + + n_mbufs = dpdk_calculate_mbufs(cfg, mtu); + + do { + /* Full DPDK memory pool name must be unique and cannot be + * longer than RTE_MEMPOOL_NAMESIZE. Note that for the shared + * mempool case this can result in one device using a mempool + * which references a different device in it's name. However as + * mempool names are hashed, the device name will not be readable + * so this is not an issue for tasks such as debugging. + */ + ret = snprintf(mp_name, RTE_MEMPOOL_NAMESIZE, + "ovs%08x%02d%05d%07u", + hash, socket_id, mtu, n_mbufs); + if (ret < 0 || ret >= RTE_MEMPOOL_NAMESIZE) { + VLOG_DBG("snprintf returned %d. " + "Failed to generate a mempool name for \"%s\". " + "Hash:0x%x, socket_id: %d, mtu:%d, mbufs:%u.", + ret, netdev_name, hash, socket_id, mtu, n_mbufs); + break; + } + + VLOG_DBG("Port %s: Requesting a mempool of %u mbufs of size %u " + "on socket %d for %d Rx and %d Tx queues, " + "cache line size of %u", + netdev_name, n_mbufs, mbuf_size, socket_id, + cfg->n_rxq, cfg->n_txq, + RTE_CACHE_LINE_SIZE); + + /* The size of the mbuf's private area (i.e. area that holds OvS' + * dp_packet data)*/ + mbuf_priv_data_len = sizeof(struct dp_packet) - + sizeof(struct rte_mbuf); + /* The size of the entire dp_packet. */ + pkt_size = sizeof(struct dp_packet) + mbuf_size; + /* mbuf size, rounded up to cacheline size. */ + aligned_mbuf_size = ROUND_UP(pkt_size, RTE_CACHE_LINE_SIZE); + /* If there is a size discrepancy, add padding to mbuf_priv_data_len. + * This maintains mbuf size cache alignment, while also honoring RX + * buffer alignment in the data portion of the mbuf. If this adjustment + * is not made, there is a possiblity later on that for an element of + * the mempool, buf, buf->data_len < (buf->buf_len - buf->data_off). + * This is problematic in the case of multi-segment mbufs, particularly + * when an mbuf segment needs to be resized (when [push|popp]ing a VLAN + * header, for example. + */ + mbuf_priv_data_len += (aligned_mbuf_size - pkt_size); + + dmp->mp = rte_pktmbuf_pool_create(mp_name, n_mbufs, MP_CACHE_SZ, + mbuf_priv_data_len, + mbuf_size, + socket_id); + + if (dmp->mp) { + VLOG_DBG("Allocated \"%s\" mempool with %u mbufs", + mp_name, n_mbufs); + /* rte_pktmbuf_pool_create has done some initialization of the + * rte_mbuf part of each dp_packet, while ovs_rte_pktmbuf_init + * initializes some OVS specific fields of dp_packet. + */ + rte_mempool_obj_iter(dmp->mp, ovs_rte_pktmbuf_init, NULL); + return dmp; + } else if (rte_errno == EEXIST) { + /* A mempool with the same name already exists. We just + * retrieve its pointer to be returned to the caller. */ + dmp->mp = rte_mempool_lookup(mp_name); + /* As the mempool create returned EEXIST we can expect the + * lookup has returned a valid pointer. If for some reason + * that's not the case we keep track of it. */ + VLOG_DBG("A mempool with name \"%s\" already exists at %p.", + mp_name, dmp->mp); + return dmp; + } else { + VLOG_DBG("Failed to create mempool \"%s\" with a request of " + "%u mbufs, retrying with %u mbufs", + mp_name, n_mbufs, n_mbufs / 2); + } + } while (!dmp->mp && rte_errno == ENOMEM && (n_mbufs /= 2) >= MIN_NB_MBUF); + + VLOG_ERR("Failed to create mempool \"%s\" with a request of %u mbufs", + mp_name, n_mbufs); + + rte_free(dmp); + return NULL; +} + +struct dpdk_mp * +dpdk_mp_get(struct dpdk_mp_config *cfg) +{ + int mtu = FRAME_LEN_TO_MTU(dpdk_buf_size(cfg->mtu)); + struct dpdk_mp *dmp = NULL, *next; + bool reuse = false; + + ovs_mutex_lock(&dpdk_mp_mutex); + /* Check if shared memory is being used, if so check existing mempools + * to see if reuse is possible. */ + if (!per_port_memory) { + /* If user has provided defined mempools, check if one is suitable + * and get new buffer size.*/ + mtu = dpdk_get_user_adjusted_mtu(mtu, cfg->mtu, + cfg->socket_id); + LIST_FOR_EACH (dmp, list_node, &dpdk_mp_list) { + if (dmp->socket_id == cfg->socket_id + && dmp->mtu == mtu) { + VLOG_DBG("Reusing mempool \"%s\"", dmp->mp->name); + dmp->refcount++; + reuse = true; + break; + } + } + } + /* Sweep mempools after reuse or before create. */ + dpdk_mp_sweep(); + + if (!reuse) { + dmp = dpdk_mp_create(cfg, mtu); + if (dmp) { + /* Shared memory will hit the reuse case above so will not + * request a mempool that already exists but we need to check + * for the EEXIST case for per port memory case. Compare the + * mempool returned by dmp to each entry in dpdk_mp_list. If a + * match is found, free dmp as a new entry is not required, set + * dmp to point to the existing entry and increment the refcount + * to avoid being freed at a later stage. + */ + if (per_port_memory && rte_errno == EEXIST) { + LIST_FOR_EACH (next, list_node, &dpdk_mp_list) { + if (dmp->mp == next->mp) { + rte_free(dmp); + dmp = next; + dmp->refcount++; + } + } + } else { + ovs_list_push_back(&dpdk_mp_list, &dmp->list_node); + } + } + } + + ovs_mutex_unlock(&dpdk_mp_mutex); + + return dmp; +} + +/* Decrement reference to a mempool. */ +void +dpdk_mp_put(struct dpdk_mp *dmp) +{ + if (!dmp) { + return; + } + + ovs_mutex_lock(&dpdk_mp_mutex); + ovs_assert(dmp->refcount); + dmp->refcount--; + ovs_mutex_unlock(&dpdk_mp_mutex); +} + +void +dpdk_mp_dump(FILE *stream, struct dpdk_mp *dmp) +{ + ovs_mutex_lock(&dpdk_mp_mutex); + if (dmp && dmp->mp) { + rte_mempool_dump(stream, dmp->mp); + fprintf(stream, " count: avail (%u), in use (%u)\n", + rte_mempool_avail_count(dmp->mp), + rte_mempool_in_use_count(dmp->mp)); + } else { + rte_mempool_list_dump(stream); + } + ovs_mutex_unlock(&dpdk_mp_mutex); +} + +static void +parse_mempool_config(const struct smap *ovs_other_config) +{ + per_port_memory = smap_get_bool(ovs_other_config, + "per-port-memory", false); + VLOG_INFO("Per port memory for DPDK devices %s.", + per_port_memory ? "enabled" : "disabled"); +} + +static void +parse_user_mempools_list(const struct smap *ovs_other_config) +{ + const char *mtus = smap_get(ovs_other_config, "shared-mempool-config"); + char *list, *copy, *key, *value; + int error = 0; + + if (!mtus) { + return; + } + + n_user_mempools = 0; + list = copy = xstrdup(mtus); + + while (ofputil_parse_key_value(&list, &key, &value)) { + int socket_id, mtu, adj_mtu; + + if (!str_to_int(key, 0, &mtu) || mtu < 0) { + error = EINVAL; + VLOG_WARN("Invalid user configured shared mempool MTU."); + break; + } + + if (!str_to_int(value, 0, &socket_id)) { + /* No socket specified. It will apply for all numas. */ + socket_id = INT_MAX; + } else if (socket_id < 0) { + error = EINVAL; + VLOG_WARN("Invalid user configured shared mempool NUMA."); + break; + } + + user_mempools = xrealloc(user_mempools, (n_user_mempools + 1) * + sizeof(struct user_mempool_config)); + adj_mtu = FRAME_LEN_TO_MTU(dpdk_buf_size(mtu)); + user_mempools[n_user_mempools].adj_mtu = adj_mtu; + user_mempools[n_user_mempools].socket_id = socket_id; + n_user_mempools++; + VLOG_INFO("User configured shared mempool set for: MTU %d, NUMA %s.", + mtu, socket_id == INT_MAX ? "ALL" : value); + } + + if (error) { + VLOG_WARN("User configured shared mempools will not be used."); + n_user_mempools = 0; + free(user_mempools); + user_mempools = NULL; + } + free(copy); +} + +void +dpdk_mp_init(const struct smap *ovs_other_config) +{ + parse_mempool_config(ovs_other_config); + parse_user_mempools_list(ovs_other_config); +} + void netdev_dpdk_common_update_ol_flags(struct netdev_dpdk_common *common) { diff --git a/lib/netdev-dpdk-common.h b/lib/netdev-dpdk-common.h index 3e25558eb5..5cc3ca8894 100644 --- a/lib/netdev-dpdk-common.h +++ b/lib/netdev-dpdk-common.h @@ -21,23 +21,58 @@ #include <stdbool.h> #include <stdint.h> +#include <stdio.h> #include <rte_ether.h> #include <rte_ethdev.h> +#include <rte_mempool.h> #include "netdev-provider.h" #include "openvswitch/compiler.h" #include "ovs-thread.h" +#include "packets.h" struct dpdk_tx_queue; struct ingress_policer; struct qos_conf; +struct smap; +extern struct ovs_mutex dpdk_mutex; +extern struct ovs_mutex dpdk_mp_mutex OVS_ACQ_AFTER(dpdk_mutex); + +/* + * need to reserve tons of extra space in the mbufs so we can align the + * DMA addresses to 4KB. + * The minimum mbuf size is limited to avoid scatter behaviour and drop in + * performance for standard Ethernet MTU. + */ #define ETHER_HDR_MAX_LEN (RTE_ETHER_HDR_LEN + RTE_ETHER_CRC_LEN \ + (2 * VLAN_HEADER_LEN)) +#define MTU_TO_FRAME_LEN(mtu) ((mtu) + RTE_ETHER_HDR_LEN + \ + RTE_ETHER_CRC_LEN) #define MTU_TO_MAX_FRAME_LEN(mtu) ((mtu) + ETHER_HDR_MAX_LEN) +#define FRAME_LEN_TO_MTU(frame_len) ((frame_len) \ + - RTE_ETHER_HDR_LEN - RTE_ETHER_CRC_LEN) +#define NETDEV_DPDK_MBUF_ALIGN 1024 #define NETDEV_DPDK_MAX_PKT_LEN 9728 +/* Max and min number of packets in the mempool. OVS tries to allocate a + * mempool with MAX_NB_MBUF: if this fails (because the system doesn't have + * enough hugepages) we keep halving the number until the allocation succeeds + * or we reach MIN_NB_MBUF */ + +#define MAX_NB_MBUF (4096 * 64) +#define MIN_NB_MBUF (4096 * 4) +#define MP_CACHE_SZ RTE_MEMPOOL_CACHE_MAX_SIZE + +struct dpdk_mp { + struct rte_mempool *mp; + int mtu; + int socket_id; + int refcount; + struct ovs_list list_node OVS_GUARDED_BY(dpdk_mp_mutex); +}; + /* Custom software stats for dpdk ports */ struct netdev_dpdk_sw_stats { /* No. of retries when unable to transmit. */ @@ -225,4 +260,29 @@ void netdev_dpdk_common_get_sw_custom_stats(struct netdev_dpdk_common *common, struct netdev_custom_stats *stats); +/* Allocates an area of 'sz' bytes from DPDK. The memory is zero'ed. + * + * Unlike xmalloc(), this function can return NULL on failure. */ +static inline void * +dpdk_zmalloc(size_t sz) +{ + return rte_zmalloc("ovs_dpdk", sz, CACHE_LINE_SIZE); +} + +struct dpdk_mp_config { + char *name; + int mtu; + int socket_id; + int n_rxq; + int rxq_size; + int n_txq; + int txq_size; +}; + +void dpdk_mp_init(const struct smap *ovs_other_config); +bool dpdk_mp_per_port_memory(void); +struct dpdk_mp * dpdk_mp_get(struct dpdk_mp_config *cfg); +void dpdk_mp_put(struct dpdk_mp *dmp); +void dpdk_mp_dump(FILE *stream, struct dpdk_mp *dmp); + #endif /* NETDEV_DPDK_COMMON_H */ diff --git a/lib/netdev-dpdk.c b/lib/netdev-dpdk.c index d6ba8f534c..56adb72eb2 100644 --- a/lib/netdev-dpdk.c +++ b/lib/netdev-dpdk.c @@ -88,43 +88,9 @@ static char *vhost_sock_dir = NULL; /* Location of vhost-user sockets */ static bool vhost_iommu_enabled = false; /* Status of vHost IOMMU support */ static bool vhost_postcopy_enabled = false; /* Status of vHost POSTCOPY * support. */ -static bool per_port_memory = false; /* Status of per port memory support */ #define DPDK_PORT_WATCHDOG_INTERVAL 5 -#define OVS_CACHE_LINE_SIZE CACHE_LINE_SIZE -#define OVS_VPORT_DPDK "ovs_dpdk" - -/* - * need to reserve tons of extra space in the mbufs so we can align the - * DMA addresses to 4KB. - * The minimum mbuf size is limited to avoid scatter behaviour and drop in - * performance for standard Ethernet MTU. - */ -#define MTU_TO_FRAME_LEN(mtu) ((mtu) + RTE_ETHER_HDR_LEN + \ - RTE_ETHER_CRC_LEN) -#define FRAME_LEN_TO_MTU(frame_len) ((frame_len) \ - - RTE_ETHER_HDR_LEN - RTE_ETHER_CRC_LEN) -#define NETDEV_DPDK_MBUF_ALIGN 1024 - -/* Max and min number of packets in the mempool. OVS tries to allocate a - * mempool with MAX_NB_MBUF: if this fails (because the system doesn't have - * enough hugepages) we keep halving the number until the allocation succeeds - * or we reach MIN_NB_MBUF */ - -#define MAX_NB_MBUF (4096 * 64) -#define MIN_NB_MBUF (4096 * 4) -#define MP_CACHE_SZ RTE_MEMPOOL_CACHE_MAX_SIZE - -/* MAX_NB_MBUF can be divided by 2 many times, until MIN_NB_MBUF */ -BUILD_ASSERT_DECL(MAX_NB_MBUF % ROUND_DOWN_POW2(MAX_NB_MBUF / MIN_NB_MBUF) - == 0); - -/* The smallest possible NB_MBUF that we're going to try should be a multiple - * of MP_CACHE_SZ. This is advised by DPDK documentation. */ -BUILD_ASSERT_DECL((MAX_NB_MBUF / ROUND_DOWN_POW2(MAX_NB_MBUF / MIN_NB_MBUF)) - % MP_CACHE_SZ == 0); - #define SOCKET0 0 /* Default size of Physical NIC RXQ */ @@ -328,35 +294,12 @@ static const struct dpdk_qos_ops *const qos_confs[] = { NULL }; -static struct ovs_mutex dpdk_mutex = OVS_MUTEX_INITIALIZER; +struct ovs_mutex dpdk_mutex = OVS_MUTEX_INITIALIZER; /* Contains all 'struct dpdk_dev's. */ static struct ovs_list dpdk_list OVS_GUARDED_BY(dpdk_mutex) = OVS_LIST_INITIALIZER(&dpdk_list); -static struct ovs_mutex dpdk_mp_mutex OVS_ACQ_AFTER(dpdk_mutex) - = OVS_MUTEX_INITIALIZER; - -/* Contains all 'struct dpdk_mp's. */ -static struct ovs_list dpdk_mp_list OVS_GUARDED_BY(dpdk_mp_mutex) - = OVS_LIST_INITIALIZER(&dpdk_mp_list); - -struct dpdk_mp { - struct rte_mempool *mp; - int mtu; - int socket_id; - int refcount; - struct ovs_list list_node OVS_GUARDED_BY(dpdk_mp_mutex); -}; - -struct user_mempool_config { - int adj_mtu; - int socket_id; -}; - -static struct user_mempool_config *user_mempools = NULL; -static int n_user_mempools; - /* There should be one 'struct dpdk_tx_queue' created for * each netdev tx queue. */ struct dpdk_tx_queue { @@ -588,69 +531,6 @@ is_dpdk_class(const struct netdev_class *class) return is_eth_class(class) || is_vhost_class(class); } -/* DPDK NIC drivers allocate RX buffers at a particular granularity, typically - * aligned at 1k or less. If a declared mbuf size is not a multiple of this - * value, insufficient buffers are allocated to accomodate the packet in its - * entirety. Furthermore, certain drivers need to ensure that there is also - * sufficient space in the Rx buffer to accommodate two VLAN tags (for QinQ - * frames). If the RX buffer is too small, then the driver enables scatter RX - * behaviour, which reduces performance. To prevent this, use a buffer size - * that is closest to 'mtu', but which satisfies the aforementioned criteria. - */ -static uint32_t -dpdk_buf_size(int mtu) -{ - return ROUND_UP(MTU_TO_MAX_FRAME_LEN(mtu), NETDEV_DPDK_MBUF_ALIGN) - + RTE_PKTMBUF_HEADROOM; -} - -static int -dpdk_get_user_adjusted_mtu(int port_adj_mtu, int port_mtu, int port_socket_id) -{ - int best_adj_user_mtu = INT_MAX; - - for (unsigned i = 0; i < n_user_mempools; i++) { - int user_adj_mtu, user_socket_id; - - user_adj_mtu = user_mempools[i].adj_mtu; - user_socket_id = user_mempools[i].socket_id; - if (port_adj_mtu > user_adj_mtu - || (user_socket_id != INT_MAX - && user_socket_id != port_socket_id)) { - continue; - } - if (user_adj_mtu < best_adj_user_mtu) { - /* This is the is the lowest valid user MTU. */ - best_adj_user_mtu = user_adj_mtu; - if (best_adj_user_mtu == port_adj_mtu) { - /* Found an exact fit, no need to keep searching. */ - break; - } - } - } - if (best_adj_user_mtu == INT_MAX) { - VLOG_DBG("No user configured shared mempool mbuf sizes found " - "suitable for port with MTU %d, NUMA %d.", port_mtu, - port_socket_id); - best_adj_user_mtu = port_adj_mtu; - } else { - VLOG_DBG("Found user configured shared mempool with mbufs " - "of size %d, suitable for port with MTU %d, NUMA %d.", - MTU_TO_FRAME_LEN(best_adj_user_mtu), port_mtu, - port_socket_id); - } - return best_adj_user_mtu; -} - -/* Allocates an area of 'sz' bytes from DPDK. The memory is zero'ed. - * - * Unlike xmalloc(), this function can return NULL on failure. */ -static void * -dpdk_rte_mzalloc(size_t sz) -{ - return rte_zmalloc(OVS_VPORT_DPDK, sz, OVS_CACHE_LINE_SIZE); -} - void free_dpdk_buf(struct dp_packet *p) { @@ -659,284 +539,6 @@ free_dpdk_buf(struct dp_packet *p) rte_pktmbuf_free(pkt); } -static void -ovs_rte_pktmbuf_init(struct rte_mempool *mp OVS_UNUSED, - void *opaque_arg OVS_UNUSED, - void *_p, - unsigned i OVS_UNUSED) -{ - struct rte_mbuf *pkt = _p; - - dp_packet_init_dpdk((struct dp_packet *) pkt); -} - -static int -dpdk_mp_full(const struct rte_mempool *mp) OVS_REQUIRES(dpdk_mp_mutex) -{ - /* At this point we want to know if all the mbufs are back - * in the mempool. rte_mempool_full() is not atomic but it's - * the best available and as we are no longer requesting mbufs - * from the mempool, it means mbufs will not move from - * 'mempool ring' --> 'mempool cache'. In rte_mempool_full() - * the ring is counted before caches, so we won't get false - * positives in this use case and we handle false negatives. - * - * If future implementations of rte_mempool_full() were to change - * it could be possible for a false positive. Even that would - * likely be ok, as there are additional checks during mempool - * freeing but it would make things racey. - */ - return rte_mempool_full(mp); -} - -/* Free unused mempools. */ -static void -dpdk_mp_sweep(void) OVS_REQUIRES(dpdk_mp_mutex) -{ - struct dpdk_mp *dmp; - - LIST_FOR_EACH_SAFE (dmp, list_node, &dpdk_mp_list) { - if (!dmp->refcount && dpdk_mp_full(dmp->mp)) { - VLOG_DBG("Freeing mempool \"%s\"", dmp->mp->name); - ovs_list_remove(&dmp->list_node); - rte_mempool_free(dmp->mp); - rte_free(dmp); - } - } -} - -struct dpdk_mp_config { - char *name; - int mtu; - int socket_id; - int n_rxq; - int rxq_size; - int n_txq; - int txq_size; -}; - -/* Calculating the required number of mbufs differs depending on the - * mempool model being used. Check if per port memory is in use before - * calculating. - */ -static uint32_t -dpdk_calculate_mbufs(struct dpdk_mp_config *cfg, int mtu) -{ - uint32_t n_mbufs; - - if (!per_port_memory) { - /* Shared memory are being used. - * XXX: this is a really rough method of provisioning memory. - * It's impossible to determine what the exact memory requirements are - * when the number of ports and rxqs that utilize a particular mempool - * can change dynamically at runtime. For now, use this rough - * heurisitic. - */ - if (mtu >= RTE_ETHER_MTU) { - n_mbufs = MAX_NB_MBUF; - } else { - n_mbufs = MIN_NB_MBUF; - } - } else { - /* Per port memory is being used. - * XXX: rough estimation of number of mbufs required for this port: - * <packets required to fill the device rxqs> - * + <packets that could be stuck on other ports txqs> - * + <packets in the pmd threads> - * + <additional memory for corner cases> - */ - n_mbufs = cfg->n_rxq * cfg->rxq_size - + cfg->n_txq * cfg->txq_size - + MIN(RTE_MAX_LCORE, cfg->n_rxq) * NETDEV_MAX_BURST - + MIN_NB_MBUF; - } - - return n_mbufs; -} - -static struct dpdk_mp * -dpdk_mp_create(struct dpdk_mp_config *cfg, int mtu) -{ - char mp_name[RTE_MEMPOOL_NAMESIZE]; - const char *netdev_name = cfg->name; - int socket_id = cfg->socket_id; - uint32_t n_mbufs = 0; - uint32_t mbuf_size = 0; - uint32_t aligned_mbuf_size = 0; - uint32_t mbuf_priv_data_len = 0; - uint32_t pkt_size = 0; - uint32_t hash = hash_string(netdev_name, 0); - struct dpdk_mp *dmp = NULL; - int ret; - - dmp = dpdk_rte_mzalloc(sizeof *dmp); - if (!dmp) { - return NULL; - } - dmp->socket_id = socket_id; - dmp->mtu = mtu; - dmp->refcount = 1; - - /* Get the size of each mbuf, based on the MTU */ - mbuf_size = MTU_TO_FRAME_LEN(mtu); - - n_mbufs = dpdk_calculate_mbufs(cfg, mtu); - - do { - /* Full DPDK memory pool name must be unique and cannot be - * longer than RTE_MEMPOOL_NAMESIZE. Note that for the shared - * mempool case this can result in one device using a mempool - * which references a different device in it's name. However as - * mempool names are hashed, the device name will not be readable - * so this is not an issue for tasks such as debugging. - */ - ret = snprintf(mp_name, RTE_MEMPOOL_NAMESIZE, - "ovs%08x%02d%05d%07u", - hash, socket_id, mtu, n_mbufs); - if (ret < 0 || ret >= RTE_MEMPOOL_NAMESIZE) { - VLOG_DBG("snprintf returned %d. " - "Failed to generate a mempool name for \"%s\". " - "Hash:0x%x, socket_id: %d, mtu:%d, mbufs:%u.", - ret, netdev_name, hash, socket_id, mtu, n_mbufs); - break; - } - - VLOG_DBG("Port %s: Requesting a mempool of %u mbufs of size %u " - "on socket %d for %d Rx and %d Tx queues, " - "cache line size of %u", - netdev_name, n_mbufs, mbuf_size, socket_id, - cfg->n_rxq, cfg->n_txq, - RTE_CACHE_LINE_SIZE); - - /* The size of the mbuf's private area (i.e. area that holds OvS' - * dp_packet data)*/ - mbuf_priv_data_len = sizeof(struct dp_packet) - - sizeof(struct rte_mbuf); - /* The size of the entire dp_packet. */ - pkt_size = sizeof(struct dp_packet) + mbuf_size; - /* mbuf size, rounded up to cacheline size. */ - aligned_mbuf_size = ROUND_UP(pkt_size, RTE_CACHE_LINE_SIZE); - /* If there is a size discrepancy, add padding to mbuf_priv_data_len. - * This maintains mbuf size cache alignment, while also honoring RX - * buffer alignment in the data portion of the mbuf. If this adjustment - * is not made, there is a possiblity later on that for an element of - * the mempool, buf, buf->data_len < (buf->buf_len - buf->data_off). - * This is problematic in the case of multi-segment mbufs, particularly - * when an mbuf segment needs to be resized (when [push|popp]ing a VLAN - * header, for example. - */ - mbuf_priv_data_len += (aligned_mbuf_size - pkt_size); - - dmp->mp = rte_pktmbuf_pool_create(mp_name, n_mbufs, MP_CACHE_SZ, - mbuf_priv_data_len, - mbuf_size, - socket_id); - - if (dmp->mp) { - VLOG_DBG("Allocated \"%s\" mempool with %u mbufs", - mp_name, n_mbufs); - /* rte_pktmbuf_pool_create has done some initialization of the - * rte_mbuf part of each dp_packet, while ovs_rte_pktmbuf_init - * initializes some OVS specific fields of dp_packet. - */ - rte_mempool_obj_iter(dmp->mp, ovs_rte_pktmbuf_init, NULL); - return dmp; - } else if (rte_errno == EEXIST) { - /* A mempool with the same name already exists. We just - * retrieve its pointer to be returned to the caller. */ - dmp->mp = rte_mempool_lookup(mp_name); - /* As the mempool create returned EEXIST we can expect the - * lookup has returned a valid pointer. If for some reason - * that's not the case we keep track of it. */ - VLOG_DBG("A mempool with name \"%s\" already exists at %p.", - mp_name, dmp->mp); - return dmp; - } else { - VLOG_DBG("Failed to create mempool \"%s\" with a request of " - "%u mbufs, retrying with %u mbufs", - mp_name, n_mbufs, n_mbufs / 2); - } - } while (!dmp->mp && rte_errno == ENOMEM && (n_mbufs /= 2) >= MIN_NB_MBUF); - - VLOG_ERR("Failed to create mempool \"%s\" with a request of %u mbufs", - mp_name, n_mbufs); - - rte_free(dmp); - return NULL; -} - -static struct dpdk_mp * -dpdk_mp_get(struct dpdk_mp_config *cfg) -{ - int mtu = FRAME_LEN_TO_MTU(dpdk_buf_size(cfg->mtu)); - struct dpdk_mp *dmp = NULL, *next; - bool reuse = false; - - ovs_mutex_lock(&dpdk_mp_mutex); - /* Check if shared memory is being used, if so check existing mempools - * to see if reuse is possible. */ - if (!per_port_memory) { - /* If user has provided defined mempools, check if one is suitable - * and get new buffer size.*/ - mtu = dpdk_get_user_adjusted_mtu(mtu, cfg->mtu, - cfg->socket_id); - LIST_FOR_EACH (dmp, list_node, &dpdk_mp_list) { - if (dmp->socket_id == cfg->socket_id - && dmp->mtu == mtu) { - VLOG_DBG("Reusing mempool \"%s\"", dmp->mp->name); - dmp->refcount++; - reuse = true; - break; - } - } - } - /* Sweep mempools after reuse or before create. */ - dpdk_mp_sweep(); - - if (!reuse) { - dmp = dpdk_mp_create(cfg, mtu); - if (dmp) { - /* Shared memory will hit the reuse case above so will not - * request a mempool that already exists but we need to check - * for the EEXIST case for per port memory case. Compare the - * mempool returned by dmp to each entry in dpdk_mp_list. If a - * match is found, free dmp as a new entry is not required, set - * dmp to point to the existing entry and increment the refcount - * to avoid being freed at a later stage. - */ - if (per_port_memory && rte_errno == EEXIST) { - LIST_FOR_EACH (next, list_node, &dpdk_mp_list) { - if (dmp->mp == next->mp) { - rte_free(dmp); - dmp = next; - dmp->refcount++; - } - } - } else { - ovs_list_push_back(&dpdk_mp_list, &dmp->list_node); - } - } - } - - ovs_mutex_unlock(&dpdk_mp_mutex); - - return dmp; -} - -/* Decrement reference to a mempool. */ -static void -dpdk_mp_put(struct dpdk_mp *dmp) -{ - if (!dmp) { - return; - } - - ovs_mutex_lock(&dpdk_mp_mutex); - ovs_assert(dmp->refcount); - dmp->refcount--; - ovs_mutex_unlock(&dpdk_mp_mutex); -} - /* Depending on the memory model being used this function tries to * identify and reuse an existing mempool or tries to allocate a new * mempool on requested_socket_id with mbuf size corresponding to the @@ -952,7 +554,7 @@ netdev_dpdk_mempool_configure(struct netdev_dpdk_common *common, /* With shared memory we do not need to configure a mempool if the MTU * and socket ID have not changed, the previous configuration is still * valid so return 0 */ - if (!per_port_memory && common->mtu == cfg->mtu + if (!dpdk_mp_per_port_memory() && common->mtu == cfg->mtu && common->socket_id == cfg->socket_id) { return ret; } @@ -1414,7 +1016,7 @@ netdev_dpdk_eth_alloc(void) { struct netdev_dpdk *dev; - dev = dpdk_rte_mzalloc(sizeof *dev); + dev = dpdk_zmalloc(sizeof *dev); if (dev) { return &dev->up; } @@ -1428,7 +1030,7 @@ netdev_dpdk_alloc_txq(unsigned int n_txqs) struct dpdk_tx_queue *txqs; unsigned i; - txqs = dpdk_rte_mzalloc(n_txqs * sizeof *txqs); + txqs = dpdk_zmalloc(n_txqs * sizeof *txqs); if (txqs) { for (i = 0; i < n_txqs; i++) { rte_spinlock_init(&txqs[i].tx_lock); @@ -1443,7 +1045,7 @@ netdev_dpdk_vhost_alloc(void) { struct netdev_dpdk *dev; - dev = dpdk_rte_mzalloc(sizeof *dev); + dev = dpdk_zmalloc(sizeof *dev); if (dev) { return &dev->up; } @@ -1500,8 +1102,8 @@ vhost_common_construct(struct netdev *netdev) int socket_id = rte_lcore_to_socket_id(rte_get_main_lcore()); struct netdev_dpdk *dev = netdev_dpdk_cast(netdev); - dev->vhost_rxq_enabled = dpdk_rte_mzalloc(OVS_VHOST_MAX_QUEUE_NUM * - sizeof *dev->vhost_rxq_enabled); + dev->vhost_rxq_enabled = dpdk_zmalloc(OVS_VHOST_MAX_QUEUE_NUM * + sizeof *dev->vhost_rxq_enabled); if (!dev->vhost_rxq_enabled) { return ENOMEM; } @@ -2508,7 +2110,7 @@ out: static struct netdev_rxq * netdev_dpdk_eth_rxq_alloc(void) { - struct netdev_rxq_dpdk *rx = dpdk_rte_mzalloc(sizeof *rx); + struct netdev_rxq_dpdk *rx = dpdk_zmalloc(sizeof *rx); if (rx) { return &rx->up; @@ -2552,7 +2154,7 @@ netdev_dpdk_eth_rxq_dealloc(struct netdev_rxq *rxq) static struct netdev_rxq * netdev_dpdk_vhost_rxq_alloc(void) { - return dpdk_rte_mzalloc(sizeof(struct netdev_rxq)); + return dpdk_zmalloc(sizeof(struct netdev_rxq)); } static int @@ -4821,23 +4423,14 @@ netdev_dpdk_get_mempool_info(struct unixctl_conn *conn, struct netdev_dpdk *dev = netdev_dpdk_cast(netdev); ovs_mutex_lock(&dev->mutex); - ovs_mutex_lock(&dpdk_mp_mutex); - if (common->dpdk_mp) { - rte_mempool_dump(stream, common->dpdk_mp->mp); - fprintf(stream, " count: avail (%u), in use (%u)\n", - rte_mempool_avail_count(common->dpdk_mp->mp), - rte_mempool_in_use_count(common->dpdk_mp->mp)); + dpdk_mp_dump(stream, common->dpdk_mp); } else { error = "Not allocated"; } - - ovs_mutex_unlock(&dpdk_mp_mutex); ovs_mutex_unlock(&dev->mutex); } else { - ovs_mutex_lock(&dpdk_mp_mutex); - rte_mempool_list_dump(stream); - ovs_mutex_unlock(&dpdk_mp_mutex); + dpdk_mp_dump(stream, NULL); } fclose(stream); @@ -6734,66 +6327,6 @@ netdev_dpdk_rte_flow_query_count(struct netdev *netdev, return ret; } -static void -parse_mempool_config(const struct smap *ovs_other_config) -{ - per_port_memory = smap_get_bool(ovs_other_config, - "per-port-memory", false); - VLOG_INFO("Per port memory for DPDK devices %s.", - per_port_memory ? "enabled" : "disabled"); -} - -static void -parse_user_mempools_list(const struct smap *ovs_other_config) -{ - const char *mtus = smap_get(ovs_other_config, "shared-mempool-config"); - char *list, *copy, *key, *value; - int error = 0; - - if (!mtus) { - return; - } - - n_user_mempools = 0; - list = copy = xstrdup(mtus); - - while (ofputil_parse_key_value(&list, &key, &value)) { - int socket_id, mtu, adj_mtu; - - if (!str_to_int(key, 0, &mtu) || mtu < 0) { - error = EINVAL; - VLOG_WARN("Invalid user configured shared mempool MTU."); - break; - } - - if (!str_to_int(value, 0, &socket_id)) { - /* No socket specified. It will apply for all numas. */ - socket_id = INT_MAX; - } else if (socket_id < 0) { - error = EINVAL; - VLOG_WARN("Invalid user configured shared mempool NUMA."); - break; - } - - user_mempools = xrealloc(user_mempools, (n_user_mempools + 1) * - sizeof(struct user_mempool_config)); - adj_mtu = FRAME_LEN_TO_MTU(dpdk_buf_size(mtu)); - user_mempools[n_user_mempools].adj_mtu = adj_mtu; - user_mempools[n_user_mempools].socket_id = socket_id; - n_user_mempools++; - VLOG_INFO("User configured shared mempool set for: MTU %d, NUMA %s.", - mtu, socket_id == INT_MAX ? "ALL" : value); - } - - if (error) { - VLOG_WARN("User configured shared mempools will not be used."); - n_user_mempools = 0; - free(user_mempools); - user_mempools = NULL; - } - free(copy); -} - static int process_vhost_flags(char *flag, const char *default_val, int size, const struct smap *ovs_other_config, @@ -6996,8 +6529,7 @@ static const struct netdev_class netdev_dpdk_vhost_client_class = { void netdev_dpdk_register(const struct smap *ovs_other_config) { - parse_mempool_config(ovs_other_config); - parse_user_mempools_list(ovs_other_config); + dpdk_mp_init(ovs_other_config); parse_vhost_config(ovs_other_config); netdev_register_provider(&netdev_dpdk_class); diff --git a/tests/system-dpdk.at b/tests/system-dpdk.at index 34df1f6762..1d4bfdec86 100644 --- a/tests/system-dpdk.at +++ b/tests/system-dpdk.at @@ -907,6 +907,7 @@ CHECK_MEMPOOL_PARAM([6000], [ALL], []) CHECK_MEMPOOL_PARAM([1500], [ALL], []) AT_CHECK(ovs-appctl vlog/set netdev_dpdk:dbg) +AT_CHECK(ovs-appctl vlog/set netdev_dpdk_common:dbg) dnl Add userspace bridge and a dpdk port AT_CHECK([ovs-vsctl add-br br10 -- set bridge br10 datapath_type=netdev]) -- 2.54.0 _______________________________________________ dev mailing list [email protected] https://mail.openvswitch.org/mailman/listinfo/ovs-dev
