With the recent commit on net-next the ct zone filtering is now
discoverable in the kernel:
  46da6029bf46 ("netfilter: conntrack: make filtering by zone discoverable")

This means we no longer need to rely on the kernel version parsing,
which may produce false negative results, especially with distribution
kernels.  And we'll be able to remove it at some point in the future.

Instead, we can probe the feature directly using the CTA_FILTER_ZONE
attribute.  Kernels that support CTA_FILTER, but not the
CTA_FILTER_ZONE will fail the request.  Kernels that do not support
CTA_FILTER will not report NLM_F_DUMP_FILTERED in the reply.
So, a successful GET request with NLM_F_DUMP_FILTERED flag set in the
reply guarantees that kernel understands CTA_FILTER_ZONE and hence
supports filtering by zone in both GET and DELETE requests.

Using the zone 60000 for probing as it is not a default zone and high
enough for OVN to not actually use it, so it is unlikely to contain a
lot of entries for the dump.  In the worst case, if the filtering is
not supported we'll dump the entire table twice on this one request.
Later requests will only dump once as before.

Kernel version parsing is preserved as a fallback for upstream kernels
between 6.8 and 7.2.

The kernel-level filtering is important for OVN deployments, as OVN
allocates separate zones per port and requests to flush them on port
additions and removals.  On systems with large conntrack tables this
may take seconds if filtering is done in user space.

Providing CTA_FILTER_ZONE for the actual deletion requests is not
necessary as the kernel allows requests without it for backwards
compatibility.  But it is cleaner if we do.

Signed-off-by: Ilya Maximets <[email protected]>
---

Unless there will be some objections, I would also propose backporting
this down to 3.7 once the kernel change hits mainline, since 3.7 is our
LTS and people will likely use it with distribution kernels older than
6.8 (all the RHEL 9 derivatives, for exmaple) for a long time.

 lib/netlink-conntrack.c | 74 ++++++++++++++++++++++++++++++++++++-----
 lib/netlink-socket.c    |  7 ++--
 lib/netlink-socket.h    |  2 ++
 3 files changed, 72 insertions(+), 11 deletions(-)

diff --git a/lib/netlink-conntrack.c b/lib/netlink-conntrack.c
index b02000253..2406b1b9a 100644
--- a/lib/netlink-conntrack.c
+++ b/lib/netlink-conntrack.c
@@ -66,6 +66,9 @@ static struct vlog_rate_limit rl = VLOG_RATE_LIMIT_INIT(1, 5);
 #define CTA_MARK_MASK     (CTA_SECMARK + 4)
 #define CTA_LABELS        (CTA_SECMARK + 5)
 #define CTA_LABELS_MASK   (CTA_SECMARK + 6)
+#define CTA_FILTER        (CTA_SECMARK + 8)
+
+#define CTA_FILTER_ZONE   3
 
 #define CTA_TIMESTAMP_START 1
 #define CTA_TIMESTAMP_STOP  2
@@ -266,10 +269,17 @@ out:
     return err;
 }
 
+enum {
+    CT_FLUSH_ZONE_UNSUPPORTED,
+    CT_FLUSH_ZONE_CTA_ZONE,
+    CT_FLUSH_ZONE_CTA_FILTER,
+};
+
 static int
-nl_ct_flush_zone_with_cta_zone(uint16_t flush_zone)
+nl_ct_flush_zone_filtered(uint16_t flush_zone, int mode)
 {
     struct ofpbuf buf;
+    size_t offset;
     int err;
 
     ofpbuf_init(&buf, NL_DUMP_BUFSIZE);
@@ -278,6 +288,12 @@ nl_ct_flush_zone_with_cta_zone(uint16_t flush_zone)
                         IPCTNL_MSG_CT_DELETE, NLM_F_REQUEST);
     nl_msg_put_be16(&buf, CTA_ZONE, htons(flush_zone));
 
+    if (mode == CT_FLUSH_ZONE_CTA_FILTER) {
+        offset = nl_msg_start_nested_with_flag(&buf, CTA_FILTER);
+        nl_msg_put_flag(&buf, CTA_FILTER_ZONE);
+        nl_msg_end_nested(&buf, offset);
+    }
+
     err = nl_transact(NETLINK_NETFILTER, &buf, NULL);
     ofpbuf_uninit(&buf);
 
@@ -285,21 +301,59 @@ nl_ct_flush_zone_with_cta_zone(uint16_t flush_zone)
 }
 
 static bool
-netlink_flush_supports_zone(void)
+nl_ct_probe_cta_filter_zone(void)
+{
+    struct ofpbuf buf, reply;
+    struct nl_dump dump;
+    size_t offset;
+    int err;
+
+    ofpbuf_init(&buf, NL_DUMP_BUFSIZE);
+
+    /* GET and DELETE support CTA_FILTER_ZONE through a common kernel path.
+     * We probe the GET to avoid accidental deletions. */
+    nl_msg_put_nfgenmsg(&buf, 0, AF_UNSPEC, NFNL_SUBSYS_CTNETLINK,
+                        IPCTNL_MSG_CT_GET, NLM_F_REQUEST);
+    /* Using a high enough zone that is unlikely to have a lot of entries. */
+    nl_msg_put_be16(&buf, CTA_ZONE, htons(60000));
+
+    offset = nl_msg_start_nested_with_flag(&buf, CTA_FILTER);
+    nl_msg_put_flag(&buf, CTA_FILTER_ZONE);
+    nl_msg_end_nested(&buf, offset);
+
+    nl_dump_start(&dump, NETLINK_NETFILTER, &buf);
+    ofpbuf_clear(&buf);
+
+    while (nl_dump_next(&dump, &reply, &buf)) {
+        /* Nothing to do. */
+    }
+
+    err = nl_dump_done(&dump);
+    ofpbuf_uninit(&buf);
+
+    return err == 0 && (dump.nl_flags & NLM_F_DUMP_FILTERED);
+}
+
+static int
+nl_ct_flush_zone_mode(void)
 {
     static struct ovsthread_once once = OVSTHREAD_ONCE_INITIALIZER;
-    static bool supported = false;
+    static int mode = CT_FLUSH_ZONE_UNSUPPORTED;
 
     if (ovsthread_once_start(&once)) {
-        if (ovs_kernel_is_version_or_newer(6, 8)) {
-            supported = true;
+        if (nl_ct_probe_cta_filter_zone()) {
+            mode = CT_FLUSH_ZONE_CTA_FILTER;
+            VLOG_DBG("Conntrack flush by zone: using CTA_FILTER_ZONE.");
+        } else if (ovs_kernel_is_version_or_newer(6, 8)) {
+            mode = CT_FLUSH_ZONE_CTA_ZONE;
+            VLOG_DBG("Conntrack flush by zone: using bare CTA_ZONE.");
         } else {
             VLOG_INFO("Disabling conntrack flush by zone. "
                       "Not supported in Linux kernel.");
         }
         ovsthread_once_done(&once);
     }
-    return supported;
+    return mode;
 }
 
 int
@@ -319,11 +373,13 @@ nl_ct_flush_zone(uint16_t flush_zone)
      * Additionally newer kernels also support flushing a zone without listing
      * it first. */
 
-    struct nl_dump dump;
     struct ofpbuf buf, reply, delete;
+    struct nl_dump dump;
+    int mode;
 
-    if (netlink_flush_supports_zone()) {
-        return nl_ct_flush_zone_with_cta_zone(flush_zone);
+    mode = nl_ct_flush_zone_mode();
+    if (mode != CT_FLUSH_ZONE_UNSUPPORTED) {
+        return nl_ct_flush_zone_filtered(flush_zone, mode);
     }
 
     ofpbuf_init(&buf, NL_DUMP_BUFSIZE);
diff --git a/lib/netlink-socket.c b/lib/netlink-socket.c
index 69a3f18b1..d7e9e5cc0 100644
--- a/lib/netlink-socket.c
+++ b/lib/netlink-socket.c
@@ -748,6 +748,7 @@ nl_dump_start(struct nl_dump *dump, int protocol, const 
struct ofpbuf *request)
                                       true);
     }
     dump->nl_seq = nl_msg_nlmsghdr(request)->nlmsg_seq;
+    dump->nl_flags = 0;
     ovs_mutex_unlock(&dump->mutex);
 }
 
@@ -790,13 +791,15 @@ nl_dump_refill(struct nl_dump *dump, struct ofpbuf 
*buffer)
 }
 
 static int
-nl_dump_next__(struct ofpbuf *reply, struct ofpbuf *buffer)
+nl_dump_next__(struct nl_dump *dump, struct ofpbuf *reply,
+               struct ofpbuf *buffer)
 {
     struct nlmsghdr *nlmsghdr = nl_msg_next(buffer, reply);
     if (!nlmsghdr) {
         VLOG_WARN_RL(&rl, "netlink dump contains message fragment");
         return EPROTO;
     } else if (nlmsghdr->nlmsg_type == NLMSG_DONE) {
+        dump->nl_flags = nlmsghdr->nlmsg_flags;
         return EOF;
     } else {
         return 0;
@@ -850,7 +853,7 @@ nl_dump_next(struct nl_dump *dump, struct ofpbuf *reply, 
struct ofpbuf *buffer)
 
     /* Fetch the next message from the buffer. */
     if (!retval) {
-        retval = nl_dump_next__(reply, buffer);
+        retval = nl_dump_next__(dump, reply, buffer);
         if (retval) {
             /* Record 'retval' as the dump status, but don't overwrite an error
              * with EOF.  */
diff --git a/lib/netlink-socket.h b/lib/netlink-socket.h
index 4d087bcae..4e0e66a51 100644
--- a/lib/netlink-socket.h
+++ b/lib/netlink-socket.h
@@ -248,6 +248,8 @@ struct nl_dump {
     int status OVS_GUARDED;     /* 0: dump in progress,
                                  * positive errno: dump completed with error,
                                  * EOF: dump completed successfully. */
+    uint32_t nl_flags;          /* nlmsg_flags from the NLMSG_DONE message.
+                                 * Only valid after nl_dump_done(). */
 
     /* 'mutex' protects 'status' and serializes access to 'sock'. */
     struct ovs_mutex mutex;     /* Protects 'status', synchronizes recv(). */
-- 
2.55.0

_______________________________________________
dev mailing list
[email protected]
https://mail.openvswitch.org/mailman/listinfo/ovs-dev

Reply via email to