By default only the chassis currently active for a distributed gateway
port advertises the routes that track it, so a failover lasts as long
as the newly active chassis needs to originate the routes and the
fabric to reconverge on them.

Add "options:dynamic-routing-standby-advertise" on Logical_Router_Port.
When set on a distributed gateway port, every member of its HA chassis
group advertises the port's routes.  northd propagates the option onto
the chassisredirect port binding, which ovn-controller consults to tell
whether it is an active or a standby advertiser.

Standby members use a priority PRIORITY_DEFAULT above the base, the
smallest band keeping every standby priority above every active one.
The routing daemon turns that into a higher metric, and hence a higher
BGP MED, so the fabric holds a precomputed backup path towards each
standby.

Reported-at: https://redhat.atlassian.net/browse/FDP-3745
Signed-off-by: Mairtin O'Loingsigh <[email protected]>
Co-Authored-By: Claude Opus 5 <[email protected]>
---
 NEWS                          |   7 +
 controller/ovn-controller.c   |   1 +
 controller/route.c            | 143 +++++++++++++++-
 controller/route.h            |   1 +
 northd/northd.c               |  16 ++
 ovn-nb.xml                    |  32 ++++
 tests/multinode-bgp-macros.at | 136 +++++++++++++++
 tests/multinode.at            | 312 ++++++++++++++++++++++------------
 tests/ovn-northd.at           |  47 +++++
 9 files changed, 582 insertions(+), 113 deletions(-)

diff --git a/NEWS b/NEWS
index f1c56dd14..987d21fe0 100644
--- a/NEWS
+++ b/NEWS
@@ -16,6 +16,13 @@ Post v26.09.0
        now written to the SB MAC_Binding table and consumed at the
        same priority as dynamic entries, making the preference option
        obsolete.
+     * Added a new "options:dynamic-routing-standby-advertise" key on
+       Logical_Router_Port.  When set on a distributed gateway port, every
+       member of its HA chassis group installs and advertises the port's
+       routes instead of only the active one; standby members use a higher
+       priority band (and hence a higher metric/BGP MED) so the fabric
+       prefers the active chassis while holding a precomputed backup path
+       (BGP PIC Edge).
 
 OVN v26.09.0 - xxx xx xxxx
 --------------------------
diff --git a/controller/ovn-controller.c b/controller/ovn-controller.c
index c601f89dc..49c1f2420 100644
--- a/controller/ovn-controller.c
+++ b/controller/ovn-controller.c
@@ -5330,6 +5330,7 @@ en_route_run(struct engine_node *node, void *data)
         .dynamic_routing_port_mapping = dynamic_routing_port_mapping,
         .local_datapaths = &rt_data->local_datapaths,
         .local_bindings = &rt_data->lbinding_data.bindings,
+        .active_tunnels = &rt_data->active_tunnels,
     };
 
     struct route_ctx_out r_ctx_out = {
diff --git a/controller/route.c b/controller/route.c
index c7df5b6ce..adc99978c 100644
--- a/controller/route.c
+++ b/controller/route.c
@@ -42,6 +42,11 @@ VLOG_DEFINE_THIS_MODULE(exchange);
 #define PRIORITY_DEFAULT 1000
 #define PRIORITY_LOCAL_BOUND 100
 
+/* Name of the Logical_Router_Port option, propagated by northd onto the
+ * chassisredirect port binding, that opts a distributed gateway port into
+ * advertising its routes from every member of its HA chassis group. */
+#define OPT_STANDBY_ADVERTISE "dynamic-routing-standby-advertise"
+
 /* Discover the veth peer interface name of 'iface' using the
  * status:peer_ifindex value that OVS populates for veth devices.
  *
@@ -111,14 +116,97 @@ route_is_distributed_lb(const struct 
sbrec_advertised_route *route)
                          OVN_AR_DISTRIBUTED_LB_ID, false);
 }
 
+/* Returns true if 'cr_pb' is a chassisredirect port belonging to an HA
+ * chassis group that has been opted into standby route advertisement.
+ *
+ * This is off by default: enabling it makes every member of the group
+ * install and advertise the port's routes, which is only desirable when the
+ * fabric is expected to pre-compute backup paths (BGP PIC Edge). */
+static bool
+route_cr_port_standby_enabled(const struct sbrec_port_binding *cr_pb)
+{
+    return cr_pb && cr_pb->ha_chassis_group &&
+           smap_get_bool(&cr_pb->options, OPT_STANDBY_ADVERTISE, false);
+}
+
+/* Resolves the chassisredirect port that governs active/standby ownership of
+ * 'route', or NULL if 'route' is not subject to HA chassis group arbitration
+ * on 'chassis'.
+ *
+ * Both candidate ports are examined because northd populates them
+ * differently depending on what generated the route: for NAT redistribution
+ * tracked_port is the NAT's distributed gateway port while logical_port is
+ * the advertising LRP (see build_nat_route_for_port() in
+ * northd/en-advertised-route-sync.c), and either may be, or resolve to, the
+ * chassisredirect port.
+ *
+ * Returns NULL when 'chassis' is not a member of the group, since a
+ * non-member is neither active nor standby for the port. */
+static const struct sbrec_port_binding *
+route_get_ha_cr_port(struct ovsdb_idl_index *sbrec_port_binding_by_name,
+                     const struct sbrec_advertised_route *route,
+                     const struct sbrec_chassis *chassis)
+{
+    const struct sbrec_port_binding *candidates[] = {
+        route->tracked_port,
+        route->logical_port,
+    };
+
+    for (size_t i = 0; i < ARRAY_SIZE(candidates); i++) {
+        const struct sbrec_port_binding *pb = candidates[i];
+        if (!pb) {
+            continue;
+        }
+
+        const struct sbrec_port_binding *cr_pb =
+            !strcmp(pb->type, "chassisredirect")
+                ? pb
+                : lport_get_cr_port(sbrec_port_binding_by_name, pb, NULL);
+
+        if (route_cr_port_standby_enabled(cr_pb) &&
+            ha_chassis_group_contains(cr_pb->ha_chassis_group, chassis)) {
+            return cr_pb;
+        }
+    }
+
+    return NULL;
+}
+
+/* Returns true if 'chassis' is a standby (i.e. non-active) member of the HA
+ * chassis group owning 'cr_pb', which must be a port returned by
+ * route_get_ha_cr_port(). */
+static bool
+route_chassis_is_standby(const struct sbrec_port_binding *cr_pb,
+                         const struct sbrec_chassis *chassis,
+                         const struct sset *active_tunnels)
+{
+    return !ha_chassis_group_is_active(cr_pb->ha_chassis_group, active_tunnels,
+                                       chassis);
+}
+
+/* Returns true if this chassis advertises 'route'.
+ *
+ * On success '*ha_cr_port', if non-NULL, receives the chassisredirect port
+ * arbitrating active/standby for 'route', or NULL if there is none.  Callers
+ * that need it get it from here rather than resolving it again: the lookup is
+ * needed either way, because a standby member advertises the port's routes
+ * even when the advertising port itself is resident elsewhere. */
 static bool
 route_advertising_port_is_local(
     const struct sbrec_advertised_route *route,
     struct ovsdb_idl_index *sbrec_port_binding_by_name,
-    const struct sbrec_chassis *chassis)
+    const struct sbrec_chassis *chassis,
+    const struct sbrec_port_binding **ha_cr_port)
 {
-    return lport_is_local(sbrec_port_binding_by_name, chassis,
-                          route->logical_port->logical_port);
+    const struct sbrec_port_binding *cr_pb =
+        route_get_ha_cr_port(sbrec_port_binding_by_name, route, chassis);
+
+    if (ha_cr_port) {
+        *ha_cr_port = cr_pb;
+    }
+
+    return cr_pb || lport_is_local(sbrec_port_binding_by_name, chassis,
+                                   route->logical_port->logical_port);
 }
 
 static bool
@@ -162,8 +250,8 @@ build_lb_route_gates(struct hmap *gates,
         if (!route->tracked_port || !route_has_health_checks(route)) {
             continue;
         }
-        if (!route_advertising_port_is_local(
-                route, sbrec_port_binding_by_name, chassis)) {
+        if (!route_advertising_port_is_local(route, sbrec_port_binding_by_name,
+                                             chassis, NULL)) {
             continue;
         }
 
@@ -412,6 +500,16 @@ route_exchange_find_port(struct ovsdb_idl_index 
*sbrec_port_binding_by_name,
             smap_get(&cr_pb->options, "dynamic-routing-port-name");
     }
 
+    /* When the port is opted into standby advertisement, let every member of
+     * the HA chassis group process its routes rather than only the resident
+     * one.  The active/standby distinction is then expressed as a route
+     * priority in route_run(), so the standby's routes are advertised with a
+     * higher metric and the fabric can pre-compute a backup path. */
+    if (route_cr_port_standby_enabled(cr_pb) &&
+        ha_chassis_group_contains(cr_pb->ha_chassis_group, chassis)) {
+        return route_exchange_relevant_port(cr_pb) ? cr_pb : NULL;
+    }
+
     if (!lport_pb_is_chassis_resident(chassis, cr_pb)) {
         return NULL;
     }
@@ -757,9 +855,14 @@ route_run(struct route_ctx_in *r_ctx_in,
             continue;
         }
 
+        /* 'cr_pb' is the chassisredirect port that arbitrates active/standby
+         * for this route, or NULL if the route is unrelated to an HA chassis
+         * group or the group did not opt in, in which case only the resident
+         * chassis reaches this point anyway. */
+        const struct sbrec_port_binding *cr_pb;
         if (!route_advertising_port_is_local(
-                route, r_ctx_in->sbrec_port_binding_by_name,
-                r_ctx_in->chassis)) {
+                route, r_ctx_in->sbrec_port_binding_by_name, r_ctx_in->chassis,
+                &cr_pb)) {
             sset_add(r_ctx_out->tracked_ports_remote,
                      route->logical_port->logical_port);
             continue;
@@ -770,6 +873,7 @@ route_run(struct route_ctx_in *r_ctx_in,
         bool distributed_lb = route_is_distributed_lb(route);
 
         unsigned int priority = PRIORITY_DEFAULT;
+
         if (route->tracked_port) {
             bool tracked_port_local;
             bool local_only_eligible = route_is_local_only_eligible(
@@ -793,6 +897,31 @@ route_run(struct route_ctx_in *r_ctx_in,
             }
         }
 
+        /* Single site where the standby band is applied, once the base
+         * priority is final. Routes installed by a standby member of the HA
+         * chassis group land in a strictly higher priority band, which the
+         * routing daemon translates into a higher metric (and hence a higher
+         * BGP MED), so the fabric pre-computes the backup path without ever
+         * preferring it over the active chassis.
+         *
+         * PRIORITY_DEFAULT is the smallest band that preserves the invariant
+         * "every standby priority exceeds every active priority": the lowest
+         * standby priority is PRIORITY_LOCAL_BOUND + PRIORITY_DEFAULT, which
+         * is still above the highest active one, PRIORITY_DEFAULT. */
+        if (cr_pb && route_chassis_is_standby(cr_pb, r_ctx_in->chassis,
+                                              r_ctx_in->active_tunnels)) {
+            unsigned int base = priority;
+
+            priority += PRIORITY_DEFAULT;
+
+            static struct vlog_rate_limit rl = VLOG_RATE_LIMIT_INIT(5, 20);
+            VLOG_DBG_RL(&rl,
+                        "Advertising route %s from standby chassis %s "
+                        "with priority %u (base priority %u)",
+                        route->ip_prefix, r_ctx_in->chassis->name, priority,
+                        base);
+        }
+
         if (distributed_lb &&
             !smap_get_bool(&route->external_ids, "enabled", true)) {
             continue;
diff --git a/controller/route.h b/controller/route.h
index b99b62d5d..b0f91e466 100644
--- a/controller/route.h
+++ b/controller/route.h
@@ -44,6 +44,7 @@ struct route_ctx_in {
     const char *dynamic_routing_port_mapping;
     const struct hmap *local_datapaths;
     struct shash *local_bindings;
+    const struct sset *active_tunnels;
 };
 
 struct route_ctx_out {
diff --git a/northd/northd.c b/northd/northd.c
index 4eb2ea44b..66e1e1bbe 100644
--- a/northd/northd.c
+++ b/northd/northd.c
@@ -4231,6 +4231,22 @@ sync_pb_for_lrp(struct ovn_port *op,
         if (always_redirect) {
             smap_add(&new, "always-redirect", "true");
         }
+
+        /* Opt the distributed gateway port into being advertised from every
+         * member of its HA chassis group, not just the active one.  Only the
+         * chassisredirect port binding carries this: it is the one
+         * ovn-controller consults to decide whether it is an active or a
+         * standby advertiser.
+         *
+         * This is deliberately not conditional on this router having dynamic
+         * routing enabled.  The redirect port is the *tracked* port of the
+         * routes in question (e.g. the gateway port of a NAT); the router
+         * that advertises them, and therefore the one that has dynamic
+         * routing enabled, is generally a different, adjacent one. */
+        if (smap_get_bool(&op->nbrp->options,
+                          "dynamic-routing-standby-advertise", false)) {
+            smap_add(&new, "dynamic-routing-standby-advertise", "true");
+        }
     } else {
         if (op->peer) {
             smap_add(&new, "peer", op->peer->key);
diff --git a/ovn-nb.xml b/ovn-nb.xml
index 078bd6068..01a339d75 100644
--- a/ovn-nb.xml
+++ b/ovn-nb.xml
@@ -4929,6 +4929,38 @@ or
         </p>
       </column>
 
+      <column name="options" key="dynamic-routing-standby-advertise"
+         type='{"type": "boolean"}'>
+        <p>
+          Only relevant on a distributed gateway port whose HA chassis group
+          has more than one member.
+        </p>
+
+        <p>
+          By default only the chassis that is currently active for this
+          distributed gateway port installs and advertises the port's routes.
+          A failover therefore requires the newly active chassis to originate
+          the routes and the fabric to converge on them, which takes time
+          proportional to the reconvergence of the routing protocol.
+        </p>
+
+        <p>
+          When set to <code>true</code> every member of the HA chassis group
+          installs and advertises the routes. The standby members do so in a
+          strictly higher priority band, which the routing daemon translates
+          into a higher metric and hence, for BGP, a higher MED. The fabric
+          consequently prefers the active chassis while already holding a
+          precomputed backup path towards each standby, which allows for a
+          fast local repair (BGP PIC Edge) when the active chassis fails.
+        </p>
+
+        <p>
+          This is off by default because it increases the number of paths the
+          fabric has to hold for each advertised prefix by the size of the HA
+          chassis group.
+        </p>
+      </column>
+
       <column name="options" key="dynamic-routing-port-name"
           type='{"type": "string"}'>
         Only relevant if <ref column="options" key="dynamic-routing"
diff --git a/tests/multinode-bgp-macros.at b/tests/multinode-bgp-macros.at
index dba811f50..ac40ab215 100644
--- a/tests/multinode-bgp-macros.at
+++ b/tests/multinode-bgp-macros.at
@@ -620,4 +620,140 @@ m_config_host_frr_router_l3() {
     m_setup_host_frr_vrf $node $vni $vxlan_ip $bgp_mac $bgp_ip
 }
 
+# m_setup_bgp_unnumbered_gws
+#
+# Sets up the pair of BGP unnumbered gateways used by the multinode BGP
+# tests: on ovn-gw-1 and ovn-gw-2 an external FRR router (the simulated
+# fabric peer) and an OVN FRR router, peering over an unnumbered (IPv6
+# link-local) session in ovnvrf10 and ovnvrf20 respectively.
+#
+# Callers should wait for the sessions to come up themselves, e.g.
+#   OVS_WAIT_UNTIL([m_as ovn-gw-1 vtysh -c 'show bgp vrf ovnvrf10 neighbors' \
+#                   | grep -qE 'Connections established 1'])
+# so that a failure is reported against the test's own line number.
+m_setup_bgp_unnumbered_gws() {
+    m_setup_external_frr_router  ovn-gw-1                        41.41.41.41/32
+    m_config_external_frr_router ovn-gw-1 4200000100 41.41.41.41 
41.41.41.41/32 41::41/64 12:fb:d6:66:99:0c
+    m_setup_ovn_frr_router       ovn-gw-1                                      
           12:fb:d6:66:99:1c 10
+    m_config_ovn_frr_router      ovn-gw-1 4210000000 14.14.14.14               
                             10
+
+    m_setup_external_frr_router  ovn-gw-2                        42.42.42.42/32
+    m_config_external_frr_router ovn-gw-2 4200000200 42.42.42.42 
42.42.42.42/32 42::42/64 22:fb:d6:66:99:0c
+    m_setup_ovn_frr_router       ovn-gw-2                                      
           22:fb:d6:66:99:2c 20
+    m_config_ovn_frr_router      ovn-gw-2 4210000000 24.24.24.24               
                             20
+}
+
+# m_add_guest_vm_and_connections NODE VRF_ID DEFAULT_ROUTE_MAC DEFAULT_ROUTE
+#                                DEFAULT_ROUTE_GW GUEST_GW_IP GUEST_IP
+#
+# Connects the OVN FRR router of NODE to the join switch, hangs a guest
+# logical switch off the shared guest router and creates a fake VM on it.
+# Adds the default route on the gateway router towards the external FRR
+# speaker (learned from the unnumbered session in ovnvrf-VRF_ID) and the
+# route back out of the guest router via the DGP.
+#
+# The gateway router port is tagged "dynamic-routing-redistribute=nat", so
+# NATs on the guest router are advertised from every node.
+#
+# Expects these shell variables to be set by m_setup_bgp_guest_topology:
+#   join_ls, lr_guest, lrp_guest_join, guest_vm_ns
+m_add_guest_vm_and_connections() {
+    local node=$1 vrf_id=$2 default_route_mac=$3 default_route=$4
+    local default_route_gw=$5 guest_gw_ip=$6 guest_ip=$7
+
+    local gw_router=$(m_ovn_frr_router_name $node)
+    local gw_router_lrp=$(m_ovn_frr_router_port_name $node)
+
+    local gw_lr=$(m_ovn_frr_router_name $node)
+    local lrp_to_join=lrp-$node-to-join
+    local lsp_join_to_lrp=join-to-lrp-$node
+
+    local ls_g=ls-guest-$node
+    local lsp_g_lrg=lsp-guest-$node-lr-guest
+    local lsp_g_iface=lsp-guest-$node-guest-vm
+    local lrp_g_lsg=lrp-guest-ls-guest-$node
+
+    local guest_gw_cidr="$guest_gw_ip/24"
+    local guest_cidr="$guest_ip/24"
+
+    # Set up connections to connect the new vm.
+    check multinode_nbctl lrp-add $gw_lr $lrp_to_join $default_route_mac
+    check multinode_nbctl lrp-set-options $lrp_to_join \
+        dynamic-routing-redistribute=nat
+    check multinode_nbctl lsp-add-router-port $join_ls $lsp_join_to_lrp 
$lrp_to_join
+
+    check multinode_nbctl ls-add $ls_g
+    check multinode_nbctl lrp-add $lr_guest $lrp_g_lsg \
+        00:16:03:01:03:03 $guest_gw_cidr
+    check multinode_nbctl lsp-add-router-port $ls_g $lsp_g_lrg $lrp_g_lsg
+    check multinode_nbctl lsp-add $ls_g $lsp_g_iface
+    check multinode_nbctl lsp-set-addresses $lsp_g_iface \
+        '00:16:01:00:02:02 '$guest_cidr''
+
+    # Create the new vm.
+    m_as $node /data/create_fake_vm.sh $lsp_g_iface $guest_vm_ns \
+        00:16:01:00:02:02 1342 $guest_ip 24 $guest_gw_ip 1000::13/64 1000::a
+    local neighbor_lla=$(m_as $node vtysh -c "show bgp vrf ovnvrf${vrf_id} 
neighbor ext0-bgp" | grep "^Foreign host:" | awk '{print $3}' | tr -d ',')
+
+    check multinode_nbctl lr-route-add $gw_router "0.0.0.0/0" \
+        $neighbor_lla $gw_router_lrp
+    check multinode_nbctl lr-route-add $lr_guest \
+        $default_route $default_route_gw $lrp_guest_join
+}
+
+# m_setup_bgp_guest_topology GW1_PRIO GW2_PRIO
+#
+# Builds the guest topology shared by the multinode BGP unnumbered tests on
+# top of the gateways created by m_setup_bgp_unnumbered_gws:
+#
+#                guest-1          guest-2
+#                       \        /
+#                        lr-guest
+#                          DGP (priority: gw-1=GW1_PRIO, gw-2=GW2_PRIO)
+#                           |
+#                        ls-join
+#                       /       \
+# tor <-> lr-ovn-gw-1-ext0*    lr-ovn-gw-2-ext0 <-> tor
+#               |                     |
+#         ls-ovn-gw-1-ext0     ls-ovn-gw-2-ext0
+#
+# The DGP is hosted by an HA chassis group holding both gateways.  Equal
+# priorities leave the choice of active chassis to OVN; unequal ones pin it,
+# which is what the standby-advertisement test needs.
+#
+# A dnat_and_snat for 172.16.10.2 on lr-guest is redistributed onto both
+# gateway routers, so both nodes advertise it into the fabric.
+#
+# Exports the topology names for the caller: join_ls, lsp_join_guest,
+# lr_guest, lrp_guest_join, guest_vm_iface, guest_vm_ns.
+m_setup_bgp_guest_topology() {
+    local gw1_prio=$1 gw2_prio=$2
+
+    join_ls="ls-join"
+    lsp_join_guest="lsp-join-guest"
+
+    lr_guest="lr-guest"
+    lrp_guest_join="lrp-guest-join-dgp"
+
+    guest_vm_iface="guest-vm"
+    guest_vm_ns="ns-guest"
+
+    check multinode_nbctl ls-add $join_ls
+
+    check multinode_nbctl lr-add $lr_guest
+    check multinode_nbctl lrp-add $lr_guest $lrp_guest_join 00:16:06:12:f0:0d
+    check multinode_nbctl lsp-add-router-port $join_ls $lsp_join_guest 
$lrp_guest_join
+    check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-1 
$gw1_prio
+    check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-2 
$gw2_prio
+
+    m_add_guest_vm_and_connections ovn-gw-1 10 00:00:ff:00:00:01 41.0.0.0/8 \
+        fe80::200:ffff:fe00:1 192.168.10.1 192.168.10.10
+    m_add_guest_vm_and_connections ovn-gw-2 20 00:00:ff:00:00:02 42.0.0.0/8 \
+        fe80::200:ffff:fe00:2 192.168.20.1 192.168.20.10
+
+    # NAT that gets advertised via BGP from both gateways.
+    check multinode_nbctl --gateway-port $lrp_guest_join --add-route \
+        lr-nat-add $lr_guest dnat_and_snat 172.16.10.2 192.168.10.10
+}
+
 OVS_END_SHELL_HELPERS
diff --git a/tests/multinode.at b/tests/multinode.at
index 0c277f5e8..02e4352ac 100644
--- a/tests/multinode.at
+++ b/tests/multinode.at
@@ -2906,132 +2906,232 @@ cleanup_multinode_resources
 
 CHECK_VRF()
 
-add_guest_vm_and_connections() {
-    node=$1
-    vrf_id=$2
-    default_route_mac=$3
-    default_route=$4
-    default_route_gw=$5
-    guest_gw_ip=$6
-    guest_ip=$7
-    gw_router=$(m_ovn_frr_router_name $node)
-    gw_router_lrp=$(m_ovn_frr_router_port_name $node)
-
-    gw_lr=$(m_ovn_frr_router_name $node)
-    lrp_to_join=lrp-$node-to-join
-    lsp_join_to_lrp=join-to-lrp-$node
-    lrp_guest=lrp-guest-$node
-
-    ls_g=ls-guest-$node
-    lsp_g_lrg=lsp-guest-$node-lr-guest
-    lsp_g_iface=lsp-guest-$node-guest-vm
-    lrp_g_lsg=lrp-guest-ls-guest-$node
-
-    guest_gw_cidr="$guest_gw_ip/24"
-    guest_cidr="$guest_ip/24"
-
-    # set up connections to connect the new vm
-    check multinode_nbctl lrp-add $gw_lr $lrp_to_join $default_route_mac
-    check multinode_nbctl lrp-set-options $lrp_to_join \
-        dynamic-routing-redistribute=nat
-    check multinode_nbctl lsp-add-router-port $join_ls $lsp_join_to_lrp 
$lrp_to_join
-
-    check multinode_nbctl ls-add $ls_g
-    check multinode_nbctl lrp-add $lr_guest $lrp_g_lsg \
-        00:16:03:01:03:03 $guest_gw_cidr
-    check multinode_nbctl lsp-add-router-port $ls_g $lsp_g_lrg $lrp_g_lsg
-    check multinode_nbctl lsp-add $ls_g $lsp_g_iface
-    check multinode_nbctl lsp-set-addresses $lsp_g_iface \
-        '00:16:01:00:02:02 '$guest_cidr''
-
-    # create the new vm
-    m_as $node /data/create_fake_vm.sh $lsp_g_iface $guest_vm_ns \
-        00:16:01:00:02:02 1342 $guest_ip 24 $guest_gw_ip 1000::13/64 1000::a
-    neighbor_lla=$(m_as $node vtysh -c "show bgp vrf ovnvrf${vrf_id} neighbor 
ext0-bgp" | grep "^Foreign host:" | awk '{print $3}' | tr -d ',')
-
-    check multinode_nbctl lr-route-add $gw_router "0.0.0.0/0" \
-        $neighbor_lla $gw_router_lrp
-    check multinode_nbctl lr-route-add $lr_guest \
-        $default_route $default_route_gw $lrp_guest_join
-}
-
-m_setup_external_frr_router  ovn-gw-1                        41.41.41.41/32
-m_config_external_frr_router ovn-gw-1 4200000100 41.41.41.41 41.41.41.41/32 
41::41/64 12:fb:d6:66:99:0c
-m_setup_ovn_frr_router       ovn-gw-1                                          
       12:fb:d6:66:99:1c 10
-m_config_ovn_frr_router      ovn-gw-1 4210000000 14.14.14.14                   
                         10
-
-m_setup_external_frr_router  ovn-gw-2                        42.42.42.42/32
-m_config_external_frr_router ovn-gw-2 4200000200 42.42.42.42 42.42.42.42/32 
42::42/64 22:fb:d6:66:99:0c
-m_setup_ovn_frr_router       ovn-gw-2                                          
       22:fb:d6:66:99:2c 20
-m_config_ovn_frr_router      ovn-gw-2 4210000000 24.24.24.24                   
                         20
+m_setup_bgp_unnumbered_gws
 
 OVS_WAIT_UNTIL([m_as ovn-gw-2 vtysh -c 'show bgp vrf ovnvrf20 neighbors' | 
grep -qE 'Connections established 1'])
 OVS_WAIT_UNTIL([m_as ovn-gw-1 vtysh -c 'show bgp vrf ovnvrf10 neighbors' | 
grep -qE 'Connections established 1'])
 
-# Tor <-> ovn-gw via bgp
-# lr-guest with distributed gateway port
-# bgp on lr-ovn-gw-2-ext0
+# Tor <-> ovn-gw via bgp, lr-guest with a distributed gateway port.  See
+# m_setup_bgp_guest_topology for the topology diagram.  Both gateways get
+# the same HA chassis group priority here.
+m_setup_bgp_guest_topology 20 20
+
+OVS_WAIT_UNTIL([m_central_as ovn-sbctl list Advertised_Route | grep -q 
172.16.10.2])
+OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ip route | grep -q 'ext1'])
+OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
+OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ip route | grep -q 'ext1'])
+OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
+
+AT_CLEANUP
+
+AT_SETUP([ovn multinode bgp ha-chassis standby routes])
+
+# Check that ovn-fake-multinode setup is up and running.
+check_fake_multinode_setup
+
+CHECK_VRF()
+
+# Delete the multinode NB and OVS resources before starting the test.
+cleanup_multinode_resources
+
+CHECK_VRF()
+
+# Verifies "options:dynamic-routing-standby-advertise" on a distributed
+# gateway port: every member of the DGP's HA chassis group advertises the
+# port's routes, and the standby members do so in a strictly higher priority
+# band so the fabric holds a precomputed backup path (BGP PIC Edge).
 #
+# Topology:
 #                guest-1          guest-2
 #                       \        /
 #                        lr-guest
-#                          DGP
+#                          DGP (priority: gw-1=30, gw-2=10)
 #                           |
 #                        ls-join
 #                       /       \
-# tor <-> lr-ovn-gw-2-ext0*    lr-ovn-gw-1-ext0* <-> tor
-#               |                     |
-#         ls-ovn-gw-2-ext0     ls-ovn-gw-1-ext0
+# tor <-> lr-ovn-gw-1-ext0*    lr-ovn-gw-2-ext0 <-> tor
+#     (AS 4200000100)   |        | (AS 4200000200)
+#    (ACTIVE pri=30)    |        | (STANDBY pri=10)
+#                 ls-ovn-gw-1  ls-ovn-gw-2
+#
+# The NAT on lr-guest is redistributed onto both lrp-ovn-gw-N-to-join ports,
+# so both chassis already see the Advertised_Route; its tracked_port is the
+# DGP, which is what ties the route to the HA chassis group.
 #
+# Expected kernel route metrics for the NAT IP:
 #
+#                             active chassis   standby chassis
+#   standby-advertise unset        100              1000
+#   standby-advertise set          100              2000
 #
 
-join_ls="ls-join"
-lsp_join_guest="lsp-join-guest"
-
-lr_guest="lr-guest"
-lrp_guest_join="lrp-guest-join-dgp"
-
-guest_vm_iface="guest-vm"
-guest_vm_ns="ns-guest"
-
-check multinode_nbctl ls-add $join_ls
-
-check multinode_nbctl lr-add $lr_guest
-check multinode_nbctl lrp-add $lr_guest $lrp_guest_join 00:16:06:12:f0:0d
-check multinode_nbctl lsp-add-router-port $join_ls $lsp_join_guest 
$lrp_guest_join
-check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-1 20
-check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-2 20
-
-vrf_id=10
-default_route_mac=00:00:ff:00:00:01
-default_route=41.0.0.0/8
-default_route_gw=fe80::200:ffff:fe00:1
-guest_gw_ip=192.168.10.1
-guest_ip=192.168.10.10
-add_guest_vm_and_connections ovn-gw-1 $vrf_id           \
-    $default_route_mac $default_route $default_route_gw \
-    $guest_gw_ip $guest_ip
-
-vrf_id=20
-default_route_mac=00:00:ff:00:00:02
-default_route=42.0.0.0/8
-default_route_gw=fe80::200:ffff:fe00:2
-guest_gw_ip=192.168.20.1
-guest_ip=192.168.20.10
-add_guest_vm_and_connections ovn-gw-2 $vrf_id           \
-    $default_route_mac $default_route $default_route_gw \
-    $guest_gw_ip $guest_ip
-
-check multinode_nbctl --gateway-port $lrp_guest_join --add-route lr-nat-add \
-    $lr_guest dnat_and_snat 172.16.10.2 192.168.10.10
+# Setup external FRR routers and OVN FRR routers with BGP unnumbered.
+m_setup_bgp_unnumbered_gws
+
+echo "Waiting for BGP sessions to establish..."
+OVS_WAIT_UNTIL([m_as ovn-gw-2 vtysh -c 'show bgp vrf ovnvrf20 neighbors' | 
grep -qE 'Connections established 1'])
+echo "BGP session established on gw-2"
+OVS_WAIT_UNTIL([m_as ovn-gw-1 vtysh -c 'show bgp vrf ovnvrf10 neighbors' | 
grep -qE 'Connections established 1'])
+echo "BGP session established on gw-1"
+
+# Create topology with unequal priorities: ovn-gw-1 = 30 (active), ovn-gw-2 = 
10 (standby).
+m_setup_bgp_guest_topology 30 10
 
+# Wait for routes to be advertised.
 OVS_WAIT_UNTIL([m_central_as ovn-sbctl list Advertised_Route | grep -q 
172.16.10.2])
-OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ip route | grep -q 'ext1'])
+
+# Get chassis UUIDs and tunnel interface names for BFD testing
+gw1_chassis=$(m_fetch_column Chassis _uuid name=ovn-gw-1)
+gw2_chassis=$(m_fetch_column Chassis _uuid name=ovn-gw-2)
+
+ip_gw1=$(m_as ovn-gw-1 ip a show dev eth1 | grep "inet " | awk '{print $2}'| 
cut -d '/' -f1)
+ip_gw2=$(m_as ovn-gw-2 ip a show dev eth1 | grep "inet " | awk '{print $2}'| 
cut -d '/' -f1)
+
+tunnel_gw1_to_gw2=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find 
interface \
+    options:remote_ip=$ip_gw2 type=geneve)
+tunnel_gw2_to_gw1=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find 
interface \
+    options:remote_ip=$ip_gw1 type=geneve)
+
+# Wait for BFD to come up between gateways
+OVS_WAIT_UNTIL([
+    state=$(m_as ovn-gw-1 ovs-vsctl get interface $tunnel_gw1_to_gw2 
bfd_status:state 2>/dev/null || echo "down")
+    echo "BFD gw-1 -> gw-2: $state"
+    test "$state" = "up"
+])
+
+OVS_WAIT_UNTIL([
+    state=$(m_as ovn-gw-2 ovs-vsctl get interface $tunnel_gw2_to_gw1 
bfd_status:state 2>/dev/null || echo "down")
+    echo "BFD gw-2 -> gw-1: $state"
+    test "$state" = "up"
+])
+
+# Verify CR port is bound to ovn-gw-1 (highest priority)
+m_wait_row_count Port_Binding 1 logical_port=cr-$lrp_guest_join 
chassis=$gw1_chassis
+
+# route_metric NODE VRF_TABLE PREFIX.
+route_metric() {
+    # The prefix is not always the first field: blackhole routes, which is
+    # what a NAT redistribution installs, render as
+    # "blackhole PREFIX proto ovn metric N".
+    m_as $1 ip route show table $2 | awk -v p="$3" '
+        { match_found = 0
+          for (i = 1; i <= NF; i++) if ($i == p) match_found = 1
+          if (match_found) {
+              for (i = 1; i <= NF; i++) if ($i == "metric") print $(i + 1)
+          } }'
+}
+
+# wait_route_metric NODE VRF_TABLE PREFIX EXPECTED.
+wait_route_metric() {
+    # OVS_WAIT_UNTIL runs its body in a context where the positional
+    # parameters are its own, so copy them out first.
+    local rm_node=$1 rm_table=$2 rm_prefix=$3 rm_expected=$4
+
+    echo "Waiting for $rm_prefix in table $rm_table on $rm_node to have" \
+         "metric $rm_expected..."
+    OVS_WAIT_UNTIL([test "$rm_expected" = \
+                        "$(route_metric $rm_node $rm_table $rm_prefix)"], [
+        echo "Routes in table $rm_table on $rm_node:"
+        m_as $rm_node ip route show table $rm_table])
+}
+
+# bgp_med NODE PREFIX.
+bgp_med() {
+    m_as $1 vtysh $(m_frr_ns_flags frr-ns) \
+        -c "show bgp ipv4 unicast $2" |
+        sed -n 's/.*metric \([[0-9]][[0-9]]*\).*/\1/p' | head -1
+}
+
+AS_BOX([Verifying route priorities with standby advertisement disabled.])
+
+# Both chassis should have routes learned via BGP in frr-ns namespace
+OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ip route | grep -q 
'172.16.10.2'])
+OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ip route | grep -q 
'172.16.10.2'])
+
+# Both chassis can ping the NAT IP
+OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
+OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
+
+# Check that both chassis have advertised the route
+m_wait_row_count Advertised_Route 2 ip_prefix=172.16.10.2
+
+# gw-1 is active and hosts the tracked DGP, so it uses PRIORITY_LOCAL_BOUND.
+# gw-2 is standby and gets PRIORITY_DEFAULT: no band, the feature is off.
+wait_route_metric ovn-gw-1 10 172.16.10.2 100
+wait_route_metric ovn-gw-2 20 172.16.10.2 1000
+
+AS_BOX([Enable standby advertisement and verify the band is applied.])
+
+check multinode_nbctl lrp-set-options $lrp_guest_join \
+    dynamic-routing-standby-advertise=true
+
+# northd must propagate the option onto the chassisredirect port binding;
+# that is what ovn-controller reads.
+m_wait_row_count Port_Binding 1 logical_port=cr-$lrp_guest_join \
+    options:dynamic-routing-standby-advertise=true
+
+# The active chassis is unchanged; the standby moves into the higher band.
+wait_route_metric ovn-gw-1 10 172.16.10.2 100
+wait_route_metric ovn-gw-2 20 172.16.10.2 2000
+
+# Both chassis still advertise exactly one route each for the prefix.
+m_wait_row_count Advertised_Route 2 ip_prefix=172.16.10.2
+
+# The kernel metric must reach the fabric as a BGP MED, otherwise the standby
+# path would be indistinguishable from the active one to the external peer.
+echo "TEST 2: Verifying BGP MED seen by the external speakers"
+OVS_WAIT_UNTIL([test "100" = "$(bgp_med ovn-gw-1 172.16.10.2/32)"], [
+    m_as ovn-gw-1 vtysh $(m_frr_ns_flags frr-ns) \
+        -c "show bgp ipv4 unicast 172.16.10.2/32"])
+OVS_WAIT_UNTIL([test "2000" = "$(bgp_med ovn-gw-2 172.16.10.2/32)"], [
+    m_as ovn-gw-2 vtysh $(m_frr_ns_flags frr-ns) \
+        -c "show bgp ipv4 unicast 172.16.10.2/32"])
+
+AS_BOX([Fail the DGP over to gw-2 and verify the bands swap.])
+check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-2 40
+
+m_wait_row_count Port_Binding 1 logical_port=cr-$lrp_guest_join 
chassis=$gw2_chassis
+
+# Verify connectivity is maintained.
+OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
+
+# Roles are now reversed: gw-2 is active and hosts the tracked DGP, gw-1 is
+# the standby and moves into the band.
+wait_route_metric ovn-gw-2 20 172.16.10.2 100
+wait_route_metric ovn-gw-1 10 172.16.10.2 2000
+
+m_wait_row_count Advertised_Route 2 ip_prefix=172.16.10.2
+
+OVS_WAIT_UNTIL([test "100" = "$(bgp_med ovn-gw-2 172.16.10.2/32)"], [
+    m_as ovn-gw-2 vtysh $(m_frr_ns_flags frr-ns) \
+        -c "show bgp ipv4 unicast 172.16.10.2/32"])
+OVS_WAIT_UNTIL([test "2000" = "$(bgp_med ovn-gw-1 172.16.10.2/32)"], [
+    m_as ovn-gw-1 vtysh $(m_frr_ns_flags frr-ns) \
+        -c "show bgp ipv4 unicast 172.16.10.2/32"])
+
+AS_BOX([Fail back to gw-1.])
+
+check multinode_nbctl lrp-set-gateway-chassis $lrp_guest_join ovn-gw-2 10
+
+m_wait_row_count Port_Binding 1 logical_port=cr-$lrp_guest_join 
chassis=$gw1_chassis
+echo "CR port migrated back to ovn-gw-1"
+
 OVS_WAIT_UNTIL([m_as ovn-gw-1 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
-OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ip route | grep -q 'ext1'])
 OVS_WAIT_UNTIL([m_as ovn-gw-2 ip netns exec frr-ns ping -W 1 -c 1 172.16.10.2])
 
+wait_route_metric ovn-gw-1 10 172.16.10.2 100
+wait_route_metric ovn-gw-2 20 172.16.10.2 2000
+
+m_wait_row_count Advertised_Route 2 ip_prefix=172.16.10.2
+
+AS_BOX([Disabling the option withdraws the band again.])
+
+check multinode_nbctl remove Logical_Router_Port $lrp_guest_join options \
+    dynamic-routing-standby-advertise
+
+wait_route_metric ovn-gw-1 10 172.16.10.2 100
+wait_route_metric ovn-gw-2 20 172.16.10.2 1000
+
 AT_CLEANUP
 
 AT_SETUP([ovn multinode dynamic-routing - BGP learned routes with router 
filter name and multiple DGPs])
diff --git a/tests/ovn-northd.at b/tests/ovn-northd.at
index 6572b1318..3b7a5567a 100644
--- a/tests/ovn-northd.at
+++ b/tests/ovn-northd.at
@@ -17470,6 +17470,53 @@ OVN_CLEANUP_NORTHD
 AT_CLEANUP
 ])
 
+OVN_FOR_EACH_NORTHD_NO_HV([
+AT_SETUP([dynamic-routing - standby advertise options])
+AT_KEYWORDS([dynamic-routing])
+ovn_start
+
+# "dynamic-routing-standby-advertise" is consumed by ovn-controller on the
+# chassisredirect port binding, since that is the port the HA chassis group
+# hangs off.  Check that northd puts it there, and only there.
+#
+# Note that lr0 deliberately does not have dynamic routing enabled: the
+# redirect port is the tracked port of routes advertised by an adjacent
+# router, so the option must be propagated regardless.
+
+check ovn-nbctl lr-add lr0
+check ovn-nbctl ls-add sw0
+check ovn-nbctl lrp-add lr0 lr0-sw0 00:00:00:00:ff:01 10.0.0.1/24
+check ovn-nbctl lsp-add sw0 sw0-lr0
+check ovn-nbctl set Logical_Switch_Port sw0-lr0 \
+    type=router options:router-port=lr0-sw0
+check ovn-nbctl --wait=sb lrp-set-gateway-chassis lr0-sw0 hv1 20
+
+cr_lrp=cr-lr0-sw0
+
+# Off by default.
+AT_CHECK([fetch_column sb:Port_Binding options logical_port=$cr_lrp | \
+    grep -qv 'dynamic-routing-standby-advertise'])
+
+check ovn-nbctl --wait=sb set Logical_Router_Port lr0-sw0 \
+    options:dynamic-routing-standby-advertise=true
+
+AT_CHECK([fetch_column sb:Port_Binding options logical_port=$cr_lrp | \
+    grep -q 'dynamic-routing-standby-advertise=true'])
+# The distributed part of the port must not carry it: only the redirect port
+# arbitrates active vs standby.
+AT_CHECK([fetch_column sb:Port_Binding options logical_port=lr0-sw0 | \
+    grep -qv 'dynamic-routing-standby-advertise'])
+
+check ovn-nbctl --wait=sb remove Logical_Router_Port lr0-sw0 options \
+    dynamic-routing-standby-advertise
+
+AT_CHECK([fetch_column sb:Port_Binding options logical_port=$cr_lrp | \
+    grep -qv 'dynamic-routing-standby-advertise'])
+
+OVN_CLEANUP_NORTHD
+AT_CLEANUP
+])
+
 OVN_FOR_EACH_NORTHD_NO_HV([
 AT_SETUP([dynamic-routing - host routes - unnumbered LRP interfaces])
 AT_KEYWORDS([dynamic-routing])
-- 
2.55.0

_______________________________________________
dev mailing list
[email protected]
https://mail.openvswitch.org/mailman/listinfo/ovs-dev

Reply via email to