On Fri, Aug 7, 2026 at 7:21 PM Xavier Simonart via dev < [email protected]> wrote:
> When an HA gateway chassis reboots, tunnel ports should be automatically > removed. > Without this, OVS might restore stale tunnel ports from the database which > can interfere with HA failover (e.g. sets BFD up while no openflow flows > are installed). > > Mark tunnel ports as transient (other_config:transient=true) when the > chassis is a member of an HA chassis group. > OVS automatically removes transient ports at system boot, when starting > ovsdb-server service. > > Reported-at: https://redhat.atlassian.net/browse/FDP-3866 > > Signed-off-by: Xavier Simonart <[email protected]> > --- > Hi Xavier, thank you for the patch. I have a few comments down below. > NEWS | 6 + > controller/encaps.c | 22 +++- > controller/encaps.h | 3 +- > controller/ovn-controller.c | 33 +++++- > tests/multinode-macros.at | 24 ++++ > tests/multinode.at | 230 +++++++++++++++++++++++++++++++++--- > tests/ovn-controller.at | 57 +++++++++ > 7 files changed, 354 insertions(+), 21 deletions(-) > > diff --git a/NEWS b/NEWS > index 44f117807..3765e870c 100644 > --- a/NEWS > +++ b/NEWS > @@ -94,6 +94,12 @@ Post v26.03.0 > - Added a new "ovn-debug lflow-pipeline-oftable-start-list" command > that > prints the starting OpenFlow table number of the logical ingress and > egress pipelines. > + - Mark tunnel ports as transient (other_config:transient=true) when the > + local chassis is a member of an HA chassis group. > + OVS removes transient ports on ovsdb-server restart, preventing stale > + tunnel ports from interfering with BFD and HA failover after a > gateway > + chassis reboots. > + > > OVN v26.03.0 - xxx xx xxxx > -------------------------- > diff --git a/controller/encaps.c b/controller/encaps.c > index 048e85c38..8bd6b31d1 100644 > --- a/controller/encaps.c > +++ b/controller/encaps.c > @@ -39,6 +39,7 @@ encaps_register_ovs_idl(struct ovsdb_idl *ovs_idl) > ovsdb_idl_track_add_column(ovs_idl, &ovsrec_port_col_name); > ovsdb_idl_track_add_column(ovs_idl, &ovsrec_port_col_interfaces); > ovsdb_idl_track_add_column(ovs_idl, &ovsrec_port_col_external_ids); > + ovsdb_idl_add_column(ovs_idl, &ovsrec_port_col_other_config); > ovsdb_idl_add_table(ovs_idl, &ovsrec_table_interface); > ovsdb_idl_track_add_column(ovs_idl, &ovsrec_interface_col_name); > ovsdb_idl_track_add_column(ovs_idl, &ovsrec_interface_col_type); > @@ -63,6 +64,7 @@ struct tunnel_ctx { > const struct ovsrec_open_vswitch_table *ovs_table; > const struct ovsrec_bridge *br_int; > const struct sbrec_chassis *this_chassis; > + bool is_ha_chassis_member; > }; > > struct tunnel_node { > @@ -209,6 +211,10 @@ tunnel_add(struct tunnel_ctx *tc, > { > struct smap options = SMAP_INITIALIZER(&options); > struct smap other_config = SMAP_INITIALIZER(&other_config); > + struct smap port_other_config = SMAP_INITIALIZER(&port_other_config); > + if (tc->is_ha_chassis_member) { > + smap_add(&port_other_config, "transient", "true"); > + } > smap_add(&options, "remote_ip", encap->ip); > smap_add(&options, "local_ip", local_ip); > smap_add(&options, "key", "flow"); > @@ -292,9 +298,9 @@ tunnel_add(struct tunnel_ctx *tc, > } > > /* If there's an existing tunnel record that does not need any change, > - * keep it. Otherwise, create a new record (if there was an existing > - * record, the new record will supplant it and encaps_run() will > delete > - * it). */ > + * (except maybe the transient flag) keep it. Otherwise, create a new > + * record (if there was an existing record, the new record will > + * supplant it and encaps_run() will delete it). */ > struct tunnel_node *tunnel = shash_find_data(&tc->tunnel, > tunnel_entry_id); > bool old_id_format = false; > @@ -306,6 +312,10 @@ tunnel_add(struct tunnel_ctx *tc, > && tunnel->port->n_interfaces == 1 > && !strcmp(tunnel->port->interfaces[0]->type, encap->type) > && smap_equal(&tunnel->port->interfaces[0]->options, &options)) { > + if (!smap_equal(&tunnel->port->other_config, &port_other_config)) > { > + ovsrec_port_set_other_config(tunnel->port, > &port_other_config); > + } > Could we use setkey/delkey instead e.g.? if (tc->is_ha_chassis_member) { ovsrec_port_update_other_config_setkey(tunnel->port, "transient", "true"); } else if (smap_get(&tunnel->port->other_config, "transient")) { ovsrec_port_update_other_config_delkey(tunnel->port, "transient"); } > + > if (old_id_format) { > /* We must be upgrading from an older version. We can reuse > the > * existing tunnel, but needs to update the tunnel's ID to > the new > @@ -345,6 +355,7 @@ tunnel_add(struct tunnel_ctx *tc, > ovsrec_port_set_interfaces(port, &iface, 1); > const struct smap id = SMAP_CONST1(&id, OVN_TUNNEL_ID, > tunnel_entry_id); > ovsrec_port_set_external_ids(port, &id); > + ovsrec_port_set_other_config(port, &port_other_config); > If we use the setkey/delkey we can use SMAP_CONST1 here. > > ovsrec_bridge_update_ports_addvalue(tc->br_int, port); > > @@ -355,6 +366,7 @@ exit: > free(tunnel_entry_id_old); > smap_destroy(&options); > smap_destroy(&other_config); > + smap_destroy(&port_other_config); > } > > static bool > @@ -734,7 +746,8 @@ encaps_run(struct ovsdb_idl_txn *ovs_idl_txn, > const struct sbrec_sb_global *sbg, > const struct ovsrec_open_vswitch_table *ovs_table, > const struct sset *transport_zones, > - const struct ovsrec_bridge_table *bridge_table) > + const struct ovsrec_bridge_table *bridge_table, > + bool is_ha_chassis_member) > { > if (!ovs_idl_txn || !ovnsb_idl_txn || !br_int) { > return; > @@ -781,6 +794,7 @@ encaps_run(struct ovsdb_idl_txn *ovs_idl_txn, > .br_int = br_int, > .this_chassis = this_chassis, > .ovs_table = ovs_table, > + .is_ha_chassis_member = is_ha_chassis_member, > }; > > tc.ovs_txn = ovs_idl_txn; > diff --git a/controller/encaps.h b/controller/encaps.h > index 0257d08c1..f29a128eb 100644 > --- a/controller/encaps.h > +++ b/controller/encaps.h > @@ -49,7 +49,8 @@ void encaps_run(struct ovsdb_idl_txn *ovs_idl_txn, > const struct sbrec_sb_global *, > const struct ovsrec_open_vswitch_table *, > const struct sset *transport_zones, > - const struct ovsrec_bridge_table *bridge_table); > + const struct ovsrec_bridge_table *bridge_table, > + bool is_ha_chassis_member); > > bool is_flow_based_tunnels_enabled( > const struct ovsrec_open_vswitch_table *ovs_table, > diff --git a/controller/ovn-controller.c b/controller/ovn-controller.c > index ea9952a19..a70eaf73c 100644 > --- a/controller/ovn-controller.c > +++ b/controller/ovn-controller.c > @@ -1043,6 +1043,33 @@ ctrl_register_ovs_idl(struct ovsdb_idl *ovs_idl) > * track that column which should be addressed in the future. */ > } > > +static bool > +chassis_is_ha_member(const struct sbrec_chassis *chassis, > + const struct sbrec_ha_chassis_group_table > *ha_grp_table) > +{ > + static bool is_ha_member = false; > This variable is not needed at all. > + const struct sbrec_ha_chassis_group *grp; > + > + if (!sbrec_ha_chassis_group_table_track_get_first(ha_grp_table)) { > + return is_ha_member; > + } > This if shouldn't be needed, the loop wouldn't run if it's empty. > + > + SBREC_HA_CHASSIS_GROUP_TABLE_FOR_EACH (grp, ha_grp_table) { > + /* If there is only one member, then it is not HA */ > + if (grp->n_ha_chassis < 2) { > + continue; > + } > + for (size_t i = 0; i < grp->n_ha_chassis; i++) { > + if (grp->ha_chassis[i]->chassis == chassis) { > + is_ha_member = true; > + return true; > + } > + } > + } > + is_ha_member = false; > + return false; > +} > + > struct ed_type_ofctrl_is_connected { > bool connected; > }; > @@ -8034,13 +8061,17 @@ main(int argc, char *argv[]) > const struct sbrec_sb_global *sbg = > sbrec_sb_global_first(ovnsb_idl_loop.idl); > if (chassis && sbg && ovs_feature_set_discovered()) { > + bool is_ha_chassis_member = chassis_is_ha_member( > + chassis, > + > sbrec_ha_chassis_group_table_get(ovnsb_idl_loop.idl)); > Do we risk a problem if we would use the en_bfd_chassis node? On the first engine run we wouldn't have this info, but we could flag all tunnels transient temporarily in case of a restart between the engine runs WDYT? > encaps_run(ovs_idl_txn, ovnsb_idl_txn, br_int, > > sbrec_chassis_table_get(ovnsb_idl_loop.idl), > chassis, > sbg, > ovs_table, > &transport_zones, > - bridge_table); > + bridge_table, > + is_ha_chassis_member); > > ovn_netlink_notifiers_run(); > > diff --git a/tests/multinode-macros.at b/tests/multinode-macros.at > index ade4d167c..88677e364 100644 > --- a/tests/multinode-macros.at > +++ b/tests/multinode-macros.at > @@ -510,6 +510,30 @@ m_is_fedora() { > m_central_as grep -qi fedora /etc/os-release > } > > +# Run ovs-vsctl using Host socket > +host_ovs_vsctl() { > + # Discover host OVS socket on first call > + if [[ -z "$HOST_OVS_SOCK" ]]; then > + for sock in /run/openvswitch/db.sock /var/run/openvswitch/db.sock > /usr/local/var/run/openvswitch/db.sock; do > + if [[ -S "$sock" ]]; then > + HOST_OVS_SOCK=$sock > + break > + fi > + done > + # Fallback on unusual prefix: discover from running process > + if [[ -z "$HOST_OVS_SOCK" ]]; then > + HOST_OVS_SOCK=$(ps aux | grep '[o]vsdb-server' | grep -oP > 'punix:\K[^, ]+' | while read s; do > + [[ -S "$s" ]] && [[ "$s" != *"$OVS_RUNDIR"* ]] && echo > "$s" && break > + done) > + fi > + if [[ -z "$HOST_OVS_SOCK" ]]; then > + echo "ERROR: Could not find host OVS socket" >&2 > + AT_FAIL_IF([:]) > + fi > + fi > + ovs-vsctl --db=unix:$HOST_OVS_SOCK "$@" > +} > + > # M_START_L4_SERVER([fake_node], [namespace], [ip_addr], [port], > [reply_string], [pidfile]) > # > # Helper to properly start l4 server in inside 'fake_node''s namespace'. > diff --git a/tests/multinode.at b/tests/multinode.at > index c69ab3cb4..1ea2ec08e 100644 > --- a/tests/multinode.at > +++ b/tests/multinode.at > @@ -3093,6 +3093,10 @@ m_as ovn-gw-1 ovn-appctl vlog/disable-rate-limit > m_as ovn-gw-2 ovn-appctl vlog/disable-rate-limit > m_as ovn-gw-3 ovn-appctl vlog/disable-rate-limit > > +# Decrease revalidation time on ovs switch simulating ToR. > +check host_ovs_vsctl set Open_vSwitch . other_config:max-revalidator=100 > +on_exit "check host_ovs_vsctl remove Open_vSwitch . other_config > max-revalidator" > + > check_fake_multinode_setup > > # Delete the multinode NB and OVS resources before starting the test. > @@ -3104,18 +3108,38 @@ ip_gw1=$(m_as ovn-gw-1 ip a show dev eth1 | grep > "inet " | awk '{print $2}'| cut > ip_gw2=$(m_as ovn-gw-2 ip a show dev eth1 | grep "inet " | awk '{print > $2}'| cut -d '/' -f1) > ip_gw3=$(m_as ovn-gw-3 ip a show dev eth1 | grep "inet " | awk '{print > $2}'| cut -d '/' -f1) > > -from_gw1_to_gw2=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw2) > -from_gw1_to_gw3=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw3) > -from_gw1_to_ch1=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch1) > -from_gw1_to_ch2=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch2) > -from_gw2_to_gw1=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw1) > -from_gw2_to_gw3=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw3) > -from_gw2_to_ch1=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch1) > -from_gw2_to_ch2=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch2) > -from_ch1_to_gw1=$(m_as ovn-chassis-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw1) > -from_ch1_to_gw2=$(m_as ovn-chassis-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw2) > -from_ch2_to_gw1=$(m_as ovn-chassis-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw1) > -from_ch2_to_gw2=$(m_as ovn-chassis-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw2) > +get_geneve_names_gw1() > +{ > + from_gw1_to_gw2=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw2) > + from_gw1_to_gw3=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw3) > + from_gw1_to_ch1=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch1) > + from_gw1_to_ch2=$(m_as ovn-gw-1 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch2) > +} > + > +get_geneve_names_gw2() > +{ > + from_gw2_to_gw1=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw1) > + from_gw2_to_gw3=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_gw3) > + from_gw2_to_ch1=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch1) > + from_gw2_to_ch2=$(m_as ovn-gw-2 ovs-vsctl --bare --columns=name find > interface options:remote_ip=$ip_ch2) > +} > + > +get_geneve_names_chassis1() > +{ > + from_ch1_to_gw1=$(m_as ovn-chassis-1 ovs-vsctl --bare --columns=name > find interface options:remote_ip=$ip_gw1) > + from_ch1_to_gw2=$(m_as ovn-chassis-1 ovs-vsctl --bare --columns=name > find interface options:remote_ip=$ip_gw2) > +} > + > +get_geneve_names_chassis2() > +{ > + from_ch2_to_gw1=$(m_as ovn-chassis-2 ovs-vsctl --bare --columns=name > find interface options:remote_ip=$ip_gw1) > + from_ch2_to_gw2=$(m_as ovn-chassis-2 ovs-vsctl --bare --columns=name > find interface options:remote_ip=$ip_gw2) > +} > + > +get_geneve_names_gw1 > +get_geneve_names_gw2 > +get_geneve_names_chassis1 > +get_geneve_names_chassis2 > > m_as ovn-chassis-1 ip link del hv1-vif1-p > m_as ovn-chassis-2 ip link del hv2-vif1-p > @@ -3128,6 +3152,10 @@ OVS_WAIT_UNTIL([m_as ovn-gw-1 ip link show | grep > -q genev_sys]) > OVS_WAIT_UNTIL([m_as ovn-gw-2 ip link show | grep -q genev_sys]) > OVS_WAIT_UNTIL([m_as ovn-gw-3 ip link show | grep -q genev_sys]) > > +# Use "aggressive" bfd parameters > +check multinode_nbctl set NB_Global . options:"bfd-min-rx"=500 > +check multinode_nbctl set NB_Global . options:"bfd-min-tx"=100 > + > check multinode_nbctl ls-add inside > check multinode_nbctl ls-add outside > check multinode_nbctl ls-add ext > @@ -3186,6 +3214,8 @@ gw1_chassis=$(m_fetch_column Chassis _uuid > name=ovn-gw-1) > gw2_chassis=$(m_fetch_column Chassis _uuid name=ovn-gw-2) > gw3_chassis=$(m_fetch_column Chassis _uuid name=ovn-gw-3) > > +gw1_eth1_ip_mask=$(m_as ovn-gw-1 ip -4 -o addr show eth1 | awk '{print > $4}') > + > wait_bfd_enabled() { > chassis=$1 > interface=$2 > @@ -3263,7 +3293,11 @@ start_tcpdump() { > [ovn-gw-2], [-neei eth2], [gw2], > [ovn-gw-2], [-neei eth2 -Q out], [gw2_out], > [ovn-gw-3], [-neei eth2], [gw3], > - [ovn-gw-3], [-neei eth2 -Q out], [gw3_out]) > + [ovn-gw-3], [-neei eth2 -Q out], [gw3_out], > + [ovn-gw-1], [-neei eth1], [gw1_eth1], > + [ovn-gw-2], [-neei eth1], [gw2_eth1], > + [ovn-chassis-1], [-neei eth1], [ch1_eth1], > + [ovn-chassis-2], [-neei eth1], [ch2_eth1]) > } > > stop_tcpdump() { > @@ -3275,7 +3309,7 @@ stop_tcpdump() { > send_background_packets() { > echo "$(date +%H:%M:%S.%03N) Sending packets in Background" > start_tcpdump > - M_NS_DAEMONIZE([ovn-chassis-3], [ext1], [ping -qf -i 0.1 > 192.168.1.1], [ping.pid]) > + M_NS_DAEMONIZE([ovn-chassis-3], [ext1], [ping -qf -i 0.01 > 192.168.1.1], [ping.pid]) > } > > stop_sending_background_packets() { > @@ -3357,6 +3391,28 @@ dump_statistics() { > echo "$((ch3_req - ch3_rep))" > } > > +add_port() { > + bridge=$1 > + interface=$2 > + address=$3 > + echo "Adding $bridge $interface $address" > + > + pid=$(podman inspect -f '{{.State.Pid}}' ovn-gw-1) > + ln -sf /proc/$pid/ns/net /var/run/netns/$pid > + port=$(host_ovs_vsctl --data=bare --no-heading --columns=name find > interface \ > + external_ids:container_id=ovn-gw-1 > external_ids:container_iface="$interface") > + port="${port:0:13}" > + ip link del "${port}_l" 2>/dev/null || true > + ip link del "${port}_c" 2>/dev/null || true > + check ip link add "${port}_l" type veth peer name "${port}_c" > + ip link set "${port}_l" up > + ip link set "${port}_c" netns $pid name "$interface" > + podman exec ovn-gw-1 ip link set "$interface" up > + if [[ -n "$address" ]]; then > + podman exec ovn-gw-1 ip addr add "$address" dev "$interface" > + fi > +} > + > prepare() { > send_background_packets > # We make sure gw1 is leader since enough time that it generated all > its garps. > @@ -3505,6 +3561,7 @@ check_migration_between_gw1_and_gw2_kill_gw2() { > > echo "$(date +%H:%M:%S.%03N) Restarting gw2 ovn-controller" > m_as ovn-gw-2 /usr/share/ovn/scripts/ovn-ctl start_controller > ${CONTROLLER_SSL_ARGS} > + get_geneve_names_gw2 > > # The network is now restored => packets should go through gw1 and > reach chassis-1. > check_packets "true" "false" "false" "true" > @@ -3543,12 +3600,130 @@ check_migration_between_gw1_and_gw2_kill_gw1() { > > echo "$(date +%H:%M:%S.%03N) Restarting gw1 ovn-controller after > killing gw1" > m_as ovn-gw-1 /usr/share/ovn/scripts/ovn-ctl start_controller > ${CONTROLLER_SSL_ARGS} > + get_geneve_names_gw1 > > # The network is now restored => packets should go through gw1 and > reach chassis-1. > check_packets "true" "false" "false" "true" > final_check "kill_gw1" $max_expected_loss2 > } > > +check_migration_between_gw1_and_gw2_reboot_gw1() { > + AS_BOX([$(date +%H:%M:%S.%03N) Rebooting ovn-gw-1]) > + max_expected_loss1=$1 > + max_expected_loss2=$2 > + prepare > + > + podman stop -t 0 ovn-gw-1 > + (exec 3>&- 4>&- 5>&- 6>&-; podman start ovn-gw-1) > + > + # As ovn-gw-1 got stopped and restarted, its ports might get deleted. > Add them back. > + add_port br-ovn-ext eth2 > + add_port br-ovn eth1 $gw1_eth1_ip_mask > + > + M_START_TCPDUMPS([ovn-gw-1], [-neei eth2], [gw1], [ovn-gw-1], [-neei > eth2 -Q out], [gw1_out]) > + check_loss_after_flap "gw1" $max_expected_loss1 > + > + # gw1 died => gw2 should generate garps. > + check_garps "false" "true" "false" > + check_packets "false" "true" "false" "true" > + > + start_openvswitch ovn-gw-1 > + M_START_TCPDUMPS([ovn-gw-1], [-neei eth1], [gw1_eth1]) > + > + # Wait some long time before restarting ovn-controller > + sleep 10 > + > + # gw2 should still be handling packets as OVN not restarted on gw1 > + check_packets "false" "true" "false" "true" > + > + echo "$(date +%H:%M:%S.%03N) Restarting gw1 ovn-controller after > rebooting gw1" > + m_as ovn-gw-1 /usr/share/ovn/scripts/ovn-ctl start_controller > ${CONTROLLER_SSL_ARGS} > + get_geneve_names_gw1 > + > + # The network is now restored => packets should go through gw1 and > reach chassis-1. > + check_packets "true" "false" "false" "true" > + final_check "reboot_gw1" $max_expected_loss2 > +} > + > +check_compute_restart() { > + AS_BOX([$(date +%H:%M:%S.%03N) Killing ovn-chassis-1 ovn-controller > and ovs-vswitchd]) > + max_expected_loss=$1 > + prepare > + > + # Kill ovn-chassis-1 > + echo "$(date +%H:%M:%S.%03N) Killing chassis-1" > + on_exit 'm_as ovn-chassis-1 /usr/share/openvswitch/scripts/ovs-ctl > status || > + m_as ovn-chassis-1 /usr/share/openvswitch/scripts/ovs-ctl > start --system-id=ovn-chassis-1' > + on_exit 'm_as ovn-chassis-1 /usr/share/ovn/scripts/ovn-ctl > status_controller || > + m_as ovn-chassis-1 /usr/share/ovn/scripts/ovn-ctl > start_controller ${CONTROLLER_SSL_ARGS}' > + > + m_as ovn-chassis-1 kill -9 $(m_as ovn-chassis-1 cat > /run/ovn/ovn-controller.pid) > + m_as ovn-chassis-1 kill -9 $(m_as ovn-chassis-1 cat > /run/openvswitch/ovs-vswitchd.pid) > + m_as ovn-chassis-1 kill -9 $(m_as ovn-chassis-1 cat > /run/openvswitch/ovsdb-server.pid) > + > + # Now restart chassis-1 > + flap_count=$(m_as ovn-gw-2 ovs-vsctl get interfac $from_gw2_to_ch1 > bfd_status | sed 's/.*flap_count=\"\([[0-9]]*\).*/\1/g') > + start_openvswitch ovn-chassis-1 > + > + echo "$(date +%H:%M:%S.%03N) Waiting for flap count between gw-1 and > chassis-1 to increase" > + OVS_WAIT_UNTIL([ > + new_flap_count=$(m_as ovn-gw-1 ovs-vsctl get interfac > $from_gw1_to_ch1 bfd_status | sed 's/.*flap_count=\"\([[0-9]]*\).*/\1/g') > + echo "Comparing $new_flap_count versus $flap_count" > + test "$new_flap_count" -gt "$((flap_count))" > + ]) > + > + echo "$(date +%H:%M:%S.%03N) Restarting ovn-chassis-1 ovn-controller." > + m_as ovn-chassis-1 /usr/share/ovn/scripts/ovn-ctl start_controller > ${CONTROLLER_SSL_ARGS} > + get_geneve_names_chassis1 > + > + wait_bfd_up ovn-chassis-1 $from_ch1_to_gw1 > + > + # Wait a long time to catch losses > + sleep 5 > + lost=0 > + final_check "compute" $max_expected_loss > +} > + > +check_update_ovn_ovs() { > + chassis=$1 > + ovn=$2 > + bfd=$3 > + max_expected_loss=$4 > + > + AS_BOX([$(date +%H:%M:%S.%03N) Check migration after restarting > $chassis ovs-vswitchd $ovn]) > + prepare > + lost=0 > + > + echo "$(date +%H:%M:%S.%03N) Restarting OVS $ovn on $chassis" > + if [[ "$ovn" == "ovn" ]]; then > + m_as $chassis /usr/share/ovn/scripts/ovn-ctl stop_controller > ${CONTROLLER_SSL_ARGS} --restart > + m_as $chassis /usr/share/openvswitch/scripts/ovs-ctl restart > --system-id=$chassis --no-ovs-vswitchd > + m_as $chassis /usr/share/openvswitch/scripts/ovs-ctl restart > --system-id=$chassis --no-ovsdb-server > + m_as $chassis /usr/share/ovn/scripts/ovn-ctl start_controller > ${CONTROLLER_SSL_ARGS} > + else > + m_as $chassis /usr/share/openvswitch/scripts/ovs-ctl restart > --system-id=$chassis > + fi > + > + if [[ "$bfd" == "bfd" ]]; then > + for c in $from_gw1_to_gw2 $from_gw1_to_gw3 $from_gw1_to_ch1 > $from_gw1_to_ch2; do > + wait_bfd_up ovn-gw-1 $c > + done > + for c in $from_ch1_to_gw1 $from_ch1_to_gw2; do > + wait_bfd_up ovn-chassis-1 $c > + done > + fi > + # The network is now restored => packets should go through gw1 and > reach chassis-1. > + # When packet loss is expected (e.g. in HA) packet might temporarily > go > + # through gw2. Ignore this. > + if [[ "$max_expected_loss" -gt 0 ]]; then > + check_packets "true" "ignore" "false" "true" > + final_check "ovs_update" $max_expected_loss > + else > + check_packets "true" "false" "false" "true" > + final_check "ovs_update" $max_expected_loss > + fi > +} > + > start_tcpdump > echo "$(date +%H:%M:%S.%03N) Sending packet from hv1-vif1(inside1) to > ext1" > M_NS_CHECK_EXEC([ovn-chassis-1], [hv1-vif1], [ping -c3 -q -i 0.1 > 192.168.0.1 | FORMAT_PING], > @@ -3579,7 +3754,32 @@ check_migration_between_gw1_and_gw2_bfd_stop 1 1 > check_migration_between_gw1_and_gw2_kill_gw2 1 1 > > # We simulate restart of both OVS & OVN gw1. gw2 should take over. > -check_migration_between_gw1_and_gw2_kill_gw1 400 200 > +# Expect around 1500 msec (mult x min_rx) + 1000 (for sending GARP) drop > time when gw1 dies. > +check_migration_between_gw1_and_gw2_kill_gw1 300 100 > + > +# We simulate death of gw1. gw2 should take over. > +check_migration_between_gw1_and_gw2_reboot_gw1 300 100 > + > +# We simulate ovs update on gw1. When ovs is stopped, flows should still > be handled by Kernel datapath. > +# When OVS is restarted, BFD should go down immediately, and gw2 might > start handling packets. > +check_update_ovn_ovs ovn-gw-1 "" bfd 300 > +check_update_ovn_ovs ovn-gw-1 ovn bfd 300 > +check_update_ovn_ovs ovn-chassis-1 "" bfd 300 > +check_update_ovn_ovs ovn-chassis-1 ovn bfd 300 > + > +# We simulate restart of ovn-chassis-1. We expect up to 3 sec loss. > +# 1 sec for chassis-1 to send Down, 1 sec for chassis-1 to send Init and > 1 sec for gw1 to send up. > +check_compute_restart 300 > + > +# Now change config tp L3 GW > +check multinode_nbctl clear Logical_Router_Port R1_outside gateway_chassis > +check multinode_nbctl set Logical_Router R1 options:chassis=ovn-gw-1 > +check multinode_nbctl --wait=hv sync > +m_check_row_count HA_Chassis_Group 0 name=R1_outside > +check_update_ovn_ovs ovn-gw-1 "" "" 0 > +check_update_ovn_ovs ovn-chassis-1 "" "" 0 > +check_update_ovn_ovs ovn-gw-1 ovn "" 0 > +check_update_ovn_ovs ovn-chassis-1 ovn "" 0 > > AT_CLEANUP > > diff --git a/tests/ovn-controller.at b/tests/ovn-controller.at > index e17ebea76..a7b79fc67 100644 > --- a/tests/ovn-controller.at > +++ b/tests/ovn-controller.at > @@ -4239,3 +4239,60 @@ done > OVN_CLEANUP([hv1]) > AT_CLEANUP > ]) > + > +OVN_FOR_EACH_NORTHD([ > +AT_SETUP([ovn-controller - transient flag for HA chassis]) > +AT_KEYWORDS([ovn]) > +ovn_start > + > +net_add n1 > +sim_add hv1 > +as hv1 > +check ovs-vsctl add-br br-phys > +ovn_attach n1 br-phys 192.168.0.1 > + > +sim_add hv2 > +as hv2 > +check ovs-vsctl add-br br-phys > +ovn_attach n1 br-phys 192.168.0.2 > + > +check ovn-nbctl ls-add ls1 > +check ovn-nbctl lr-add lr1 > +check ovn-nbctl lrp-add lr1 lr1-ls1 00:00:00:00:00:01 10.0.0.1/24 > +check ovn-nbctl lsp-add ls1 ls1-lr1 -- lsp-set-type ls1-lr1 router \ > + -- lsp-set-addresses ls1-lr1 router -- lsp-set-options ls1-lr1 > router-port=lr1-ls1 > + > +OVS_WAIT_UNTIL([as hv1 ovs-vsctl get interface ovn-hv2-0 type]) > +OVS_WAIT_UNTIL([as hv2 ovs-vsctl get interface ovn-hv1-0 type]) > + > +# Verify tunnel ports do not have transient flag set. > +AT_CHECK([as hv1 ovs-vsctl --bare get port ovn-hv2-0 other_config], [0], > [dnl > +{} > +]) > +AT_CHECK([as hv2 ovs-vsctl --bare get port ovn-hv1-0 other_config], [0], > [dnl > +{} > +]) > + > +check ovn-nbctl ha-chassis-group-add hagrp > +check ovn-nbctl ha-chassis-group-add-chassis hagrp hv1 10 > +check ovn-nbctl ha-chassis-group-add-chassis hagrp hv2 20 > + > +hagrp_uuid=$(ovn-nbctl --bare --columns _uuid find ha_chassis_group > name=hagrp) > +check ovn-nbctl set logical_router_port lr1-ls1 > ha_chassis_group=$hagrp_uuid > + > +wait_row_count HA_Chassis_Group 1 name=hagrp > +wait_row_count HA_Chassis 2 > + > +# Verify tunnel ports now have transient=true. > +OVS_WAIT_UNTIL([test "$(as hv1 ovs-vsctl get port ovn-hv2-0 > other_config:transient)" = '"true"']) > +OVS_WAIT_UNTIL([test "$(as hv2 ovs-vsctl get port ovn-hv1-0 > other_config:transient)" = '"true"']) > + > +check ovn-nbctl ha-chassis-group-remove-chassis hagrp hv2 > + > +# Verify transient flag is cleared on both HVs. > +OVS_WAIT_UNTIL([test "$(as hv1 ovs-vsctl --bare get port ovn-hv2-0 > other_config)" = '{}']) > +OVS_WAIT_UNTIL([test "$(as hv2 ovs-vsctl --bare get port ovn-hv1-0 > other_config)" = '{}']) > + > +OVN_CLEANUP([hv1],[hv2]) > +AT_CLEANUP > +]) > -- > 2.47.1 > > _______________________________________________ > dev mailing list > [email protected] > https://mail.openvswitch.org/mailman/listinfo/ovs-dev > > Regards, Ales _______________________________________________ dev mailing list [email protected] https://mail.openvswitch.org/mailman/listinfo/ovs-dev
