Hi Gregory,

Please find full kernel backtrace below.


------------------------------------------------------------------------------------------
BUG: spinlock recursion on CPU#1, rx-thread1/2787

 lock: 0x800000010d442508, .magic: dead4ead, .owner: rx-thread1/2787,
.owner_cpu: 1
CPU: 1 PID: 2787 Comm: rxthread1 Tainted: G           O 3.10.87-rt80- #1

Stack : ffffffff82120000 000000001400dce1 0000000004208040
80000000bd85cf18
          0000000000000000 0000000000000000 ffffffff82120000
ffffffff80ea5640
          0000000000000004 0000000000000000 0000000000000654
202020204f20332e
          0000000000000003 ffffffff82110208 ffffffff82120000
0000000000000000
          800000010d442508 800000010d442400 80000001061b8800
000000000000000e
          800000010641d200 ffffffffc0a80101 0000000000000001
0000000000000001
          0000000000000003 0000000000000001 0000000000000000
ffffffff80cdad4c
          800000010632c000 800000010632f220 0000000000000000
ffffffff80ad1628
          800000010619f120 ffffffff80e16710 0000000000000001
0000000000000ae3
          0000000000000000 ffffffff80864cd4 0000000000000000
0000000000000000
          ...

Call Trace:

[<ffffffff80864cd4>] show_stack+0x6c/0xf8

[<ffffffff80ad1628>] do_raw_spin_lock+0x168/0x170

[<ffffffff80bf7b1c>] dev_queue_xmit+0x43c/0x470

[<ffffffff80c32c08>] ip_finish_output+0x250/0x490

[<ffffffffc0115664>] rpl_iptunnel_xmit+0x134/0x218 [openvswitch]

[<ffffffffc0120f28>] rpl_vxlan_xmit+0x430/0x538 [openvswitch]

[<ffffffffc00f9de0>] do_execute_actions+0x18f8/0x19e8 [openvswitch]

[<ffffffffc00fa2b0>] ovs_execute_actions+0x90/0x208 [openvswitch]

[<ffffffffc0101860>] ovs_dp_process_packet+0xb0/0x1a8 [openvswitch]

[<ffffffffc010c5d8>] ovs_vport_receive+0x78/0x130 [openvswitch]

[<ffffffffc010ce6c>] internal_dev_xmit+0x34/0x98 [openvswitch]

[<ffffffff80bf74d0>] dev_hard_start_xmit+0x2e8/0x4f8

[<ffffffff80c10e48>] sch_direct_xmit+0xf0/0x238

[<ffffffff80bf78b8>] dev_queue_xmit+0x1d8/0x470

[<ffffffff80c5ffe4>] arp_process+0x614/0x628

[<ffffffff80bf0cb0>] __netif_receive_skb_core+0x2e8/0x5d8

[<ffffffff80bf4770>] process_backlog+0xc0/0x1b0

[<ffffffff80bf501c>] net_rx_action+0x154/0x240

[<ffffffff8088d130>] __do_softirq+0x1d0/0x218

[<ffffffff8088d240>] do_softirq+0x68/0x70

[<ffffffff8088d3a0>] local_bh_enable+0xa8/0xb0

[<ffffffff80bf5c88>] netif_rx_ni+0x20/0x30
[<ffffffffc0019ba4>] rx_work_handler+0x634/0x1078 [rx_ethernet]
[<ffffffff808a8e8c>] kthread+0xac/0xb8
[<ffffffff8085ffd0>] ret_from_kernel_thread+0x14/0x1c

BUG: spinlock lockup suspected on CPU#2, rx-thread2/2788
BUG: spinlock lockup suspected on CPU#3, handler406/6296
 lock: 0x800000010d442508, .magic: dead4ead, .owner: rx-thread1/2787,
.owner_cpu: 1
CPU: 3 PID: 6296 Comm: handler406 Tainted: G           O 3.10.87-rt80- #1
Stack : ffffffff80f10000 000000001400dce1 0000000000404140 0000000000002bf0
          0000000000000000 0000000000000000 ffffffff82120000
1b6c02ad5eabfa7f
          0000001b6c02ad5e 643a204720202020 202020202020204f
20332e31302e3837
          0000000000000006 ffffffff82110da0 ffffffff82120000
0000000000000000
          800000010d442508 0000000000010000 0000000059682f00
0000000059682f00
          800000010641d200 ffffffffc0a80101 0000000000000001
0000000000000001
          0000000000000006 0000000000000001 000000ffee0e36c0
0f00000004196607
          800000010d0dc000 800000010d0df410 0000000000000000
ffffffff80ad15dc
          800000010c9e21f0 ffffffff80e16710 0000000000000003
0000000000001898
          0000000000000000 ffffffff80864cd4 0000000000000000
0000000000000000
          ...
Call Trace:
[<ffffffff80864cd4>] show_stack+0x6c/0xf8
[<ffffffff80ad15dc>] do_raw_spin_lock+0x11c/0x170
[<ffffffff80bf7b1c>] dev_queue_xmit+0x43c/0x470
[<ffffffff80c32c08>] ip_finish_output+0x250/0x490
[<ffffffffc0115664>] rpl_iptunnel_xmit+0x134/0x218 [openvswitch]
[<ffffffffc0120f28>] rpl_vxlan_xmit+0x430/0x538 [openvswitch]
[<ffffffffc00f9de0>] do_execute_actions+0x18f8/0x19e8 [openvswitch]
[<ffffffffc00fa2b0>] ovs_execute_actions+0x90/0x208 [openvswitch]
[<ffffffffc0100028>] ovs_packet_cmd_execute+0x258/0x2a8 [openvswitch]
[<ffffffff80c16b5c>] genl_family_rcv_msg+0x32c/0x368
[<ffffffff80c16c2c>] genl_rcv_msg+0x94/0xd8
[<ffffffff80c15ca8>] netlink_rcv_skb+0x158/0x168
[<ffffffff80c15f00>] genl_rcv+0x30/0x48
[<ffffffff80c15544>] netlink_unicast+0x1ac/0x238
[<ffffffff80c15998>] netlink_sendmsg+0x300/0x368
[<ffffffff80bdba68>] sock_sendmsg+0xc0/0xf8
[<ffffffff80bdbdd4>] ___sys_sendmsg+0x2fc/0x308
[<ffffffff80bdf0c8>] __sys_sendmsg+0x48/0x98
[<ffffffff80869984>] handle_sys64+0x44/0x68
-------------------------------------------------------------------------------------------------

I have prepared the below patch to fix the issue.


--- a/openvswitch-cvm-2.8.1/datapath/linux/compat/vxlan.c
+++ b/openvswitch-cvm-2.8.1/datapath/linux/compat/vxlan.c
@@ -1114,7 +1114,8 @@ static void vxlan_xmit_one(struct sk_buff *skb,
struct net_device *dev,
                        goto tx_error;

                }



-               if (rt->dst.dev == dev) {

+               if ((rt->dst.dev == dev) ||

+                   (OVS_CB(skb)->input_vport->dev == rt->dst.dev)) {

                        netdev_dbg(dev, "circular route to %pI4\n",

                                   &dst->sin.sin_addr.s_addr);

                        dev->stats.collisions++;

@@ -1173,7 +1174,8 @@ static void vxlan_xmit_one(struct sk_buff *skb,
struct net_device *dev,
                        goto tx_error;

                }



-               if (ndst->dev == dev) {

+               if ((ndst->dev == dev) ||

+                   (OVS_CB(skb)->input_vport->dev == ndst->dev)) {

                        netdev_dbg(dev, "circular route to %pI6\n",

                                   &dst->sin6.sin6_addr);

                        dst_release(ndst);



This patch fixes the issue with the configuration given in the first mail.
It fixes only the first loop in case of vxlan.
What about if there are many levels in the loop.

I would like to know, how ovs handles these loops. Please share your
thoughts.





On Fri, May 4, 2018 at 1:21 AM, Gregory Rose <[email protected]> wrote:

> On 5/1/2018 10:35 PM, Neelakantam Gaddam wrote:
>
> Hi All,
>
> The issue here is we are trying to send the packets on the same device in
> a loop. While sending on a device, a spinlock for the tx queue has to be
> acquired in dev_queue_xmit function. This is where we are trying to acquire
> the same lock again, which is leading to the kernel crash. This issue
> becomes worse if internal ports are involved in the configuration.
>
> I think, we should avoid these loops in the vport send functions. But in
> case of tunneling, the tunnel send function should take care of these
> checks.
>
> Please share your thoughts on this issue.
>
>
> Hello Neelakantam,
>
> Please provide the full kernel panic backtrace and I'll have a look at the
> problem.  I'm rather busy
> dealing with some other critical issues but will try to get back to you
> next week.
>
> Thanks,
>
> - Greg
>
>
>
> On Mon, Apr 30, 2018 at 11:17 AM, Neelakantam Gaddam <
> [email protected]> wrote:
>
>> Hi All,
>>
>>
>>
>> OVS misconfiguration leading to spinlock recursion in dev_queue_xmit.
>>
>>
>>
>> We are running ovs-2.8.1 with openvswitch kernel modules on two hosts
>> connected back to back. We are running OVS on MIPS64 platform.
>>
>>
>>
>> We are using the below configuration.
>>
>>
>>
>> ovs-vsctl add-br br0
>>
>> ovs-vsctl add-bond br0 bond0 p1p1 p1p2
>>
>> ovs-vsctl set port bond0 lacp=active bond_mode=balance-tcp
>>
>> ifconfig br0 100.0.0.1 up
>>
>> ovs-vsctl add-port br0 veth0
>>
>> ovs-vsctl add-port br0 vx0 -- set interface vx0 type=vxlan
>> options:local_ip=100.0.0.1 options:remote_ip=100.0.0.2 option:key=flow
>>
>>
>>
>> ovs-ofctl add-flow br0 "table=0, priority=1, cookie=100, tun_id=100,
>> in_port=4, action=output:3"
>>
>> ovs-ofctl add-flow br0 "table=0, priority=1, cookie=100, in_port=3,
>> actions=set_field:100->tun_id output:4"
>>
>>
>>
>>
>>
>> When this configuration is applied on both hosts, we are seeing the below
>> spinlock recursion bug.
>>
>>
>>
>> [<ffffffff80864cd4>] show_stack+0x6c/0xf8
>>
>>
>> [<ffffffff80ad1628>] do_raw_spin_lock+0x168/0x170
>>
>>
>> [<ffffffff80bf7b1c>] dev_queue_xmit+0x43c/0x470
>>
>>
>> [<ffffffff80c32c08>] ip_finish_output+0x250/0x490
>>
>>
>> [<ffffffffc0115664>] rpl_iptunnel_xmit+0x134/0x218 [openvswitch]
>>
>>
>> [<ffffffffc0120f28>] rpl_vxlan_xmit+0x430/0x538 [openvswitch]
>>
>>
>> [<ffffffffc00f9de0>] do_execute_actions+0x18f8/0x19e8 [openvswitch]
>>
>>
>> [<ffffffffc00fa2b0>] ovs_execute_actions+0x90/0x208 [openvswitch]
>>
>>
>> [<ffffffffc0101860>] ovs_dp_process_packet+0xb0/0x1a8 [openvswitch]
>>
>>
>> [<ffffffffc010c5d8>] ovs_vport_receive+0x78/0x130 [openvswitch]
>>
>>
>> [<ffffffffc010ce6c>] internal_dev_xmit+0x34/0x98 [openvswitch]
>>
>>
>> [<ffffffff80bf74d0>] dev_hard_start_xmit+0x2e8/0x4f8
>>
>>
>> [<ffffffff80c10e48>] sch_direct_xmit+0xf0/0x238
>>
>>
>> [<ffffffff80bf78b8>] dev_queue_xmit+0x1d8/0x470
>>
>>
>> [<ffffffff80c5ffe4>] arp_process+0x614/0x628
>>
>>
>> [<ffffffff80bf0cb0>] __netif_receive_skb_core+0x2e8/0x5d8
>>
>>
>> [<ffffffff80bf4770>] process_backlog+0xc0/0x1b0
>>
>>
>> [<ffffffff80bf501c>] net_rx_action+0x154/0x240
>>
>>
>> [<ffffffff8088d130>] __do_softirq+0x1d0/0x218
>>
>>
>> [<ffffffff8088d240>] do_softirq+0x68/0x70
>>
>>
>> [<ffffffff8088d3a0>] local_bh_enable+0xa8/0xb0
>>
>>
>> [<ffffffff80bf5c88>] netif_rx_ni+0x20/0x30
>>
>>
>>
>> The packet path traced is : netif_rx->arp->dev_queue_xmit(internal
>> port)->vxlan_xmit->dev_queue_xmit(internal port). According to the
>> configuration, this packet path is valid. But we should not hit the crash.
>>
>>
>> Questions:
>>
>>
>>    - Is it a kernel bug or ovs bug ?
>>    - How OVS handles these kind of misconfigurations especially packet
>>    loops involved?
>>
>>
>>
>> Any suggestion or help is greatly appreciated.
>>
>>
>>
>>
>>
>>
>>
>> Thanks
>>
>>
>> --
>> Thanks & Regards
>> Neelakantam Gaddam
>>
>
>
>
> --
> Thanks & Regards
> Neelakantam Gaddam
>
>
> _______________________________________________
> discuss mailing 
> [email protected]https://mail.openvswitch.org/mailman/listinfo/ovs-discuss
>
>
>


-- 
Thanks & Regards
Neelakantam Gaddam
_______________________________________________
discuss mailing list
[email protected]
https://mail.openvswitch.org/mailman/listinfo/ovs-discuss

Reply via email to