summaryrefslogtreecommitdiffstats
path: root/test/test_sixrd.py
blob: c6b3c0885168afb48abb0d3fdeb344e9dcc1a382 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
#!/usr/bin/env python
""" 6RD RFC5969 functional tests """

import unittest
from scapy.layers.inet import IP, UDP, Ether
from scapy.layers.inet6 import IPv6
from scapy.packet import Raw
from framework import VppTestCase, VppTestRunner
from vpp_ip_route import VppIpRoute, VppRoutePath, DpoProto
from socket import AF_INET, AF_INET6, inet_pton

""" Test6rd is a subclass of  VPPTestCase classes.

6RD tests.

"""


class Test6RD(VppTestCase):
    """ 6RD Test Case """

    @classmethod
    def setUpClass(cls):
        super(Test6RD, cls).setUpClass()
        cls.create_pg_interfaces(range(2))
        cls.interfaces = list(cls.pg_interfaces)

    def setUp(cls):
        super(Test6RD, cls).setUp()
        for i in cls.interfaces:
            i.admin_up()
            i.config_ip4()
            i.config_ip6()
            i.disable_ipv6_ra()
            i.resolve_arp()
            i.resolve_ndp()

    def tearDown(self):
        super(Test6RD, self).tearDown()
        if not self.vpp_dead:
            for i in self.pg_interfaces:
                i.unconfig_ip4()
                i.unconfig_ip6()
                i.admin_down()
            if type(self.tunnel_index) is list:
                for sw_if_index in self.tunnel_index:
                    self.vapi.ipip_6rd_del_tunnel(sw_if_index)
            else:
                self.vapi.ipip_6rd_del_tunnel(self.tunnel_index)

    def validate_6in4(self, rx, expected):
        if IP not in rx:
            self.fail()
        if IPv6 not in rx:
            self.fail()

        self.assertEqual(rx[IP].src, expected[IP].src)
        self.assertEqual(rx[IP].dst, expected[IP].dst)
        self.assertEqual(rx[IP].proto, expected[IP].proto)
        self.assertEqual(rx[IPv6].src, expected[IPv6].src)
        self.assertEqual(rx[IPv6].dst, expected[IPv6].dst)

    def validate_4in6(self, rx, expected):
        if IPv6 not in rx:
            self.fail()
        if IP in rx:
            self.fail()

        self.assertTrue(rx[IPv6].src == expected[IPv6].src)
        self.assertTrue(rx[IPv6].dst == expected[IPv6].dst)
        self.assertTrue(rx[IPv6].nh == expected[IPv6].nh)

    def payload(self, len):
        return 'x' * len

    def test_6rd_ip6_to_ip4(self):
        """ ip6 -> ip4 (encap) 6rd test """
        p_ether = Ether(src=self.pg0.remote_mac, dst=self.pg0.local_mac)
        p_ip6 = IPv6(src="1::1", dst="2002:AC10:0202::1", nh='UDP')

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'), 16,
            inet_pton(AF_INET, '0.0.0.0'), 0,
            self.pg0.local_ip4n, True)
        self.tunnel_index = rv.sw_if_index

        self.vapi.cli("show ip6 fib")
        p_payload = UDP(sport=1234, dport=1234)
        p = (p_ether / p_ip6 / p_payload)

        p_reply = (IP(src=self.pg0.local_ip4, dst=self.pg1.remote_ip4,
                      proto='ipv6') / p_ip6)

        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_6in4(p, p_reply)

        # MTU tests (default is 1480)
        plen = 1481 - 40 - 8
        p_ip6 = IPv6(src="1::1", dst="2002:AC10:0202::1")
        p_payload = UDP(sport=1234, dport=1234) / Raw(self.payload(plen))
        p = (p_ether / p_ip6 / p_payload)

        p_reply = (IP(src=self.pg0.local_ip4, dst=self.pg1.remote_ip4,
                      proto='ipv6') / p_ip6)

        rx = self.send_and_assert_no_replies(self.pg0, p*10)

    def test_6rd_ip4_to_ip6(self):
        """ ip4 -> ip6 (decap) 6rd test """

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg0.local_ip4n, True)
        self.tunnel_index = rv.sw_if_index
        rv = self.vapi.ipip_6rd_del_tunnel(rv.sw_if_index)
        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg0.local_ip4n, True)
        self.tunnel_index = rv.sw_if_index

        p_ip6 = (IPv6(src="2002:AC10:0202::1", dst=self.pg1.remote_ip6) /
                 UDP(sport=1234, dport=1234))

        p = (Ether(src=self.pg0.remote_mac,
                   dst=self.pg0.local_mac) /
             IP(src=self.pg1.remote_ip4, dst=self.pg0.local_ip4) /
             p_ip6)

        p_reply = p_ip6

        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_reply)

    def test_6rd_ip4_to_ip6_multiple(self):
        """ ip4 -> ip6 (decap) 6rd test """

        self.tunnel_index = []
        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg0.local_ip4n, True)
        self.tunnel_index.append(rv.sw_if_index)

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2003::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg1.local_ip4n, True)
        self.tunnel_index.append(rv.sw_if_index)

        self.vapi.cli("show ip6 fib")
        p_ether = Ether(src=self.pg0.remote_mac, dst=self.pg0.local_mac)
        p_ip4 = IP(src=self.pg1.remote_ip4, dst=self.pg0.local_ip4)
        p_ip6_1 = (IPv6(src="2002:AC10:0202::1", dst=self.pg1.remote_ip6) /
                   UDP(sport=1234, dport=1234))
        p_ip6_2 = (IPv6(src="2003:AC10:0202::1", dst=self.pg1.remote_ip6) /
                   UDP(sport=1234, dport=1234))

        p = (p_ether / p_ip4 / p_ip6_1)
        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_ip6_1)

        p = (p_ether / p_ip4 / p_ip6_2)
        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_ip6_2)

    def test_6rd_ip4_to_ip6_suffix(self):
        """ ip4 -> ip6 (decap) 6rd test """

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'), 16,
            inet_pton(AF_INET, '172.0.0.0'), 8,
            self.pg0.local_ip4n, True)

        self.tunnel_index = rv.sw_if_index

        self.vapi.cli("show ip6 fib")
        p_ether = Ether(src=self.pg0.remote_mac, dst=self.pg0.local_mac)
        p_ip4 = IP(src=self.pg1.remote_ip4, dst=self.pg0.local_ip4)
        p_ip6 = (IPv6(src="2002:1002:0200::1", dst=self.pg1.remote_ip6) /
                 UDP(sport=1234, dport=1234))

        p = (p_ether / p_ip4 / p_ip6)
        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_ip6)

    def test_6rd_ip4_to_ip6_sec_check(self):
        """ ip4 -> ip6 (decap) security check 6rd test """

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg0.local_ip4n, True)
        self.tunnel_index = rv.sw_if_index

        self.vapi.cli("show ip6 fib")
        p_ip6 = (IPv6(src="2002:AC10:0202::1", dst=self.pg1.remote_ip6) /
                 UDP(sport=1234, dport=1234))
        p_ip6_fail = (IPv6(src="2002:DEAD:0202::1", dst=self.pg1.remote_ip6) /
                      UDP(sport=1234, dport=1234))

        p = (Ether(src=self.pg0.remote_mac,
                   dst=self.pg0.local_mac) /
             IP(src=self.pg1.remote_ip4, dst=self.pg0.local_ip4) /
             p_ip6)

        p_reply = p_ip6

        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_reply)

        p = (Ether(src=self.pg0.remote_mac,
                   dst=self.pg0.local_mac) /
             IP(src=self.pg1.remote_ip4, dst=self.pg0.local_ip4) /
             p_ip6_fail)
        rx = self.send_and_assert_no_replies(self.pg0, p*10)

    def test_6rd_bgp_tunnel(self):
        """ 6rd BGP tunnel """

        rv = self.vapi.ipip_6rd_add_tunnel(
            0, inet_pton(AF_INET6, '2002::'),
            16, inet_pton(AF_INET, '0.0.0.0'),
            0, self.pg0.local_ip4n, False)
        self.tunnel_index = rv.sw_if_index

        default_route = VppIpRoute(
            self, "DEAD::", 16, [VppRoutePath("2002:0808:0808::",
                                              self.tunnel_index,
                                              proto=DpoProto.DPO_PROTO_IP6)],
            is_ip6=1)
        default_route.add_vpp_config()

        ip4_route = VppIpRoute(self, "8.0.0.0", 8,
                               [VppRoutePath(self.pg1.remote_ip4, 0xFFFFFFFF)])
        ip4_route.add_vpp_config()

        # Via recursive route 6 -> 4
        p = (Ether(src=self.pg0.remote_mac,
                   dst=self.pg0.local_mac) /
             IPv6(src="1::1", dst="DEAD:BEEF::1") /
             UDP(sport=1234, dport=1234))

        p_reply = (IP(src=self.pg0.local_ip4, dst="8.8.8.8",
                      proto='ipv6') /
                   IPv6(src='1::1', dst='DEAD:BEEF::1', nh='UDP'))

        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_6in4(p, p_reply)

        # Via recursive route 4 -> 6 (Security check must be disabled)
        p_ip6 = (IPv6(src="DEAD:BEEF::1", dst=self.pg1.remote_ip6) /
                 UDP(sport=1234, dport=1234))
        p = (Ether(src=self.pg0.remote_mac,
                   dst=self.pg0.local_mac) /
             IP(src="8.8.8.8", dst=self.pg0.local_ip4) /
             p_ip6)

        p_reply = p_ip6

        rx = self.send_and_expect(self.pg0, p*10, self.pg1)
        for p in rx:
            self.validate_4in6(p, p_reply)


if __name__ == '__main__':
    unittest.main(testRunner=VppTestRunner)
">node) - STRUCT_OFFSET_OF (vxlan_tunnel_t, node))); } /** * Function definition to backwalk a FIB node - * Here we will restack the new dpo of VXLAN DIP to encap node. */ static fib_node_back_walk_rc_t vxlan_tunnel_back_walk (fib_node_t * node, fib_node_back_walk_ctx_t * ctx) { vxlan_tunnel_restack_dpo (vxlan_tunnel_from_fib_node (node)); return (FIB_NODE_BACK_WALK_CONTINUE); } /** * Function definition to get a FIB node from its index */ static fib_node_t * vxlan_tunnel_fib_node_get (fib_node_index_t index) { vxlan_tunnel_t *t; vxlan_main_t *vxm = &vxlan_main; t = pool_elt_at_index (vxm->tunnels, index); return (&t->node); } /** * Function definition to inform the FIB node that its last lock has gone. */ static void vxlan_tunnel_last_lock_gone (fib_node_t * node) { /* * The VXLAN tunnel is a root of the graph. As such * it never has children and thus is never locked. */ ASSERT (0); } /* * Virtual function table registered by VXLAN tunnels * for participation in the FIB object graph. */ const static fib_node_vft_t vxlan_vft = { .fnv_get = vxlan_tunnel_fib_node_get, .fnv_last_lock = vxlan_tunnel_last_lock_gone, .fnv_back_walk = vxlan_tunnel_back_walk, }; #define foreach_copy_field \ _(vni) \ _(mcast_sw_if_index) \ _(encap_fib_index) \ _(decap_next_index) \ _(src) \ _(dst) static void vxlan_rewrite (vxlan_tunnel_t * t, bool is_ip6) { union { ip4_vxlan_header_t h4; ip6_vxlan_header_t h6; } h; int len = is_ip6 ? sizeof h.h6 : sizeof h.h4; udp_header_t *udp; vxlan_header_t *vxlan; /* Fixed portion of the (outer) ip header */ clib_memset (&h, 0, sizeof (h)); if (!is_ip6) { ip4_header_t *ip = &h.h4.ip4; udp = &h.h4.udp, vxlan = &h.h4.vxlan; ip->ip_version_and_header_length = 0x45; ip->ttl = 254; ip->protocol = IP_PROTOCOL_UDP; ip->src_address = t->src.ip4; ip->dst_address = t->dst.ip4; /* we fix up the ip4 header length and checksum after-the-fact */ ip->checksum = ip4_header_checksum (ip); } else { ip6_header_t *ip = &h.h6.ip6; udp = &h.h6.udp, vxlan = &h.h6.vxlan; ip->ip_version_traffic_class_and_flow_label = clib_host_to_net_u32 (6 << 28); ip->hop_limit = 255; ip->protocol = IP_PROTOCOL_UDP; ip->src_address = t->src.ip6; ip->dst_address = t->dst.ip6; } /* UDP header, randomize src port on something, maybe? */ udp->src_port = clib_host_to_net_u16 (4789); udp->dst_port = clib_host_to_net_u16 (UDP_DST_PORT_vxlan); /* VXLAN header */ vnet_set_vni_and_flags (vxlan, t->vni); vnet_rewrite_set_data (*t, &h, len); } static bool vxlan_decap_next_is_valid (vxlan_main_t * vxm, u32 is_ip6, u32 decap_next_index) { vlib_main_t *vm = vxm->vlib_main; u32 input_idx = (!is_ip6) ? vxlan4_input_node.index : vxlan6_input_node.index; vlib_node_runtime_t *r = vlib_node_get_runtime (vm, input_idx); return decap_next_index < r->n_next_nodes; } static uword vtep_addr_ref (ip46_address_t * ip) { uword *vtep = ip46_address_is_ip4 (ip) ? hash_get (vxlan_main.vtep4, ip->ip4.as_u32) : hash_get_mem (vxlan_main.vtep6, &ip->ip6); if (vtep) return ++(*vtep); ip46_address_is_ip4 (ip) ? hash_set (vxlan_main.vtep4, ip->ip4.as_u32, 1) : hash_set_mem_alloc (&vxlan_main.vtep6, &ip->ip6, 1); return 1; } static uword vtep_addr_unref (ip46_address_t * ip) { uword *vtep = ip46_address_is_ip4 (ip) ? hash_get (vxlan_main.vtep4, ip->ip4.as_u32) : hash_get_mem (vxlan_main.vtep6, &ip->ip6); ASSERT (vtep); if (--(*vtep) != 0) return *vtep; ip46_address_is_ip4 (ip) ? hash_unset (vxlan_main.vtep4, ip->ip4.as_u32) : hash_unset_mem_free (&vxlan_main.vtep6, &ip->ip6); return 0; } /* *INDENT-OFF* */ typedef CLIB_PACKED(union { struct { fib_node_index_t mfib_entry_index; adj_index_t mcast_adj_index; }; u64 as_u64; }) mcast_shared_t; /* *INDENT-ON* */ static inline mcast_shared_t mcast_shared_get (ip46_address_t * ip) { ASSERT (ip46_address_is_multicast (ip)); uword *p = hash_get_mem (vxlan_main.mcast_shared, ip); ASSERT (p); mcast_shared_t ret = {.as_u64 = *p }; return ret; } static inline void mcast_shared_add (ip46_address_t * dst, fib_node_index_t mfei, adj_index_t ai) { mcast_shared_t new_ep = { .mcast_adj_index = ai, .mfib_entry_index = mfei, }; hash_set_mem_alloc (&vxlan_main.mcast_shared, dst, new_ep.as_u64); } static inline void mcast_shared_remove (ip46_address_t * dst) { mcast_shared_t ep = mcast_shared_get (dst); adj_unlock (ep.mcast_adj_index); mfib_table_entry_delete_index (ep.mfib_entry_index, MFIB_SOURCE_VXLAN); hash_unset_mem_free (&vxlan_main.mcast_shared, dst); } int vnet_vxlan_add_del_tunnel (vnet_vxlan_add_del_tunnel_args_t * a, u32 * sw_if_indexp) { vxlan_main_t *vxm = &vxlan_main; vnet_main_t *vnm = vxm->vnet_main; vxlan_decap_info_t *p; u32 sw_if_index = ~0; vxlan4_tunnel_key_t key4; vxlan6_tunnel_key_t key6; u32 is_ip6 = a->is_ip6; int not_found; if (!is_ip6) { /* ip4 mcast is indexed by mcast addr only */ key4.key[0] = ip46_address_is_multicast (&a->dst) ? a->dst.ip4.as_u32 : a->dst.ip4.as_u32 | (((u64) a->src.ip4.as_u32) << 32); key4.key[1] = (((u64) a->encap_fib_index) << 32) | clib_host_to_net_u32 (a->vni << 8); not_found = clib_bihash_search_inline_16_8 (&vxm->vxlan4_tunnel_by_key, &key4); p = (void *) &key4.value; } else { key6.key[0] = a->dst.ip6.as_u64[0]; key6.key[1] = a->dst.ip6.as_u64[1]; key6.key[2] = (((u64) a->encap_fib_index) << 32) | clib_host_to_net_u32 (a->vni << 8); not_found = clib_bihash_search_inline_24_8 (&vxm->vxlan6_tunnel_by_key, &key6); p = (void *) &key6.value; } if (not_found) p = 0; if (a->is_add) { l2input_main_t *l2im = &l2input_main; u32 dev_instance; /* real dev instance tunnel index */ u32 user_instance; /* request and actual instance number */ /* adding a tunnel: tunnel must not already exist */ if (p) return VNET_API_ERROR_TUNNEL_EXIST; /*if not set explicitly, default to l2 */ if (a->decap_next_index == ~0) a->decap_next_index = VXLAN_INPUT_NEXT_L2_INPUT; if (!vxlan_decap_next_is_valid (vxm, is_ip6, a->decap_next_index)) return VNET_API_ERROR_INVALID_DECAP_NEXT; vxlan_tunnel_t *t; pool_get_aligned (vxm->tunnels, t, CLIB_CACHE_LINE_BYTES); clib_memset (t, 0, sizeof (*t)); dev_instance = t - vxm->tunnels; /* copy from arg structure */ #define _(x) t->x = a->x; foreach_copy_field; #undef _ vxlan_rewrite (t, is_ip6); /* * Reconcile the real dev_instance and a possible requested instance. */ user_instance = a->instance; if (user_instance == ~0) user_instance = dev_instance; if (hash_get (vxm->instance_used, user_instance)) { pool_put (vxm->tunnels, t); return VNET_API_ERROR_INSTANCE_IN_USE; } hash_set (vxm->instance_used, user_instance, 1); t->dev_instance = dev_instance; /* actual */ t->user_instance = user_instance; /* name */ t->flow_index = ~0; t->hw_if_index = vnet_register_interface (vnm, vxlan_device_class.index, dev_instance, vxlan_hw_class.index, dev_instance); vnet_hw_interface_t *hi = vnet_get_hw_interface (vnm, t->hw_if_index); /* Set vxlan tunnel output node */ u32 encap_index = !is_ip6 ? vxlan4_encap_node.index : vxlan6_encap_node.index; vnet_set_interface_output_node (vnm, t->hw_if_index, encap_index); t->sw_if_index = sw_if_index = hi->sw_if_index; /* copy the key */ int add_failed; if (is_ip6) { key6.value = (u64) dev_instance; add_failed = clib_bihash_add_del_24_8 (&vxm->vxlan6_tunnel_by_key, &key6, 1 /*add */ ); } else { vxlan_decap_info_t di = {.sw_if_index = t->sw_if_index, }; if (ip46_address_is_multicast (&t->dst)) di.local_ip = t->src.ip4; else di.next_index = t->decap_next_index; key4.value = di.as_u64; add_failed = clib_bihash_add_del_16_8 (&vxm->vxlan4_tunnel_by_key, &key4, 1 /*add */ ); } if (add_failed) { vnet_delete_hw_interface (vnm, t->hw_if_index); hash_unset (vxm->instance_used, t->user_instance); pool_put (vxm->tunnels, t); return VNET_API_ERROR_INVALID_REGISTRATION; } vec_validate_init_empty (vxm->tunnel_index_by_sw_if_index, sw_if_index, ~0); vxm->tunnel_index_by_sw_if_index[sw_if_index] = dev_instance; /* setup l2 input config with l2 feature and bd 0 to drop packet */ vec_validate (l2im->configs, sw_if_index); l2im->configs[sw_if_index].feature_bitmap = L2INPUT_FEAT_DROP; l2im->configs[sw_if_index].bd_index = 0; vnet_sw_interface_t *si = vnet_get_sw_interface (vnm, sw_if_index); si->flags &= ~VNET_SW_INTERFACE_FLAG_HIDDEN; vnet_sw_interface_set_flags (vnm, sw_if_index, VNET_SW_INTERFACE_FLAG_ADMIN_UP); fib_node_init (&t->node, FIB_NODE_TYPE_VXLAN_TUNNEL); fib_prefix_t tun_dst_pfx; vnet_flood_class_t flood_class = VNET_FLOOD_CLASS_TUNNEL_NORMAL; fib_prefix_from_ip46_addr (&t->dst, &tun_dst_pfx); if (!ip46_address_is_multicast (&t->dst)) { /* Unicast tunnel - * source the FIB entry for the tunnel's destination * and become a child thereof. The tunnel will then get poked * when the forwarding for the entry updates, and the tunnel can * re-stack accordingly */ vtep_addr_ref (&t->src); t->fib_entry_index = fib_table_entry_special_add (t->encap_fib_index, &tun_dst_pfx, FIB_SOURCE_RR, FIB_ENTRY_FLAG_NONE); t->sibling_index = fib_entry_child_add (t->fib_entry_index, FIB_NODE_TYPE_VXLAN_TUNNEL, dev_instance); vxlan_tunnel_restack_dpo (t); } else { /* Multicast tunnel - * as the same mcast group can be used for multiple mcast tunnels * with different VNIs, create the output fib adjacency only if * it does not already exist */ fib_protocol_t fp = fib_ip_proto (is_ip6); if (vtep_addr_ref (&t->dst) == 1) { fib_node_index_t mfei; adj_index_t ai; fib_route_path_t path = { .frp_proto = fib_proto_to_dpo (fp), .frp_addr = zero_addr, .frp_sw_if_index = 0xffffffff, .frp_fib_index = ~0, .frp_weight = 0, .frp_flags = FIB_ROUTE_PATH_LOCAL, }; const mfib_prefix_t mpfx = { .fp_proto = fp, .fp_len = (is_ip6 ? 128 : 32), .fp_grp_addr = tun_dst_pfx.fp_addr, }; /* * Setup the (*,G) to receive traffic on the mcast group * - the forwarding interface is for-us * - the accepting interface is that from the API */ mfib_table_entry_path_update (t->encap_fib_index, &mpfx, MFIB_SOURCE_VXLAN, &path, MFIB_ITF_FLAG_FORWARD); path.frp_sw_if_index = a->mcast_sw_if_index; path.frp_flags = FIB_ROUTE_PATH_FLAG_NONE; mfei = mfib_table_entry_path_update (t->encap_fib_index, &mpfx, MFIB_SOURCE_VXLAN, &path, MFIB_ITF_FLAG_ACCEPT); /* * Create the mcast adjacency to send traffic to the group */ ai = adj_mcast_add_or_lock (fp, fib_proto_to_link (fp), a->mcast_sw_if_index); /* * create a new end-point */ mcast_shared_add (&t->dst, mfei, ai); } dpo_id_t dpo = DPO_INVALID; mcast_shared_t ep = mcast_shared_get (&t->dst); /* Stack shared mcast dst mac addr rewrite on encap */ dpo_set (&dpo, DPO_ADJACENCY_MCAST, fib_proto_to_dpo (fp), ep.mcast_adj_index); dpo_stack_from_node (encap_index, &t->next_dpo, &dpo); dpo_reset (&dpo); flood_class = VNET_FLOOD_CLASS_TUNNEL_MASTER; } vnet_get_sw_interface (vnet_get_main (), sw_if_index)->flood_class = flood_class; } else { /* deleting a tunnel: tunnel must exist */ if (!p) return VNET_API_ERROR_NO_SUCH_ENTRY; u32 instance = is_ip6 ? key6.value : vxm->tunnel_index_by_sw_if_index[p->sw_if_index]; vxlan_tunnel_t *t = pool_elt_at_index (vxm->tunnels, instance); sw_if_index = t->sw_if_index; vnet_sw_interface_set_flags (vnm, sw_if_index, 0 /* down */ ); vxm->tunnel_index_by_sw_if_index[sw_if_index] = ~0; if (!is_ip6) clib_bihash_add_del_16_8 (&vxm->vxlan4_tunnel_by_key, &key4, 0 /*del */ ); else clib_bihash_add_del_24_8 (&vxm->vxlan6_tunnel_by_key, &key6, 0 /*del */ ); if (!ip46_address_is_multicast (&t->dst)) { if (t->flow_index != ~0) vnet_flow_del (vnm, t->flow_index); vtep_addr_unref (&t->src); fib_entry_child_remove (t->fib_entry_index, t->sibling_index); fib_table_entry_delete_index (t->fib_entry_index, FIB_SOURCE_RR); } else if (vtep_addr_unref (&t->dst) == 0) { mcast_shared_remove (&t->dst); } vnet_delete_hw_interface (vnm, t->hw_if_index); hash_unset (vxm->instance_used, t->user_instance); fib_node_deinit (&t->node); pool_put (vxm->tunnels, t); } if (sw_if_indexp) *sw_if_indexp = sw_if_index; return 0; } static uword get_decap_next_for_node (u32 node_index, u32 ipv4_set) { vxlan_main_t *vxm = &vxlan_main; vlib_main_t *vm = vxm->vlib_main; uword input_node = (ipv4_set) ? vxlan4_input_node.index : vxlan6_input_node.index; return vlib_node_add_next (vm, input_node, node_index); } static uword unformat_decap_next (unformat_input_t * input, va_list * args) { u32 *result = va_arg (*args, u32 *); u32 ipv4_set = va_arg (*args, int); vxlan_main_t *vxm = &vxlan_main; vlib_main_t *vm = vxm->vlib_main; u32 node_index; u32 tmp; if (unformat (input, "l2")) *result = VXLAN_INPUT_NEXT_L2_INPUT; else if (unformat (input, "node %U", unformat_vlib_node, vm, &node_index)) *result = get_decap_next_for_node (node_index, ipv4_set); else if (unformat (input, "%d", &tmp)) *result = tmp; else return 0; return 1; } static clib_error_t * vxlan_add_del_tunnel_command_fn (vlib_main_t * vm, unformat_input_t * input, vlib_cli_command_t * cmd) { unformat_input_t _line_input, *line_input = &_line_input; ip46_address_t src = ip46_address_initializer, dst = ip46_address_initializer; u8 is_add = 1; u8 src_set = 0; u8 dst_set = 0; u8 grp_set = 0; u8 ipv4_set = 0; u8 ipv6_set = 0; u32 instance = ~0; u32 encap_fib_index = 0; u32 mcast_sw_if_index = ~0; u32 decap_next_index = VXLAN_INPUT_NEXT_L2_INPUT; u32 vni = 0; u32 table_id; clib_error_t *parse_error = NULL; /* Get a line of input. */ if (!unformat_user (input, unformat_line_input, line_input)) return 0; while (unformat_check_input (line_input) != UNFORMAT_END_OF_INPUT) { if (unformat (line_input, "del")) { is_add = 0; } else if (unformat (line_input, "instance %d", &instance)) ; else if (unformat (line_input, "src %U", unformat_ip46_address, &src, IP46_TYPE_ANY)) { src_set = 1; ip46_address_is_ip4 (&src) ? (ipv4_set = 1) : (ipv6_set = 1); } else if (unformat (line_input, "dst %U", unformat_ip46_address, &dst, IP46_TYPE_ANY)) { dst_set = 1; ip46_address_is_ip4 (&dst) ? (ipv4_set = 1) : (ipv6_set = 1); } else if (unformat (line_input, "group %U %U", unformat_ip46_address, &dst, IP46_TYPE_ANY, unformat_vnet_sw_interface, vnet_get_main (), &mcast_sw_if_index)) { grp_set = dst_set = 1; ip46_address_is_ip4 (&dst) ? (ipv4_set = 1) : (ipv6_set = 1); } else if (unformat (line_input, "encap-vrf-id %d", &table_id)) { encap_fib_index = fib_table_find (fib_ip_proto (ipv6_set), table_id); } else if (unformat (line_input, "decap-next %U", unformat_decap_next, &decap_next_index, ipv4_set)) ; else if (unformat (line_input, "vni %d", &vni)) ; else { parse_error = clib_error_return (0, "parse error: '%U'", format_unformat_error, line_input); break; } } unformat_free (line_input); if (parse_error) return parse_error; if (encap_fib_index == ~0) return clib_error_return (0, "nonexistent encap-vrf-id %d", table_id); if (src_set == 0) return clib_error_return (0, "tunnel src address not specified"); if (dst_set == 0) return clib_error_return (0, "tunnel dst address not specified"); if (grp_set && !ip46_address_is_multicast (&dst)) return clib_error_return (0, "tunnel group address not multicast"); if (grp_set == 0 && ip46_address_is_multicast (&dst)) return clib_error_return (0, "dst address must be unicast"); if (grp_set && mcast_sw_if_index == ~0) return clib_error_return (0, "tunnel nonexistent multicast device"); if (ipv4_set && ipv6_set) return clib_error_return (0, "both IPv4 and IPv6 addresses specified"); if (ip46_address_cmp (&src, &dst) == 0) return clib_error_return (0, "src and dst addresses are identical"); if (decap_next_index == ~0) return clib_error_return (0, "next node not found"); if (vni == 0) return clib_error_return (0, "vni not specified"); if (vni >> 24) return clib_error_return (0, "vni %d out of range", vni); vnet_vxlan_add_del_tunnel_args_t a = { .is_add = is_add, .is_ip6 = ipv6_set, .instance = instance, #define _(x) .x = x, foreach_copy_field #undef _ }; u32 tunnel_sw_if_index; int rv = vnet_vxlan_add_del_tunnel (&a, &tunnel_sw_if_index); switch (rv) { case 0: if (is_add) vlib_cli_output (vm, "%U\n", format_vnet_sw_if_index_name, vnet_get_main (), tunnel_sw_if_index); break; case VNET_API_ERROR_TUNNEL_EXIST: return clib_error_return (0, "tunnel already exists..."); case VNET_API_ERROR_NO_SUCH_ENTRY: return clib_error_return (0, "tunnel does not exist..."); case VNET_API_ERROR_INSTANCE_IN_USE: return clib_error_return (0, "Instance is in use"); default: return clib_error_return (0, "vnet_vxlan_add_del_tunnel returned %d", rv); } return 0; } /*? * Add or delete a VXLAN Tunnel. * * VXLAN provides the features needed to allow L2 bridge domains (BDs) * to span multiple servers. This is done by building an L2 overlay on * top of an L3 network underlay using VXLAN tunnels. * * This makes it possible for servers to be co-located in the same data * center or be separated geographically as long as they are reachable * through the underlay L3 network. * * You can refer to this kind of L2 overlay bridge domain as a VXLAN * (Virtual eXtensible VLAN) segment. * * @cliexpar * Example of how to create a VXLAN Tunnel: * @cliexcmd{create vxlan tunnel src 10.0.3.1 dst 10.0.3.3 vni 13 encap-vrf-id 7} * Example of how to create a VXLAN Tunnel with a known name, vxlan_tunnel42: * @cliexcmd{create vxlan tunnel src 10.0.3.1 dst 10.0.3.3 instance 42} * Example of how to create a multicast VXLAN Tunnel with a known name, vxlan_tunnel23: * @cliexcmd{create vxlan tunnel src 10.0.3.1 group 239.1.1.1 GigabitEthernet0/8/0 instance 23} * Example of how to delete a VXLAN Tunnel: * @cliexcmd{create vxlan tunnel src 10.0.3.1 dst 10.0.3.3 vni 13 del} ?*/ /* *INDENT-OFF* */ VLIB_CLI_COMMAND (create_vxlan_tunnel_command, static) = { .path = "create vxlan tunnel", .short_help = "create vxlan tunnel src <local-vtep-addr>" " {dst <remote-vtep-addr>|group <mcast-vtep-addr> <intf-name>} vni <nn>" " [instance <id>]" " [encap-vrf-id <nn>] [decap-next [l2|node <name>]] [del]", .function = vxlan_add_del_tunnel_command_fn, }; /* *INDENT-ON* */ static clib_error_t * show_vxlan_tunnel_command_fn (vlib_main_t * vm, unformat_input_t * input, vlib_cli_command_t * cmd) { vxlan_main_t *vxm = &vxlan_main; vxlan_tunnel_t *t; int raw = 0; while (unformat_check_input (input) != UNFORMAT_END_OF_INPUT) { if (unformat (input, "raw")) raw = 1; else return clib_error_return (0, "parse error: '%U'", format_unformat_error, input); } if (pool_elts (vxm->tunnels) == 0) vlib_cli_output (vm, "No vxlan tunnels configured..."); /* *INDENT-OFF* */ pool_foreach (t, vxm->tunnels, ({ vlib_cli_output (vm, "%U", format_vxlan_tunnel, t); })); /* *INDENT-ON* */ if (raw) { vlib_cli_output (vm, "Raw IPv4 Hash Table:\n%U\n", format_bihash_16_8, &vxm->vxlan4_tunnel_by_key, 1 /* verbose */ ); vlib_cli_output (vm, "Raw IPv6 Hash Table:\n%U\n", format_bihash_24_8, &vxm->vxlan6_tunnel_by_key, 1 /* verbose */ ); } return 0; } /*? * Display all the VXLAN Tunnel entries. * * @cliexpar * Example of how to display the VXLAN Tunnel entries: * @cliexstart{show vxlan tunnel} * [0] src 10.0.3.1 dst 10.0.3.3 vni 13 encap_fib_index 0 sw_if_index 5 decap_next l2 * @cliexend ?*/ /* *INDENT-OFF* */ VLIB_CLI_COMMAND (show_vxlan_tunnel_command, static) = { .path = "show vxlan tunnel", .short_help = "show vxlan tunnel [raw]", .function = show_vxlan_tunnel_command_fn, }; /* *INDENT-ON* */ void vnet_int_vxlan_bypass_mode (u32 sw_if_index, u8 is_ip6, u8 is_enable) { vxlan_main_t *vxm = &vxlan_main; if (pool_is_free_index (vxm->vnet_main->interface_main.sw_interfaces, sw_if_index)) return; is_enable = ! !is_enable; if (is_ip6) { if (clib_bitmap_get (vxm->bm_ip6_bypass_enabled_by_sw_if, sw_if_index) != is_enable) { vnet_feature_enable_disable ("ip6-unicast", "ip6-vxlan-bypass", sw_if_index, is_enable, 0, 0); vxm->bm_ip6_bypass_enabled_by_sw_if = clib_bitmap_set (vxm->bm_ip6_bypass_enabled_by_sw_if, sw_if_index, is_enable); } } else { if (clib_bitmap_get (vxm->bm_ip4_bypass_enabled_by_sw_if, sw_if_index) != is_enable) { vnet_feature_enable_disable ("ip4-unicast", "ip4-vxlan-bypass", sw_if_index, is_enable, 0, 0); vxm->bm_ip4_bypass_enabled_by_sw_if = clib_bitmap_set (vxm->bm_ip4_bypass_enabled_by_sw_if, sw_if_index, is_enable); } } } static clib_error_t * set_ip_vxlan_bypass (u32 is_ip6, unformat_input_t * input, vlib_cli_command_t * cmd) { unformat_input_t _line_input, *line_input = &_line_input; vnet_main_t *vnm = vnet_get_main (); clib_error_t *error = 0; u32 sw_if_index, is_enable; sw_if_index = ~0; is_enable = 1; if (!unformat_user (input, unformat_line_input, line_input)) return 0; while (unformat_check_input (line_input) != UNFORMAT_END_OF_INPUT) { if (unformat_user (line_input, unformat_vnet_sw_interface, vnm, &sw_if_index)) ; else if (unformat (line_input, "del")) is_enable = 0; else { error = unformat_parse_error (line_input); goto done; } } if (~0 == sw_if_index) { error = clib_error_return (0, "unknown interface `%U'", format_unformat_error, line_input); goto done; } vnet_int_vxlan_bypass_mode (sw_if_index, is_ip6, is_enable); done: unformat_free (line_input); return error; } static clib_error_t * set_ip4_vxlan_bypass (vlib_main_t * vm, unformat_input_t * input, vlib_cli_command_t * cmd) { return set_ip_vxlan_bypass (0, input, cmd); } /*? * This command adds the 'ip4-vxlan-bypass' graph node for a given interface. * By adding the IPv4 vxlan-bypass graph node to an interface, the node checks * for and validate input vxlan packet and bypass ip4-lookup, ip4-local, * ip4-udp-lookup nodes to speedup vxlan packet forwarding. This node will * cause extra overhead to for non-vxlan packets which is kept at a minimum. * * @cliexpar * @parblock * Example of graph node before ip4-vxlan-bypass is enabled: * @cliexstart{show vlib graph ip4-vxlan-bypass} * Name Next Previous * ip4-vxlan-bypass error-drop [0] * vxlan4-input [1] * ip4-lookup [2] * @cliexend * * Example of how to enable ip4-vxlan-bypass on an interface: * @cliexcmd{set interface ip vxlan-bypass GigabitEthernet2/0/0} * * Example of graph node after ip4-vxlan-bypass is enabled: * @cliexstart{show vlib graph ip4-vxlan-bypass} * Name Next Previous * ip4-vxlan-bypass error-drop [0] ip4-input * vxlan4-input [1] ip4-input-no-checksum * ip4-lookup [2] * @cliexend * * Example of how to display the feature enabled on an interface: * @cliexstart{show ip interface features GigabitEthernet2/0/0} * IP feature paths configured on GigabitEthernet2/0/0... * ... * ipv4 unicast: * ip4-vxlan-bypass * ip4-lookup * ... * @cliexend * * Example of how to disable ip4-vxlan-bypass on an interface: * @cliexcmd{set interface ip vxlan-bypass GigabitEthernet2/0/0 del} * @endparblock ?*/ /* *INDENT-OFF* */ VLIB_CLI_COMMAND (set_interface_ip_vxlan_bypass_command, static) = { .path = "set interface ip vxlan-bypass", .function = set_ip4_vxlan_bypass, .short_help = "set interface ip vxlan-bypass <interface> [del]", }; /* *INDENT-ON* */ static clib_error_t * set_ip6_vxlan_bypass (vlib_main_t * vm, unformat_input_t * input, vlib_cli_command_t * cmd) { return set_ip_vxlan_bypass (1, input, cmd); } /*? * This command adds the 'ip6-vxlan-bypass' graph node for a given interface. * By adding the IPv6 vxlan-bypass graph node to an interface, the node checks * for and validate input vxlan packet and bypass ip6-lookup, ip6-local, * ip6-udp-lookup nodes to speedup vxlan packet forwarding. This node will * cause extra overhead to for non-vxlan packets which is kept at a minimum. * * @cliexpar * @parblock * Example of graph node before ip6-vxlan-bypass is enabled: * @cliexstart{show vlib graph ip6-vxlan-bypass} * Name Next Previous * ip6-vxlan-bypass error-drop [0] * vxlan6-input [1] * ip6-lookup [2] * @cliexend * * Example of how to enable ip6-vxlan-bypass on an interface: * @cliexcmd{set interface ip6 vxlan-bypass GigabitEthernet2/0/0} * * Example of graph node after ip6-vxlan-bypass is enabled: * @cliexstart{show vlib graph ip6-vxlan-bypass} * Name Next Previous * ip6-vxlan-bypass error-drop [0] ip6-input * vxlan6-input [1] ip4-input-no-checksum * ip6-lookup [2] * @cliexend * * Example of how to display the feature enabled on an interface: * @cliexstart{show ip interface features GigabitEthernet2/0/0} * IP feature paths configured on GigabitEthernet2/0/0... * ... * ipv6 unicast: * ip6-vxlan-bypass * ip6-lookup * ... * @cliexend * * Example of how to disable ip6-vxlan-bypass on an interface: * @cliexcmd{set interface ip6 vxlan-bypass GigabitEthernet2/0/0 del} * @endparblock ?*/ /* *INDENT-OFF* */ VLIB_CLI_COMMAND (set_interface_ip6_vxlan_bypass_command, static) = { .path = "set interface ip6 vxlan-bypass", .function = set_ip6_vxlan_bypass, .short_help = "set interface ip vxlan-bypass <interface> [del]", }; /* *INDENT-ON* */ int vnet_vxlan_add_del_rx_flow (u32 hw_if_index, u32 t_index, int is_add) { vxlan_main_t *vxm = &vxlan_main; vxlan_tunnel_t *t = pool_elt_at_index (vxm->tunnels, t_index); vnet_main_t *vnm = vnet_get_main (); if (is_add) { if (t->flow_index == ~0) { vxlan_main_t *vxm = &vxlan_main; vnet_flow_t flow = { .actions = VNET_FLOW_ACTION_REDIRECT_TO_NODE | VNET_FLOW_ACTION_MARK, .mark_flow_id = t->dev_instance + vxm->flow_id_start, .redirect_node_index = vxlan4_flow_input_node.index, .type = VNET_FLOW_TYPE_IP4_VXLAN, .ip4_vxlan = { .src_addr = t->dst.ip4, .dst_addr = t->src.ip4, .dst_port = UDP_DST_PORT_vxlan, .vni = t->vni, } , }; vnet_flow_add (vnm, &flow, &t->flow_index); } return vnet_flow_enable (vnm, t->flow_index, hw_if_index); } /* flow index is removed when the tunnel is deleted */ return vnet_flow_disable (vnm, t->flow_index, hw_if_index); } u32 vnet_vxlan_get_tunnel_index (u32 sw_if_index) { vxlan_main_t *vxm = &vxlan_main; if (sw_if_index >= vec_len (vxm->tunnel_index_by_sw_if_index)) return ~0; return vxm->tunnel_index_by_sw_if_index[sw_if_index]; } static clib_error_t * vxlan_offload_command_fn (vlib_main_t * vm, unformat_input_t * input, vlib_cli_command_t * cmd) { unformat_input_t _line_input, *line_input = &_line_input; /* Get a line of input. */ if (!unformat_user (input, unformat_line_input, line_input)) return 0; vnet_main_t *vnm = vnet_get_main (); u32 rx_sw_if_index = ~0; u32 hw_if_index = ~0; int is_add = 1; while (unformat_check_input (line_input) != UNFORMAT_END_OF_INPUT) { if (unformat (line_input, "hw %U", unformat_vnet_hw_interface, vnm, &hw_if_index)) continue; if (unformat (line_input, "rx %U", unformat_vnet_sw_interface, vnm, &rx_sw_if_index)) continue; if (unformat (line_input, "del")) { is_add = 0; continue; } return clib_error_return (0, "unknown input `%U'", format_unformat_error, line_input); } if (rx_sw_if_index == ~0) return clib_error_return (0, "missing rx interface"); if (hw_if_index == ~0) return clib_error_return (0, "missing hw interface"); u32 t_index = vnet_vxlan_get_tunnel_index (rx_sw_if_index);; if (t_index == ~0) return clib_error_return (0, "%U is not a vxlan tunnel", format_vnet_sw_if_index_name, vnm, rx_sw_if_index); vxlan_main_t *vxm = &vxlan_main; vxlan_tunnel_t *t = pool_elt_at_index (vxm->tunnels, t_index); if (!ip46_address_is_ip4 (&t->dst)) return clib_error_return (0, "currently only IPV4 tunnels are supported"); vnet_hw_interface_t *hw_if = vnet_get_hw_interface (vnm, hw_if_index); ip4_main_t *im = &ip4_main; u32 rx_fib_index = vec_elt (im->fib_index_by_sw_if_index, hw_if->sw_if_index); if (t->encap_fib_index != rx_fib_index) return clib_error_return (0, "interface/tunnel fib mismatch"); if (vnet_vxlan_add_del_rx_flow (hw_if_index, t_index, is_add)) return clib_error_return (0, "error %s flow", is_add ? "enabling" : "disabling"); return 0; } /* *INDENT-OFF* */ VLIB_CLI_COMMAND (vxlan_offload_command, static) = { .path = "set flow-offload vxlan", .short_help = "set flow-offload vxlan hw <interface-name> rx <tunnel-name> [del]", .function = vxlan_offload_command_fn, }; /* *INDENT-ON* */ #define VXLAN_HASH_NUM_BUCKETS (2 * 1024) #define VXLAN_HASH_MEMORY_SIZE (1 << 20) clib_error_t * vxlan_init (vlib_main_t * vm) { vxlan_main_t *vxm = &vxlan_main; vxm->vnet_main = vnet_get_main (); vxm->vlib_main = vm; vnet_flow_get_range (vxm->vnet_main, "vxlan", 1024 * 1024, &vxm->flow_id_start); vxm->bm_ip4_bypass_enabled_by_sw_if = 0; vxm->bm_ip6_bypass_enabled_by_sw_if = 0; /* initialize the ip6 hash */ clib_bihash_init_16_8 (&vxm->vxlan4_tunnel_by_key, "vxlan4", VXLAN_HASH_NUM_BUCKETS, VXLAN_HASH_MEMORY_SIZE); clib_bihash_init_24_8 (&vxm->vxlan6_tunnel_by_key, "vxlan6", VXLAN_HASH_NUM_BUCKETS, VXLAN_HASH_MEMORY_SIZE); vxm->vtep6 = hash_create_mem (0, sizeof (ip6_address_t), sizeof (uword)); vxm->mcast_shared = hash_create_mem (0, sizeof (ip46_address_t), sizeof (mcast_shared_t)); udp_register_dst_port (vm, UDP_DST_PORT_vxlan, vxlan4_input_node.index, /* is_ip4 */ 1); udp_register_dst_port (vm, UDP_DST_PORT_vxlan6, vxlan6_input_node.index, /* is_ip4 */ 0); fib_node_register_type (FIB_NODE_TYPE_VXLAN_TUNNEL, &vxlan_vft); return 0; } VLIB_INIT_FUNCTION (vxlan_init); /* * fd.io coding-style-patch-verification: ON * * Local Variables: * eval: (c-set-style "gnu") * End: */