xref: /openbmc/linux/net/sched/cls_flower.c (revision 28efb0046512e8a13ed9f9bdf0d68d10bbfbe9cf)
1 /*
2  * net/sched/cls_flower.c		Flower classifier
3  *
4  * Copyright (c) 2015 Jiri Pirko <jiri@resnulli.us>
5  *
6  * This program is free software; you can redistribute it and/or modify
7  * it under the terms of the GNU General Public License as published by
8  * the Free Software Foundation; either version 2 of the License, or
9  * (at your option) any later version.
10  */
11 
12 #include <linux/kernel.h>
13 #include <linux/init.h>
14 #include <linux/module.h>
15 #include <linux/rhashtable.h>
16 #include <linux/workqueue.h>
17 
18 #include <linux/if_ether.h>
19 #include <linux/in6.h>
20 #include <linux/ip.h>
21 #include <linux/mpls.h>
22 
23 #include <net/sch_generic.h>
24 #include <net/pkt_cls.h>
25 #include <net/ip.h>
26 #include <net/flow_dissector.h>
27 
28 #include <net/dst.h>
29 #include <net/dst_metadata.h>
30 
31 struct fl_flow_key {
32 	int	indev_ifindex;
33 	struct flow_dissector_key_control control;
34 	struct flow_dissector_key_control enc_control;
35 	struct flow_dissector_key_basic basic;
36 	struct flow_dissector_key_eth_addrs eth;
37 	struct flow_dissector_key_vlan vlan;
38 	union {
39 		struct flow_dissector_key_ipv4_addrs ipv4;
40 		struct flow_dissector_key_ipv6_addrs ipv6;
41 	};
42 	struct flow_dissector_key_ports tp;
43 	struct flow_dissector_key_icmp icmp;
44 	struct flow_dissector_key_arp arp;
45 	struct flow_dissector_key_keyid enc_key_id;
46 	union {
47 		struct flow_dissector_key_ipv4_addrs enc_ipv4;
48 		struct flow_dissector_key_ipv6_addrs enc_ipv6;
49 	};
50 	struct flow_dissector_key_ports enc_tp;
51 	struct flow_dissector_key_mpls mpls;
52 	struct flow_dissector_key_tcp tcp;
53 	struct flow_dissector_key_ip ip;
54 } __aligned(BITS_PER_LONG / 8); /* Ensure that we can do comparisons as longs. */
55 
56 struct fl_flow_mask_range {
57 	unsigned short int start;
58 	unsigned short int end;
59 };
60 
61 struct fl_flow_mask {
62 	struct fl_flow_key key;
63 	struct fl_flow_mask_range range;
64 	struct rcu_head	rcu;
65 };
66 
67 struct cls_fl_head {
68 	struct rhashtable ht;
69 	struct fl_flow_mask mask;
70 	struct flow_dissector dissector;
71 	bool mask_assigned;
72 	struct list_head filters;
73 	struct rhashtable_params ht_params;
74 	union {
75 		struct work_struct work;
76 		struct rcu_head	rcu;
77 	};
78 	struct idr handle_idr;
79 };
80 
81 struct cls_fl_filter {
82 	struct rhash_head ht_node;
83 	struct fl_flow_key mkey;
84 	struct tcf_exts exts;
85 	struct tcf_result res;
86 	struct fl_flow_key key;
87 	struct list_head list;
88 	u32 handle;
89 	u32 flags;
90 	struct rcu_head	rcu;
91 	struct net_device *hw_dev;
92 };
93 
94 static unsigned short int fl_mask_range(const struct fl_flow_mask *mask)
95 {
96 	return mask->range.end - mask->range.start;
97 }
98 
99 static void fl_mask_update_range(struct fl_flow_mask *mask)
100 {
101 	const u8 *bytes = (const u8 *) &mask->key;
102 	size_t size = sizeof(mask->key);
103 	size_t i, first = 0, last = size - 1;
104 
105 	for (i = 0; i < sizeof(mask->key); i++) {
106 		if (bytes[i]) {
107 			if (!first && i)
108 				first = i;
109 			last = i;
110 		}
111 	}
112 	mask->range.start = rounddown(first, sizeof(long));
113 	mask->range.end = roundup(last + 1, sizeof(long));
114 }
115 
116 static void *fl_key_get_start(struct fl_flow_key *key,
117 			      const struct fl_flow_mask *mask)
118 {
119 	return (u8 *) key + mask->range.start;
120 }
121 
122 static void fl_set_masked_key(struct fl_flow_key *mkey, struct fl_flow_key *key,
123 			      struct fl_flow_mask *mask)
124 {
125 	const long *lkey = fl_key_get_start(key, mask);
126 	const long *lmask = fl_key_get_start(&mask->key, mask);
127 	long *lmkey = fl_key_get_start(mkey, mask);
128 	int i;
129 
130 	for (i = 0; i < fl_mask_range(mask); i += sizeof(long))
131 		*lmkey++ = *lkey++ & *lmask++;
132 }
133 
134 static void fl_clear_masked_range(struct fl_flow_key *key,
135 				  struct fl_flow_mask *mask)
136 {
137 	memset(fl_key_get_start(key, mask), 0, fl_mask_range(mask));
138 }
139 
140 static struct cls_fl_filter *fl_lookup(struct cls_fl_head *head,
141 				       struct fl_flow_key *mkey)
142 {
143 	return rhashtable_lookup_fast(&head->ht,
144 				      fl_key_get_start(mkey, &head->mask),
145 				      head->ht_params);
146 }
147 
148 static int fl_classify(struct sk_buff *skb, const struct tcf_proto *tp,
149 		       struct tcf_result *res)
150 {
151 	struct cls_fl_head *head = rcu_dereference_bh(tp->root);
152 	struct cls_fl_filter *f;
153 	struct fl_flow_key skb_key;
154 	struct fl_flow_key skb_mkey;
155 
156 	if (!atomic_read(&head->ht.nelems))
157 		return -1;
158 
159 	fl_clear_masked_range(&skb_key, &head->mask);
160 
161 	skb_key.indev_ifindex = skb->skb_iif;
162 	/* skb_flow_dissect() does not set n_proto in case an unknown protocol,
163 	 * so do it rather here.
164 	 */
165 	skb_key.basic.n_proto = skb->protocol;
166 	skb_flow_dissect(skb, &head->dissector, &skb_key, 0);
167 
168 	fl_set_masked_key(&skb_mkey, &skb_key, &head->mask);
169 
170 	f = fl_lookup(head, &skb_mkey);
171 	if (f && !tc_skip_sw(f->flags)) {
172 		*res = f->res;
173 		return tcf_exts_exec(skb, &f->exts, res);
174 	}
175 	return -1;
176 }
177 
178 static int fl_init(struct tcf_proto *tp)
179 {
180 	struct cls_fl_head *head;
181 
182 	head = kzalloc(sizeof(*head), GFP_KERNEL);
183 	if (!head)
184 		return -ENOBUFS;
185 
186 	INIT_LIST_HEAD_RCU(&head->filters);
187 	rcu_assign_pointer(tp->root, head);
188 	idr_init(&head->handle_idr);
189 
190 	return 0;
191 }
192 
193 static void fl_destroy_filter(struct rcu_head *head)
194 {
195 	struct cls_fl_filter *f = container_of(head, struct cls_fl_filter, rcu);
196 
197 	tcf_exts_destroy(&f->exts);
198 	kfree(f);
199 }
200 
201 static void fl_hw_destroy_filter(struct tcf_proto *tp, struct cls_fl_filter *f)
202 {
203 	struct tc_cls_flower_offload cls_flower = {};
204 	struct net_device *dev = f->hw_dev;
205 
206 	if (!tc_can_offload(dev))
207 		return;
208 
209 	tc_cls_common_offload_init(&cls_flower.common, tp);
210 	cls_flower.command = TC_CLSFLOWER_DESTROY;
211 	cls_flower.cookie = (unsigned long) f;
212 
213 	dev->netdev_ops->ndo_setup_tc(dev, TC_SETUP_CLSFLOWER, &cls_flower);
214 }
215 
216 static int fl_hw_replace_filter(struct tcf_proto *tp,
217 				struct flow_dissector *dissector,
218 				struct fl_flow_key *mask,
219 				struct cls_fl_filter *f)
220 {
221 	struct net_device *dev = tp->q->dev_queue->dev;
222 	struct tc_cls_flower_offload cls_flower = {};
223 	int err;
224 
225 	if (!tc_can_offload(dev)) {
226 		if (tcf_exts_get_dev(dev, &f->exts, &f->hw_dev) ||
227 		    (f->hw_dev && !tc_can_offload(f->hw_dev))) {
228 			f->hw_dev = dev;
229 			return tc_skip_sw(f->flags) ? -EINVAL : 0;
230 		}
231 		dev = f->hw_dev;
232 		cls_flower.egress_dev = true;
233 	} else {
234 		f->hw_dev = dev;
235 	}
236 
237 	tc_cls_common_offload_init(&cls_flower.common, tp);
238 	cls_flower.command = TC_CLSFLOWER_REPLACE;
239 	cls_flower.cookie = (unsigned long) f;
240 	cls_flower.dissector = dissector;
241 	cls_flower.mask = mask;
242 	cls_flower.key = &f->mkey;
243 	cls_flower.exts = &f->exts;
244 
245 	err = dev->netdev_ops->ndo_setup_tc(dev, TC_SETUP_CLSFLOWER,
246 					    &cls_flower);
247 	if (!err)
248 		f->flags |= TCA_CLS_FLAGS_IN_HW;
249 
250 	if (tc_skip_sw(f->flags))
251 		return err;
252 	return 0;
253 }
254 
255 static void fl_hw_update_stats(struct tcf_proto *tp, struct cls_fl_filter *f)
256 {
257 	struct tc_cls_flower_offload cls_flower = {};
258 	struct net_device *dev = f->hw_dev;
259 
260 	if (!tc_can_offload(dev))
261 		return;
262 
263 	tc_cls_common_offload_init(&cls_flower.common, tp);
264 	cls_flower.command = TC_CLSFLOWER_STATS;
265 	cls_flower.cookie = (unsigned long) f;
266 	cls_flower.exts = &f->exts;
267 
268 	dev->netdev_ops->ndo_setup_tc(dev, TC_SETUP_CLSFLOWER,
269 				      &cls_flower);
270 }
271 
272 static void __fl_delete(struct tcf_proto *tp, struct cls_fl_filter *f)
273 {
274 	struct cls_fl_head *head = rtnl_dereference(tp->root);
275 
276 	idr_remove_ext(&head->handle_idr, f->handle);
277 	list_del_rcu(&f->list);
278 	if (!tc_skip_hw(f->flags))
279 		fl_hw_destroy_filter(tp, f);
280 	tcf_unbind_filter(tp, &f->res);
281 	call_rcu(&f->rcu, fl_destroy_filter);
282 }
283 
284 static void fl_destroy_sleepable(struct work_struct *work)
285 {
286 	struct cls_fl_head *head = container_of(work, struct cls_fl_head,
287 						work);
288 	if (head->mask_assigned)
289 		rhashtable_destroy(&head->ht);
290 	kfree(head);
291 	module_put(THIS_MODULE);
292 }
293 
294 static void fl_destroy_rcu(struct rcu_head *rcu)
295 {
296 	struct cls_fl_head *head = container_of(rcu, struct cls_fl_head, rcu);
297 
298 	INIT_WORK(&head->work, fl_destroy_sleepable);
299 	schedule_work(&head->work);
300 }
301 
302 static void fl_destroy(struct tcf_proto *tp)
303 {
304 	struct cls_fl_head *head = rtnl_dereference(tp->root);
305 	struct cls_fl_filter *f, *next;
306 
307 	list_for_each_entry_safe(f, next, &head->filters, list)
308 		__fl_delete(tp, f);
309 	idr_destroy(&head->handle_idr);
310 
311 	__module_get(THIS_MODULE);
312 	call_rcu(&head->rcu, fl_destroy_rcu);
313 }
314 
315 static void *fl_get(struct tcf_proto *tp, u32 handle)
316 {
317 	struct cls_fl_head *head = rtnl_dereference(tp->root);
318 
319 	return idr_find_ext(&head->handle_idr, handle);
320 }
321 
322 static const struct nla_policy fl_policy[TCA_FLOWER_MAX + 1] = {
323 	[TCA_FLOWER_UNSPEC]		= { .type = NLA_UNSPEC },
324 	[TCA_FLOWER_CLASSID]		= { .type = NLA_U32 },
325 	[TCA_FLOWER_INDEV]		= { .type = NLA_STRING,
326 					    .len = IFNAMSIZ },
327 	[TCA_FLOWER_KEY_ETH_DST]	= { .len = ETH_ALEN },
328 	[TCA_FLOWER_KEY_ETH_DST_MASK]	= { .len = ETH_ALEN },
329 	[TCA_FLOWER_KEY_ETH_SRC]	= { .len = ETH_ALEN },
330 	[TCA_FLOWER_KEY_ETH_SRC_MASK]	= { .len = ETH_ALEN },
331 	[TCA_FLOWER_KEY_ETH_TYPE]	= { .type = NLA_U16 },
332 	[TCA_FLOWER_KEY_IP_PROTO]	= { .type = NLA_U8 },
333 	[TCA_FLOWER_KEY_IPV4_SRC]	= { .type = NLA_U32 },
334 	[TCA_FLOWER_KEY_IPV4_SRC_MASK]	= { .type = NLA_U32 },
335 	[TCA_FLOWER_KEY_IPV4_DST]	= { .type = NLA_U32 },
336 	[TCA_FLOWER_KEY_IPV4_DST_MASK]	= { .type = NLA_U32 },
337 	[TCA_FLOWER_KEY_IPV6_SRC]	= { .len = sizeof(struct in6_addr) },
338 	[TCA_FLOWER_KEY_IPV6_SRC_MASK]	= { .len = sizeof(struct in6_addr) },
339 	[TCA_FLOWER_KEY_IPV6_DST]	= { .len = sizeof(struct in6_addr) },
340 	[TCA_FLOWER_KEY_IPV6_DST_MASK]	= { .len = sizeof(struct in6_addr) },
341 	[TCA_FLOWER_KEY_TCP_SRC]	= { .type = NLA_U16 },
342 	[TCA_FLOWER_KEY_TCP_DST]	= { .type = NLA_U16 },
343 	[TCA_FLOWER_KEY_UDP_SRC]	= { .type = NLA_U16 },
344 	[TCA_FLOWER_KEY_UDP_DST]	= { .type = NLA_U16 },
345 	[TCA_FLOWER_KEY_VLAN_ID]	= { .type = NLA_U16 },
346 	[TCA_FLOWER_KEY_VLAN_PRIO]	= { .type = NLA_U8 },
347 	[TCA_FLOWER_KEY_VLAN_ETH_TYPE]	= { .type = NLA_U16 },
348 	[TCA_FLOWER_KEY_ENC_KEY_ID]	= { .type = NLA_U32 },
349 	[TCA_FLOWER_KEY_ENC_IPV4_SRC]	= { .type = NLA_U32 },
350 	[TCA_FLOWER_KEY_ENC_IPV4_SRC_MASK] = { .type = NLA_U32 },
351 	[TCA_FLOWER_KEY_ENC_IPV4_DST]	= { .type = NLA_U32 },
352 	[TCA_FLOWER_KEY_ENC_IPV4_DST_MASK] = { .type = NLA_U32 },
353 	[TCA_FLOWER_KEY_ENC_IPV6_SRC]	= { .len = sizeof(struct in6_addr) },
354 	[TCA_FLOWER_KEY_ENC_IPV6_SRC_MASK] = { .len = sizeof(struct in6_addr) },
355 	[TCA_FLOWER_KEY_ENC_IPV6_DST]	= { .len = sizeof(struct in6_addr) },
356 	[TCA_FLOWER_KEY_ENC_IPV6_DST_MASK] = { .len = sizeof(struct in6_addr) },
357 	[TCA_FLOWER_KEY_TCP_SRC_MASK]	= { .type = NLA_U16 },
358 	[TCA_FLOWER_KEY_TCP_DST_MASK]	= { .type = NLA_U16 },
359 	[TCA_FLOWER_KEY_UDP_SRC_MASK]	= { .type = NLA_U16 },
360 	[TCA_FLOWER_KEY_UDP_DST_MASK]	= { .type = NLA_U16 },
361 	[TCA_FLOWER_KEY_SCTP_SRC_MASK]	= { .type = NLA_U16 },
362 	[TCA_FLOWER_KEY_SCTP_DST_MASK]	= { .type = NLA_U16 },
363 	[TCA_FLOWER_KEY_SCTP_SRC]	= { .type = NLA_U16 },
364 	[TCA_FLOWER_KEY_SCTP_DST]	= { .type = NLA_U16 },
365 	[TCA_FLOWER_KEY_ENC_UDP_SRC_PORT]	= { .type = NLA_U16 },
366 	[TCA_FLOWER_KEY_ENC_UDP_SRC_PORT_MASK]	= { .type = NLA_U16 },
367 	[TCA_FLOWER_KEY_ENC_UDP_DST_PORT]	= { .type = NLA_U16 },
368 	[TCA_FLOWER_KEY_ENC_UDP_DST_PORT_MASK]	= { .type = NLA_U16 },
369 	[TCA_FLOWER_KEY_FLAGS]		= { .type = NLA_U32 },
370 	[TCA_FLOWER_KEY_FLAGS_MASK]	= { .type = NLA_U32 },
371 	[TCA_FLOWER_KEY_ICMPV4_TYPE]	= { .type = NLA_U8 },
372 	[TCA_FLOWER_KEY_ICMPV4_TYPE_MASK] = { .type = NLA_U8 },
373 	[TCA_FLOWER_KEY_ICMPV4_CODE]	= { .type = NLA_U8 },
374 	[TCA_FLOWER_KEY_ICMPV4_CODE_MASK] = { .type = NLA_U8 },
375 	[TCA_FLOWER_KEY_ICMPV6_TYPE]	= { .type = NLA_U8 },
376 	[TCA_FLOWER_KEY_ICMPV6_TYPE_MASK] = { .type = NLA_U8 },
377 	[TCA_FLOWER_KEY_ICMPV6_CODE]	= { .type = NLA_U8 },
378 	[TCA_FLOWER_KEY_ICMPV6_CODE_MASK] = { .type = NLA_U8 },
379 	[TCA_FLOWER_KEY_ARP_SIP]	= { .type = NLA_U32 },
380 	[TCA_FLOWER_KEY_ARP_SIP_MASK]	= { .type = NLA_U32 },
381 	[TCA_FLOWER_KEY_ARP_TIP]	= { .type = NLA_U32 },
382 	[TCA_FLOWER_KEY_ARP_TIP_MASK]	= { .type = NLA_U32 },
383 	[TCA_FLOWER_KEY_ARP_OP]		= { .type = NLA_U8 },
384 	[TCA_FLOWER_KEY_ARP_OP_MASK]	= { .type = NLA_U8 },
385 	[TCA_FLOWER_KEY_ARP_SHA]	= { .len = ETH_ALEN },
386 	[TCA_FLOWER_KEY_ARP_SHA_MASK]	= { .len = ETH_ALEN },
387 	[TCA_FLOWER_KEY_ARP_THA]	= { .len = ETH_ALEN },
388 	[TCA_FLOWER_KEY_ARP_THA_MASK]	= { .len = ETH_ALEN },
389 	[TCA_FLOWER_KEY_MPLS_TTL]	= { .type = NLA_U8 },
390 	[TCA_FLOWER_KEY_MPLS_BOS]	= { .type = NLA_U8 },
391 	[TCA_FLOWER_KEY_MPLS_TC]	= { .type = NLA_U8 },
392 	[TCA_FLOWER_KEY_MPLS_LABEL]	= { .type = NLA_U32 },
393 	[TCA_FLOWER_KEY_TCP_FLAGS]	= { .type = NLA_U16 },
394 	[TCA_FLOWER_KEY_TCP_FLAGS_MASK]	= { .type = NLA_U16 },
395 	[TCA_FLOWER_KEY_IP_TOS]		= { .type = NLA_U8 },
396 	[TCA_FLOWER_KEY_IP_TOS_MASK]	= { .type = NLA_U8 },
397 	[TCA_FLOWER_KEY_IP_TTL]		= { .type = NLA_U8 },
398 	[TCA_FLOWER_KEY_IP_TTL_MASK]	= { .type = NLA_U8 },
399 };
400 
401 static void fl_set_key_val(struct nlattr **tb,
402 			   void *val, int val_type,
403 			   void *mask, int mask_type, int len)
404 {
405 	if (!tb[val_type])
406 		return;
407 	memcpy(val, nla_data(tb[val_type]), len);
408 	if (mask_type == TCA_FLOWER_UNSPEC || !tb[mask_type])
409 		memset(mask, 0xff, len);
410 	else
411 		memcpy(mask, nla_data(tb[mask_type]), len);
412 }
413 
414 static int fl_set_key_mpls(struct nlattr **tb,
415 			   struct flow_dissector_key_mpls *key_val,
416 			   struct flow_dissector_key_mpls *key_mask)
417 {
418 	if (tb[TCA_FLOWER_KEY_MPLS_TTL]) {
419 		key_val->mpls_ttl = nla_get_u8(tb[TCA_FLOWER_KEY_MPLS_TTL]);
420 		key_mask->mpls_ttl = MPLS_TTL_MASK;
421 	}
422 	if (tb[TCA_FLOWER_KEY_MPLS_BOS]) {
423 		u8 bos = nla_get_u8(tb[TCA_FLOWER_KEY_MPLS_BOS]);
424 
425 		if (bos & ~MPLS_BOS_MASK)
426 			return -EINVAL;
427 		key_val->mpls_bos = bos;
428 		key_mask->mpls_bos = MPLS_BOS_MASK;
429 	}
430 	if (tb[TCA_FLOWER_KEY_MPLS_TC]) {
431 		u8 tc = nla_get_u8(tb[TCA_FLOWER_KEY_MPLS_TC]);
432 
433 		if (tc & ~MPLS_TC_MASK)
434 			return -EINVAL;
435 		key_val->mpls_tc = tc;
436 		key_mask->mpls_tc = MPLS_TC_MASK;
437 	}
438 	if (tb[TCA_FLOWER_KEY_MPLS_LABEL]) {
439 		u32 label = nla_get_u32(tb[TCA_FLOWER_KEY_MPLS_LABEL]);
440 
441 		if (label & ~MPLS_LABEL_MASK)
442 			return -EINVAL;
443 		key_val->mpls_label = label;
444 		key_mask->mpls_label = MPLS_LABEL_MASK;
445 	}
446 	return 0;
447 }
448 
449 static void fl_set_key_vlan(struct nlattr **tb,
450 			    struct flow_dissector_key_vlan *key_val,
451 			    struct flow_dissector_key_vlan *key_mask)
452 {
453 #define VLAN_PRIORITY_MASK	0x7
454 
455 	if (tb[TCA_FLOWER_KEY_VLAN_ID]) {
456 		key_val->vlan_id =
457 			nla_get_u16(tb[TCA_FLOWER_KEY_VLAN_ID]) & VLAN_VID_MASK;
458 		key_mask->vlan_id = VLAN_VID_MASK;
459 	}
460 	if (tb[TCA_FLOWER_KEY_VLAN_PRIO]) {
461 		key_val->vlan_priority =
462 			nla_get_u8(tb[TCA_FLOWER_KEY_VLAN_PRIO]) &
463 			VLAN_PRIORITY_MASK;
464 		key_mask->vlan_priority = VLAN_PRIORITY_MASK;
465 	}
466 }
467 
468 static void fl_set_key_flag(u32 flower_key, u32 flower_mask,
469 			    u32 *dissector_key, u32 *dissector_mask,
470 			    u32 flower_flag_bit, u32 dissector_flag_bit)
471 {
472 	if (flower_mask & flower_flag_bit) {
473 		*dissector_mask |= dissector_flag_bit;
474 		if (flower_key & flower_flag_bit)
475 			*dissector_key |= dissector_flag_bit;
476 	}
477 }
478 
479 static int fl_set_key_flags(struct nlattr **tb,
480 			    u32 *flags_key, u32 *flags_mask)
481 {
482 	u32 key, mask;
483 
484 	/* mask is mandatory for flags */
485 	if (!tb[TCA_FLOWER_KEY_FLAGS_MASK])
486 		return -EINVAL;
487 
488 	key = be32_to_cpu(nla_get_u32(tb[TCA_FLOWER_KEY_FLAGS]));
489 	mask = be32_to_cpu(nla_get_u32(tb[TCA_FLOWER_KEY_FLAGS_MASK]));
490 
491 	*flags_key  = 0;
492 	*flags_mask = 0;
493 
494 	fl_set_key_flag(key, mask, flags_key, flags_mask,
495 			TCA_FLOWER_KEY_FLAGS_IS_FRAGMENT, FLOW_DIS_IS_FRAGMENT);
496 
497 	return 0;
498 }
499 
500 static void fl_set_key_ip(struct nlattr **tb,
501 			  struct flow_dissector_key_ip *key,
502 			  struct flow_dissector_key_ip *mask)
503 {
504 		fl_set_key_val(tb, &key->tos, TCA_FLOWER_KEY_IP_TOS,
505 			       &mask->tos, TCA_FLOWER_KEY_IP_TOS_MASK,
506 			       sizeof(key->tos));
507 
508 		fl_set_key_val(tb, &key->ttl, TCA_FLOWER_KEY_IP_TTL,
509 			       &mask->ttl, TCA_FLOWER_KEY_IP_TTL_MASK,
510 			       sizeof(key->ttl));
511 }
512 
513 static int fl_set_key(struct net *net, struct nlattr **tb,
514 		      struct fl_flow_key *key, struct fl_flow_key *mask)
515 {
516 	__be16 ethertype;
517 	int ret = 0;
518 #ifdef CONFIG_NET_CLS_IND
519 	if (tb[TCA_FLOWER_INDEV]) {
520 		int err = tcf_change_indev(net, tb[TCA_FLOWER_INDEV]);
521 		if (err < 0)
522 			return err;
523 		key->indev_ifindex = err;
524 		mask->indev_ifindex = 0xffffffff;
525 	}
526 #endif
527 
528 	fl_set_key_val(tb, key->eth.dst, TCA_FLOWER_KEY_ETH_DST,
529 		       mask->eth.dst, TCA_FLOWER_KEY_ETH_DST_MASK,
530 		       sizeof(key->eth.dst));
531 	fl_set_key_val(tb, key->eth.src, TCA_FLOWER_KEY_ETH_SRC,
532 		       mask->eth.src, TCA_FLOWER_KEY_ETH_SRC_MASK,
533 		       sizeof(key->eth.src));
534 
535 	if (tb[TCA_FLOWER_KEY_ETH_TYPE]) {
536 		ethertype = nla_get_be16(tb[TCA_FLOWER_KEY_ETH_TYPE]);
537 
538 		if (ethertype == htons(ETH_P_8021Q)) {
539 			fl_set_key_vlan(tb, &key->vlan, &mask->vlan);
540 			fl_set_key_val(tb, &key->basic.n_proto,
541 				       TCA_FLOWER_KEY_VLAN_ETH_TYPE,
542 				       &mask->basic.n_proto, TCA_FLOWER_UNSPEC,
543 				       sizeof(key->basic.n_proto));
544 		} else {
545 			key->basic.n_proto = ethertype;
546 			mask->basic.n_proto = cpu_to_be16(~0);
547 		}
548 	}
549 
550 	if (key->basic.n_proto == htons(ETH_P_IP) ||
551 	    key->basic.n_proto == htons(ETH_P_IPV6)) {
552 		fl_set_key_val(tb, &key->basic.ip_proto, TCA_FLOWER_KEY_IP_PROTO,
553 			       &mask->basic.ip_proto, TCA_FLOWER_UNSPEC,
554 			       sizeof(key->basic.ip_proto));
555 		fl_set_key_ip(tb, &key->ip, &mask->ip);
556 	}
557 
558 	if (tb[TCA_FLOWER_KEY_IPV4_SRC] || tb[TCA_FLOWER_KEY_IPV4_DST]) {
559 		key->control.addr_type = FLOW_DISSECTOR_KEY_IPV4_ADDRS;
560 		mask->control.addr_type = ~0;
561 		fl_set_key_val(tb, &key->ipv4.src, TCA_FLOWER_KEY_IPV4_SRC,
562 			       &mask->ipv4.src, TCA_FLOWER_KEY_IPV4_SRC_MASK,
563 			       sizeof(key->ipv4.src));
564 		fl_set_key_val(tb, &key->ipv4.dst, TCA_FLOWER_KEY_IPV4_DST,
565 			       &mask->ipv4.dst, TCA_FLOWER_KEY_IPV4_DST_MASK,
566 			       sizeof(key->ipv4.dst));
567 	} else if (tb[TCA_FLOWER_KEY_IPV6_SRC] || tb[TCA_FLOWER_KEY_IPV6_DST]) {
568 		key->control.addr_type = FLOW_DISSECTOR_KEY_IPV6_ADDRS;
569 		mask->control.addr_type = ~0;
570 		fl_set_key_val(tb, &key->ipv6.src, TCA_FLOWER_KEY_IPV6_SRC,
571 			       &mask->ipv6.src, TCA_FLOWER_KEY_IPV6_SRC_MASK,
572 			       sizeof(key->ipv6.src));
573 		fl_set_key_val(tb, &key->ipv6.dst, TCA_FLOWER_KEY_IPV6_DST,
574 			       &mask->ipv6.dst, TCA_FLOWER_KEY_IPV6_DST_MASK,
575 			       sizeof(key->ipv6.dst));
576 	}
577 
578 	if (key->basic.ip_proto == IPPROTO_TCP) {
579 		fl_set_key_val(tb, &key->tp.src, TCA_FLOWER_KEY_TCP_SRC,
580 			       &mask->tp.src, TCA_FLOWER_KEY_TCP_SRC_MASK,
581 			       sizeof(key->tp.src));
582 		fl_set_key_val(tb, &key->tp.dst, TCA_FLOWER_KEY_TCP_DST,
583 			       &mask->tp.dst, TCA_FLOWER_KEY_TCP_DST_MASK,
584 			       sizeof(key->tp.dst));
585 		fl_set_key_val(tb, &key->tcp.flags, TCA_FLOWER_KEY_TCP_FLAGS,
586 			       &mask->tcp.flags, TCA_FLOWER_KEY_TCP_FLAGS_MASK,
587 			       sizeof(key->tcp.flags));
588 	} else if (key->basic.ip_proto == IPPROTO_UDP) {
589 		fl_set_key_val(tb, &key->tp.src, TCA_FLOWER_KEY_UDP_SRC,
590 			       &mask->tp.src, TCA_FLOWER_KEY_UDP_SRC_MASK,
591 			       sizeof(key->tp.src));
592 		fl_set_key_val(tb, &key->tp.dst, TCA_FLOWER_KEY_UDP_DST,
593 			       &mask->tp.dst, TCA_FLOWER_KEY_UDP_DST_MASK,
594 			       sizeof(key->tp.dst));
595 	} else if (key->basic.ip_proto == IPPROTO_SCTP) {
596 		fl_set_key_val(tb, &key->tp.src, TCA_FLOWER_KEY_SCTP_SRC,
597 			       &mask->tp.src, TCA_FLOWER_KEY_SCTP_SRC_MASK,
598 			       sizeof(key->tp.src));
599 		fl_set_key_val(tb, &key->tp.dst, TCA_FLOWER_KEY_SCTP_DST,
600 			       &mask->tp.dst, TCA_FLOWER_KEY_SCTP_DST_MASK,
601 			       sizeof(key->tp.dst));
602 	} else if (key->basic.n_proto == htons(ETH_P_IP) &&
603 		   key->basic.ip_proto == IPPROTO_ICMP) {
604 		fl_set_key_val(tb, &key->icmp.type, TCA_FLOWER_KEY_ICMPV4_TYPE,
605 			       &mask->icmp.type,
606 			       TCA_FLOWER_KEY_ICMPV4_TYPE_MASK,
607 			       sizeof(key->icmp.type));
608 		fl_set_key_val(tb, &key->icmp.code, TCA_FLOWER_KEY_ICMPV4_CODE,
609 			       &mask->icmp.code,
610 			       TCA_FLOWER_KEY_ICMPV4_CODE_MASK,
611 			       sizeof(key->icmp.code));
612 	} else if (key->basic.n_proto == htons(ETH_P_IPV6) &&
613 		   key->basic.ip_proto == IPPROTO_ICMPV6) {
614 		fl_set_key_val(tb, &key->icmp.type, TCA_FLOWER_KEY_ICMPV6_TYPE,
615 			       &mask->icmp.type,
616 			       TCA_FLOWER_KEY_ICMPV6_TYPE_MASK,
617 			       sizeof(key->icmp.type));
618 		fl_set_key_val(tb, &key->icmp.code, TCA_FLOWER_KEY_ICMPV6_CODE,
619 			       &mask->icmp.code,
620 			       TCA_FLOWER_KEY_ICMPV6_CODE_MASK,
621 			       sizeof(key->icmp.code));
622 	} else if (key->basic.n_proto == htons(ETH_P_MPLS_UC) ||
623 		   key->basic.n_proto == htons(ETH_P_MPLS_MC)) {
624 		ret = fl_set_key_mpls(tb, &key->mpls, &mask->mpls);
625 		if (ret)
626 			return ret;
627 	} else if (key->basic.n_proto == htons(ETH_P_ARP) ||
628 		   key->basic.n_proto == htons(ETH_P_RARP)) {
629 		fl_set_key_val(tb, &key->arp.sip, TCA_FLOWER_KEY_ARP_SIP,
630 			       &mask->arp.sip, TCA_FLOWER_KEY_ARP_SIP_MASK,
631 			       sizeof(key->arp.sip));
632 		fl_set_key_val(tb, &key->arp.tip, TCA_FLOWER_KEY_ARP_TIP,
633 			       &mask->arp.tip, TCA_FLOWER_KEY_ARP_TIP_MASK,
634 			       sizeof(key->arp.tip));
635 		fl_set_key_val(tb, &key->arp.op, TCA_FLOWER_KEY_ARP_OP,
636 			       &mask->arp.op, TCA_FLOWER_KEY_ARP_OP_MASK,
637 			       sizeof(key->arp.op));
638 		fl_set_key_val(tb, key->arp.sha, TCA_FLOWER_KEY_ARP_SHA,
639 			       mask->arp.sha, TCA_FLOWER_KEY_ARP_SHA_MASK,
640 			       sizeof(key->arp.sha));
641 		fl_set_key_val(tb, key->arp.tha, TCA_FLOWER_KEY_ARP_THA,
642 			       mask->arp.tha, TCA_FLOWER_KEY_ARP_THA_MASK,
643 			       sizeof(key->arp.tha));
644 	}
645 
646 	if (tb[TCA_FLOWER_KEY_ENC_IPV4_SRC] ||
647 	    tb[TCA_FLOWER_KEY_ENC_IPV4_DST]) {
648 		key->enc_control.addr_type = FLOW_DISSECTOR_KEY_IPV4_ADDRS;
649 		mask->enc_control.addr_type = ~0;
650 		fl_set_key_val(tb, &key->enc_ipv4.src,
651 			       TCA_FLOWER_KEY_ENC_IPV4_SRC,
652 			       &mask->enc_ipv4.src,
653 			       TCA_FLOWER_KEY_ENC_IPV4_SRC_MASK,
654 			       sizeof(key->enc_ipv4.src));
655 		fl_set_key_val(tb, &key->enc_ipv4.dst,
656 			       TCA_FLOWER_KEY_ENC_IPV4_DST,
657 			       &mask->enc_ipv4.dst,
658 			       TCA_FLOWER_KEY_ENC_IPV4_DST_MASK,
659 			       sizeof(key->enc_ipv4.dst));
660 	}
661 
662 	if (tb[TCA_FLOWER_KEY_ENC_IPV6_SRC] ||
663 	    tb[TCA_FLOWER_KEY_ENC_IPV6_DST]) {
664 		key->enc_control.addr_type = FLOW_DISSECTOR_KEY_IPV6_ADDRS;
665 		mask->enc_control.addr_type = ~0;
666 		fl_set_key_val(tb, &key->enc_ipv6.src,
667 			       TCA_FLOWER_KEY_ENC_IPV6_SRC,
668 			       &mask->enc_ipv6.src,
669 			       TCA_FLOWER_KEY_ENC_IPV6_SRC_MASK,
670 			       sizeof(key->enc_ipv6.src));
671 		fl_set_key_val(tb, &key->enc_ipv6.dst,
672 			       TCA_FLOWER_KEY_ENC_IPV6_DST,
673 			       &mask->enc_ipv6.dst,
674 			       TCA_FLOWER_KEY_ENC_IPV6_DST_MASK,
675 			       sizeof(key->enc_ipv6.dst));
676 	}
677 
678 	fl_set_key_val(tb, &key->enc_key_id.keyid, TCA_FLOWER_KEY_ENC_KEY_ID,
679 		       &mask->enc_key_id.keyid, TCA_FLOWER_UNSPEC,
680 		       sizeof(key->enc_key_id.keyid));
681 
682 	fl_set_key_val(tb, &key->enc_tp.src, TCA_FLOWER_KEY_ENC_UDP_SRC_PORT,
683 		       &mask->enc_tp.src, TCA_FLOWER_KEY_ENC_UDP_SRC_PORT_MASK,
684 		       sizeof(key->enc_tp.src));
685 
686 	fl_set_key_val(tb, &key->enc_tp.dst, TCA_FLOWER_KEY_ENC_UDP_DST_PORT,
687 		       &mask->enc_tp.dst, TCA_FLOWER_KEY_ENC_UDP_DST_PORT_MASK,
688 		       sizeof(key->enc_tp.dst));
689 
690 	if (tb[TCA_FLOWER_KEY_FLAGS])
691 		ret = fl_set_key_flags(tb, &key->control.flags, &mask->control.flags);
692 
693 	return ret;
694 }
695 
696 static bool fl_mask_eq(struct fl_flow_mask *mask1,
697 		       struct fl_flow_mask *mask2)
698 {
699 	const long *lmask1 = fl_key_get_start(&mask1->key, mask1);
700 	const long *lmask2 = fl_key_get_start(&mask2->key, mask2);
701 
702 	return !memcmp(&mask1->range, &mask2->range, sizeof(mask1->range)) &&
703 	       !memcmp(lmask1, lmask2, fl_mask_range(mask1));
704 }
705 
706 static const struct rhashtable_params fl_ht_params = {
707 	.key_offset = offsetof(struct cls_fl_filter, mkey), /* base offset */
708 	.head_offset = offsetof(struct cls_fl_filter, ht_node),
709 	.automatic_shrinking = true,
710 };
711 
712 static int fl_init_hashtable(struct cls_fl_head *head,
713 			     struct fl_flow_mask *mask)
714 {
715 	head->ht_params = fl_ht_params;
716 	head->ht_params.key_len = fl_mask_range(mask);
717 	head->ht_params.key_offset += mask->range.start;
718 
719 	return rhashtable_init(&head->ht, &head->ht_params);
720 }
721 
722 #define FL_KEY_MEMBER_OFFSET(member) offsetof(struct fl_flow_key, member)
723 #define FL_KEY_MEMBER_SIZE(member) (sizeof(((struct fl_flow_key *) 0)->member))
724 
725 #define FL_KEY_IS_MASKED(mask, member)						\
726 	memchr_inv(((char *)mask) + FL_KEY_MEMBER_OFFSET(member),		\
727 		   0, FL_KEY_MEMBER_SIZE(member))				\
728 
729 #define FL_KEY_SET(keys, cnt, id, member)					\
730 	do {									\
731 		keys[cnt].key_id = id;						\
732 		keys[cnt].offset = FL_KEY_MEMBER_OFFSET(member);		\
733 		cnt++;								\
734 	} while(0);
735 
736 #define FL_KEY_SET_IF_MASKED(mask, keys, cnt, id, member)			\
737 	do {									\
738 		if (FL_KEY_IS_MASKED(mask, member))				\
739 			FL_KEY_SET(keys, cnt, id, member);			\
740 	} while(0);
741 
742 static void fl_init_dissector(struct cls_fl_head *head,
743 			      struct fl_flow_mask *mask)
744 {
745 	struct flow_dissector_key keys[FLOW_DISSECTOR_KEY_MAX];
746 	size_t cnt = 0;
747 
748 	FL_KEY_SET(keys, cnt, FLOW_DISSECTOR_KEY_CONTROL, control);
749 	FL_KEY_SET(keys, cnt, FLOW_DISSECTOR_KEY_BASIC, basic);
750 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
751 			     FLOW_DISSECTOR_KEY_ETH_ADDRS, eth);
752 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
753 			     FLOW_DISSECTOR_KEY_IPV4_ADDRS, ipv4);
754 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
755 			     FLOW_DISSECTOR_KEY_IPV6_ADDRS, ipv6);
756 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
757 			     FLOW_DISSECTOR_KEY_PORTS, tp);
758 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
759 			     FLOW_DISSECTOR_KEY_IP, ip);
760 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
761 			     FLOW_DISSECTOR_KEY_TCP, tcp);
762 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
763 			     FLOW_DISSECTOR_KEY_ICMP, icmp);
764 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
765 			     FLOW_DISSECTOR_KEY_ARP, arp);
766 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
767 			     FLOW_DISSECTOR_KEY_MPLS, mpls);
768 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
769 			     FLOW_DISSECTOR_KEY_VLAN, vlan);
770 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
771 			     FLOW_DISSECTOR_KEY_ENC_KEYID, enc_key_id);
772 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
773 			     FLOW_DISSECTOR_KEY_ENC_IPV4_ADDRS, enc_ipv4);
774 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
775 			     FLOW_DISSECTOR_KEY_ENC_IPV6_ADDRS, enc_ipv6);
776 	if (FL_KEY_IS_MASKED(&mask->key, enc_ipv4) ||
777 	    FL_KEY_IS_MASKED(&mask->key, enc_ipv6))
778 		FL_KEY_SET(keys, cnt, FLOW_DISSECTOR_KEY_ENC_CONTROL,
779 			   enc_control);
780 	FL_KEY_SET_IF_MASKED(&mask->key, keys, cnt,
781 			     FLOW_DISSECTOR_KEY_ENC_PORTS, enc_tp);
782 
783 	skb_flow_dissector_init(&head->dissector, keys, cnt);
784 }
785 
786 static int fl_check_assign_mask(struct cls_fl_head *head,
787 				struct fl_flow_mask *mask)
788 {
789 	int err;
790 
791 	if (head->mask_assigned) {
792 		if (!fl_mask_eq(&head->mask, mask))
793 			return -EINVAL;
794 		else
795 			return 0;
796 	}
797 
798 	/* Mask is not assigned yet. So assign it and init hashtable
799 	 * according to that.
800 	 */
801 	err = fl_init_hashtable(head, mask);
802 	if (err)
803 		return err;
804 	memcpy(&head->mask, mask, sizeof(head->mask));
805 	head->mask_assigned = true;
806 
807 	fl_init_dissector(head, mask);
808 
809 	return 0;
810 }
811 
812 static int fl_set_parms(struct net *net, struct tcf_proto *tp,
813 			struct cls_fl_filter *f, struct fl_flow_mask *mask,
814 			unsigned long base, struct nlattr **tb,
815 			struct nlattr *est, bool ovr)
816 {
817 	int err;
818 
819 	err = tcf_exts_validate(net, tp, tb, est, &f->exts, ovr);
820 	if (err < 0)
821 		return err;
822 
823 	if (tb[TCA_FLOWER_CLASSID]) {
824 		f->res.classid = nla_get_u32(tb[TCA_FLOWER_CLASSID]);
825 		tcf_bind_filter(tp, &f->res, base);
826 	}
827 
828 	err = fl_set_key(net, tb, &f->key, &mask->key);
829 	if (err)
830 		return err;
831 
832 	fl_mask_update_range(mask);
833 	fl_set_masked_key(&f->mkey, &f->key, mask);
834 
835 	return 0;
836 }
837 
838 static int fl_change(struct net *net, struct sk_buff *in_skb,
839 		     struct tcf_proto *tp, unsigned long base,
840 		     u32 handle, struct nlattr **tca,
841 		     void **arg, bool ovr)
842 {
843 	struct cls_fl_head *head = rtnl_dereference(tp->root);
844 	struct cls_fl_filter *fold = *arg;
845 	struct cls_fl_filter *fnew;
846 	struct nlattr **tb;
847 	struct fl_flow_mask mask = {};
848 	unsigned long idr_index;
849 	int err;
850 
851 	if (!tca[TCA_OPTIONS])
852 		return -EINVAL;
853 
854 	tb = kcalloc(TCA_FLOWER_MAX + 1, sizeof(struct nlattr *), GFP_KERNEL);
855 	if (!tb)
856 		return -ENOBUFS;
857 
858 	err = nla_parse_nested(tb, TCA_FLOWER_MAX, tca[TCA_OPTIONS],
859 			       fl_policy, NULL);
860 	if (err < 0)
861 		goto errout_tb;
862 
863 	if (fold && handle && fold->handle != handle) {
864 		err = -EINVAL;
865 		goto errout_tb;
866 	}
867 
868 	fnew = kzalloc(sizeof(*fnew), GFP_KERNEL);
869 	if (!fnew) {
870 		err = -ENOBUFS;
871 		goto errout_tb;
872 	}
873 
874 	err = tcf_exts_init(&fnew->exts, TCA_FLOWER_ACT, 0);
875 	if (err < 0)
876 		goto errout;
877 
878 	if (!handle) {
879 		err = idr_alloc_ext(&head->handle_idr, fnew, &idr_index,
880 				    1, 0x80000000, GFP_KERNEL);
881 		if (err)
882 			goto errout;
883 		fnew->handle = idr_index;
884 	}
885 
886 	/* user specifies a handle and it doesn't exist */
887 	if (handle && !fold) {
888 		err = idr_alloc_ext(&head->handle_idr, fnew, &idr_index,
889 				    handle, handle + 1, GFP_KERNEL);
890 		if (err)
891 			goto errout;
892 		fnew->handle = idr_index;
893 	}
894 
895 	if (tb[TCA_FLOWER_FLAGS]) {
896 		fnew->flags = nla_get_u32(tb[TCA_FLOWER_FLAGS]);
897 
898 		if (!tc_flags_valid(fnew->flags)) {
899 			err = -EINVAL;
900 			goto errout_idr;
901 		}
902 	}
903 
904 	err = fl_set_parms(net, tp, fnew, &mask, base, tb, tca[TCA_RATE], ovr);
905 	if (err)
906 		goto errout_idr;
907 
908 	err = fl_check_assign_mask(head, &mask);
909 	if (err)
910 		goto errout_idr;
911 
912 	if (!tc_skip_sw(fnew->flags)) {
913 		if (!fold && fl_lookup(head, &fnew->mkey)) {
914 			err = -EEXIST;
915 			goto errout_idr;
916 		}
917 
918 		err = rhashtable_insert_fast(&head->ht, &fnew->ht_node,
919 					     head->ht_params);
920 		if (err)
921 			goto errout_idr;
922 	}
923 
924 	if (!tc_skip_hw(fnew->flags)) {
925 		err = fl_hw_replace_filter(tp,
926 					   &head->dissector,
927 					   &mask.key,
928 					   fnew);
929 		if (err)
930 			goto errout_idr;
931 	}
932 
933 	if (!tc_in_hw(fnew->flags))
934 		fnew->flags |= TCA_CLS_FLAGS_NOT_IN_HW;
935 
936 	if (fold) {
937 		if (!tc_skip_sw(fold->flags))
938 			rhashtable_remove_fast(&head->ht, &fold->ht_node,
939 					       head->ht_params);
940 		if (!tc_skip_hw(fold->flags))
941 			fl_hw_destroy_filter(tp, fold);
942 	}
943 
944 	*arg = fnew;
945 
946 	if (fold) {
947 		fnew->handle = handle;
948 		idr_replace_ext(&head->handle_idr, fnew, fnew->handle);
949 		list_replace_rcu(&fold->list, &fnew->list);
950 		tcf_unbind_filter(tp, &fold->res);
951 		call_rcu(&fold->rcu, fl_destroy_filter);
952 	} else {
953 		list_add_tail_rcu(&fnew->list, &head->filters);
954 	}
955 
956 	kfree(tb);
957 	return 0;
958 
959 errout_idr:
960 	if (fnew->handle)
961 		idr_remove_ext(&head->handle_idr, fnew->handle);
962 errout:
963 	tcf_exts_destroy(&fnew->exts);
964 	kfree(fnew);
965 errout_tb:
966 	kfree(tb);
967 	return err;
968 }
969 
970 static int fl_delete(struct tcf_proto *tp, void *arg, bool *last)
971 {
972 	struct cls_fl_head *head = rtnl_dereference(tp->root);
973 	struct cls_fl_filter *f = arg;
974 
975 	if (!tc_skip_sw(f->flags))
976 		rhashtable_remove_fast(&head->ht, &f->ht_node,
977 				       head->ht_params);
978 	__fl_delete(tp, f);
979 	*last = list_empty(&head->filters);
980 	return 0;
981 }
982 
983 static void fl_walk(struct tcf_proto *tp, struct tcf_walker *arg)
984 {
985 	struct cls_fl_head *head = rtnl_dereference(tp->root);
986 	struct cls_fl_filter *f;
987 
988 	list_for_each_entry_rcu(f, &head->filters, list) {
989 		if (arg->count < arg->skip)
990 			goto skip;
991 		if (arg->fn(tp, f, arg) < 0) {
992 			arg->stop = 1;
993 			break;
994 		}
995 skip:
996 		arg->count++;
997 	}
998 }
999 
1000 static int fl_dump_key_val(struct sk_buff *skb,
1001 			   void *val, int val_type,
1002 			   void *mask, int mask_type, int len)
1003 {
1004 	int err;
1005 
1006 	if (!memchr_inv(mask, 0, len))
1007 		return 0;
1008 	err = nla_put(skb, val_type, len, val);
1009 	if (err)
1010 		return err;
1011 	if (mask_type != TCA_FLOWER_UNSPEC) {
1012 		err = nla_put(skb, mask_type, len, mask);
1013 		if (err)
1014 			return err;
1015 	}
1016 	return 0;
1017 }
1018 
1019 static int fl_dump_key_mpls(struct sk_buff *skb,
1020 			    struct flow_dissector_key_mpls *mpls_key,
1021 			    struct flow_dissector_key_mpls *mpls_mask)
1022 {
1023 	int err;
1024 
1025 	if (!memchr_inv(mpls_mask, 0, sizeof(*mpls_mask)))
1026 		return 0;
1027 	if (mpls_mask->mpls_ttl) {
1028 		err = nla_put_u8(skb, TCA_FLOWER_KEY_MPLS_TTL,
1029 				 mpls_key->mpls_ttl);
1030 		if (err)
1031 			return err;
1032 	}
1033 	if (mpls_mask->mpls_tc) {
1034 		err = nla_put_u8(skb, TCA_FLOWER_KEY_MPLS_TC,
1035 				 mpls_key->mpls_tc);
1036 		if (err)
1037 			return err;
1038 	}
1039 	if (mpls_mask->mpls_label) {
1040 		err = nla_put_u32(skb, TCA_FLOWER_KEY_MPLS_LABEL,
1041 				  mpls_key->mpls_label);
1042 		if (err)
1043 			return err;
1044 	}
1045 	if (mpls_mask->mpls_bos) {
1046 		err = nla_put_u8(skb, TCA_FLOWER_KEY_MPLS_BOS,
1047 				 mpls_key->mpls_bos);
1048 		if (err)
1049 			return err;
1050 	}
1051 	return 0;
1052 }
1053 
1054 static int fl_dump_key_ip(struct sk_buff *skb,
1055 			  struct flow_dissector_key_ip *key,
1056 			  struct flow_dissector_key_ip *mask)
1057 {
1058 	if (fl_dump_key_val(skb, &key->tos, TCA_FLOWER_KEY_IP_TOS, &mask->tos,
1059 			    TCA_FLOWER_KEY_IP_TOS_MASK, sizeof(key->tos)) ||
1060 	    fl_dump_key_val(skb, &key->ttl, TCA_FLOWER_KEY_IP_TTL, &mask->ttl,
1061 			    TCA_FLOWER_KEY_IP_TTL_MASK, sizeof(key->ttl)))
1062 		return -1;
1063 
1064 	return 0;
1065 }
1066 
1067 static int fl_dump_key_vlan(struct sk_buff *skb,
1068 			    struct flow_dissector_key_vlan *vlan_key,
1069 			    struct flow_dissector_key_vlan *vlan_mask)
1070 {
1071 	int err;
1072 
1073 	if (!memchr_inv(vlan_mask, 0, sizeof(*vlan_mask)))
1074 		return 0;
1075 	if (vlan_mask->vlan_id) {
1076 		err = nla_put_u16(skb, TCA_FLOWER_KEY_VLAN_ID,
1077 				  vlan_key->vlan_id);
1078 		if (err)
1079 			return err;
1080 	}
1081 	if (vlan_mask->vlan_priority) {
1082 		err = nla_put_u8(skb, TCA_FLOWER_KEY_VLAN_PRIO,
1083 				 vlan_key->vlan_priority);
1084 		if (err)
1085 			return err;
1086 	}
1087 	return 0;
1088 }
1089 
1090 static void fl_get_key_flag(u32 dissector_key, u32 dissector_mask,
1091 			    u32 *flower_key, u32 *flower_mask,
1092 			    u32 flower_flag_bit, u32 dissector_flag_bit)
1093 {
1094 	if (dissector_mask & dissector_flag_bit) {
1095 		*flower_mask |= flower_flag_bit;
1096 		if (dissector_key & dissector_flag_bit)
1097 			*flower_key |= flower_flag_bit;
1098 	}
1099 }
1100 
1101 static int fl_dump_key_flags(struct sk_buff *skb, u32 flags_key, u32 flags_mask)
1102 {
1103 	u32 key, mask;
1104 	__be32 _key, _mask;
1105 	int err;
1106 
1107 	if (!memchr_inv(&flags_mask, 0, sizeof(flags_mask)))
1108 		return 0;
1109 
1110 	key = 0;
1111 	mask = 0;
1112 
1113 	fl_get_key_flag(flags_key, flags_mask, &key, &mask,
1114 			TCA_FLOWER_KEY_FLAGS_IS_FRAGMENT, FLOW_DIS_IS_FRAGMENT);
1115 
1116 	_key = cpu_to_be32(key);
1117 	_mask = cpu_to_be32(mask);
1118 
1119 	err = nla_put(skb, TCA_FLOWER_KEY_FLAGS, 4, &_key);
1120 	if (err)
1121 		return err;
1122 
1123 	return nla_put(skb, TCA_FLOWER_KEY_FLAGS_MASK, 4, &_mask);
1124 }
1125 
1126 static int fl_dump(struct net *net, struct tcf_proto *tp, void *fh,
1127 		   struct sk_buff *skb, struct tcmsg *t)
1128 {
1129 	struct cls_fl_head *head = rtnl_dereference(tp->root);
1130 	struct cls_fl_filter *f = fh;
1131 	struct nlattr *nest;
1132 	struct fl_flow_key *key, *mask;
1133 
1134 	if (!f)
1135 		return skb->len;
1136 
1137 	t->tcm_handle = f->handle;
1138 
1139 	nest = nla_nest_start(skb, TCA_OPTIONS);
1140 	if (!nest)
1141 		goto nla_put_failure;
1142 
1143 	if (f->res.classid &&
1144 	    nla_put_u32(skb, TCA_FLOWER_CLASSID, f->res.classid))
1145 		goto nla_put_failure;
1146 
1147 	key = &f->key;
1148 	mask = &head->mask.key;
1149 
1150 	if (mask->indev_ifindex) {
1151 		struct net_device *dev;
1152 
1153 		dev = __dev_get_by_index(net, key->indev_ifindex);
1154 		if (dev && nla_put_string(skb, TCA_FLOWER_INDEV, dev->name))
1155 			goto nla_put_failure;
1156 	}
1157 
1158 	if (!tc_skip_hw(f->flags))
1159 		fl_hw_update_stats(tp, f);
1160 
1161 	if (fl_dump_key_val(skb, key->eth.dst, TCA_FLOWER_KEY_ETH_DST,
1162 			    mask->eth.dst, TCA_FLOWER_KEY_ETH_DST_MASK,
1163 			    sizeof(key->eth.dst)) ||
1164 	    fl_dump_key_val(skb, key->eth.src, TCA_FLOWER_KEY_ETH_SRC,
1165 			    mask->eth.src, TCA_FLOWER_KEY_ETH_SRC_MASK,
1166 			    sizeof(key->eth.src)) ||
1167 	    fl_dump_key_val(skb, &key->basic.n_proto, TCA_FLOWER_KEY_ETH_TYPE,
1168 			    &mask->basic.n_proto, TCA_FLOWER_UNSPEC,
1169 			    sizeof(key->basic.n_proto)))
1170 		goto nla_put_failure;
1171 
1172 	if (fl_dump_key_mpls(skb, &key->mpls, &mask->mpls))
1173 		goto nla_put_failure;
1174 
1175 	if (fl_dump_key_vlan(skb, &key->vlan, &mask->vlan))
1176 		goto nla_put_failure;
1177 
1178 	if ((key->basic.n_proto == htons(ETH_P_IP) ||
1179 	     key->basic.n_proto == htons(ETH_P_IPV6)) &&
1180 	    (fl_dump_key_val(skb, &key->basic.ip_proto, TCA_FLOWER_KEY_IP_PROTO,
1181 			    &mask->basic.ip_proto, TCA_FLOWER_UNSPEC,
1182 			    sizeof(key->basic.ip_proto)) ||
1183 	    fl_dump_key_ip(skb, &key->ip, &mask->ip)))
1184 		goto nla_put_failure;
1185 
1186 	if (key->control.addr_type == FLOW_DISSECTOR_KEY_IPV4_ADDRS &&
1187 	    (fl_dump_key_val(skb, &key->ipv4.src, TCA_FLOWER_KEY_IPV4_SRC,
1188 			     &mask->ipv4.src, TCA_FLOWER_KEY_IPV4_SRC_MASK,
1189 			     sizeof(key->ipv4.src)) ||
1190 	     fl_dump_key_val(skb, &key->ipv4.dst, TCA_FLOWER_KEY_IPV4_DST,
1191 			     &mask->ipv4.dst, TCA_FLOWER_KEY_IPV4_DST_MASK,
1192 			     sizeof(key->ipv4.dst))))
1193 		goto nla_put_failure;
1194 	else if (key->control.addr_type == FLOW_DISSECTOR_KEY_IPV6_ADDRS &&
1195 		 (fl_dump_key_val(skb, &key->ipv6.src, TCA_FLOWER_KEY_IPV6_SRC,
1196 				  &mask->ipv6.src, TCA_FLOWER_KEY_IPV6_SRC_MASK,
1197 				  sizeof(key->ipv6.src)) ||
1198 		  fl_dump_key_val(skb, &key->ipv6.dst, TCA_FLOWER_KEY_IPV6_DST,
1199 				  &mask->ipv6.dst, TCA_FLOWER_KEY_IPV6_DST_MASK,
1200 				  sizeof(key->ipv6.dst))))
1201 		goto nla_put_failure;
1202 
1203 	if (key->basic.ip_proto == IPPROTO_TCP &&
1204 	    (fl_dump_key_val(skb, &key->tp.src, TCA_FLOWER_KEY_TCP_SRC,
1205 			     &mask->tp.src, TCA_FLOWER_KEY_TCP_SRC_MASK,
1206 			     sizeof(key->tp.src)) ||
1207 	     fl_dump_key_val(skb, &key->tp.dst, TCA_FLOWER_KEY_TCP_DST,
1208 			     &mask->tp.dst, TCA_FLOWER_KEY_TCP_DST_MASK,
1209 			     sizeof(key->tp.dst)) ||
1210 	     fl_dump_key_val(skb, &key->tcp.flags, TCA_FLOWER_KEY_TCP_FLAGS,
1211 			     &mask->tcp.flags, TCA_FLOWER_KEY_TCP_FLAGS_MASK,
1212 			     sizeof(key->tcp.flags))))
1213 		goto nla_put_failure;
1214 	else if (key->basic.ip_proto == IPPROTO_UDP &&
1215 		 (fl_dump_key_val(skb, &key->tp.src, TCA_FLOWER_KEY_UDP_SRC,
1216 				  &mask->tp.src, TCA_FLOWER_KEY_UDP_SRC_MASK,
1217 				  sizeof(key->tp.src)) ||
1218 		  fl_dump_key_val(skb, &key->tp.dst, TCA_FLOWER_KEY_UDP_DST,
1219 				  &mask->tp.dst, TCA_FLOWER_KEY_UDP_DST_MASK,
1220 				  sizeof(key->tp.dst))))
1221 		goto nla_put_failure;
1222 	else if (key->basic.ip_proto == IPPROTO_SCTP &&
1223 		 (fl_dump_key_val(skb, &key->tp.src, TCA_FLOWER_KEY_SCTP_SRC,
1224 				  &mask->tp.src, TCA_FLOWER_KEY_SCTP_SRC_MASK,
1225 				  sizeof(key->tp.src)) ||
1226 		  fl_dump_key_val(skb, &key->tp.dst, TCA_FLOWER_KEY_SCTP_DST,
1227 				  &mask->tp.dst, TCA_FLOWER_KEY_SCTP_DST_MASK,
1228 				  sizeof(key->tp.dst))))
1229 		goto nla_put_failure;
1230 	else if (key->basic.n_proto == htons(ETH_P_IP) &&
1231 		 key->basic.ip_proto == IPPROTO_ICMP &&
1232 		 (fl_dump_key_val(skb, &key->icmp.type,
1233 				  TCA_FLOWER_KEY_ICMPV4_TYPE, &mask->icmp.type,
1234 				  TCA_FLOWER_KEY_ICMPV4_TYPE_MASK,
1235 				  sizeof(key->icmp.type)) ||
1236 		  fl_dump_key_val(skb, &key->icmp.code,
1237 				  TCA_FLOWER_KEY_ICMPV4_CODE, &mask->icmp.code,
1238 				  TCA_FLOWER_KEY_ICMPV4_CODE_MASK,
1239 				  sizeof(key->icmp.code))))
1240 		goto nla_put_failure;
1241 	else if (key->basic.n_proto == htons(ETH_P_IPV6) &&
1242 		 key->basic.ip_proto == IPPROTO_ICMPV6 &&
1243 		 (fl_dump_key_val(skb, &key->icmp.type,
1244 				  TCA_FLOWER_KEY_ICMPV6_TYPE, &mask->icmp.type,
1245 				  TCA_FLOWER_KEY_ICMPV6_TYPE_MASK,
1246 				  sizeof(key->icmp.type)) ||
1247 		  fl_dump_key_val(skb, &key->icmp.code,
1248 				  TCA_FLOWER_KEY_ICMPV6_CODE, &mask->icmp.code,
1249 				  TCA_FLOWER_KEY_ICMPV6_CODE_MASK,
1250 				  sizeof(key->icmp.code))))
1251 		goto nla_put_failure;
1252 	else if ((key->basic.n_proto == htons(ETH_P_ARP) ||
1253 		  key->basic.n_proto == htons(ETH_P_RARP)) &&
1254 		 (fl_dump_key_val(skb, &key->arp.sip,
1255 				  TCA_FLOWER_KEY_ARP_SIP, &mask->arp.sip,
1256 				  TCA_FLOWER_KEY_ARP_SIP_MASK,
1257 				  sizeof(key->arp.sip)) ||
1258 		  fl_dump_key_val(skb, &key->arp.tip,
1259 				  TCA_FLOWER_KEY_ARP_TIP, &mask->arp.tip,
1260 				  TCA_FLOWER_KEY_ARP_TIP_MASK,
1261 				  sizeof(key->arp.tip)) ||
1262 		  fl_dump_key_val(skb, &key->arp.op,
1263 				  TCA_FLOWER_KEY_ARP_OP, &mask->arp.op,
1264 				  TCA_FLOWER_KEY_ARP_OP_MASK,
1265 				  sizeof(key->arp.op)) ||
1266 		  fl_dump_key_val(skb, key->arp.sha, TCA_FLOWER_KEY_ARP_SHA,
1267 				  mask->arp.sha, TCA_FLOWER_KEY_ARP_SHA_MASK,
1268 				  sizeof(key->arp.sha)) ||
1269 		  fl_dump_key_val(skb, key->arp.tha, TCA_FLOWER_KEY_ARP_THA,
1270 				  mask->arp.tha, TCA_FLOWER_KEY_ARP_THA_MASK,
1271 				  sizeof(key->arp.tha))))
1272 		goto nla_put_failure;
1273 
1274 	if (key->enc_control.addr_type == FLOW_DISSECTOR_KEY_IPV4_ADDRS &&
1275 	    (fl_dump_key_val(skb, &key->enc_ipv4.src,
1276 			    TCA_FLOWER_KEY_ENC_IPV4_SRC, &mask->enc_ipv4.src,
1277 			    TCA_FLOWER_KEY_ENC_IPV4_SRC_MASK,
1278 			    sizeof(key->enc_ipv4.src)) ||
1279 	     fl_dump_key_val(skb, &key->enc_ipv4.dst,
1280 			     TCA_FLOWER_KEY_ENC_IPV4_DST, &mask->enc_ipv4.dst,
1281 			     TCA_FLOWER_KEY_ENC_IPV4_DST_MASK,
1282 			     sizeof(key->enc_ipv4.dst))))
1283 		goto nla_put_failure;
1284 	else if (key->enc_control.addr_type == FLOW_DISSECTOR_KEY_IPV6_ADDRS &&
1285 		 (fl_dump_key_val(skb, &key->enc_ipv6.src,
1286 			    TCA_FLOWER_KEY_ENC_IPV6_SRC, &mask->enc_ipv6.src,
1287 			    TCA_FLOWER_KEY_ENC_IPV6_SRC_MASK,
1288 			    sizeof(key->enc_ipv6.src)) ||
1289 		 fl_dump_key_val(skb, &key->enc_ipv6.dst,
1290 				 TCA_FLOWER_KEY_ENC_IPV6_DST,
1291 				 &mask->enc_ipv6.dst,
1292 				 TCA_FLOWER_KEY_ENC_IPV6_DST_MASK,
1293 			    sizeof(key->enc_ipv6.dst))))
1294 		goto nla_put_failure;
1295 
1296 	if (fl_dump_key_val(skb, &key->enc_key_id, TCA_FLOWER_KEY_ENC_KEY_ID,
1297 			    &mask->enc_key_id, TCA_FLOWER_UNSPEC,
1298 			    sizeof(key->enc_key_id)) ||
1299 	    fl_dump_key_val(skb, &key->enc_tp.src,
1300 			    TCA_FLOWER_KEY_ENC_UDP_SRC_PORT,
1301 			    &mask->enc_tp.src,
1302 			    TCA_FLOWER_KEY_ENC_UDP_SRC_PORT_MASK,
1303 			    sizeof(key->enc_tp.src)) ||
1304 	    fl_dump_key_val(skb, &key->enc_tp.dst,
1305 			    TCA_FLOWER_KEY_ENC_UDP_DST_PORT,
1306 			    &mask->enc_tp.dst,
1307 			    TCA_FLOWER_KEY_ENC_UDP_DST_PORT_MASK,
1308 			    sizeof(key->enc_tp.dst)))
1309 		goto nla_put_failure;
1310 
1311 	if (fl_dump_key_flags(skb, key->control.flags, mask->control.flags))
1312 		goto nla_put_failure;
1313 
1314 	if (f->flags && nla_put_u32(skb, TCA_FLOWER_FLAGS, f->flags))
1315 		goto nla_put_failure;
1316 
1317 	if (tcf_exts_dump(skb, &f->exts))
1318 		goto nla_put_failure;
1319 
1320 	nla_nest_end(skb, nest);
1321 
1322 	if (tcf_exts_dump_stats(skb, &f->exts) < 0)
1323 		goto nla_put_failure;
1324 
1325 	return skb->len;
1326 
1327 nla_put_failure:
1328 	nla_nest_cancel(skb, nest);
1329 	return -1;
1330 }
1331 
1332 static void fl_bind_class(void *fh, u32 classid, unsigned long cl)
1333 {
1334 	struct cls_fl_filter *f = fh;
1335 
1336 	if (f && f->res.classid == classid)
1337 		f->res.class = cl;
1338 }
1339 
1340 static struct tcf_proto_ops cls_fl_ops __read_mostly = {
1341 	.kind		= "flower",
1342 	.classify	= fl_classify,
1343 	.init		= fl_init,
1344 	.destroy	= fl_destroy,
1345 	.get		= fl_get,
1346 	.change		= fl_change,
1347 	.delete		= fl_delete,
1348 	.walk		= fl_walk,
1349 	.dump		= fl_dump,
1350 	.bind_class	= fl_bind_class,
1351 	.owner		= THIS_MODULE,
1352 };
1353 
1354 static int __init cls_fl_init(void)
1355 {
1356 	return register_tcf_proto_ops(&cls_fl_ops);
1357 }
1358 
1359 static void __exit cls_fl_exit(void)
1360 {
1361 	unregister_tcf_proto_ops(&cls_fl_ops);
1362 }
1363 
1364 module_init(cls_fl_init);
1365 module_exit(cls_fl_exit);
1366 
1367 MODULE_AUTHOR("Jiri Pirko <jiri@resnulli.us>");
1368 MODULE_DESCRIPTION("Flower classifier");
1369 MODULE_LICENSE("GPL v2");
1370