sch_multiq.c 9.2 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13
/*
 * Copyright (c) 2008, Intel Corporation.
 *
 * This program is free software; you can redistribute it and/or modify it
 * under the terms and conditions of the GNU General Public License,
 * version 2, as published by the Free Software Foundation.
 *
 * This program is distributed in the hope it will be useful, but WITHOUT
 * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
 * FITNESS FOR A PARTICULAR PURPOSE.  See the GNU General Public License for
 * more details.
 *
 * You should have received a copy of the GNU General Public License along with
14
 * this program; if not, see <http://www.gnu.org/licenses/>.
15 16 17 18 19
 *
 * Author: Alexander Duyck <alexander.h.duyck@intel.com>
 */

#include <linux/module.h>
20
#include <linux/slab.h>
21 22 23 24 25 26 27
#include <linux/types.h>
#include <linux/kernel.h>
#include <linux/string.h>
#include <linux/errno.h>
#include <linux/skbuff.h>
#include <net/netlink.h>
#include <net/pkt_sched.h>
28
#include <net/pkt_cls.h>
29 30 31 32 33

struct multiq_sched_data {
	u16 bands;
	u16 max_bands;
	u16 curband;
J
John Fastabend 已提交
34
	struct tcf_proto __rcu *filter_list;
35
	struct tcf_block *block;
36 37 38 39 40 41 42 43 44 45
	struct Qdisc **queues;
};


static struct Qdisc *
multiq_classify(struct sk_buff *skb, struct Qdisc *sch, int *qerr)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	u32 band;
	struct tcf_result res;
J
John Fastabend 已提交
46
	struct tcf_proto *fl = rcu_dereference_bh(q->filter_list);
47 48 49
	int err;

	*qerr = NET_XMIT_SUCCESS | __NET_XMIT_BYPASS;
50
	err = tcf_classify(skb, fl, &res, false);
51 52 53 54
#ifdef CONFIG_NET_CLS_ACT
	switch (err) {
	case TC_ACT_STOLEN:
	case TC_ACT_QUEUED:
55
	case TC_ACT_TRAP:
56
		*qerr = NET_XMIT_SUCCESS | __NET_XMIT_STOLEN;
57
		/* fall through */
58 59 60 61 62 63 64 65 66 67 68 69 70
	case TC_ACT_SHOT:
		return NULL;
	}
#endif
	band = skb_get_queue_mapping(skb);

	if (band >= q->bands)
		return q->queues[0];

	return q->queues[band];
}

static int
71 72
multiq_enqueue(struct sk_buff *skb, struct Qdisc *sch,
	       struct sk_buff **to_free)
73 74 75 76 77 78 79 80 81
{
	struct Qdisc *qdisc;
	int ret;

	qdisc = multiq_classify(skb, sch, &ret);
#ifdef CONFIG_NET_CLS_ACT
	if (qdisc == NULL) {

		if (ret & __NET_XMIT_BYPASS)
82
			qdisc_qstats_drop(sch);
83
		__qdisc_drop(skb, to_free);
84 85 86 87
		return ret;
	}
#endif

88
	ret = qdisc_enqueue(skb, qdisc, to_free);
89 90 91 92 93
	if (ret == NET_XMIT_SUCCESS) {
		sch->q.qlen++;
		return NET_XMIT_SUCCESS;
	}
	if (net_xmit_drop_count(ret))
94
		qdisc_qstats_drop(sch);
95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111
	return ret;
}

static struct sk_buff *multiq_dequeue(struct Qdisc *sch)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	struct Qdisc *qdisc;
	struct sk_buff *skb;
	int band;

	for (band = 0; band < q->bands; band++) {
		/* cycle through bands to ensure fairness */
		q->curband++;
		if (q->curband >= q->bands)
			q->curband = 0;

		/* Check that target subqueue is available before
112
		 * pulling an skb to avoid head-of-line blocking.
113
		 */
114 115
		if (!netif_xmit_stopped(
		    netdev_get_tx_queue(qdisc_dev(sch), q->curband))) {
116 117 118
			qdisc = q->queues[q->curband];
			skb = qdisc->dequeue(qdisc);
			if (skb) {
119
				qdisc_bstats_update(sch, skb);
120 121 122 123 124 125 126 127 128
				sch->q.qlen--;
				return skb;
			}
		}
	}
	return NULL;

}

129 130 131 132 133 134 135 136 137 138 139 140 141 142 143
static struct sk_buff *multiq_peek(struct Qdisc *sch)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	unsigned int curband = q->curband;
	struct Qdisc *qdisc;
	struct sk_buff *skb;
	int band;

	for (band = 0; band < q->bands; band++) {
		/* cycle through bands to ensure fairness */
		curband++;
		if (curband >= q->bands)
			curband = 0;

		/* Check that target subqueue is available before
144
		 * pulling an skb to avoid head-of-line blocking.
145
		 */
146 147
		if (!netif_xmit_stopped(
		    netdev_get_tx_queue(qdisc_dev(sch), curband))) {
148 149 150 151 152 153 154 155 156 157
			qdisc = q->queues[curband];
			skb = qdisc->ops->peek(qdisc);
			if (skb)
				return skb;
		}
	}
	return NULL;

}

158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175
static void
multiq_reset(struct Qdisc *sch)
{
	u16 band;
	struct multiq_sched_data *q = qdisc_priv(sch);

	for (band = 0; band < q->bands; band++)
		qdisc_reset(q->queues[band]);
	sch->q.qlen = 0;
	q->curband = 0;
}

static void
multiq_destroy(struct Qdisc *sch)
{
	int band;
	struct multiq_sched_data *q = qdisc_priv(sch);

176
	tcf_block_put(q->block);
177 178 179 180 181 182 183 184 185 186 187 188 189
	for (band = 0; band < q->bands; band++)
		qdisc_destroy(q->queues[band]);

	kfree(q->queues);
}

static int multiq_tune(struct Qdisc *sch, struct nlattr *opt)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	struct tc_multiq_qopt *qopt;
	int i;

	if (!netif_is_multiqueue(qdisc_dev(sch)))
190
		return -EOPNOTSUPP;
191 192 193 194 195 196 197 198 199 200
	if (nla_len(opt) < sizeof(*qopt))
		return -EINVAL;

	qopt = nla_data(opt);

	qopt->bands = qdisc_dev(sch)->real_num_tx_queues;

	sch_tree_lock(sch);
	q->bands = qopt->bands;
	for (i = q->bands; i < q->max_bands; i++) {
201
		if (q->queues[i] != &noop_qdisc) {
202 203
			struct Qdisc *child = q->queues[i];
			q->queues[i] = &noop_qdisc;
204 205
			qdisc_tree_reduce_backlog(child, child->q.qlen,
						  child->qstats.backlog);
206 207 208 209 210 211 212 213
			qdisc_destroy(child);
		}
	}

	sch_tree_unlock(sch);

	for (i = 0; i < q->bands; i++) {
		if (q->queues[i] == &noop_qdisc) {
214
			struct Qdisc *child, *old;
215
			child = qdisc_create_dflt(sch->dev_queue,
216 217 218 219 220
						  &pfifo_qdisc_ops,
						  TC_H_MAKE(sch->handle,
							    i + 1));
			if (child) {
				sch_tree_lock(sch);
221 222
				old = q->queues[i];
				q->queues[i] = child;
223 224
				if (child != &noop_qdisc)
					qdisc_hash_add(child, true);
225

226
				if (old != &noop_qdisc) {
227 228 229
					qdisc_tree_reduce_backlog(old,
								  old->q.qlen,
								  old->qstats.backlog);
230
					qdisc_destroy(old);
231 232 233 234 235 236 237 238
				}
				sch_tree_unlock(sch);
			}
		}
	}
	return 0;
}

239 240
static int multiq_init(struct Qdisc *sch, struct nlattr *opt,
		       struct netlink_ext_ack *extack)
241 242
{
	struct multiq_sched_data *q = qdisc_priv(sch);
243
	int i, err;
244 245 246

	q->queues = NULL;

247
	if (!opt)
248 249
		return -EINVAL;

250
	err = tcf_block_get(&q->block, &q->filter_list, sch);
251 252 253
	if (err)
		return err;

254 255 256 257 258 259 260 261
	q->max_bands = qdisc_dev(sch)->num_tx_queues;

	q->queues = kcalloc(q->max_bands, sizeof(struct Qdisc *), GFP_KERNEL);
	if (!q->queues)
		return -ENOBUFS;
	for (i = 0; i < q->max_bands; i++)
		q->queues[i] = &noop_qdisc;

262
	return multiq_tune(sch, opt);
263 264 265 266 267 268 269 270 271 272 273
}

static int multiq_dump(struct Qdisc *sch, struct sk_buff *skb)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	unsigned char *b = skb_tail_pointer(skb);
	struct tc_multiq_qopt opt;

	opt.bands = q->bands;
	opt.max_bands = q->max_bands;

274 275
	if (nla_put(skb, TCA_OPTIONS, sizeof(opt), &opt))
		goto nla_put_failure;
276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292

	return skb->len;

nla_put_failure:
	nlmsg_trim(skb, b);
	return -1;
}

static int multiq_graft(struct Qdisc *sch, unsigned long arg, struct Qdisc *new,
		      struct Qdisc **old)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	unsigned long band = arg - 1;

	if (new == NULL)
		new = &noop_qdisc;

293
	*old = qdisc_replace(sch, new, &q->queues[band]);
294 295 296 297 298 299 300 301 302 303 304 305
	return 0;
}

static struct Qdisc *
multiq_leaf(struct Qdisc *sch, unsigned long arg)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	unsigned long band = arg - 1;

	return q->queues[band];
}

306
static unsigned long multiq_find(struct Qdisc *sch, u32 classid)
307 308 309 310 311 312 313 314 315 316 317 318
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	unsigned long band = TC_H_MIN(classid);

	if (band - 1 >= q->bands)
		return 0;
	return band;
}

static unsigned long multiq_bind(struct Qdisc *sch, unsigned long parent,
				 u32 classid)
{
319
	return multiq_find(sch, classid);
320 321 322
}


323
static void multiq_unbind(struct Qdisc *q, unsigned long cl)
324 325 326 327 328 329 330 331 332
{
}

static int multiq_dump_class(struct Qdisc *sch, unsigned long cl,
			     struct sk_buff *skb, struct tcmsg *tcm)
{
	struct multiq_sched_data *q = qdisc_priv(sch);

	tcm->tcm_handle |= TC_H_MIN(cl);
E
Eric Dumazet 已提交
333
	tcm->tcm_info = q->queues[cl - 1]->handle;
334 335 336 337 338 339 340 341 342 343
	return 0;
}

static int multiq_dump_class_stats(struct Qdisc *sch, unsigned long cl,
				 struct gnet_dump *d)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	struct Qdisc *cl_q;

	cl_q = q->queues[cl - 1];
344 345
	if (gnet_stats_copy_basic(qdisc_root_sleeping_running(sch),
				  d, NULL, &cl_q->bstats) < 0 ||
346
	    gnet_stats_copy_queue(d, NULL, &cl_q->qstats, cl_q->q.qlen) < 0)
347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364
		return -1;

	return 0;
}

static void multiq_walk(struct Qdisc *sch, struct qdisc_walker *arg)
{
	struct multiq_sched_data *q = qdisc_priv(sch);
	int band;

	if (arg->stop)
		return;

	for (band = 0; band < q->bands; band++) {
		if (arg->count < arg->skip) {
			arg->count++;
			continue;
		}
E
Eric Dumazet 已提交
365
		if (arg->fn(sch, band + 1, arg) < 0) {
366 367 368 369 370 371 372
			arg->stop = 1;
			break;
		}
		arg->count++;
	}
}

373
static struct tcf_block *multiq_tcf_block(struct Qdisc *sch, unsigned long cl)
374 375 376 377 378
{
	struct multiq_sched_data *q = qdisc_priv(sch);

	if (cl)
		return NULL;
379
	return q->block;
380 381 382 383 384
}

static const struct Qdisc_class_ops multiq_class_ops = {
	.graft		=	multiq_graft,
	.leaf		=	multiq_leaf,
385
	.find		=	multiq_find,
386
	.walk		=	multiq_walk,
387
	.tcf_block	=	multiq_tcf_block,
388
	.bind_tcf	=	multiq_bind,
389
	.unbind_tcf	=	multiq_unbind,
390 391 392 393 394 395 396 397 398 399 400
	.dump		=	multiq_dump_class,
	.dump_stats	=	multiq_dump_class_stats,
};

static struct Qdisc_ops multiq_qdisc_ops __read_mostly = {
	.next		=	NULL,
	.cl_ops		=	&multiq_class_ops,
	.id		=	"multiq",
	.priv_size	=	sizeof(struct multiq_sched_data),
	.enqueue	=	multiq_enqueue,
	.dequeue	=	multiq_dequeue,
401
	.peek		=	multiq_peek,
402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423
	.init		=	multiq_init,
	.reset		=	multiq_reset,
	.destroy	=	multiq_destroy,
	.change		=	multiq_tune,
	.dump		=	multiq_dump,
	.owner		=	THIS_MODULE,
};

static int __init multiq_module_init(void)
{
	return register_qdisc(&multiq_qdisc_ops);
}

static void __exit multiq_module_exit(void)
{
	unregister_qdisc(&multiq_qdisc_ops);
}

module_init(multiq_module_init)
module_exit(multiq_module_exit)

MODULE_LICENSE("GPL");