xref: /freebsd-13.1/contrib/ofed/libmlx5/mlx5.c (revision dc411487)
1 /*
2  * Copyright (c) 2012 Mellanox Technologies, Inc.  All rights reserved.
3  *
4  * This software is available to you under a choice of one of two
5  * licenses.  You may choose to be licensed under the terms of the GNU
6  * General Public License (GPL) Version 2, available from the file
7  * COPYING in the main directory of this source tree, or the
8  * OpenIB.org BSD license below:
9  *
10  *     Redistribution and use in source and binary forms, with or
11  *     without modification, are permitted provided that the following
12  *     conditions are met:
13  *
14  *      - Redistributions of source code must retain the above
15  *        copyright notice, this list of conditions and the following
16  *        disclaimer.
17  *
18  *      - Redistributions in binary form must reproduce the above
19  *        copyright notice, this list of conditions and the following
20  *        disclaimer in the documentation and/or other materials
21  *        provided with the distribution.
22  *
23  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
24  * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
25  * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
26  * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
27  * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
28  * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
29  * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
30  * SOFTWARE.
31  */
32 #define _GNU_SOURCE
33 #include <config.h>
34 
35 #include <stdio.h>
36 #include <stdlib.h>
37 #include <unistd.h>
38 #include <errno.h>
39 #include <sys/mman.h>
40 #include <pthread.h>
41 #include <string.h>
42 #include <sched.h>
43 #include <sys/param.h>
44 #include <sys/cpuset.h>
45 
46 #include "mlx5.h"
47 #include "mlx5-abi.h"
48 
49 #ifndef PCI_VENDOR_ID_MELLANOX
50 #define PCI_VENDOR_ID_MELLANOX			0x15b3
51 #endif
52 
53 #ifndef CPU_OR
54 #define CPU_OR(x, y, z) do {} while (0)
55 #endif
56 
57 #ifndef CPU_EQUAL
58 #define CPU_EQUAL(x, y) 1
59 #endif
60 
61 
62 #define HCA(v, d) \
63 	{ .vendor = PCI_VENDOR_ID_##v,			\
64 	  .device = d }
65 
66 static struct {
67 	unsigned		vendor;
68 	unsigned		device;
69 } hca_table[] = {
70 	HCA(MELLANOX, 4113),	/* MT4113 Connect-IB */
71 	HCA(MELLANOX, 4114),	/* Connect-IB Virtual Function */
72 	HCA(MELLANOX, 4115),	/* ConnectX-4 */
73 	HCA(MELLANOX, 4116),	/* ConnectX-4 Virtual Function */
74 	HCA(MELLANOX, 4117),	/* ConnectX-4LX */
75 	HCA(MELLANOX, 4118),	/* ConnectX-4LX Virtual Function */
76 	HCA(MELLANOX, 4119),	/* ConnectX-5, PCIe 3.0 */
77 	HCA(MELLANOX, 4120),	/* ConnectX-5 Virtual Function */
78 	HCA(MELLANOX, 4121),    /* ConnectX-5 Ex */
79 	HCA(MELLANOX, 4122),	/* ConnectX-5 Ex VF */
80 	HCA(MELLANOX, 4123),    /* ConnectX-6 */
81 	HCA(MELLANOX, 4124),	/* ConnectX-6 VF */
82 	HCA(MELLANOX, 4125),	/* ConnectX-6 DX */
83 	HCA(MELLANOX, 4126),	/* ConnectX family mlx5Gen Virtual Function */
84 	HCA(MELLANOX, 41682),	/* BlueField integrated ConnectX-5 network controller */
85 	HCA(MELLANOX, 41683),	/* BlueField integrated ConnectX-5 network controller VF */
86 };
87 
88 uint32_t mlx5_debug_mask = 0;
89 int mlx5_freeze_on_error_cqe;
90 
91 static struct ibv_context_ops mlx5_ctx_ops = {
92 	.query_device  = mlx5_query_device,
93 	.query_port    = mlx5_query_port,
94 	.alloc_pd      = mlx5_alloc_pd,
95 	.dealloc_pd    = mlx5_free_pd,
96 	.reg_mr	       = mlx5_reg_mr,
97 	.rereg_mr      = mlx5_rereg_mr,
98 	.dereg_mr      = mlx5_dereg_mr,
99 	.alloc_mw      = mlx5_alloc_mw,
100 	.dealloc_mw    = mlx5_dealloc_mw,
101 	.bind_mw       = mlx5_bind_mw,
102 	.create_cq     = mlx5_create_cq,
103 	.poll_cq       = mlx5_poll_cq,
104 	.req_notify_cq = mlx5_arm_cq,
105 	.cq_event      = mlx5_cq_event,
106 	.resize_cq     = mlx5_resize_cq,
107 	.destroy_cq    = mlx5_destroy_cq,
108 	.create_srq    = mlx5_create_srq,
109 	.modify_srq    = mlx5_modify_srq,
110 	.query_srq     = mlx5_query_srq,
111 	.destroy_srq   = mlx5_destroy_srq,
112 	.post_srq_recv = mlx5_post_srq_recv,
113 	.create_qp     = mlx5_create_qp,
114 	.query_qp      = mlx5_query_qp,
115 	.modify_qp     = mlx5_modify_qp,
116 	.destroy_qp    = mlx5_destroy_qp,
117 	.post_send     = mlx5_post_send,
118 	.post_recv     = mlx5_post_recv,
119 	.create_ah     = mlx5_create_ah,
120 	.destroy_ah    = mlx5_destroy_ah,
121 	.attach_mcast  = mlx5_attach_mcast,
122 	.detach_mcast  = mlx5_detach_mcast
123 };
124 
read_number_from_line(const char * line,int * value)125 static int read_number_from_line(const char *line, int *value)
126 {
127 	const char *ptr;
128 
129 	ptr = strchr(line, ':');
130 	if (!ptr)
131 		return 1;
132 
133 	++ptr;
134 
135 	*value = atoi(ptr);
136 	return 0;
137 }
138 /**
139  * The function looks for the first free user-index in all the
140  * user-index tables. If all are used, returns -1, otherwise
141  * a valid user-index.
142  * In case the reference count of the table is zero, it means the
143  * table is not in use and wasn't allocated yet, therefore the
144  * mlx5_store_uidx allocates the table, and increment the reference
145  * count on the table.
146  */
get_free_uidx(struct mlx5_context * ctx)147 static int32_t get_free_uidx(struct mlx5_context *ctx)
148 {
149 	int32_t tind;
150 	int32_t i;
151 
152 	for (tind = 0; tind < MLX5_UIDX_TABLE_SIZE; tind++) {
153 		if (ctx->uidx_table[tind].refcnt < MLX5_UIDX_TABLE_MASK)
154 			break;
155 	}
156 
157 	if (tind == MLX5_UIDX_TABLE_SIZE)
158 		return -1;
159 
160 	if (!ctx->uidx_table[tind].refcnt)
161 		return tind << MLX5_UIDX_TABLE_SHIFT;
162 
163 	for (i = 0; i < MLX5_UIDX_TABLE_MASK + 1; i++) {
164 		if (!ctx->uidx_table[tind].table[i])
165 			break;
166 	}
167 
168 	return (tind << MLX5_UIDX_TABLE_SHIFT) | i;
169 }
170 
mlx5_store_uidx(struct mlx5_context * ctx,void * rsc)171 int32_t mlx5_store_uidx(struct mlx5_context *ctx, void *rsc)
172 {
173 	int32_t tind;
174 	int32_t ret = -1;
175 	int32_t uidx;
176 
177 	pthread_mutex_lock(&ctx->uidx_table_mutex);
178 	uidx = get_free_uidx(ctx);
179 	if (uidx < 0)
180 		goto out;
181 
182 	tind = uidx >> MLX5_UIDX_TABLE_SHIFT;
183 
184 	if (!ctx->uidx_table[tind].refcnt) {
185 		ctx->uidx_table[tind].table = calloc(MLX5_UIDX_TABLE_MASK + 1,
186 						     sizeof(struct mlx5_resource *));
187 		if (!ctx->uidx_table[tind].table)
188 			goto out;
189 	}
190 
191 	++ctx->uidx_table[tind].refcnt;
192 	ctx->uidx_table[tind].table[uidx & MLX5_UIDX_TABLE_MASK] = rsc;
193 	ret = uidx;
194 
195 out:
196 	pthread_mutex_unlock(&ctx->uidx_table_mutex);
197 	return ret;
198 }
199 
mlx5_clear_uidx(struct mlx5_context * ctx,uint32_t uidx)200 void mlx5_clear_uidx(struct mlx5_context *ctx, uint32_t uidx)
201 {
202 	int tind = uidx >> MLX5_UIDX_TABLE_SHIFT;
203 
204 	pthread_mutex_lock(&ctx->uidx_table_mutex);
205 
206 	if (!--ctx->uidx_table[tind].refcnt)
207 		free(ctx->uidx_table[tind].table);
208 	else
209 		ctx->uidx_table[tind].table[uidx & MLX5_UIDX_TABLE_MASK] = NULL;
210 
211 	pthread_mutex_unlock(&ctx->uidx_table_mutex);
212 }
213 
mlx5_is_sandy_bridge(int * num_cores)214 static int mlx5_is_sandy_bridge(int *num_cores)
215 {
216 	char line[128];
217 	FILE *fd;
218 	int rc = 0;
219 	int cur_cpu_family = -1;
220 	int cur_cpu_model = -1;
221 
222 	fd = fopen("/proc/cpuinfo", "r");
223 	if (!fd)
224 		return 0;
225 
226 	*num_cores = 0;
227 
228 	while (fgets(line, 128, fd)) {
229 		int value;
230 
231 		/* if this is information on new processor */
232 		if (!strncmp(line, "processor", 9)) {
233 			++*num_cores;
234 
235 			cur_cpu_family = -1;
236 			cur_cpu_model  = -1;
237 		} else if (!strncmp(line, "cpu family", 10)) {
238 			if ((cur_cpu_family < 0) && (!read_number_from_line(line, &value)))
239 				cur_cpu_family = value;
240 		} else if (!strncmp(line, "model", 5)) {
241 			if ((cur_cpu_model < 0) && (!read_number_from_line(line, &value)))
242 				cur_cpu_model = value;
243 		}
244 
245 		/* if this is a Sandy Bridge CPU */
246 		if ((cur_cpu_family == 6) &&
247 		    (cur_cpu_model == 0x2A || (cur_cpu_model == 0x2D) ))
248 			rc = 1;
249 	}
250 
251 	fclose(fd);
252 	return rc;
253 }
254 
255 /*
256 man cpuset
257 
258   This format displays each 32-bit word in hexadecimal (using ASCII characters "0" - "9" and "a" - "f"); words
259   are filled with leading zeros, if required. For masks longer than one word, a comma separator is used between
260   words. Words are displayed in big-endian order, which has the most significant bit first. The hex digits
261   within a word are also in big-endian order.
262 
263   The number of 32-bit words displayed is the minimum number needed to display all bits of the bitmask, based on
264   the size of the bitmask.
265 
266   Examples of the Mask Format:
267 
268      00000001                        # just bit 0 set
269      40000000,00000000,00000000      # just bit 94 set
270      000000ff,00000000               # bits 32-39 set
271      00000000,000E3862               # 1,5,6,11-13,17-19 set
272 
273   A mask with bits 0, 1, 2, 4, 8, 16, 32, and 64 set displays as:
274 
275      00000001,00000001,00010117
276 
277   The first "1" is for bit 64, the second for bit 32, the third for bit 16, the fourth for bit 8, the fifth for
278   bit 4, and the "7" is for bits 2, 1, and 0.
279 */
mlx5_local_cpu_set(struct ibv_device * ibdev,cpuset_t * cpu_set)280 static void mlx5_local_cpu_set(struct ibv_device *ibdev, cpuset_t *cpu_set)
281 {
282 	char *p, buf[1024];
283 	char *env_value;
284 	uint32_t word;
285 	int i, k;
286 
287 	env_value = getenv("MLX5_LOCAL_CPUS");
288 	if (env_value)
289 		strncpy(buf, env_value, sizeof(buf));
290 	else {
291 		char fname[MAXPATHLEN];
292 
293 		snprintf(fname, MAXPATHLEN, "/sys/class/infiniband/%s",
294 			 ibv_get_device_name(ibdev));
295 
296 		if (ibv_read_sysfs_file(fname, "device/local_cpus", buf, sizeof(buf))) {
297 			fprintf(stderr, PFX "Warning: can not get local cpu set: failed to open %s\n", fname);
298 			return;
299 		}
300 	}
301 
302 	p = strrchr(buf, ',');
303 	if (!p)
304 		p = buf;
305 
306 	i = 0;
307 	do {
308 		if (*p == ',') {
309 			*p = 0;
310 			p ++;
311 		}
312 
313 		word = strtoul(p, NULL, 16);
314 
315 		for (k = 0; word; ++k, word >>= 1)
316 			if (word & 1)
317 				CPU_SET(k+i, cpu_set);
318 
319 		if (p == buf)
320 			break;
321 
322 		p = strrchr(buf, ',');
323 		if (!p)
324 			p = buf;
325 
326 		i += 32;
327 	} while (i < CPU_SETSIZE);
328 }
329 
mlx5_enable_sandy_bridge_fix(struct ibv_device * ibdev)330 static int mlx5_enable_sandy_bridge_fix(struct ibv_device *ibdev)
331 {
332 	cpuset_t my_cpus, dev_local_cpus, result_set;
333 	int stall_enable;
334 	int ret;
335 	int num_cores;
336 
337 	if (!mlx5_is_sandy_bridge(&num_cores))
338 		return 0;
339 
340 	/* by default enable stall on sandy bridge arch */
341 	stall_enable = 1;
342 
343 	/*
344 	 * check if app is bound to cpu set that is inside
345 	 * of device local cpu set. Disable stalling if true
346 	 */
347 
348 	/* use static cpu set - up to CPU_SETSIZE (1024) cpus/node */
349 	CPU_ZERO(&my_cpus);
350 	CPU_ZERO(&dev_local_cpus);
351 	CPU_ZERO(&result_set);
352 	ret = cpuset_getaffinity(CPU_LEVEL_WHICH, CPU_WHICH_PID, -1,
353 	    sizeof(my_cpus), &my_cpus);
354 	if (ret == -1) {
355 		if (errno == EINVAL)
356 			fprintf(stderr, PFX "Warning: my cpu set is too small\n");
357 		else
358 			fprintf(stderr, PFX "Warning: failed to get my cpu set\n");
359 		goto out;
360 	}
361 
362 	/* get device local cpu set */
363 	mlx5_local_cpu_set(ibdev, &dev_local_cpus);
364 
365 	/* check if my cpu set is in dev cpu */
366 	CPU_OR(&result_set, &my_cpus, &dev_local_cpus);
367 	stall_enable = CPU_EQUAL(&result_set, &dev_local_cpus) ? 0 : 1;
368 
369 out:
370 	return stall_enable;
371 }
372 
mlx5_read_env(struct ibv_device * ibdev,struct mlx5_context * ctx)373 static void mlx5_read_env(struct ibv_device *ibdev, struct mlx5_context *ctx)
374 {
375 	char *env_value;
376 
377 	env_value = getenv("MLX5_STALL_CQ_POLL");
378 	if (env_value)
379 		/* check if cq stall is enforced by user */
380 		ctx->stall_enable = (strcmp(env_value, "0")) ? 1 : 0;
381 	else
382 		/* autodetect if we need to do cq polling */
383 		ctx->stall_enable = mlx5_enable_sandy_bridge_fix(ibdev);
384 
385 	env_value = getenv("MLX5_STALL_NUM_LOOP");
386 	if (env_value)
387 		mlx5_stall_num_loop = atoi(env_value);
388 
389 	env_value = getenv("MLX5_STALL_CQ_POLL_MIN");
390 	if (env_value)
391 		mlx5_stall_cq_poll_min = atoi(env_value);
392 
393 	env_value = getenv("MLX5_STALL_CQ_POLL_MAX");
394 	if (env_value)
395 		mlx5_stall_cq_poll_max = atoi(env_value);
396 
397 	env_value = getenv("MLX5_STALL_CQ_INC_STEP");
398 	if (env_value)
399 		mlx5_stall_cq_inc_step = atoi(env_value);
400 
401 	env_value = getenv("MLX5_STALL_CQ_DEC_STEP");
402 	if (env_value)
403 		mlx5_stall_cq_dec_step = atoi(env_value);
404 
405 	ctx->stall_adaptive_enable = 0;
406 	ctx->stall_cycles = 0;
407 
408 	if (mlx5_stall_num_loop < 0) {
409 		ctx->stall_adaptive_enable = 1;
410 		ctx->stall_cycles = mlx5_stall_cq_poll_min;
411 	}
412 
413 }
414 
get_total_uuars(int page_size)415 static int get_total_uuars(int page_size)
416 {
417 	int size = MLX5_DEF_TOT_UUARS;
418 	int uuars_in_page;
419 	char *env;
420 
421 	env = getenv("MLX5_TOTAL_UUARS");
422 	if (env)
423 		size = atoi(env);
424 
425 	if (size < 1)
426 		return -EINVAL;
427 
428 	uuars_in_page = page_size / MLX5_ADAPTER_PAGE_SIZE * MLX5_NUM_NON_FP_BFREGS_PER_UAR;
429 	size = max(uuars_in_page, size);
430 	size = align(size, MLX5_NUM_NON_FP_BFREGS_PER_UAR);
431 	if (size > MLX5_MAX_BFREGS)
432 		return -ENOMEM;
433 
434 	return size;
435 }
436 
open_debug_file(struct mlx5_context * ctx)437 static void open_debug_file(struct mlx5_context *ctx)
438 {
439 	char *env;
440 
441 	env = getenv("MLX5_DEBUG_FILE");
442 	if (!env) {
443 		ctx->dbg_fp = stderr;
444 		return;
445 	}
446 
447 	ctx->dbg_fp = fopen(env, "aw+");
448 	if (!ctx->dbg_fp) {
449 		fprintf(stderr, "Failed opening debug file %s, using stderr\n", env);
450 		ctx->dbg_fp = stderr;
451 		return;
452 	}
453 }
454 
close_debug_file(struct mlx5_context * ctx)455 static void close_debug_file(struct mlx5_context *ctx)
456 {
457 	if (ctx->dbg_fp && ctx->dbg_fp != stderr)
458 		fclose(ctx->dbg_fp);
459 }
460 
set_debug_mask(void)461 static void set_debug_mask(void)
462 {
463 	char *env;
464 
465 	env = getenv("MLX5_DEBUG_MASK");
466 	if (env)
467 		mlx5_debug_mask = strtol(env, NULL, 0);
468 }
469 
set_freeze_on_error(void)470 static void set_freeze_on_error(void)
471 {
472 	char *env;
473 
474 	env = getenv("MLX5_FREEZE_ON_ERROR_CQE");
475 	if (env)
476 		mlx5_freeze_on_error_cqe = strtol(env, NULL, 0);
477 }
478 
get_always_bf(void)479 static int get_always_bf(void)
480 {
481 	char *env;
482 
483 	env = getenv("MLX5_POST_SEND_PREFER_BF");
484 	if (!env)
485 		return 1;
486 
487 	return strcmp(env, "0") ? 1 : 0;
488 }
489 
get_shut_up_bf(void)490 static int get_shut_up_bf(void)
491 {
492 	char *env;
493 
494 	env = getenv("MLX5_SHUT_UP_BF");
495 	if (!env)
496 		return 0;
497 
498 	return strcmp(env, "0") ? 1 : 0;
499 }
500 
get_num_low_lat_uuars(int tot_uuars)501 static int get_num_low_lat_uuars(int tot_uuars)
502 {
503 	char *env;
504 	int num = 4;
505 
506 	env = getenv("MLX5_NUM_LOW_LAT_UUARS");
507 	if (env)
508 		num = atoi(env);
509 
510 	if (num < 0)
511 		return -EINVAL;
512 
513 	num = max(num, tot_uuars - MLX5_MED_BFREGS_TSHOLD);
514 	return num;
515 }
516 
517 /* The library allocates an array of uuar contexts. The one in index zero does
518  * not to execersize odd/even policy so it can avoid a lock but it may not use
519  * blue flame. The upper ones, low_lat_uuars can use blue flame with no lock
520  * since they are assigned to one QP only. The rest can use blue flame but since
521  * they are shared they need a lock
522  */
need_uuar_lock(struct mlx5_context * ctx,int uuarn)523 static int need_uuar_lock(struct mlx5_context *ctx, int uuarn)
524 {
525 	if (uuarn == 0 || mlx5_single_threaded)
526 		return 0;
527 
528 	if (uuarn >= (ctx->tot_uuars - ctx->low_lat_uuars) * 2)
529 		return 0;
530 
531 	return 1;
532 }
533 
single_threaded_app(void)534 static int single_threaded_app(void)
535 {
536 
537 	char *env;
538 
539 	env = getenv("MLX5_SINGLE_THREADED");
540 	if (env)
541 		return strcmp(env, "1") ? 0 : 1;
542 
543 	return 0;
544 }
545 
mlx5_cmd_get_context(struct mlx5_context * context,struct mlx5_alloc_ucontext * req,size_t req_len,struct mlx5_alloc_ucontext_resp * resp,size_t resp_len)546 static int mlx5_cmd_get_context(struct mlx5_context *context,
547 				struct mlx5_alloc_ucontext *req,
548 				size_t req_len,
549 				struct mlx5_alloc_ucontext_resp *resp,
550 				size_t resp_len)
551 {
552 	if (!ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
553 				 req_len, &resp->ibv_resp, resp_len))
554 		return 0;
555 
556 	/* The ibv_cmd_get_context fails in older kernels when passing
557 	 * a request length that the kernel doesn't know.
558 	 * To avoid breaking compatibility of new libmlx5 and older
559 	 * kernels, when ibv_cmd_get_context fails with the full
560 	 * request length, we try once again with the legacy length.
561 	 * We repeat this process while reducing requested size based
562 	 * on the feature input size. To avoid this in the future, we
563 	 * will remove the check in kernel that requires fields unknown
564 	 * to the kernel to be cleared. This will require that any new
565 	 * feature that involves extending struct mlx5_alloc_ucontext
566 	 * will be accompanied by an indication in the form of one or
567 	 * more fields in struct mlx5_alloc_ucontext_resp. If the
568 	 * response value can be interpreted as feature not supported
569 	 * when the returned value is zero, this will suffice to
570 	 * indicate to the library that the request was ignored by the
571 	 * kernel, either because it is unaware or because it decided
572 	 * to do so. If zero is a valid response, we will add a new
573 	 * field that indicates whether the request was handled.
574 	 */
575 	if (!ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
576 				 offsetof(struct mlx5_alloc_ucontext, lib_caps),
577 				 &resp->ibv_resp, resp_len))
578 		return 0;
579 
580 	return ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
581 				   offsetof(struct mlx5_alloc_ucontext,
582 					    cqe_version),
583 				   &resp->ibv_resp, resp_len);
584 }
585 
mlx5_map_internal_clock(struct mlx5_device * mdev,struct ibv_context * ibv_ctx)586 static int mlx5_map_internal_clock(struct mlx5_device *mdev,
587 				   struct ibv_context *ibv_ctx)
588 {
589 	struct mlx5_context *context = to_mctx(ibv_ctx);
590 	void *hca_clock_page;
591 	off_t offset = 0;
592 
593 	set_command(MLX5_MMAP_GET_CORE_CLOCK_CMD, &offset);
594 	hca_clock_page = mmap(NULL, mdev->page_size,
595 			      PROT_READ, MAP_SHARED, ibv_ctx->cmd_fd,
596 			      mdev->page_size * offset);
597 
598 	if (hca_clock_page == MAP_FAILED) {
599 		fprintf(stderr, PFX
600 			"Warning: Timestamp available,\n"
601 			"but failed to mmap() hca core clock page.\n");
602 		return -1;
603 	}
604 
605 	context->hca_core_clock = hca_clock_page +
606 		(context->core_clock.offset & (mdev->page_size - 1));
607 	return 0;
608 }
609 
mlx5dv_query_device(struct ibv_context * ctx_in,struct mlx5dv_context * attrs_out)610 int mlx5dv_query_device(struct ibv_context *ctx_in,
611 			 struct mlx5dv_context *attrs_out)
612 {
613 	struct mlx5_context *mctx = to_mctx(ctx_in);
614 	uint64_t comp_mask_out = 0;
615 
616 	attrs_out->version   = 0;
617 	attrs_out->flags     = 0;
618 
619 	if (mctx->cqe_version == MLX5_CQE_VERSION_V1)
620 		attrs_out->flags |= MLX5DV_CONTEXT_FLAGS_CQE_V1;
621 
622 	if (mctx->vendor_cap_flags & MLX5_VENDOR_CAP_FLAGS_MPW)
623 		attrs_out->flags |= MLX5DV_CONTEXT_FLAGS_MPW;
624 
625 	if (attrs_out->comp_mask & MLX5DV_CONTEXT_MASK_CQE_COMPRESION) {
626 		attrs_out->cqe_comp_caps = mctx->cqe_comp_caps;
627 		comp_mask_out |= MLX5DV_CONTEXT_MASK_CQE_COMPRESION;
628 	}
629 
630 	attrs_out->comp_mask = comp_mask_out;
631 
632 	return 0;
633 }
634 
mlx5dv_get_qp(struct ibv_qp * qp_in,struct mlx5dv_qp * qp_out)635 static int mlx5dv_get_qp(struct ibv_qp *qp_in,
636 			 struct mlx5dv_qp *qp_out)
637 {
638 	struct mlx5_qp *mqp = to_mqp(qp_in);
639 
640 	qp_out->comp_mask = 0;
641 	qp_out->dbrec     = mqp->db;
642 
643 	if (mqp->sq_buf_size)
644 		/* IBV_QPT_RAW_PACKET */
645 		qp_out->sq.buf = (void *)((uintptr_t)mqp->sq_buf.buf);
646 	else
647 		qp_out->sq.buf = (void *)((uintptr_t)mqp->buf.buf + mqp->sq.offset);
648 	qp_out->sq.wqe_cnt = mqp->sq.wqe_cnt;
649 	qp_out->sq.stride  = 1 << mqp->sq.wqe_shift;
650 
651 	qp_out->rq.buf     = (void *)((uintptr_t)mqp->buf.buf + mqp->rq.offset);
652 	qp_out->rq.wqe_cnt = mqp->rq.wqe_cnt;
653 	qp_out->rq.stride  = 1 << mqp->rq.wqe_shift;
654 
655 	qp_out->bf.reg    = mqp->bf->reg;
656 
657 	if (mqp->bf->uuarn > 0)
658 		qp_out->bf.size = mqp->bf->buf_size;
659 	else
660 		qp_out->bf.size = 0;
661 
662 	return 0;
663 }
664 
mlx5dv_get_cq(struct ibv_cq * cq_in,struct mlx5dv_cq * cq_out)665 static int mlx5dv_get_cq(struct ibv_cq *cq_in,
666 			 struct mlx5dv_cq *cq_out)
667 {
668 	struct mlx5_cq *mcq = to_mcq(cq_in);
669 	struct mlx5_context *mctx = to_mctx(cq_in->context);
670 
671 	cq_out->comp_mask = 0;
672 	cq_out->cqn       = mcq->cqn;
673 	cq_out->cqe_cnt   = mcq->ibv_cq.cqe + 1;
674 	cq_out->cqe_size  = mcq->cqe_sz;
675 	cq_out->buf       = mcq->active_buf->buf;
676 	cq_out->dbrec     = mcq->dbrec;
677 	cq_out->uar	  = mctx->uar;
678 
679 	mcq->flags	 |= MLX5_CQ_FLAGS_DV_OWNED;
680 
681 	return 0;
682 }
683 
mlx5dv_get_rwq(struct ibv_wq * wq_in,struct mlx5dv_rwq * rwq_out)684 static int mlx5dv_get_rwq(struct ibv_wq *wq_in,
685 			  struct mlx5dv_rwq *rwq_out)
686 {
687 	struct mlx5_rwq *mrwq = to_mrwq(wq_in);
688 
689 	rwq_out->comp_mask = 0;
690 	rwq_out->buf       = mrwq->pbuff;
691 	rwq_out->dbrec     = mrwq->recv_db;
692 	rwq_out->wqe_cnt   = mrwq->rq.wqe_cnt;
693 	rwq_out->stride    = 1 << mrwq->rq.wqe_shift;
694 
695 	return 0;
696 }
697 
mlx5dv_get_srq(struct ibv_srq * srq_in,struct mlx5dv_srq * srq_out)698 static int mlx5dv_get_srq(struct ibv_srq *srq_in,
699 			  struct mlx5dv_srq *srq_out)
700 {
701 	struct mlx5_srq *msrq;
702 
703 	msrq = container_of(srq_in, struct mlx5_srq, vsrq.srq);
704 
705 	srq_out->comp_mask = 0;
706 	srq_out->buf       = msrq->buf.buf;
707 	srq_out->dbrec     = msrq->db;
708 	srq_out->stride    = 1 << msrq->wqe_shift;
709 	srq_out->head      = msrq->head;
710 	srq_out->tail      = msrq->tail;
711 
712 	return 0;
713 }
714 
mlx5dv_init_obj(struct mlx5dv_obj * obj,uint64_t obj_type)715 int mlx5dv_init_obj(struct mlx5dv_obj *obj, uint64_t obj_type)
716 {
717 	int ret = 0;
718 
719 	if (obj_type & MLX5DV_OBJ_QP)
720 		ret = mlx5dv_get_qp(obj->qp.in, obj->qp.out);
721 	if (!ret && (obj_type & MLX5DV_OBJ_CQ))
722 		ret = mlx5dv_get_cq(obj->cq.in, obj->cq.out);
723 	if (!ret && (obj_type & MLX5DV_OBJ_SRQ))
724 		ret = mlx5dv_get_srq(obj->srq.in, obj->srq.out);
725 	if (!ret && (obj_type & MLX5DV_OBJ_RWQ))
726 		ret = mlx5dv_get_rwq(obj->rwq.in, obj->rwq.out);
727 
728 	return ret;
729 }
730 
adjust_uar_info(struct mlx5_device * mdev,struct mlx5_context * context,struct mlx5_alloc_ucontext_resp resp)731 static void adjust_uar_info(struct mlx5_device *mdev,
732 			    struct mlx5_context *context,
733 			    struct mlx5_alloc_ucontext_resp resp)
734 {
735 	if (!resp.log_uar_size && !resp.num_uars_per_page) {
736 		/* old kernel */
737 		context->uar_size = mdev->page_size;
738 		context->num_uars_per_page = 1;
739 		return;
740 	}
741 
742 	context->uar_size = 1 << resp.log_uar_size;
743 	context->num_uars_per_page = resp.num_uars_per_page;
744 }
745 
mlx5_init_context(struct verbs_device * vdev,struct ibv_context * ctx,int cmd_fd)746 static int mlx5_init_context(struct verbs_device *vdev,
747 			     struct ibv_context *ctx, int cmd_fd)
748 {
749 	struct mlx5_context	       *context;
750 	struct mlx5_alloc_ucontext	req;
751 	struct mlx5_alloc_ucontext_resp resp;
752 	int				i;
753 	int				page_size;
754 	int				tot_uuars;
755 	int				low_lat_uuars;
756 	int				gross_uuars;
757 	int				j;
758 	off_t				offset;
759 	struct mlx5_device	       *mdev;
760 	struct verbs_context	       *v_ctx;
761 	struct ibv_port_attr		port_attr;
762 	struct ibv_device_attr_ex	device_attr;
763 	int				k;
764 	int				bfi;
765 	int				num_sys_page_map;
766 
767 	mdev = to_mdev(&vdev->device);
768 	v_ctx = verbs_get_ctx(ctx);
769 	page_size = mdev->page_size;
770 	mlx5_single_threaded = single_threaded_app();
771 
772 	context = to_mctx(ctx);
773 	context->ibv_ctx.cmd_fd = cmd_fd;
774 
775 	open_debug_file(context);
776 	set_debug_mask();
777 	set_freeze_on_error();
778 	if (gethostname(context->hostname, sizeof(context->hostname)))
779 		strcpy(context->hostname, "host_unknown");
780 
781 	tot_uuars = get_total_uuars(page_size);
782 	if (tot_uuars < 0) {
783 		errno = -tot_uuars;
784 		goto err_free;
785 	}
786 
787 	low_lat_uuars = get_num_low_lat_uuars(tot_uuars);
788 	if (low_lat_uuars < 0) {
789 		errno = -low_lat_uuars;
790 		goto err_free;
791 	}
792 
793 	if (low_lat_uuars > tot_uuars - 1) {
794 		errno = ENOMEM;
795 		goto err_free;
796 	}
797 
798 	memset(&req, 0, sizeof(req));
799 	memset(&resp, 0, sizeof(resp));
800 
801 	req.total_num_uuars = tot_uuars;
802 	req.num_low_latency_uuars = low_lat_uuars;
803 	req.cqe_version = MLX5_CQE_VERSION_V1;
804 	req.lib_caps |= MLX5_LIB_CAP_4K_UAR;
805 
806 	if (mlx5_cmd_get_context(context, &req, sizeof(req), &resp,
807 				 sizeof(resp)))
808 		goto err_free;
809 
810 	context->max_num_qps		= resp.qp_tab_size;
811 	context->bf_reg_size		= resp.bf_reg_size;
812 	context->tot_uuars		= resp.tot_uuars;
813 	context->low_lat_uuars		= low_lat_uuars;
814 	context->cache_line_size	= resp.cache_line_size;
815 	context->max_sq_desc_sz = resp.max_sq_desc_sz;
816 	context->max_rq_desc_sz = resp.max_rq_desc_sz;
817 	context->max_send_wqebb	= resp.max_send_wqebb;
818 	context->num_ports	= resp.num_ports;
819 	context->max_recv_wr	= resp.max_recv_wr;
820 	context->max_srq_recv_wr = resp.max_srq_recv_wr;
821 
822 	context->cqe_version = resp.cqe_version;
823 	if (context->cqe_version) {
824 		if (context->cqe_version == MLX5_CQE_VERSION_V1)
825 			mlx5_ctx_ops.poll_cq = mlx5_poll_cq_v1;
826 		else
827 			goto err_free;
828 	}
829 
830 	adjust_uar_info(mdev, context, resp);
831 
832 	gross_uuars = context->tot_uuars / MLX5_NUM_NON_FP_BFREGS_PER_UAR * NUM_BFREGS_PER_UAR;
833 	context->bfs = calloc(gross_uuars, sizeof(*context->bfs));
834 	if (!context->bfs) {
835 		errno = ENOMEM;
836 		goto err_free;
837 	}
838 
839 	context->cmds_supp_uhw = resp.cmds_supp_uhw;
840 	context->vendor_cap_flags = 0;
841 
842 	pthread_mutex_init(&context->qp_table_mutex, NULL);
843 	pthread_mutex_init(&context->srq_table_mutex, NULL);
844 	pthread_mutex_init(&context->uidx_table_mutex, NULL);
845 	for (i = 0; i < MLX5_QP_TABLE_SIZE; ++i)
846 		context->qp_table[i].refcnt = 0;
847 
848 	for (i = 0; i < MLX5_QP_TABLE_SIZE; ++i)
849 		context->uidx_table[i].refcnt = 0;
850 
851 	context->db_list = NULL;
852 
853 	pthread_mutex_init(&context->db_list_mutex, NULL);
854 
855 	num_sys_page_map = context->tot_uuars / (context->num_uars_per_page * MLX5_NUM_NON_FP_BFREGS_PER_UAR);
856 	for (i = 0; i < num_sys_page_map; ++i) {
857 		offset = 0;
858 		set_command(MLX5_MMAP_GET_REGULAR_PAGES_CMD, &offset);
859 		set_index(i, &offset);
860 		context->uar[i] = mmap(NULL, page_size, PROT_WRITE, MAP_SHARED,
861 				       cmd_fd, page_size * offset);
862 		if (context->uar[i] == MAP_FAILED) {
863 			context->uar[i] = NULL;
864 			goto err_free_bf;
865 		}
866 	}
867 
868 	for (i = 0; i < num_sys_page_map; i++) {
869 		for (j = 0; j < context->num_uars_per_page; j++) {
870 			for (k = 0; k < NUM_BFREGS_PER_UAR; k++) {
871 				bfi = (i * context->num_uars_per_page + j) * NUM_BFREGS_PER_UAR + k;
872 				context->bfs[bfi].reg = context->uar[i] + MLX5_ADAPTER_PAGE_SIZE * j +
873 							MLX5_BF_OFFSET + k * context->bf_reg_size;
874 				context->bfs[bfi].need_lock = need_uuar_lock(context, bfi);
875 				mlx5_spinlock_init(&context->bfs[bfi].lock);
876 				context->bfs[bfi].offset = 0;
877 				if (bfi)
878 					context->bfs[bfi].buf_size = context->bf_reg_size / 2;
879 				context->bfs[bfi].uuarn = bfi;
880 			}
881 		}
882 	}
883 	context->hca_core_clock = NULL;
884 	if (resp.response_length + sizeof(resp.ibv_resp) >=
885 	    offsetof(struct mlx5_alloc_ucontext_resp, hca_core_clock_offset) +
886 	    sizeof(resp.hca_core_clock_offset) &&
887 	    resp.comp_mask & MLX5_IB_ALLOC_UCONTEXT_RESP_MASK_CORE_CLOCK_OFFSET) {
888 		context->core_clock.offset = resp.hca_core_clock_offset;
889 		mlx5_map_internal_clock(mdev, ctx);
890 	}
891 
892 	mlx5_spinlock_init(&context->lock32);
893 
894 	context->prefer_bf = get_always_bf();
895 	context->shut_up_bf = get_shut_up_bf();
896 	mlx5_read_env(&vdev->device, context);
897 
898 	mlx5_spinlock_init(&context->hugetlb_lock);
899 	TAILQ_INIT(&context->hugetlb_list);
900 
901 	context->ibv_ctx.ops = mlx5_ctx_ops;
902 
903 	verbs_set_ctx_op(v_ctx, create_qp_ex, mlx5_create_qp_ex);
904 	verbs_set_ctx_op(v_ctx, open_xrcd, mlx5_open_xrcd);
905 	verbs_set_ctx_op(v_ctx, close_xrcd, mlx5_close_xrcd);
906 	verbs_set_ctx_op(v_ctx, create_srq_ex, mlx5_create_srq_ex);
907 	verbs_set_ctx_op(v_ctx, get_srq_num, mlx5_get_srq_num);
908 	verbs_set_ctx_op(v_ctx, query_device_ex, mlx5_query_device_ex);
909 	verbs_set_ctx_op(v_ctx, query_rt_values, mlx5_query_rt_values);
910 	verbs_set_ctx_op(v_ctx, ibv_create_flow, ibv_cmd_create_flow);
911 	verbs_set_ctx_op(v_ctx, ibv_destroy_flow, ibv_cmd_destroy_flow);
912 	verbs_set_ctx_op(v_ctx, create_cq_ex, mlx5_create_cq_ex);
913 	verbs_set_ctx_op(v_ctx, create_wq, mlx5_create_wq);
914 	verbs_set_ctx_op(v_ctx, modify_wq, mlx5_modify_wq);
915 	verbs_set_ctx_op(v_ctx, destroy_wq, mlx5_destroy_wq);
916 	verbs_set_ctx_op(v_ctx, create_rwq_ind_table, mlx5_create_rwq_ind_table);
917 	verbs_set_ctx_op(v_ctx, destroy_rwq_ind_table, mlx5_destroy_rwq_ind_table);
918 
919 	memset(&device_attr, 0, sizeof(device_attr));
920 	if (!mlx5_query_device_ex(ctx, NULL, &device_attr,
921 				  sizeof(struct ibv_device_attr_ex))) {
922 		context->cached_device_cap_flags =
923 			device_attr.orig_attr.device_cap_flags;
924 		context->atomic_cap = device_attr.orig_attr.atomic_cap;
925 		context->cached_tso_caps = device_attr.tso_caps;
926 	}
927 
928 	for (j = 0; j < min(MLX5_MAX_PORTS_NUM, context->num_ports); ++j) {
929 		memset(&port_attr, 0, sizeof(port_attr));
930 		if (!mlx5_query_port(ctx, j + 1, &port_attr))
931 			context->cached_link_layer[j] = port_attr.link_layer;
932 	}
933 
934 	return 0;
935 
936 err_free_bf:
937 	free(context->bfs);
938 
939 err_free:
940 	for (i = 0; i < MLX5_MAX_UARS; ++i) {
941 		if (context->uar[i])
942 			munmap(context->uar[i], page_size);
943 	}
944 	close_debug_file(context);
945 	return errno;
946 }
947 
mlx5_cleanup_context(struct verbs_device * device,struct ibv_context * ibctx)948 static void mlx5_cleanup_context(struct verbs_device *device,
949 				 struct ibv_context *ibctx)
950 {
951 	struct mlx5_context *context = to_mctx(ibctx);
952 	int page_size = to_mdev(ibctx->device)->page_size;
953 	int i;
954 
955 	free(context->bfs);
956 	for (i = 0; i < MLX5_MAX_UARS; ++i) {
957 		if (context->uar[i])
958 			munmap(context->uar[i], page_size);
959 	}
960 	if (context->hca_core_clock)
961 		munmap(context->hca_core_clock - context->core_clock.offset,
962 		       page_size);
963 	close_debug_file(context);
964 }
965 
966 static struct verbs_device_ops mlx5_dev_ops = {
967 	.init_context = mlx5_init_context,
968 	.uninit_context = mlx5_cleanup_context,
969 };
970 
mlx5_driver_init(const char * uverbs_sys_path,int abi_version)971 static struct verbs_device *mlx5_driver_init(const char *uverbs_sys_path,
972 					     int abi_version)
973 {
974 	char			value[8];
975 	struct mlx5_device     *dev;
976 	unsigned		vendor, device;
977 	int			i;
978 
979 	if (ibv_read_sysfs_file(uverbs_sys_path, "device/vendor",
980 				value, sizeof value) < 0)
981 		return NULL;
982 	sscanf(value, "%i", &vendor);
983 
984 	if (ibv_read_sysfs_file(uverbs_sys_path, "device/device",
985 				value, sizeof value) < 0)
986 		return NULL;
987 	sscanf(value, "%i", &device);
988 
989 	for (i = 0; i < sizeof hca_table / sizeof hca_table[0]; ++i)
990 		if (vendor == hca_table[i].vendor &&
991 		    device == hca_table[i].device)
992 			goto found;
993 
994 	return NULL;
995 
996 found:
997 	if (abi_version < MLX5_UVERBS_MIN_ABI_VERSION ||
998 	    abi_version > MLX5_UVERBS_MAX_ABI_VERSION) {
999 		fprintf(stderr, PFX "Fatal: ABI version %d of %s is not supported "
1000 			"(min supported %d, max supported %d)\n",
1001 			abi_version, uverbs_sys_path,
1002 			MLX5_UVERBS_MIN_ABI_VERSION,
1003 			MLX5_UVERBS_MAX_ABI_VERSION);
1004 		return NULL;
1005 	}
1006 
1007 	dev = calloc(1, sizeof *dev);
1008 	if (!dev) {
1009 		fprintf(stderr, PFX "Fatal: couldn't allocate device for %s\n",
1010 			uverbs_sys_path);
1011 		return NULL;
1012 	}
1013 
1014 	dev->page_size   = sysconf(_SC_PAGESIZE);
1015 	dev->driver_abi_ver = abi_version;
1016 
1017 	dev->verbs_dev.ops = &mlx5_dev_ops;
1018 	dev->verbs_dev.sz = sizeof(*dev);
1019 	dev->verbs_dev.size_of_context = sizeof(struct mlx5_context) -
1020 		sizeof(struct ibv_context);
1021 
1022 	return &dev->verbs_dev;
1023 }
1024 
mlx5_register_driver(void)1025 static __attribute__((constructor)) void mlx5_register_driver(void)
1026 {
1027 	verbs_register_driver("mlx5", mlx5_driver_init);
1028 }
1029