1 /*
2 * Copyright (c) 2012 Mellanox Technologies, Inc. All rights reserved.
3 *
4 * This software is available to you under a choice of one of two
5 * licenses. You may choose to be licensed under the terms of the GNU
6 * General Public License (GPL) Version 2, available from the file
7 * COPYING in the main directory of this source tree, or the
8 * OpenIB.org BSD license below:
9 *
10 * Redistribution and use in source and binary forms, with or
11 * without modification, are permitted provided that the following
12 * conditions are met:
13 *
14 * - Redistributions of source code must retain the above
15 * copyright notice, this list of conditions and the following
16 * disclaimer.
17 *
18 * - Redistributions in binary form must reproduce the above
19 * copyright notice, this list of conditions and the following
20 * disclaimer in the documentation and/or other materials
21 * provided with the distribution.
22 *
23 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
24 * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
25 * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
26 * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
27 * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
28 * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
29 * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
30 * SOFTWARE.
31 */
32 #define _GNU_SOURCE
33 #include <config.h>
34
35 #include <stdio.h>
36 #include <stdlib.h>
37 #include <unistd.h>
38 #include <errno.h>
39 #include <sys/mman.h>
40 #include <pthread.h>
41 #include <string.h>
42 #include <sched.h>
43 #include <sys/param.h>
44 #include <sys/cpuset.h>
45
46 #include "mlx5.h"
47 #include "mlx5-abi.h"
48
49 #ifndef PCI_VENDOR_ID_MELLANOX
50 #define PCI_VENDOR_ID_MELLANOX 0x15b3
51 #endif
52
53 #ifndef CPU_OR
54 #define CPU_OR(x, y, z) do {} while (0)
55 #endif
56
57 #ifndef CPU_EQUAL
58 #define CPU_EQUAL(x, y) 1
59 #endif
60
61
62 #define HCA(v, d) \
63 { .vendor = PCI_VENDOR_ID_##v, \
64 .device = d }
65
66 static struct {
67 unsigned vendor;
68 unsigned device;
69 } hca_table[] = {
70 HCA(MELLANOX, 4113), /* MT4113 Connect-IB */
71 HCA(MELLANOX, 4114), /* Connect-IB Virtual Function */
72 HCA(MELLANOX, 4115), /* ConnectX-4 */
73 HCA(MELLANOX, 4116), /* ConnectX-4 Virtual Function */
74 HCA(MELLANOX, 4117), /* ConnectX-4LX */
75 HCA(MELLANOX, 4118), /* ConnectX-4LX Virtual Function */
76 HCA(MELLANOX, 4119), /* ConnectX-5, PCIe 3.0 */
77 HCA(MELLANOX, 4120), /* ConnectX-5 Virtual Function */
78 HCA(MELLANOX, 4121), /* ConnectX-5 Ex */
79 HCA(MELLANOX, 4122), /* ConnectX-5 Ex VF */
80 HCA(MELLANOX, 4123), /* ConnectX-6 */
81 HCA(MELLANOX, 4124), /* ConnectX-6 VF */
82 HCA(MELLANOX, 4125), /* ConnectX-6 DX */
83 HCA(MELLANOX, 4126), /* ConnectX family mlx5Gen Virtual Function */
84 HCA(MELLANOX, 41682), /* BlueField integrated ConnectX-5 network controller */
85 HCA(MELLANOX, 41683), /* BlueField integrated ConnectX-5 network controller VF */
86 };
87
88 uint32_t mlx5_debug_mask = 0;
89 int mlx5_freeze_on_error_cqe;
90
91 static struct ibv_context_ops mlx5_ctx_ops = {
92 .query_device = mlx5_query_device,
93 .query_port = mlx5_query_port,
94 .alloc_pd = mlx5_alloc_pd,
95 .dealloc_pd = mlx5_free_pd,
96 .reg_mr = mlx5_reg_mr,
97 .rereg_mr = mlx5_rereg_mr,
98 .dereg_mr = mlx5_dereg_mr,
99 .alloc_mw = mlx5_alloc_mw,
100 .dealloc_mw = mlx5_dealloc_mw,
101 .bind_mw = mlx5_bind_mw,
102 .create_cq = mlx5_create_cq,
103 .poll_cq = mlx5_poll_cq,
104 .req_notify_cq = mlx5_arm_cq,
105 .cq_event = mlx5_cq_event,
106 .resize_cq = mlx5_resize_cq,
107 .destroy_cq = mlx5_destroy_cq,
108 .create_srq = mlx5_create_srq,
109 .modify_srq = mlx5_modify_srq,
110 .query_srq = mlx5_query_srq,
111 .destroy_srq = mlx5_destroy_srq,
112 .post_srq_recv = mlx5_post_srq_recv,
113 .create_qp = mlx5_create_qp,
114 .query_qp = mlx5_query_qp,
115 .modify_qp = mlx5_modify_qp,
116 .destroy_qp = mlx5_destroy_qp,
117 .post_send = mlx5_post_send,
118 .post_recv = mlx5_post_recv,
119 .create_ah = mlx5_create_ah,
120 .destroy_ah = mlx5_destroy_ah,
121 .attach_mcast = mlx5_attach_mcast,
122 .detach_mcast = mlx5_detach_mcast
123 };
124
read_number_from_line(const char * line,int * value)125 static int read_number_from_line(const char *line, int *value)
126 {
127 const char *ptr;
128
129 ptr = strchr(line, ':');
130 if (!ptr)
131 return 1;
132
133 ++ptr;
134
135 *value = atoi(ptr);
136 return 0;
137 }
138 /**
139 * The function looks for the first free user-index in all the
140 * user-index tables. If all are used, returns -1, otherwise
141 * a valid user-index.
142 * In case the reference count of the table is zero, it means the
143 * table is not in use and wasn't allocated yet, therefore the
144 * mlx5_store_uidx allocates the table, and increment the reference
145 * count on the table.
146 */
get_free_uidx(struct mlx5_context * ctx)147 static int32_t get_free_uidx(struct mlx5_context *ctx)
148 {
149 int32_t tind;
150 int32_t i;
151
152 for (tind = 0; tind < MLX5_UIDX_TABLE_SIZE; tind++) {
153 if (ctx->uidx_table[tind].refcnt < MLX5_UIDX_TABLE_MASK)
154 break;
155 }
156
157 if (tind == MLX5_UIDX_TABLE_SIZE)
158 return -1;
159
160 if (!ctx->uidx_table[tind].refcnt)
161 return tind << MLX5_UIDX_TABLE_SHIFT;
162
163 for (i = 0; i < MLX5_UIDX_TABLE_MASK + 1; i++) {
164 if (!ctx->uidx_table[tind].table[i])
165 break;
166 }
167
168 return (tind << MLX5_UIDX_TABLE_SHIFT) | i;
169 }
170
mlx5_store_uidx(struct mlx5_context * ctx,void * rsc)171 int32_t mlx5_store_uidx(struct mlx5_context *ctx, void *rsc)
172 {
173 int32_t tind;
174 int32_t ret = -1;
175 int32_t uidx;
176
177 pthread_mutex_lock(&ctx->uidx_table_mutex);
178 uidx = get_free_uidx(ctx);
179 if (uidx < 0)
180 goto out;
181
182 tind = uidx >> MLX5_UIDX_TABLE_SHIFT;
183
184 if (!ctx->uidx_table[tind].refcnt) {
185 ctx->uidx_table[tind].table = calloc(MLX5_UIDX_TABLE_MASK + 1,
186 sizeof(struct mlx5_resource *));
187 if (!ctx->uidx_table[tind].table)
188 goto out;
189 }
190
191 ++ctx->uidx_table[tind].refcnt;
192 ctx->uidx_table[tind].table[uidx & MLX5_UIDX_TABLE_MASK] = rsc;
193 ret = uidx;
194
195 out:
196 pthread_mutex_unlock(&ctx->uidx_table_mutex);
197 return ret;
198 }
199
mlx5_clear_uidx(struct mlx5_context * ctx,uint32_t uidx)200 void mlx5_clear_uidx(struct mlx5_context *ctx, uint32_t uidx)
201 {
202 int tind = uidx >> MLX5_UIDX_TABLE_SHIFT;
203
204 pthread_mutex_lock(&ctx->uidx_table_mutex);
205
206 if (!--ctx->uidx_table[tind].refcnt)
207 free(ctx->uidx_table[tind].table);
208 else
209 ctx->uidx_table[tind].table[uidx & MLX5_UIDX_TABLE_MASK] = NULL;
210
211 pthread_mutex_unlock(&ctx->uidx_table_mutex);
212 }
213
mlx5_is_sandy_bridge(int * num_cores)214 static int mlx5_is_sandy_bridge(int *num_cores)
215 {
216 char line[128];
217 FILE *fd;
218 int rc = 0;
219 int cur_cpu_family = -1;
220 int cur_cpu_model = -1;
221
222 fd = fopen("/proc/cpuinfo", "r");
223 if (!fd)
224 return 0;
225
226 *num_cores = 0;
227
228 while (fgets(line, 128, fd)) {
229 int value;
230
231 /* if this is information on new processor */
232 if (!strncmp(line, "processor", 9)) {
233 ++*num_cores;
234
235 cur_cpu_family = -1;
236 cur_cpu_model = -1;
237 } else if (!strncmp(line, "cpu family", 10)) {
238 if ((cur_cpu_family < 0) && (!read_number_from_line(line, &value)))
239 cur_cpu_family = value;
240 } else if (!strncmp(line, "model", 5)) {
241 if ((cur_cpu_model < 0) && (!read_number_from_line(line, &value)))
242 cur_cpu_model = value;
243 }
244
245 /* if this is a Sandy Bridge CPU */
246 if ((cur_cpu_family == 6) &&
247 (cur_cpu_model == 0x2A || (cur_cpu_model == 0x2D) ))
248 rc = 1;
249 }
250
251 fclose(fd);
252 return rc;
253 }
254
255 /*
256 man cpuset
257
258 This format displays each 32-bit word in hexadecimal (using ASCII characters "0" - "9" and "a" - "f"); words
259 are filled with leading zeros, if required. For masks longer than one word, a comma separator is used between
260 words. Words are displayed in big-endian order, which has the most significant bit first. The hex digits
261 within a word are also in big-endian order.
262
263 The number of 32-bit words displayed is the minimum number needed to display all bits of the bitmask, based on
264 the size of the bitmask.
265
266 Examples of the Mask Format:
267
268 00000001 # just bit 0 set
269 40000000,00000000,00000000 # just bit 94 set
270 000000ff,00000000 # bits 32-39 set
271 00000000,000E3862 # 1,5,6,11-13,17-19 set
272
273 A mask with bits 0, 1, 2, 4, 8, 16, 32, and 64 set displays as:
274
275 00000001,00000001,00010117
276
277 The first "1" is for bit 64, the second for bit 32, the third for bit 16, the fourth for bit 8, the fifth for
278 bit 4, and the "7" is for bits 2, 1, and 0.
279 */
mlx5_local_cpu_set(struct ibv_device * ibdev,cpuset_t * cpu_set)280 static void mlx5_local_cpu_set(struct ibv_device *ibdev, cpuset_t *cpu_set)
281 {
282 char *p, buf[1024];
283 char *env_value;
284 uint32_t word;
285 int i, k;
286
287 env_value = getenv("MLX5_LOCAL_CPUS");
288 if (env_value)
289 strncpy(buf, env_value, sizeof(buf));
290 else {
291 char fname[MAXPATHLEN];
292
293 snprintf(fname, MAXPATHLEN, "/sys/class/infiniband/%s",
294 ibv_get_device_name(ibdev));
295
296 if (ibv_read_sysfs_file(fname, "device/local_cpus", buf, sizeof(buf))) {
297 fprintf(stderr, PFX "Warning: can not get local cpu set: failed to open %s\n", fname);
298 return;
299 }
300 }
301
302 p = strrchr(buf, ',');
303 if (!p)
304 p = buf;
305
306 i = 0;
307 do {
308 if (*p == ',') {
309 *p = 0;
310 p ++;
311 }
312
313 word = strtoul(p, NULL, 16);
314
315 for (k = 0; word; ++k, word >>= 1)
316 if (word & 1)
317 CPU_SET(k+i, cpu_set);
318
319 if (p == buf)
320 break;
321
322 p = strrchr(buf, ',');
323 if (!p)
324 p = buf;
325
326 i += 32;
327 } while (i < CPU_SETSIZE);
328 }
329
mlx5_enable_sandy_bridge_fix(struct ibv_device * ibdev)330 static int mlx5_enable_sandy_bridge_fix(struct ibv_device *ibdev)
331 {
332 cpuset_t my_cpus, dev_local_cpus, result_set;
333 int stall_enable;
334 int ret;
335 int num_cores;
336
337 if (!mlx5_is_sandy_bridge(&num_cores))
338 return 0;
339
340 /* by default enable stall on sandy bridge arch */
341 stall_enable = 1;
342
343 /*
344 * check if app is bound to cpu set that is inside
345 * of device local cpu set. Disable stalling if true
346 */
347
348 /* use static cpu set - up to CPU_SETSIZE (1024) cpus/node */
349 CPU_ZERO(&my_cpus);
350 CPU_ZERO(&dev_local_cpus);
351 CPU_ZERO(&result_set);
352 ret = cpuset_getaffinity(CPU_LEVEL_WHICH, CPU_WHICH_PID, -1,
353 sizeof(my_cpus), &my_cpus);
354 if (ret == -1) {
355 if (errno == EINVAL)
356 fprintf(stderr, PFX "Warning: my cpu set is too small\n");
357 else
358 fprintf(stderr, PFX "Warning: failed to get my cpu set\n");
359 goto out;
360 }
361
362 /* get device local cpu set */
363 mlx5_local_cpu_set(ibdev, &dev_local_cpus);
364
365 /* check if my cpu set is in dev cpu */
366 CPU_OR(&result_set, &my_cpus, &dev_local_cpus);
367 stall_enable = CPU_EQUAL(&result_set, &dev_local_cpus) ? 0 : 1;
368
369 out:
370 return stall_enable;
371 }
372
mlx5_read_env(struct ibv_device * ibdev,struct mlx5_context * ctx)373 static void mlx5_read_env(struct ibv_device *ibdev, struct mlx5_context *ctx)
374 {
375 char *env_value;
376
377 env_value = getenv("MLX5_STALL_CQ_POLL");
378 if (env_value)
379 /* check if cq stall is enforced by user */
380 ctx->stall_enable = (strcmp(env_value, "0")) ? 1 : 0;
381 else
382 /* autodetect if we need to do cq polling */
383 ctx->stall_enable = mlx5_enable_sandy_bridge_fix(ibdev);
384
385 env_value = getenv("MLX5_STALL_NUM_LOOP");
386 if (env_value)
387 mlx5_stall_num_loop = atoi(env_value);
388
389 env_value = getenv("MLX5_STALL_CQ_POLL_MIN");
390 if (env_value)
391 mlx5_stall_cq_poll_min = atoi(env_value);
392
393 env_value = getenv("MLX5_STALL_CQ_POLL_MAX");
394 if (env_value)
395 mlx5_stall_cq_poll_max = atoi(env_value);
396
397 env_value = getenv("MLX5_STALL_CQ_INC_STEP");
398 if (env_value)
399 mlx5_stall_cq_inc_step = atoi(env_value);
400
401 env_value = getenv("MLX5_STALL_CQ_DEC_STEP");
402 if (env_value)
403 mlx5_stall_cq_dec_step = atoi(env_value);
404
405 ctx->stall_adaptive_enable = 0;
406 ctx->stall_cycles = 0;
407
408 if (mlx5_stall_num_loop < 0) {
409 ctx->stall_adaptive_enable = 1;
410 ctx->stall_cycles = mlx5_stall_cq_poll_min;
411 }
412
413 }
414
get_total_uuars(int page_size)415 static int get_total_uuars(int page_size)
416 {
417 int size = MLX5_DEF_TOT_UUARS;
418 int uuars_in_page;
419 char *env;
420
421 env = getenv("MLX5_TOTAL_UUARS");
422 if (env)
423 size = atoi(env);
424
425 if (size < 1)
426 return -EINVAL;
427
428 uuars_in_page = page_size / MLX5_ADAPTER_PAGE_SIZE * MLX5_NUM_NON_FP_BFREGS_PER_UAR;
429 size = max(uuars_in_page, size);
430 size = align(size, MLX5_NUM_NON_FP_BFREGS_PER_UAR);
431 if (size > MLX5_MAX_BFREGS)
432 return -ENOMEM;
433
434 return size;
435 }
436
open_debug_file(struct mlx5_context * ctx)437 static void open_debug_file(struct mlx5_context *ctx)
438 {
439 char *env;
440
441 env = getenv("MLX5_DEBUG_FILE");
442 if (!env) {
443 ctx->dbg_fp = stderr;
444 return;
445 }
446
447 ctx->dbg_fp = fopen(env, "aw+");
448 if (!ctx->dbg_fp) {
449 fprintf(stderr, "Failed opening debug file %s, using stderr\n", env);
450 ctx->dbg_fp = stderr;
451 return;
452 }
453 }
454
close_debug_file(struct mlx5_context * ctx)455 static void close_debug_file(struct mlx5_context *ctx)
456 {
457 if (ctx->dbg_fp && ctx->dbg_fp != stderr)
458 fclose(ctx->dbg_fp);
459 }
460
set_debug_mask(void)461 static void set_debug_mask(void)
462 {
463 char *env;
464
465 env = getenv("MLX5_DEBUG_MASK");
466 if (env)
467 mlx5_debug_mask = strtol(env, NULL, 0);
468 }
469
set_freeze_on_error(void)470 static void set_freeze_on_error(void)
471 {
472 char *env;
473
474 env = getenv("MLX5_FREEZE_ON_ERROR_CQE");
475 if (env)
476 mlx5_freeze_on_error_cqe = strtol(env, NULL, 0);
477 }
478
get_always_bf(void)479 static int get_always_bf(void)
480 {
481 char *env;
482
483 env = getenv("MLX5_POST_SEND_PREFER_BF");
484 if (!env)
485 return 1;
486
487 return strcmp(env, "0") ? 1 : 0;
488 }
489
get_shut_up_bf(void)490 static int get_shut_up_bf(void)
491 {
492 char *env;
493
494 env = getenv("MLX5_SHUT_UP_BF");
495 if (!env)
496 return 0;
497
498 return strcmp(env, "0") ? 1 : 0;
499 }
500
get_num_low_lat_uuars(int tot_uuars)501 static int get_num_low_lat_uuars(int tot_uuars)
502 {
503 char *env;
504 int num = 4;
505
506 env = getenv("MLX5_NUM_LOW_LAT_UUARS");
507 if (env)
508 num = atoi(env);
509
510 if (num < 0)
511 return -EINVAL;
512
513 num = max(num, tot_uuars - MLX5_MED_BFREGS_TSHOLD);
514 return num;
515 }
516
517 /* The library allocates an array of uuar contexts. The one in index zero does
518 * not to execersize odd/even policy so it can avoid a lock but it may not use
519 * blue flame. The upper ones, low_lat_uuars can use blue flame with no lock
520 * since they are assigned to one QP only. The rest can use blue flame but since
521 * they are shared they need a lock
522 */
need_uuar_lock(struct mlx5_context * ctx,int uuarn)523 static int need_uuar_lock(struct mlx5_context *ctx, int uuarn)
524 {
525 if (uuarn == 0 || mlx5_single_threaded)
526 return 0;
527
528 if (uuarn >= (ctx->tot_uuars - ctx->low_lat_uuars) * 2)
529 return 0;
530
531 return 1;
532 }
533
single_threaded_app(void)534 static int single_threaded_app(void)
535 {
536
537 char *env;
538
539 env = getenv("MLX5_SINGLE_THREADED");
540 if (env)
541 return strcmp(env, "1") ? 0 : 1;
542
543 return 0;
544 }
545
mlx5_cmd_get_context(struct mlx5_context * context,struct mlx5_alloc_ucontext * req,size_t req_len,struct mlx5_alloc_ucontext_resp * resp,size_t resp_len)546 static int mlx5_cmd_get_context(struct mlx5_context *context,
547 struct mlx5_alloc_ucontext *req,
548 size_t req_len,
549 struct mlx5_alloc_ucontext_resp *resp,
550 size_t resp_len)
551 {
552 if (!ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
553 req_len, &resp->ibv_resp, resp_len))
554 return 0;
555
556 /* The ibv_cmd_get_context fails in older kernels when passing
557 * a request length that the kernel doesn't know.
558 * To avoid breaking compatibility of new libmlx5 and older
559 * kernels, when ibv_cmd_get_context fails with the full
560 * request length, we try once again with the legacy length.
561 * We repeat this process while reducing requested size based
562 * on the feature input size. To avoid this in the future, we
563 * will remove the check in kernel that requires fields unknown
564 * to the kernel to be cleared. This will require that any new
565 * feature that involves extending struct mlx5_alloc_ucontext
566 * will be accompanied by an indication in the form of one or
567 * more fields in struct mlx5_alloc_ucontext_resp. If the
568 * response value can be interpreted as feature not supported
569 * when the returned value is zero, this will suffice to
570 * indicate to the library that the request was ignored by the
571 * kernel, either because it is unaware or because it decided
572 * to do so. If zero is a valid response, we will add a new
573 * field that indicates whether the request was handled.
574 */
575 if (!ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
576 offsetof(struct mlx5_alloc_ucontext, lib_caps),
577 &resp->ibv_resp, resp_len))
578 return 0;
579
580 return ibv_cmd_get_context(&context->ibv_ctx, &req->ibv_req,
581 offsetof(struct mlx5_alloc_ucontext,
582 cqe_version),
583 &resp->ibv_resp, resp_len);
584 }
585
mlx5_map_internal_clock(struct mlx5_device * mdev,struct ibv_context * ibv_ctx)586 static int mlx5_map_internal_clock(struct mlx5_device *mdev,
587 struct ibv_context *ibv_ctx)
588 {
589 struct mlx5_context *context = to_mctx(ibv_ctx);
590 void *hca_clock_page;
591 off_t offset = 0;
592
593 set_command(MLX5_MMAP_GET_CORE_CLOCK_CMD, &offset);
594 hca_clock_page = mmap(NULL, mdev->page_size,
595 PROT_READ, MAP_SHARED, ibv_ctx->cmd_fd,
596 mdev->page_size * offset);
597
598 if (hca_clock_page == MAP_FAILED) {
599 fprintf(stderr, PFX
600 "Warning: Timestamp available,\n"
601 "but failed to mmap() hca core clock page.\n");
602 return -1;
603 }
604
605 context->hca_core_clock = hca_clock_page +
606 (context->core_clock.offset & (mdev->page_size - 1));
607 return 0;
608 }
609
mlx5dv_query_device(struct ibv_context * ctx_in,struct mlx5dv_context * attrs_out)610 int mlx5dv_query_device(struct ibv_context *ctx_in,
611 struct mlx5dv_context *attrs_out)
612 {
613 struct mlx5_context *mctx = to_mctx(ctx_in);
614 uint64_t comp_mask_out = 0;
615
616 attrs_out->version = 0;
617 attrs_out->flags = 0;
618
619 if (mctx->cqe_version == MLX5_CQE_VERSION_V1)
620 attrs_out->flags |= MLX5DV_CONTEXT_FLAGS_CQE_V1;
621
622 if (mctx->vendor_cap_flags & MLX5_VENDOR_CAP_FLAGS_MPW)
623 attrs_out->flags |= MLX5DV_CONTEXT_FLAGS_MPW;
624
625 if (attrs_out->comp_mask & MLX5DV_CONTEXT_MASK_CQE_COMPRESION) {
626 attrs_out->cqe_comp_caps = mctx->cqe_comp_caps;
627 comp_mask_out |= MLX5DV_CONTEXT_MASK_CQE_COMPRESION;
628 }
629
630 attrs_out->comp_mask = comp_mask_out;
631
632 return 0;
633 }
634
mlx5dv_get_qp(struct ibv_qp * qp_in,struct mlx5dv_qp * qp_out)635 static int mlx5dv_get_qp(struct ibv_qp *qp_in,
636 struct mlx5dv_qp *qp_out)
637 {
638 struct mlx5_qp *mqp = to_mqp(qp_in);
639
640 qp_out->comp_mask = 0;
641 qp_out->dbrec = mqp->db;
642
643 if (mqp->sq_buf_size)
644 /* IBV_QPT_RAW_PACKET */
645 qp_out->sq.buf = (void *)((uintptr_t)mqp->sq_buf.buf);
646 else
647 qp_out->sq.buf = (void *)((uintptr_t)mqp->buf.buf + mqp->sq.offset);
648 qp_out->sq.wqe_cnt = mqp->sq.wqe_cnt;
649 qp_out->sq.stride = 1 << mqp->sq.wqe_shift;
650
651 qp_out->rq.buf = (void *)((uintptr_t)mqp->buf.buf + mqp->rq.offset);
652 qp_out->rq.wqe_cnt = mqp->rq.wqe_cnt;
653 qp_out->rq.stride = 1 << mqp->rq.wqe_shift;
654
655 qp_out->bf.reg = mqp->bf->reg;
656
657 if (mqp->bf->uuarn > 0)
658 qp_out->bf.size = mqp->bf->buf_size;
659 else
660 qp_out->bf.size = 0;
661
662 return 0;
663 }
664
mlx5dv_get_cq(struct ibv_cq * cq_in,struct mlx5dv_cq * cq_out)665 static int mlx5dv_get_cq(struct ibv_cq *cq_in,
666 struct mlx5dv_cq *cq_out)
667 {
668 struct mlx5_cq *mcq = to_mcq(cq_in);
669 struct mlx5_context *mctx = to_mctx(cq_in->context);
670
671 cq_out->comp_mask = 0;
672 cq_out->cqn = mcq->cqn;
673 cq_out->cqe_cnt = mcq->ibv_cq.cqe + 1;
674 cq_out->cqe_size = mcq->cqe_sz;
675 cq_out->buf = mcq->active_buf->buf;
676 cq_out->dbrec = mcq->dbrec;
677 cq_out->uar = mctx->uar;
678
679 mcq->flags |= MLX5_CQ_FLAGS_DV_OWNED;
680
681 return 0;
682 }
683
mlx5dv_get_rwq(struct ibv_wq * wq_in,struct mlx5dv_rwq * rwq_out)684 static int mlx5dv_get_rwq(struct ibv_wq *wq_in,
685 struct mlx5dv_rwq *rwq_out)
686 {
687 struct mlx5_rwq *mrwq = to_mrwq(wq_in);
688
689 rwq_out->comp_mask = 0;
690 rwq_out->buf = mrwq->pbuff;
691 rwq_out->dbrec = mrwq->recv_db;
692 rwq_out->wqe_cnt = mrwq->rq.wqe_cnt;
693 rwq_out->stride = 1 << mrwq->rq.wqe_shift;
694
695 return 0;
696 }
697
mlx5dv_get_srq(struct ibv_srq * srq_in,struct mlx5dv_srq * srq_out)698 static int mlx5dv_get_srq(struct ibv_srq *srq_in,
699 struct mlx5dv_srq *srq_out)
700 {
701 struct mlx5_srq *msrq;
702
703 msrq = container_of(srq_in, struct mlx5_srq, vsrq.srq);
704
705 srq_out->comp_mask = 0;
706 srq_out->buf = msrq->buf.buf;
707 srq_out->dbrec = msrq->db;
708 srq_out->stride = 1 << msrq->wqe_shift;
709 srq_out->head = msrq->head;
710 srq_out->tail = msrq->tail;
711
712 return 0;
713 }
714
mlx5dv_init_obj(struct mlx5dv_obj * obj,uint64_t obj_type)715 int mlx5dv_init_obj(struct mlx5dv_obj *obj, uint64_t obj_type)
716 {
717 int ret = 0;
718
719 if (obj_type & MLX5DV_OBJ_QP)
720 ret = mlx5dv_get_qp(obj->qp.in, obj->qp.out);
721 if (!ret && (obj_type & MLX5DV_OBJ_CQ))
722 ret = mlx5dv_get_cq(obj->cq.in, obj->cq.out);
723 if (!ret && (obj_type & MLX5DV_OBJ_SRQ))
724 ret = mlx5dv_get_srq(obj->srq.in, obj->srq.out);
725 if (!ret && (obj_type & MLX5DV_OBJ_RWQ))
726 ret = mlx5dv_get_rwq(obj->rwq.in, obj->rwq.out);
727
728 return ret;
729 }
730
adjust_uar_info(struct mlx5_device * mdev,struct mlx5_context * context,struct mlx5_alloc_ucontext_resp resp)731 static void adjust_uar_info(struct mlx5_device *mdev,
732 struct mlx5_context *context,
733 struct mlx5_alloc_ucontext_resp resp)
734 {
735 if (!resp.log_uar_size && !resp.num_uars_per_page) {
736 /* old kernel */
737 context->uar_size = mdev->page_size;
738 context->num_uars_per_page = 1;
739 return;
740 }
741
742 context->uar_size = 1 << resp.log_uar_size;
743 context->num_uars_per_page = resp.num_uars_per_page;
744 }
745
mlx5_init_context(struct verbs_device * vdev,struct ibv_context * ctx,int cmd_fd)746 static int mlx5_init_context(struct verbs_device *vdev,
747 struct ibv_context *ctx, int cmd_fd)
748 {
749 struct mlx5_context *context;
750 struct mlx5_alloc_ucontext req;
751 struct mlx5_alloc_ucontext_resp resp;
752 int i;
753 int page_size;
754 int tot_uuars;
755 int low_lat_uuars;
756 int gross_uuars;
757 int j;
758 off_t offset;
759 struct mlx5_device *mdev;
760 struct verbs_context *v_ctx;
761 struct ibv_port_attr port_attr;
762 struct ibv_device_attr_ex device_attr;
763 int k;
764 int bfi;
765 int num_sys_page_map;
766
767 mdev = to_mdev(&vdev->device);
768 v_ctx = verbs_get_ctx(ctx);
769 page_size = mdev->page_size;
770 mlx5_single_threaded = single_threaded_app();
771
772 context = to_mctx(ctx);
773 context->ibv_ctx.cmd_fd = cmd_fd;
774
775 open_debug_file(context);
776 set_debug_mask();
777 set_freeze_on_error();
778 if (gethostname(context->hostname, sizeof(context->hostname)))
779 strcpy(context->hostname, "host_unknown");
780
781 tot_uuars = get_total_uuars(page_size);
782 if (tot_uuars < 0) {
783 errno = -tot_uuars;
784 goto err_free;
785 }
786
787 low_lat_uuars = get_num_low_lat_uuars(tot_uuars);
788 if (low_lat_uuars < 0) {
789 errno = -low_lat_uuars;
790 goto err_free;
791 }
792
793 if (low_lat_uuars > tot_uuars - 1) {
794 errno = ENOMEM;
795 goto err_free;
796 }
797
798 memset(&req, 0, sizeof(req));
799 memset(&resp, 0, sizeof(resp));
800
801 req.total_num_uuars = tot_uuars;
802 req.num_low_latency_uuars = low_lat_uuars;
803 req.cqe_version = MLX5_CQE_VERSION_V1;
804 req.lib_caps |= MLX5_LIB_CAP_4K_UAR;
805
806 if (mlx5_cmd_get_context(context, &req, sizeof(req), &resp,
807 sizeof(resp)))
808 goto err_free;
809
810 context->max_num_qps = resp.qp_tab_size;
811 context->bf_reg_size = resp.bf_reg_size;
812 context->tot_uuars = resp.tot_uuars;
813 context->low_lat_uuars = low_lat_uuars;
814 context->cache_line_size = resp.cache_line_size;
815 context->max_sq_desc_sz = resp.max_sq_desc_sz;
816 context->max_rq_desc_sz = resp.max_rq_desc_sz;
817 context->max_send_wqebb = resp.max_send_wqebb;
818 context->num_ports = resp.num_ports;
819 context->max_recv_wr = resp.max_recv_wr;
820 context->max_srq_recv_wr = resp.max_srq_recv_wr;
821
822 context->cqe_version = resp.cqe_version;
823 if (context->cqe_version) {
824 if (context->cqe_version == MLX5_CQE_VERSION_V1)
825 mlx5_ctx_ops.poll_cq = mlx5_poll_cq_v1;
826 else
827 goto err_free;
828 }
829
830 adjust_uar_info(mdev, context, resp);
831
832 gross_uuars = context->tot_uuars / MLX5_NUM_NON_FP_BFREGS_PER_UAR * NUM_BFREGS_PER_UAR;
833 context->bfs = calloc(gross_uuars, sizeof(*context->bfs));
834 if (!context->bfs) {
835 errno = ENOMEM;
836 goto err_free;
837 }
838
839 context->cmds_supp_uhw = resp.cmds_supp_uhw;
840 context->vendor_cap_flags = 0;
841
842 pthread_mutex_init(&context->qp_table_mutex, NULL);
843 pthread_mutex_init(&context->srq_table_mutex, NULL);
844 pthread_mutex_init(&context->uidx_table_mutex, NULL);
845 for (i = 0; i < MLX5_QP_TABLE_SIZE; ++i)
846 context->qp_table[i].refcnt = 0;
847
848 for (i = 0; i < MLX5_QP_TABLE_SIZE; ++i)
849 context->uidx_table[i].refcnt = 0;
850
851 context->db_list = NULL;
852
853 pthread_mutex_init(&context->db_list_mutex, NULL);
854
855 num_sys_page_map = context->tot_uuars / (context->num_uars_per_page * MLX5_NUM_NON_FP_BFREGS_PER_UAR);
856 for (i = 0; i < num_sys_page_map; ++i) {
857 offset = 0;
858 set_command(MLX5_MMAP_GET_REGULAR_PAGES_CMD, &offset);
859 set_index(i, &offset);
860 context->uar[i] = mmap(NULL, page_size, PROT_WRITE, MAP_SHARED,
861 cmd_fd, page_size * offset);
862 if (context->uar[i] == MAP_FAILED) {
863 context->uar[i] = NULL;
864 goto err_free_bf;
865 }
866 }
867
868 for (i = 0; i < num_sys_page_map; i++) {
869 for (j = 0; j < context->num_uars_per_page; j++) {
870 for (k = 0; k < NUM_BFREGS_PER_UAR; k++) {
871 bfi = (i * context->num_uars_per_page + j) * NUM_BFREGS_PER_UAR + k;
872 context->bfs[bfi].reg = context->uar[i] + MLX5_ADAPTER_PAGE_SIZE * j +
873 MLX5_BF_OFFSET + k * context->bf_reg_size;
874 context->bfs[bfi].need_lock = need_uuar_lock(context, bfi);
875 mlx5_spinlock_init(&context->bfs[bfi].lock);
876 context->bfs[bfi].offset = 0;
877 if (bfi)
878 context->bfs[bfi].buf_size = context->bf_reg_size / 2;
879 context->bfs[bfi].uuarn = bfi;
880 }
881 }
882 }
883 context->hca_core_clock = NULL;
884 if (resp.response_length + sizeof(resp.ibv_resp) >=
885 offsetof(struct mlx5_alloc_ucontext_resp, hca_core_clock_offset) +
886 sizeof(resp.hca_core_clock_offset) &&
887 resp.comp_mask & MLX5_IB_ALLOC_UCONTEXT_RESP_MASK_CORE_CLOCK_OFFSET) {
888 context->core_clock.offset = resp.hca_core_clock_offset;
889 mlx5_map_internal_clock(mdev, ctx);
890 }
891
892 mlx5_spinlock_init(&context->lock32);
893
894 context->prefer_bf = get_always_bf();
895 context->shut_up_bf = get_shut_up_bf();
896 mlx5_read_env(&vdev->device, context);
897
898 mlx5_spinlock_init(&context->hugetlb_lock);
899 TAILQ_INIT(&context->hugetlb_list);
900
901 context->ibv_ctx.ops = mlx5_ctx_ops;
902
903 verbs_set_ctx_op(v_ctx, create_qp_ex, mlx5_create_qp_ex);
904 verbs_set_ctx_op(v_ctx, open_xrcd, mlx5_open_xrcd);
905 verbs_set_ctx_op(v_ctx, close_xrcd, mlx5_close_xrcd);
906 verbs_set_ctx_op(v_ctx, create_srq_ex, mlx5_create_srq_ex);
907 verbs_set_ctx_op(v_ctx, get_srq_num, mlx5_get_srq_num);
908 verbs_set_ctx_op(v_ctx, query_device_ex, mlx5_query_device_ex);
909 verbs_set_ctx_op(v_ctx, query_rt_values, mlx5_query_rt_values);
910 verbs_set_ctx_op(v_ctx, ibv_create_flow, ibv_cmd_create_flow);
911 verbs_set_ctx_op(v_ctx, ibv_destroy_flow, ibv_cmd_destroy_flow);
912 verbs_set_ctx_op(v_ctx, create_cq_ex, mlx5_create_cq_ex);
913 verbs_set_ctx_op(v_ctx, create_wq, mlx5_create_wq);
914 verbs_set_ctx_op(v_ctx, modify_wq, mlx5_modify_wq);
915 verbs_set_ctx_op(v_ctx, destroy_wq, mlx5_destroy_wq);
916 verbs_set_ctx_op(v_ctx, create_rwq_ind_table, mlx5_create_rwq_ind_table);
917 verbs_set_ctx_op(v_ctx, destroy_rwq_ind_table, mlx5_destroy_rwq_ind_table);
918
919 memset(&device_attr, 0, sizeof(device_attr));
920 if (!mlx5_query_device_ex(ctx, NULL, &device_attr,
921 sizeof(struct ibv_device_attr_ex))) {
922 context->cached_device_cap_flags =
923 device_attr.orig_attr.device_cap_flags;
924 context->atomic_cap = device_attr.orig_attr.atomic_cap;
925 context->cached_tso_caps = device_attr.tso_caps;
926 }
927
928 for (j = 0; j < min(MLX5_MAX_PORTS_NUM, context->num_ports); ++j) {
929 memset(&port_attr, 0, sizeof(port_attr));
930 if (!mlx5_query_port(ctx, j + 1, &port_attr))
931 context->cached_link_layer[j] = port_attr.link_layer;
932 }
933
934 return 0;
935
936 err_free_bf:
937 free(context->bfs);
938
939 err_free:
940 for (i = 0; i < MLX5_MAX_UARS; ++i) {
941 if (context->uar[i])
942 munmap(context->uar[i], page_size);
943 }
944 close_debug_file(context);
945 return errno;
946 }
947
mlx5_cleanup_context(struct verbs_device * device,struct ibv_context * ibctx)948 static void mlx5_cleanup_context(struct verbs_device *device,
949 struct ibv_context *ibctx)
950 {
951 struct mlx5_context *context = to_mctx(ibctx);
952 int page_size = to_mdev(ibctx->device)->page_size;
953 int i;
954
955 free(context->bfs);
956 for (i = 0; i < MLX5_MAX_UARS; ++i) {
957 if (context->uar[i])
958 munmap(context->uar[i], page_size);
959 }
960 if (context->hca_core_clock)
961 munmap(context->hca_core_clock - context->core_clock.offset,
962 page_size);
963 close_debug_file(context);
964 }
965
966 static struct verbs_device_ops mlx5_dev_ops = {
967 .init_context = mlx5_init_context,
968 .uninit_context = mlx5_cleanup_context,
969 };
970
mlx5_driver_init(const char * uverbs_sys_path,int abi_version)971 static struct verbs_device *mlx5_driver_init(const char *uverbs_sys_path,
972 int abi_version)
973 {
974 char value[8];
975 struct mlx5_device *dev;
976 unsigned vendor, device;
977 int i;
978
979 if (ibv_read_sysfs_file(uverbs_sys_path, "device/vendor",
980 value, sizeof value) < 0)
981 return NULL;
982 sscanf(value, "%i", &vendor);
983
984 if (ibv_read_sysfs_file(uverbs_sys_path, "device/device",
985 value, sizeof value) < 0)
986 return NULL;
987 sscanf(value, "%i", &device);
988
989 for (i = 0; i < sizeof hca_table / sizeof hca_table[0]; ++i)
990 if (vendor == hca_table[i].vendor &&
991 device == hca_table[i].device)
992 goto found;
993
994 return NULL;
995
996 found:
997 if (abi_version < MLX5_UVERBS_MIN_ABI_VERSION ||
998 abi_version > MLX5_UVERBS_MAX_ABI_VERSION) {
999 fprintf(stderr, PFX "Fatal: ABI version %d of %s is not supported "
1000 "(min supported %d, max supported %d)\n",
1001 abi_version, uverbs_sys_path,
1002 MLX5_UVERBS_MIN_ABI_VERSION,
1003 MLX5_UVERBS_MAX_ABI_VERSION);
1004 return NULL;
1005 }
1006
1007 dev = calloc(1, sizeof *dev);
1008 if (!dev) {
1009 fprintf(stderr, PFX "Fatal: couldn't allocate device for %s\n",
1010 uverbs_sys_path);
1011 return NULL;
1012 }
1013
1014 dev->page_size = sysconf(_SC_PAGESIZE);
1015 dev->driver_abi_ver = abi_version;
1016
1017 dev->verbs_dev.ops = &mlx5_dev_ops;
1018 dev->verbs_dev.sz = sizeof(*dev);
1019 dev->verbs_dev.size_of_context = sizeof(struct mlx5_context) -
1020 sizeof(struct ibv_context);
1021
1022 return &dev->verbs_dev;
1023 }
1024
mlx5_register_driver(void)1025 static __attribute__((constructor)) void mlx5_register_driver(void)
1026 {
1027 verbs_register_driver("mlx5", mlx5_driver_init);
1028 }
1029