1 /*******************************************************************
2 * This file is part of the Emulex RoCE Device Driver for *
3 * RoCE (RDMA over Converged Ethernet) adapters. *
4 * Copyright (C) 2008-2012 Emulex. All rights reserved. *
5 * EMULEX and SLI are trademarks of Emulex. *
8 * This program is free software; you can redistribute it and/or *
9 * modify it under the terms of version 2 of the GNU General *
10 * Public License as published by the Free Software Foundation. *
11 * This program is distributed in the hope that it will be useful. *
12 * ALL EXPRESS OR IMPLIED CONDITIONS, REPRESENTATIONS AND *
13 * WARRANTIES, INCLUDING ANY IMPLIED WARRANTY OF MERCHANTABILITY, *
14 * FITNESS FOR A PARTICULAR PURPOSE, OR NON-INFRINGEMENT, ARE *
15 * DISCLAIMED, EXCEPT TO THE EXTENT THAT SUCH DISCLAIMERS ARE HELD *
16 * TO BE LEGALLY INVALID. See the GNU General Public License for *
17 * more details, a copy of which can be found in the file COPYING *
18 * included with this package. *
20 * Contact Information:
21 * linux-drivers@emulex.com
25 * Costa Mesa, CA 92626
26 *******************************************************************/
28 #include <linux/module.h>
29 #include <linux/idr.h>
30 #include <rdma/ib_verbs.h>
31 #include <rdma/ib_user_verbs.h>
32 #include <rdma/ib_addr.h>
34 #include <linux/netdevice.h>
35 #include <net/addrconf.h>
38 #include "ocrdma_verbs.h"
39 #include "ocrdma_ah.h"
41 #include "ocrdma_hw.h"
42 #include "ocrdma_stats.h"
43 #include "ocrdma_abi.h"
45 MODULE_VERSION(OCRDMA_ROCE_DRV_VERSION);
46 MODULE_DESCRIPTION(OCRDMA_ROCE_DRV_DESC " " OCRDMA_ROCE_DRV_VERSION);
47 MODULE_AUTHOR("Emulex Corporation");
48 MODULE_LICENSE("GPL");
50 static LIST_HEAD(ocrdma_dev_list);
51 static DEFINE_SPINLOCK(ocrdma_devlist_lock);
52 static DEFINE_IDR(ocrdma_dev_id);
54 static union ib_gid ocrdma_zero_sgid;
56 void ocrdma_get_guid(struct ocrdma_dev *dev, u8 *guid)
60 memcpy(&mac_addr[0], &dev->nic_info.mac_addr[0], ETH_ALEN);
61 guid[0] = mac_addr[0] ^ 2;
62 guid[1] = mac_addr[1];
63 guid[2] = mac_addr[2];
66 guid[5] = mac_addr[3];
67 guid[6] = mac_addr[4];
68 guid[7] = mac_addr[5];
71 static bool ocrdma_add_sgid(struct ocrdma_dev *dev, union ib_gid *new_sgid)
76 memset(&ocrdma_zero_sgid, 0, sizeof(union ib_gid));
79 spin_lock_irqsave(&dev->sgid_lock, flags);
80 for (i = 0; i < OCRDMA_MAX_SGID; i++) {
81 if (!memcmp(&dev->sgid_tbl[i], &ocrdma_zero_sgid,
82 sizeof(union ib_gid))) {
83 /* found free entry */
84 memcpy(&dev->sgid_tbl[i], new_sgid,
85 sizeof(union ib_gid));
86 spin_unlock_irqrestore(&dev->sgid_lock, flags);
88 } else if (!memcmp(&dev->sgid_tbl[i], new_sgid,
89 sizeof(union ib_gid))) {
90 /* entry already present, no addition is required. */
91 spin_unlock_irqrestore(&dev->sgid_lock, flags);
95 spin_unlock_irqrestore(&dev->sgid_lock, flags);
99 static bool ocrdma_del_sgid(struct ocrdma_dev *dev, union ib_gid *sgid)
106 spin_lock_irqsave(&dev->sgid_lock, flags);
107 /* first is default sgid, which cannot be deleted. */
108 for (i = 1; i < OCRDMA_MAX_SGID; i++) {
109 if (!memcmp(&dev->sgid_tbl[i], sgid, sizeof(union ib_gid))) {
110 /* found matching entry */
111 memset(&dev->sgid_tbl[i], 0, sizeof(union ib_gid));
116 spin_unlock_irqrestore(&dev->sgid_lock, flags);
120 static int ocrdma_addr_event(unsigned long event, struct net_device *netdev,
123 struct ib_event gid_event;
124 struct ocrdma_dev *dev;
126 bool updated = false;
127 bool is_vlan = false;
129 is_vlan = netdev->priv_flags & IFF_802_1Q_VLAN;
131 netdev = rdma_vlan_dev_real_dev(netdev);
134 list_for_each_entry_rcu(dev, &ocrdma_dev_list, entry) {
135 if (dev->nic_info.netdev == netdev) {
145 mutex_lock(&dev->dev_lock);
148 updated = ocrdma_add_sgid(dev, gid);
151 updated = ocrdma_del_sgid(dev, gid);
157 /* GID table updated, notify the consumers about it */
158 gid_event.device = &dev->ibdev;
159 gid_event.element.port_num = 1;
160 gid_event.event = IB_EVENT_GID_CHANGE;
161 ib_dispatch_event(&gid_event);
163 mutex_unlock(&dev->dev_lock);
167 static int ocrdma_inetaddr_event(struct notifier_block *notifier,
168 unsigned long event, void *ptr)
170 struct in_ifaddr *ifa = ptr;
172 struct net_device *netdev = ifa->ifa_dev->dev;
174 ipv6_addr_set_v4mapped(ifa->ifa_address, (struct in6_addr *)&gid);
175 return ocrdma_addr_event(event, netdev, &gid);
178 static struct notifier_block ocrdma_inetaddr_notifier = {
179 .notifier_call = ocrdma_inetaddr_event
182 #if IS_ENABLED(CONFIG_IPV6)
184 static int ocrdma_inet6addr_event(struct notifier_block *notifier,
185 unsigned long event, void *ptr)
187 struct inet6_ifaddr *ifa = (struct inet6_ifaddr *)ptr;
188 union ib_gid *gid = (union ib_gid *)&ifa->addr;
189 struct net_device *netdev = ifa->idev->dev;
190 return ocrdma_addr_event(event, netdev, gid);
193 static struct notifier_block ocrdma_inet6addr_notifier = {
194 .notifier_call = ocrdma_inet6addr_event
197 #endif /* IPV6 and VLAN */
199 static enum rdma_link_layer ocrdma_link_layer(struct ib_device *device,
202 return IB_LINK_LAYER_ETHERNET;
205 static int ocrdma_register_device(struct ocrdma_dev *dev)
207 strlcpy(dev->ibdev.name, "ocrdma%d", IB_DEVICE_NAME_MAX);
208 ocrdma_get_guid(dev, (u8 *)&dev->ibdev.node_guid);
209 memcpy(dev->ibdev.node_desc, OCRDMA_NODE_DESC,
210 sizeof(OCRDMA_NODE_DESC));
211 dev->ibdev.owner = THIS_MODULE;
212 dev->ibdev.uverbs_abi_ver = OCRDMA_ABI_VERSION;
213 dev->ibdev.uverbs_cmd_mask =
214 OCRDMA_UVERBS(GET_CONTEXT) |
215 OCRDMA_UVERBS(QUERY_DEVICE) |
216 OCRDMA_UVERBS(QUERY_PORT) |
217 OCRDMA_UVERBS(ALLOC_PD) |
218 OCRDMA_UVERBS(DEALLOC_PD) |
219 OCRDMA_UVERBS(REG_MR) |
220 OCRDMA_UVERBS(DEREG_MR) |
221 OCRDMA_UVERBS(CREATE_COMP_CHANNEL) |
222 OCRDMA_UVERBS(CREATE_CQ) |
223 OCRDMA_UVERBS(RESIZE_CQ) |
224 OCRDMA_UVERBS(DESTROY_CQ) |
225 OCRDMA_UVERBS(REQ_NOTIFY_CQ) |
226 OCRDMA_UVERBS(CREATE_QP) |
227 OCRDMA_UVERBS(MODIFY_QP) |
228 OCRDMA_UVERBS(QUERY_QP) |
229 OCRDMA_UVERBS(DESTROY_QP) |
230 OCRDMA_UVERBS(POLL_CQ) |
231 OCRDMA_UVERBS(POST_SEND) |
232 OCRDMA_UVERBS(POST_RECV);
234 dev->ibdev.uverbs_cmd_mask |=
235 OCRDMA_UVERBS(CREATE_AH) |
236 OCRDMA_UVERBS(MODIFY_AH) |
237 OCRDMA_UVERBS(QUERY_AH) |
238 OCRDMA_UVERBS(DESTROY_AH);
240 dev->ibdev.node_type = RDMA_NODE_IB_CA;
241 dev->ibdev.phys_port_cnt = 1;
242 dev->ibdev.num_comp_vectors = dev->eq_cnt;
244 /* mandatory verbs. */
245 dev->ibdev.query_device = ocrdma_query_device;
246 dev->ibdev.query_port = ocrdma_query_port;
247 dev->ibdev.modify_port = ocrdma_modify_port;
248 dev->ibdev.query_gid = ocrdma_query_gid;
249 dev->ibdev.get_link_layer = ocrdma_link_layer;
250 dev->ibdev.alloc_pd = ocrdma_alloc_pd;
251 dev->ibdev.dealloc_pd = ocrdma_dealloc_pd;
253 dev->ibdev.create_cq = ocrdma_create_cq;
254 dev->ibdev.destroy_cq = ocrdma_destroy_cq;
255 dev->ibdev.resize_cq = ocrdma_resize_cq;
257 dev->ibdev.create_qp = ocrdma_create_qp;
258 dev->ibdev.modify_qp = ocrdma_modify_qp;
259 dev->ibdev.query_qp = ocrdma_query_qp;
260 dev->ibdev.destroy_qp = ocrdma_destroy_qp;
262 dev->ibdev.query_pkey = ocrdma_query_pkey;
263 dev->ibdev.create_ah = ocrdma_create_ah;
264 dev->ibdev.destroy_ah = ocrdma_destroy_ah;
265 dev->ibdev.query_ah = ocrdma_query_ah;
266 dev->ibdev.modify_ah = ocrdma_modify_ah;
268 dev->ibdev.poll_cq = ocrdma_poll_cq;
269 dev->ibdev.post_send = ocrdma_post_send;
270 dev->ibdev.post_recv = ocrdma_post_recv;
271 dev->ibdev.req_notify_cq = ocrdma_arm_cq;
273 dev->ibdev.get_dma_mr = ocrdma_get_dma_mr;
274 dev->ibdev.reg_phys_mr = ocrdma_reg_kernel_mr;
275 dev->ibdev.dereg_mr = ocrdma_dereg_mr;
276 dev->ibdev.reg_user_mr = ocrdma_reg_user_mr;
278 dev->ibdev.alloc_fast_reg_mr = ocrdma_alloc_frmr;
279 dev->ibdev.alloc_fast_reg_page_list = ocrdma_alloc_frmr_page_list;
280 dev->ibdev.free_fast_reg_page_list = ocrdma_free_frmr_page_list;
282 /* mandatory to support user space verbs consumer. */
283 dev->ibdev.alloc_ucontext = ocrdma_alloc_ucontext;
284 dev->ibdev.dealloc_ucontext = ocrdma_dealloc_ucontext;
285 dev->ibdev.mmap = ocrdma_mmap;
286 dev->ibdev.dma_device = &dev->nic_info.pdev->dev;
288 dev->ibdev.process_mad = ocrdma_process_mad;
290 if (ocrdma_get_asic_type(dev) == OCRDMA_ASIC_GEN_SKH_R) {
291 dev->ibdev.uverbs_cmd_mask |=
292 OCRDMA_UVERBS(CREATE_SRQ) |
293 OCRDMA_UVERBS(MODIFY_SRQ) |
294 OCRDMA_UVERBS(QUERY_SRQ) |
295 OCRDMA_UVERBS(DESTROY_SRQ) |
296 OCRDMA_UVERBS(POST_SRQ_RECV);
298 dev->ibdev.create_srq = ocrdma_create_srq;
299 dev->ibdev.modify_srq = ocrdma_modify_srq;
300 dev->ibdev.query_srq = ocrdma_query_srq;
301 dev->ibdev.destroy_srq = ocrdma_destroy_srq;
302 dev->ibdev.post_srq_recv = ocrdma_post_srq_recv;
304 return ib_register_device(&dev->ibdev, NULL);
307 static int ocrdma_alloc_resources(struct ocrdma_dev *dev)
309 mutex_init(&dev->dev_lock);
310 dev->sgid_tbl = kzalloc(sizeof(union ib_gid) *
311 OCRDMA_MAX_SGID, GFP_KERNEL);
314 spin_lock_init(&dev->sgid_lock);
316 dev->cq_tbl = kzalloc(sizeof(struct ocrdma_cq *) *
317 OCRDMA_MAX_CQ, GFP_KERNEL);
321 if (dev->attr.max_qp) {
322 dev->qp_tbl = kzalloc(sizeof(struct ocrdma_qp *) *
323 OCRDMA_MAX_QP, GFP_KERNEL);
328 dev->stag_arr = kzalloc(sizeof(u64) * OCRDMA_MAX_STAG, GFP_KERNEL);
329 if (dev->stag_arr == NULL)
332 ocrdma_alloc_pd_pool(dev);
334 spin_lock_init(&dev->av_tbl.lock);
335 spin_lock_init(&dev->flush_q_lock);
338 pr_err("%s(%d) error.\n", __func__, dev->id);
342 static void ocrdma_free_resources(struct ocrdma_dev *dev)
344 kfree(dev->stag_arr);
347 kfree(dev->sgid_tbl);
350 /* OCRDMA sysfs interface */
351 static ssize_t show_rev(struct device *device, struct device_attribute *attr,
354 struct ocrdma_dev *dev = dev_get_drvdata(device);
356 return scnprintf(buf, PAGE_SIZE, "0x%x\n", dev->nic_info.pdev->vendor);
359 static ssize_t show_fw_ver(struct device *device, struct device_attribute *attr,
362 struct ocrdma_dev *dev = dev_get_drvdata(device);
364 return scnprintf(buf, PAGE_SIZE, "%s\n", &dev->attr.fw_ver[0]);
367 static ssize_t show_hca_type(struct device *device,
368 struct device_attribute *attr, char *buf)
370 struct ocrdma_dev *dev = dev_get_drvdata(device);
372 return scnprintf(buf, PAGE_SIZE, "%s\n", &dev->model_number[0]);
375 static DEVICE_ATTR(hw_rev, S_IRUGO, show_rev, NULL);
376 static DEVICE_ATTR(fw_ver, S_IRUGO, show_fw_ver, NULL);
377 static DEVICE_ATTR(hca_type, S_IRUGO, show_hca_type, NULL);
379 static struct device_attribute *ocrdma_attributes[] = {
385 static void ocrdma_remove_sysfiles(struct ocrdma_dev *dev)
389 for (i = 0; i < ARRAY_SIZE(ocrdma_attributes); i++)
390 device_remove_file(&dev->ibdev.dev, ocrdma_attributes[i]);
393 static void ocrdma_add_default_sgid(struct ocrdma_dev *dev)
395 /* GID Index 0 - Invariant manufacturer-assigned EUI-64 */
396 union ib_gid *sgid = &dev->sgid_tbl[0];
398 sgid->global.subnet_prefix = cpu_to_be64(0xfe80000000000000LL);
399 ocrdma_get_guid(dev, &sgid->raw[8]);
402 static void ocrdma_init_ipv4_gids(struct ocrdma_dev *dev,
403 struct net_device *net)
405 struct in_device *in_dev;
407 in_dev = in_dev_get(net);
410 ipv6_addr_set_v4mapped(ifa->ifa_address,
411 (struct in6_addr *)&gid);
412 ocrdma_add_sgid(dev, &gid);
419 static void ocrdma_init_ipv6_gids(struct ocrdma_dev *dev,
420 struct net_device *net)
422 #if IS_ENABLED(CONFIG_IPV6)
423 struct inet6_dev *in6_dev;
425 struct inet6_ifaddr *ifp;
426 in6_dev = in6_dev_get(net);
428 read_lock_bh(&in6_dev->lock);
429 list_for_each_entry(ifp, &in6_dev->addr_list, if_list) {
430 pgid = (union ib_gid *)&ifp->addr;
431 ocrdma_add_sgid(dev, pgid);
433 read_unlock_bh(&in6_dev->lock);
434 in6_dev_put(in6_dev);
439 static void ocrdma_init_gid_table(struct ocrdma_dev *dev)
441 struct net_device *net_dev;
443 for_each_netdev(&init_net, net_dev) {
444 struct net_device *real_dev = rdma_vlan_dev_real_dev(net_dev) ?
445 rdma_vlan_dev_real_dev(net_dev) : net_dev;
447 if (real_dev == dev->nic_info.netdev) {
448 ocrdma_add_default_sgid(dev);
449 ocrdma_init_ipv4_gids(dev, net_dev);
450 ocrdma_init_ipv6_gids(dev, net_dev);
455 static struct ocrdma_dev *ocrdma_add(struct be_dev_info *dev_info)
458 struct ocrdma_dev *dev;
460 dev = (struct ocrdma_dev *)ib_alloc_device(sizeof(struct ocrdma_dev));
462 pr_err("Unable to allocate ib device\n");
465 dev->mbx_cmd = kzalloc(sizeof(struct ocrdma_mqe_emb_cmd), GFP_KERNEL);
469 memcpy(&dev->nic_info, dev_info, sizeof(*dev_info));
470 dev->id = idr_alloc(&ocrdma_dev_id, NULL, 0, 0, GFP_KERNEL);
474 status = ocrdma_init_hw(dev);
478 status = ocrdma_alloc_resources(dev);
482 ocrdma_init_service_level(dev);
483 ocrdma_init_gid_table(dev);
484 status = ocrdma_register_device(dev);
488 for (i = 0; i < ARRAY_SIZE(ocrdma_attributes); i++)
489 if (device_create_file(&dev->ibdev.dev, ocrdma_attributes[i]))
491 spin_lock(&ocrdma_devlist_lock);
492 list_add_tail_rcu(&dev->entry, &ocrdma_dev_list);
493 spin_unlock(&ocrdma_devlist_lock);
495 ocrdma_add_port_stats(dev);
496 /* Interrupt Moderation */
497 INIT_DELAYED_WORK(&dev->eqd_work, ocrdma_eqd_set_task);
498 schedule_delayed_work(&dev->eqd_work, msecs_to_jiffies(1000));
500 pr_info("%s %s: %s \"%s\" port %d\n",
501 dev_name(&dev->nic_info.pdev->dev), hca_name(dev),
502 port_speed_string(dev), dev->model_number,
504 pr_info("%s ocrdma%d driver loaded successfully\n",
505 dev_name(&dev->nic_info.pdev->dev), dev->id);
509 ocrdma_remove_sysfiles(dev);
511 ocrdma_free_resources(dev);
512 ocrdma_cleanup_hw(dev);
514 idr_remove(&ocrdma_dev_id, dev->id);
517 ib_dealloc_device(&dev->ibdev);
518 pr_err("%s() leaving. ret=%d\n", __func__, status);
522 static void ocrdma_remove_free(struct rcu_head *rcu)
524 struct ocrdma_dev *dev = container_of(rcu, struct ocrdma_dev, rcu);
526 idr_remove(&ocrdma_dev_id, dev->id);
528 ib_dealloc_device(&dev->ibdev);
531 static void ocrdma_remove(struct ocrdma_dev *dev)
533 /* first unregister with stack to stop all the active traffic
534 * of the registered clients.
536 cancel_delayed_work_sync(&dev->eqd_work);
537 ocrdma_remove_sysfiles(dev);
538 ib_unregister_device(&dev->ibdev);
540 ocrdma_rem_port_stats(dev);
542 spin_lock(&ocrdma_devlist_lock);
543 list_del_rcu(&dev->entry);
544 spin_unlock(&ocrdma_devlist_lock);
546 ocrdma_free_resources(dev);
547 ocrdma_cleanup_hw(dev);
549 call_rcu(&dev->rcu, ocrdma_remove_free);
552 static int ocrdma_open(struct ocrdma_dev *dev)
554 struct ib_event port_event;
556 port_event.event = IB_EVENT_PORT_ACTIVE;
557 port_event.element.port_num = 1;
558 port_event.device = &dev->ibdev;
559 ib_dispatch_event(&port_event);
563 static int ocrdma_close(struct ocrdma_dev *dev)
566 struct ocrdma_qp *qp, **cur_qp;
567 struct ib_event err_event;
568 struct ib_qp_attr attrs;
569 int attr_mask = IB_QP_STATE;
571 attrs.qp_state = IB_QPS_ERR;
572 mutex_lock(&dev->dev_lock);
574 cur_qp = dev->qp_tbl;
575 for (i = 0; i < OCRDMA_MAX_QP; i++) {
577 if (qp && qp->ibqp.qp_type != IB_QPT_GSI) {
578 /* change the QP state to ERROR */
579 _ocrdma_modify_qp(&qp->ibqp, &attrs, attr_mask);
581 err_event.event = IB_EVENT_QP_FATAL;
582 err_event.element.qp = &qp->ibqp;
583 err_event.device = &dev->ibdev;
584 ib_dispatch_event(&err_event);
588 mutex_unlock(&dev->dev_lock);
590 err_event.event = IB_EVENT_PORT_ERR;
591 err_event.element.port_num = 1;
592 err_event.device = &dev->ibdev;
593 ib_dispatch_event(&err_event);
597 static void ocrdma_shutdown(struct ocrdma_dev *dev)
603 /* event handling via NIC driver ensures that all the NIC specific
604 * initialization done before RoCE driver notifies
607 static void ocrdma_event_handler(struct ocrdma_dev *dev, u32 event)
616 case BE_DEV_SHUTDOWN:
617 ocrdma_shutdown(dev);
622 static struct ocrdma_driver ocrdma_drv = {
623 .name = "ocrdma_driver",
625 .remove = ocrdma_remove,
626 .state_change_handler = ocrdma_event_handler,
627 .be_abi_version = OCRDMA_BE_ROCE_ABI_VERSION,
630 static void ocrdma_unregister_inet6addr_notifier(void)
632 #if IS_ENABLED(CONFIG_IPV6)
633 unregister_inet6addr_notifier(&ocrdma_inet6addr_notifier);
637 static void ocrdma_unregister_inetaddr_notifier(void)
639 unregister_inetaddr_notifier(&ocrdma_inetaddr_notifier);
642 static int __init ocrdma_init_module(void)
646 ocrdma_init_debugfs();
648 status = register_inetaddr_notifier(&ocrdma_inetaddr_notifier);
652 #if IS_ENABLED(CONFIG_IPV6)
653 status = register_inet6addr_notifier(&ocrdma_inet6addr_notifier);
658 status = be_roce_register_driver(&ocrdma_drv);
665 #if IS_ENABLED(CONFIG_IPV6)
666 ocrdma_unregister_inet6addr_notifier();
669 ocrdma_unregister_inetaddr_notifier();
673 static void __exit ocrdma_exit_module(void)
675 be_roce_unregister_driver(&ocrdma_drv);
676 ocrdma_unregister_inet6addr_notifier();
677 ocrdma_unregister_inetaddr_notifier();
678 ocrdma_rem_debugfs();
681 module_init(ocrdma_init_module);
682 module_exit(ocrdma_exit_module);