* [PATCH v3 1/2] eal: add uevent api for hot plug
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
@ 2017-06-29 4:37 ` Jeff Guo
2017-06-30 3:38 ` Wu, Jingjing
2017-06-29 4:37 ` [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e Jeff Guo
` (22 subsequent siblings)
23 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2017-06-29 4:37 UTC (permalink / raw)
To: helin.zhang, jingjing.wu; +Cc: dev, jia.guo
From: "Guo, Jia" <jia.guo@intel.com>
This patch aim to add a variable "uevent_fd" in structure
"rte_intr_handle" for enable kernel object uevent monitoring,
and add some uevent API in rte eal interrupt, that is
“rte_uevent_connect” and “rte_uevent_get”, so that all driver
could use these API to monitor and read out the uevent, then
corresponding to handle these uevent, such as detach or attach
the device.
Signed-off-by: Guo, Jia <jia.guo@intel.com>
---
v3->v2: refine some return error
refine the string searching logic to aviod memory issue
---
lib/librte_eal/common/eal_common_pci_uio.c | 6 +-
lib/librte_eal/linuxapp/eal/eal_interrupts.c | 136 ++++++++++++++++++++-
lib/librte_eal/linuxapp/eal/eal_pci_uio.c | 6 +
.../linuxapp/eal/include/exec-env/rte_interrupts.h | 37 ++++++
4 files changed, 182 insertions(+), 3 deletions(-)
diff --git a/lib/librte_eal/common/eal_common_pci_uio.c b/lib/librte_eal/common/eal_common_pci_uio.c
index 367a681..5b62f70 100644
--- a/lib/librte_eal/common/eal_common_pci_uio.c
+++ b/lib/librte_eal/common/eal_common_pci_uio.c
@@ -117,6 +117,7 @@
dev->intr_handle.fd = -1;
dev->intr_handle.uio_cfg_fd = -1;
+ dev->intr_handle.uevent_fd = -1;
dev->intr_handle.type = RTE_INTR_HANDLE_UNKNOWN;
/* secondary processes - use already recorded details */
@@ -227,7 +228,10 @@
close(dev->intr_handle.uio_cfg_fd);
dev->intr_handle.uio_cfg_fd = -1;
}
-
+ if (dev->intr_handle.uevent_fd >= 0) {
+ close(dev->intr_handle.uevent_fd);
+ dev->intr_handle.uevent_fd = -1;
+ }
dev->intr_handle.fd = -1;
dev->intr_handle.type = RTE_INTR_HANDLE_UNKNOWN;
}
diff --git a/lib/librte_eal/linuxapp/eal/eal_interrupts.c b/lib/librte_eal/linuxapp/eal/eal_interrupts.c
index 2e3bd12..2c4a3fb 100644
--- a/lib/librte_eal/linuxapp/eal/eal_interrupts.c
+++ b/lib/librte_eal/linuxapp/eal/eal_interrupts.c
@@ -65,6 +65,10 @@
#include <rte_errno.h>
#include <rte_spinlock.h>
+#include <sys/socket.h>
+#include <linux/netlink.h>
+#include <sys/epoll.h>
+
#include "eal_private.h"
#include "eal_vfio.h"
#include "eal_thread.h"
@@ -669,10 +673,13 @@ struct rte_intr_source {
RTE_SET_USED(r);
return -1;
}
+
rte_spinlock_lock(&intr_lock);
TAILQ_FOREACH(src, &intr_sources, next)
- if (src->intr_handle.fd ==
- events[n].data.fd)
+ if ((src->intr_handle.fd ==
+ events[n].data.fd) ||
+ (src->intr_handle.uevent_fd ==
+ events[n].data.fd))
break;
if (src == NULL){
rte_spinlock_unlock(&intr_lock);
@@ -858,7 +865,24 @@ static __attribute__((noreturn)) void *
}
else
numfds++;
+
+ /**
+ * add device uevent file descriptor
+ * into wait list for uevent monitoring.
+ */
+ ev.events = EPOLLIN | EPOLLPRI | EPOLLRDHUP | EPOLLHUP;
+ ev.data.fd = src->intr_handle.uevent_fd;
+ if (epoll_ctl(pfd, EPOLL_CTL_ADD,
+ src->intr_handle.uevent_fd, &ev) < 0){
+ rte_panic("Error adding uevent_fd %d epoll_ctl"
+ ", %s\n",
+ src->intr_handle.uevent_fd,
+ strerror(errno));
+ } else
+ numfds++;
}
+
+
rte_spinlock_unlock(&intr_lock);
/* serve the interrupt */
eal_intr_handle_interrupts(pfd, numfds);
@@ -1255,3 +1279,111 @@ static __attribute__((noreturn)) void *
return 0;
}
+
+int
+rte_uevent_connect(void)
+{
+ struct sockaddr_nl addr;
+ int ret;
+ int netlink_fd = -1;
+ int size = 64 * 1024;
+ int nonblock = 1;
+ memset(&addr, 0, sizeof(addr));
+ addr.nl_family = AF_NETLINK;
+ addr.nl_pid = 0;
+ addr.nl_groups = 0xffffffff;
+
+ netlink_fd = socket(PF_NETLINK, SOCK_DGRAM, NETLINK_KOBJECT_UEVENT);
+ if (netlink_fd < 0)
+ return -1;
+
+ setsockopt(netlink_fd, SOL_SOCKET, SO_RCVBUFFORCE, &size, sizeof(size));
+
+ ret = ioctl(netlink_fd, FIONBIO, &nonblock);
+ if (ret != 0) {
+ RTE_LOG(ERR, EAL,
+ "ioctl(FIONBIO) failed\n");
+ close(netlink_fd);
+ return -1;
+ }
+
+ if (bind(netlink_fd, (struct sockaddr *) &addr, sizeof(addr)) < 0) {
+ close(netlink_fd);
+ return -1;
+ }
+
+ return netlink_fd;
+}
+
+static int
+parse_event(const char *buf, struct rte_uevent *event)
+{
+ char action[RTE_UEVENT_MSG_LEN];
+ char subsystem[RTE_UEVENT_MSG_LEN];
+ char dev_path[RTE_UEVENT_MSG_LEN];
+ int i = 0;
+
+ memset(action, 0, RTE_UEVENT_MSG_LEN);
+ memset(subsystem, 0, RTE_UEVENT_MSG_LEN);
+ memset(dev_path, 0, RTE_UEVENT_MSG_LEN);
+
+ while (i < RTE_UEVENT_MSG_LEN) {
+ for (; i < RTE_UEVENT_MSG_LEN; i++) {
+ if (*buf)
+ break;
+ buf++;
+ }
+ if (!strncmp(buf, "ACTION=", 7)) {
+ buf += 7;
+ i += 7;
+ snprintf(action, sizeof(action), "%s", buf);
+ } else if (!strncmp(buf, "DEVPATH=", 8)) {
+ buf += 8;
+ i += 8;
+ snprintf(dev_path, sizeof(dev_path), "%s", buf);
+ } else if (!strncmp(buf, "SUBSYSTEM=", 10)) {
+ buf += 10;
+ i += 10;
+ snprintf(subsystem, sizeof(subsystem), "%s", buf);
+ }
+ for (; i < RTE_UEVENT_MSG_LEN; i++) {
+ if (*buf == '\0')
+ break;
+ buf++;
+ }
+ }
+
+ if (!strncmp(subsystem, "uio", 3)) {
+
+ event->subsystem = RTE_UEVENT_SUBSYSTEM_UIO;
+ if (!strncmp(action, "add", 3))
+ event->action = RTE_UEVENT_ADD;
+ if (!strncmp(action, "remove", 6))
+ event->action = RTE_UEVENT_REMOVE;
+ return 0;
+ }
+
+ return -1;
+}
+
+int
+rte_uevent_get(int fd, struct rte_uevent *uevent)
+{
+ int ret;
+ char buf[RTE_UEVENT_MSG_LEN];
+
+ memset(uevent, 0, sizeof(struct rte_uevent));
+ memset(buf, 0, RTE_UEVENT_MSG_LEN);
+
+ ret = recv(fd, buf, RTE_UEVENT_MSG_LEN - 1, MSG_DONTWAIT);
+ if (ret > 0)
+ return parse_event(buf, uevent);
+ else if (ret < 0) {
+ RTE_LOG(ERR, EAL,
+ "Socket read error(%d): %s\n",
+ errno, strerror(errno));
+ return -1;
+ } else
+ /* connection closed */
+ return -1;
+}
diff --git a/lib/librte_eal/linuxapp/eal/eal_pci_uio.c b/lib/librte_eal/linuxapp/eal/eal_pci_uio.c
index fa10329..eae9cd5 100644
--- a/lib/librte_eal/linuxapp/eal/eal_pci_uio.c
+++ b/lib/librte_eal/linuxapp/eal/eal_pci_uio.c
@@ -231,6 +231,10 @@
close(dev->intr_handle.uio_cfg_fd);
dev->intr_handle.uio_cfg_fd = -1;
}
+ if (dev->intr_handle.uevent_fd >= 0) {
+ close(dev->intr_handle.uevent_fd);
+ dev->intr_handle.uevent_fd = -1;
+ }
if (dev->intr_handle.fd >= 0) {
close(dev->intr_handle.fd);
dev->intr_handle.fd = -1;
@@ -276,6 +280,8 @@
goto error;
}
+ dev->intr_handle.uevent_fd = rte_uevent_connect();
+
if (dev->kdrv == RTE_KDRV_IGB_UIO)
dev->intr_handle.type = RTE_INTR_HANDLE_UIO;
else {
diff --git a/lib/librte_eal/linuxapp/eal/include/exec-env/rte_interrupts.h b/lib/librte_eal/linuxapp/eal/include/exec-env/rte_interrupts.h
index 6daffeb..0b31a22 100644
--- a/lib/librte_eal/linuxapp/eal/include/exec-env/rte_interrupts.h
+++ b/lib/librte_eal/linuxapp/eal/include/exec-env/rte_interrupts.h
@@ -90,6 +90,7 @@ struct rte_intr_handle {
for uio_pci_generic */
};
int fd; /**< interrupt event file descriptor */
+ int uevent_fd; /**< uevent file descriptor */
enum rte_intr_handle_type type; /**< handle type */
uint32_t max_intr; /**< max interrupt requested */
uint32_t nb_efd; /**< number of available efd(event fd) */
@@ -99,6 +100,19 @@ struct rte_intr_handle {
int *intr_vec; /**< intr vector number array */
};
+#define RTE_UEVENT_MSG_LEN 4096
+#define RTE_UEVENT_SUBSYSTEM_UIO 1
+
+enum rte_uevent_action {
+ RTE_UEVENT_ADD = 0, /**< uevent type of device add */
+ RTE_UEVENT_REMOVE = 1, /**< uevent type of device remove*/
+};
+
+struct rte_uevent {
+ enum rte_uevent_action action; /**< uevent action type */
+ int subsystem; /**< subsystem id */
+};
+
#define RTE_EPOLL_PER_THREAD -1 /**< to hint using per thread epfd */
/**
@@ -236,4 +250,27 @@ struct rte_intr_handle {
int
rte_intr_cap_multiple(struct rte_intr_handle *intr_handle);
+/**
+ * It read out the uevent from the specific file descriptor.
+ *
+ * @param fd
+ * The fd which the uevent associated to
+ * @param uevent
+ * Pointer to the uevent which read from the monitoring fd.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+rte_uevent_get(int fd, struct rte_uevent *uevent);
+
+/**
+ * Connect to the device uevent file descriptor.
+ * @return
+ * - On success, the connected uevent fd.
+ * - On failure, a negative value.
+ */
+int
+rte_uevent_connect(void);
+
#endif /* _RTE_LINUXAPP_INTERRUPTS_H_ */
--
1.8.3.1
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH v3 1/2] eal: add uevent api for hot plug
2017-06-29 4:37 ` [PATCH v3 1/2] eal: " Jeff Guo
@ 2017-06-30 3:38 ` Wu, Jingjing
0 siblings, 0 replies; 494+ messages in thread
From: Wu, Jingjing @ 2017-06-30 3:38 UTC (permalink / raw)
To: Guo, Jia, Zhang, Helin; +Cc: dev@dpdk.org
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, June 29, 2017 12:38 PM
> To: Zhang, Helin <helin.zhang@intel.com>; Wu, Jingjing
> <jingjing.wu@intel.com>
> Cc: dev@dpdk.org; Guo, Jia <jia.guo@intel.com>
> Subject: [PATCH v3 1/2] eal: add uevent api for hot plug
>
> From: "Guo, Jia" <jia.guo@intel.com>
>
> This patch aim to add a variable "uevent_fd" in structure "rte_intr_handle" for
> enable kernel object uevent monitoring, and add some uevent API in rte eal
> interrupt, that is “rte_uevent_connect” and “rte_uevent_get”, so that all driver
> could use these API to monitor and read out the uevent, then corresponding to
> handle these uevent, such as detach or attach the device.
>
> Signed-off-by: Guo, Jia <jia.guo@intel.com>
Looks fine from me.
Reviewed-by: Jingjing Wu <jingjing.wu@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
2017-06-29 4:37 ` [PATCH v3 1/2] eal: " Jeff Guo
@ 2017-06-29 4:37 ` Jeff Guo
2017-06-30 3:38 ` Wu, Jingjing
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
` (21 subsequent siblings)
23 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2017-06-29 4:37 UTC (permalink / raw)
To: helin.zhang, jingjing.wu; +Cc: dev, jia.guo
From: "Guo, Jia" <jia.guo@intel.com>
This patch enable the hot plug feature in i40e, by monitoring the
hot plug uevent of the device. When remove event got, call the app
callback function to handle the detach process.
Signed-off-by: Guo, Jia <jia.guo@intel.com>
---
v3->v2: refine the return issue if device remove
---
drivers/net/i40e/i40e_ethdev.c | 19 +++++++++++++++++++
1 file changed, 19 insertions(+)
diff --git a/drivers/net/i40e/i40e_ethdev.c b/drivers/net/i40e/i40e_ethdev.c
index 4ee1113..67ffc14 100644
--- a/drivers/net/i40e/i40e_ethdev.c
+++ b/drivers/net/i40e/i40e_ethdev.c
@@ -1283,6 +1283,7 @@ static inline void i40e_GLQF_reg_init(struct i40e_hw *hw)
/* enable uio intr after callback register */
rte_intr_enable(intr_handle);
+
/*
* Add an ethertype filter to drop all flow control frames transmitted
* from VSIs. By doing so, we stop VF from sending out PAUSE or PFC
@@ -5832,11 +5833,29 @@ struct i40e_vsi *
{
struct rte_eth_dev *dev = (struct rte_eth_dev *)param;
struct i40e_hw *hw = I40E_DEV_PRIVATE_TO_HW(dev->data->dev_private);
+ struct rte_uevent event;
uint32_t icr0;
+ struct rte_pci_device *pci_dev;
+ struct rte_intr_handle *intr_handle;
+
+ pci_dev = RTE_ETH_DEV_TO_PCI(dev);
+ intr_handle = &pci_dev->intr_handle;
/* Disable interrupt */
i40e_pf_disable_irq0(hw);
+ /* check device uevent */
+ if (rte_uevent_get(intr_handle->uevent_fd, &event) == 0) {
+ if (event.subsystem == RTE_UEVENT_SUBSYSTEM_UIO) {
+ if (event.action == RTE_UEVENT_REMOVE) {
+ _rte_eth_dev_callback_process(dev,
+ RTE_ETH_EVENT_INTR_RMV, NULL);
+ return;
+ }
+ }
+ goto done;
+ }
+
/* read out interrupt causes */
icr0 = I40E_READ_REG(hw, I40E_PFINT_ICR0);
--
1.8.3.1
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e
2017-06-29 4:37 ` [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e Jeff Guo
@ 2017-06-30 3:38 ` Wu, Jingjing
0 siblings, 0 replies; 494+ messages in thread
From: Wu, Jingjing @ 2017-06-30 3:38 UTC (permalink / raw)
To: Guo, Jia, Zhang, Helin; +Cc: dev@dpdk.org
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, June 29, 2017 12:38 PM
> To: Zhang, Helin <helin.zhang@intel.com>; Wu, Jingjing
> <jingjing.wu@intel.com>
> Cc: dev@dpdk.org; Guo, Jia <jia.guo@intel.com>
> Subject: [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e
>
> From: "Guo, Jia" <jia.guo@intel.com>
>
> This patch enable the hot plug feature in i40e, by monitoring the hot plug
> uevent of the device. When remove event got, call the app callback function to
> handle the detach process.
>
> Signed-off-by: Guo, Jia <jia.guo@intel.com>
Acked-by: Jingjing Wu <jingjing.wu@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V22 0/4] add device event monitor framework
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
2017-06-29 4:37 ` [PATCH v3 1/2] eal: " Jeff Guo
2017-06-29 4:37 ` [PATCH v3 2/2] net/i40e: add hot plug monitor in i40e Jeff Guo
@ 2018-04-13 8:30 ` Jeff Guo
2018-04-13 8:30 ` [PATCH V22 1/4] eal: add device event handle in interrupt thread Jeff Guo
` (4 more replies)
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
` (20 subsequent siblings)
23 siblings, 5 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-13 8:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
About hot plug in dpdk, We already have proactive way to add/remove devices
through APIs (rte_eal_hotplug_add/remove), and also have fail-safe driver
to offload the fail-safe work from the app user. But there are still lack
of a general mechanism to monitor hotplug event for all driver, now the
hotplug interrupt event is diversity between each device and driver, such
as mlx4, pci driver and others.
Use the hot removal event for example, pci drivers not all exposure the
remove interrupt, so in order to make user to easy use the hot plug
feature for pci driver, something must be done to detect the remove event
at the kernel level and offer a new line of interrupt to the user land.
Base on the uevent of kobject mechanism in kernel, we could use it to
benefit for monitoring the hot plug status of the device which not only
uio/vfio of pci bus devices, but also other, such as cpu/usb/pci-express bus devices.
The idea is comming as bellow.
a.The uevent message form FD monitoring like below.
remove@/devices/pci0000:80/0000:80:02.2/0000:82:00.0/0000:83:03.0/0000:84:00.2/uio/uio2
ACTION=remove
DEVPATH=/devices/pci0000:80/0000:80:02.2/0000:82:00.0/0000:83:03.0/0000:84:00.2/uio/uio2
SUBSYSTEM=uio
MAJOR=243
MINOR=2
DEVNAME=uio2
SEQNUM=11366
b.add device event monitor framework:
add several general api to enable uevent monitoring.
c.show example how to use uevent monitor
enable uevent monitoring in testpmd to show device event monitor machenism usage.
TODO: failure handler mechanism for hot plug and driver auto bind for hot insertion.
that would let the next hot plug patch set to cover.
patchset history:
v22->v21:
fix clang compile issue and doc style
v21->v20:
refine release note and some code cleaning.
v20->v19:
add more detail note and socket error handler.
v19->18:
fix some typo and misunderstanding part
v18->v17:
1.add feature announcement in release document, fix bsp compile issue.
2.refine socket configuration.
3.remove hotplug policy and detach/attach process from testpmd, let it
focus on the device event monitoring which the patch set introduced.
v17->v16:
1.add related part of the interrupt handle type adding.
2.add new API into map, fix typo issue, add (void*)-1 value for unregister all callback
3.add new file into meson.build, modify coding sytle and add print info, delete unused part.
4.unregister all user's callback when stop event monitor
v16->v15:
1.remove some linux related code out of eal common layer
2.fix some uneasy readble issue.
v15->v14:
1.use exist eal interrupt epoll to replace of rte service usage for monitor thread,
2.add new device event handle type in eal interrupt.
3.remove the uevent type check and any policy from eal,
let it check and management in user's callback.
4.add "--hot-plug" configure parameter in testpmd to switch the hotplug feature.
v14->v13:
1.add __rte_experimental on function defind and fix bsd build issue
v13->v12:
1.fix some logic issue and null check issue
2.fix monitor stop func issue
v12->v11:
1.identify null param in callback for monitor all devices uevent
v11->v10:
1:modify some typo and add experimental tag in new file.
2:modify callback register calling.
v10->v9:
1.fix prefix issue.
2.use a common callback lists for all device and all type to replace
add callback parameter into device struct.
3.delete some unuse part.
v9->v8:
split the patch set into small and explicit patch
v8->v7:
1.use rte_service to replace pthread management.
2.fix defind issue and copyright issue
3.fix some lock issue
v7->v6:
1.modify vdev part according to the vdev rework
2.re-define and split the func into common and bus specific code
3.fix some incorrect issue.
4.fix the system hung after send packcet issue.
v6->v5:
1.add hot plug policy, in eal, default handle to prepare hot plug work for
all pci device, then let app to manage to deside which device need to
hot plug.
2.modify to manage event callback in each device.
3.fix some system hung issue when igb_uioome typo error.release.
4.modify the pci part to the bus-pci base on the bus rework.
5.add hot plug policy in app, show example to use hotplug list to manage
to deside which device need to hot plug.
v5->v4:
1.Move uevent monitor epolling from eal interrupt to eal device layer.
2.Redefine the eal device API for common, and distinguish between linux and bsd
3.Add failure handler helper api in bus layer.Add function of find device by name.
4.Replace of individual fd bind with single device, use a common fd to polling all device.
5.Add to register hot insertion monitoring and process, add function to auto bind driver befor user add device
6.Refine some coding style and typos issue
7.add new callback to process hot insertion
v4->v3:
1.move uevent monitor api from eal interrupt to eal device layer.
2.create uevent type and struct in eal device.
3.move uevent handler for each driver to eal layer.
4.add uevent failure handler to process signal fault issue.
5.add example for request and use uevent monitoring in testpmd.
v3->v2:
1.refine some return error
2.refine the string searching logic to avoid memory issue
v2->v1:
1.remove global variables of hotplug_fd, add uevent_fd
in rte_intr_handle to let each pci device self maintain it fd,
to fix dual device fd issue.
2.refine some typo error.
Jeff Guo (4):
eal: add device event handle in interrupt thread
eal: add device event monitor framework
eal/linux: uevent parse and process
app/testpmd: enable device hotplug monitoring
app/test-pmd/parameters.c | 5 +-
app/test-pmd/testpmd.c | 101 +++++++++-
app/test-pmd/testpmd.h | 2 +
doc/guides/rel_notes/release_18_05.rst | 12 ++
doc/guides/testpmd_app_ug/run_app.rst | 4 +
lib/librte_eal/bsdapp/eal/Makefile | 1 +
lib/librte_eal/bsdapp/eal/eal_dev.c | 21 ++
lib/librte_eal/bsdapp/eal/meson.build | 1 +
lib/librte_eal/common/eal_common_dev.c | 161 +++++++++++++++
lib/librte_eal/common/eal_private.h | 15 ++
lib/librte_eal/common/include/rte_dev.h | 94 +++++++++
lib/librte_eal/common/include/rte_eal_interrupts.h | 1 +
lib/librte_eal/linuxapp/eal/Makefile | 1 +
lib/librte_eal/linuxapp/eal/eal_dev.c | 223 +++++++++++++++++++++
lib/librte_eal/linuxapp/eal/eal_interrupts.c | 11 +-
lib/librte_eal/linuxapp/eal/meson.build | 1 +
lib/librte_eal/rte_eal_version.map | 4 +
test/test/test_interrupts.c | 39 +++-
18 files changed, 692 insertions(+), 5 deletions(-)
create mode 100644 lib/librte_eal/bsdapp/eal/eal_dev.c
create mode 100644 lib/librte_eal/linuxapp/eal/eal_dev.c
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V22 1/4] eal: add device event handle in interrupt thread
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
@ 2018-04-13 8:30 ` Jeff Guo
2018-04-13 8:30 ` [PATCH V22 2/4] eal: add device event monitor framework Jeff Guo
` (3 subsequent siblings)
4 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-13 8:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Add new interrupt handle type of RTE_INTR_HANDLE_DEV_EVENT, for
device event interrupt monitor.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Reviewed-by: Jianfeng Tan <jianfeng.tan@intel.com>
---
v22->v21:
no change
---
lib/librte_eal/common/include/rte_eal_interrupts.h | 1 +
lib/librte_eal/linuxapp/eal/eal_interrupts.c | 11 +++++-
test/test/test_interrupts.c | 39 ++++++++++++++++++++--
3 files changed, 48 insertions(+), 3 deletions(-)
diff --git a/lib/librte_eal/common/include/rte_eal_interrupts.h b/lib/librte_eal/common/include/rte_eal_interrupts.h
index 3f792a9..6eb4932 100644
--- a/lib/librte_eal/common/include/rte_eal_interrupts.h
+++ b/lib/librte_eal/common/include/rte_eal_interrupts.h
@@ -34,6 +34,7 @@ enum rte_intr_handle_type {
RTE_INTR_HANDLE_ALARM, /**< alarm handle */
RTE_INTR_HANDLE_EXT, /**< external handler */
RTE_INTR_HANDLE_VDEV, /**< virtual device */
+ RTE_INTR_HANDLE_DEV_EVENT, /**< device event handle */
RTE_INTR_HANDLE_MAX /**< count of elements */
};
diff --git a/lib/librte_eal/linuxapp/eal/eal_interrupts.c b/lib/librte_eal/linuxapp/eal/eal_interrupts.c
index f86f22f..58e9328 100644
--- a/lib/librte_eal/linuxapp/eal/eal_interrupts.c
+++ b/lib/librte_eal/linuxapp/eal/eal_interrupts.c
@@ -559,6 +559,9 @@ rte_intr_enable(const struct rte_intr_handle *intr_handle)
return -1;
break;
#endif
+ /* not used at this moment */
+ case RTE_INTR_HANDLE_DEV_EVENT:
+ return -1;
/* unknown handle type */
default:
RTE_LOG(ERR, EAL,
@@ -606,6 +609,9 @@ rte_intr_disable(const struct rte_intr_handle *intr_handle)
return -1;
break;
#endif
+ /* not used at this moment */
+ case RTE_INTR_HANDLE_DEV_EVENT:
+ return -1;
/* unknown handle type */
default:
RTE_LOG(ERR, EAL,
@@ -674,7 +680,10 @@ eal_intr_process_interrupts(struct epoll_event *events, int nfds)
bytes_read = 0;
call = true;
break;
-
+ case RTE_INTR_HANDLE_DEV_EVENT:
+ bytes_read = 0;
+ call = true;
+ break;
default:
bytes_read = 1;
break;
diff --git a/test/test/test_interrupts.c b/test/test/test_interrupts.c
index 31a70a0..dc19175 100644
--- a/test/test/test_interrupts.c
+++ b/test/test/test_interrupts.c
@@ -20,6 +20,7 @@ enum test_interrupt_handle_type {
TEST_INTERRUPT_HANDLE_VALID,
TEST_INTERRUPT_HANDLE_VALID_UIO,
TEST_INTERRUPT_HANDLE_VALID_ALARM,
+ TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT,
TEST_INTERRUPT_HANDLE_CASE1,
TEST_INTERRUPT_HANDLE_MAX
};
@@ -80,6 +81,10 @@ test_interrupt_init(void)
intr_handles[TEST_INTERRUPT_HANDLE_VALID_ALARM].type =
RTE_INTR_HANDLE_ALARM;
+ intr_handles[TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT].fd = pfds.readfd;
+ intr_handles[TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT].type =
+ RTE_INTR_HANDLE_DEV_EVENT;
+
intr_handles[TEST_INTERRUPT_HANDLE_CASE1].fd = pfds.writefd;
intr_handles[TEST_INTERRUPT_HANDLE_CASE1].type = RTE_INTR_HANDLE_UIO;
@@ -250,6 +255,14 @@ test_interrupt_enable(void)
return -1;
}
+ /* check with specific valid intr_handle */
+ test_intr_handle = intr_handles[TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT];
+ if (rte_intr_enable(&test_intr_handle) == 0) {
+ printf("unexpectedly enable a specific intr_handle "
+ "successfully\n");
+ return -1;
+ }
+
/* check with valid handler and its type */
test_intr_handle = intr_handles[TEST_INTERRUPT_HANDLE_CASE1];
if (rte_intr_enable(&test_intr_handle) < 0) {
@@ -306,6 +319,14 @@ test_interrupt_disable(void)
return -1;
}
+ /* check with specific valid intr_handle */
+ test_intr_handle = intr_handles[TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT];
+ if (rte_intr_disable(&test_intr_handle) == 0) {
+ printf("unexpectedly disable a specific intr_handle "
+ "successfully\n");
+ return -1;
+ }
+
/* check with valid handler and its type */
test_intr_handle = intr_handles[TEST_INTERRUPT_HANDLE_CASE1];
if (rte_intr_disable(&test_intr_handle) < 0) {
@@ -393,9 +414,17 @@ test_interrupt(void)
goto out;
}
+ printf("Check valid device event interrupt full path\n");
+ if (test_interrupt_full_path_check(
+ TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT) < 0) {
+ printf("failure occurred during checking valid device event "
+ "interrupt full path\n");
+ goto out;
+ }
+
printf("Check valid alarm interrupt full path\n");
- if (test_interrupt_full_path_check(TEST_INTERRUPT_HANDLE_VALID_ALARM)
- < 0) {
+ if (test_interrupt_full_path_check(
+ TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT) < 0) {
printf("failure occurred during checking valid alarm "
"interrupt full path\n");
goto out;
@@ -513,6 +542,12 @@ test_interrupt(void)
rte_intr_callback_unregister(&test_intr_handle,
test_interrupt_callback_1, (void *)-1);
+ test_intr_handle = intr_handles[TEST_INTERRUPT_HANDLE_VALID_DEV_EVENT];
+ rte_intr_callback_unregister(&test_intr_handle,
+ test_interrupt_callback, (void *)-1);
+ rte_intr_callback_unregister(&test_intr_handle,
+ test_interrupt_callback_1, (void *)-1);
+
rte_delay_ms(2 * TEST_INTERRUPT_CHECK_INTERVAL);
/* deinit */
test_interrupt_deinit();
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V22 2/4] eal: add device event monitor framework
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
2018-04-13 8:30 ` [PATCH V22 1/4] eal: add device event handle in interrupt thread Jeff Guo
@ 2018-04-13 8:30 ` Jeff Guo
2018-04-13 8:30 ` [PATCH V22 3/4] eal/linux: uevent parse and process Jeff Guo
` (2 subsequent siblings)
4 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-13 8:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch aims to add a general device event monitor framework at
EAL device layer, for device hotplug awareness and actions adopted
accordingly. It could also expand for all other types of device event
monitor, but not in this scope at the stage.
To get started, users firstly call below new added APIs to enable/disable
the device event monitor mechanism:
- rte_dev_event_monitor_start
- rte_dev_event_monitor_stop
Then users shell register or unregister callbacks through the new added
APIs. Callbacks can be some device specific, or for all devices.
-rte_dev_event_callback_register
-rte_dev_event_callback_unregister
Use hotplug case for example, when device hotplug insertion or hotplug
removal, we will get notified from kernel, then call user's callbacks
accordingly to handle it, such as detach or attach the device from the
bus, and could benefit further fail-safe or live-migration.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Reviewed-by: Jianfeng Tan <jianfeng.tan@intel.com>
---
v22->v21:
fix clang compile issue
---
doc/guides/rel_notes/release_18_05.rst | 10 ++
lib/librte_eal/bsdapp/eal/Makefile | 1 +
lib/librte_eal/bsdapp/eal/eal_dev.c | 21 +++++
lib/librte_eal/bsdapp/eal/meson.build | 1 +
lib/librte_eal/common/eal_common_dev.c | 161 ++++++++++++++++++++++++++++++++
lib/librte_eal/common/eal_private.h | 15 +++
lib/librte_eal/common/include/rte_dev.h | 94 +++++++++++++++++++
lib/librte_eal/linuxapp/eal/Makefile | 1 +
lib/librte_eal/linuxapp/eal/eal_dev.c | 22 +++++
lib/librte_eal/linuxapp/eal/meson.build | 1 +
lib/librte_eal/rte_eal_version.map | 4 +
11 files changed, 331 insertions(+)
create mode 100644 lib/librte_eal/bsdapp/eal/eal_dev.c
create mode 100644 lib/librte_eal/linuxapp/eal/eal_dev.c
diff --git a/doc/guides/rel_notes/release_18_05.rst b/doc/guides/rel_notes/release_18_05.rst
index 563c2f3..071ec91 100644
--- a/doc/guides/rel_notes/release_18_05.rst
+++ b/doc/guides/rel_notes/release_18_05.rst
@@ -58,6 +58,16 @@ New Features
* Added support for NVGRE, VXLAN and GENEVE filters in flow API.
* Added support for DROP action in flow API.
+* **Added device event monitor framework.**
+
+ Added a general device event monitor framework at EAL, for device dynamic management.
+ Such as device hotplug awareness and actions adopted accordingly. The list of new APIs:
+
+ * ``rte_dev_event_monitor_start`` and ``rte_dev_event_monitor_stop`` are for
+ the event monitor enable and disable.
+ * ``rte_dev_event_callback_register`` and ``rte_dev_event_callback_unregister``
+ are for the user's callbacks register and unregister.
+
API Changes
-----------
diff --git a/lib/librte_eal/bsdapp/eal/Makefile b/lib/librte_eal/bsdapp/eal/Makefile
index 250d5c1..200285e 100644
--- a/lib/librte_eal/bsdapp/eal/Makefile
+++ b/lib/librte_eal/bsdapp/eal/Makefile
@@ -34,6 +34,7 @@ SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_lcore.c
SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_timer.c
SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_interrupts.c
SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_alarm.c
+SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_dev.c
# from common dir
SRCS-$(CONFIG_RTE_EXEC_ENV_BSDAPP) += eal_common_lcore.c
diff --git a/lib/librte_eal/bsdapp/eal/eal_dev.c b/lib/librte_eal/bsdapp/eal/eal_dev.c
new file mode 100644
index 0000000..1c6c51b
--- /dev/null
+++ b/lib/librte_eal/bsdapp/eal/eal_dev.c
@@ -0,0 +1,21 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2018 Intel Corporation
+ */
+
+#include <rte_log.h>
+#include <rte_compat.h>
+#include <rte_dev.h>
+
+int __rte_experimental
+rte_dev_event_monitor_start(void)
+{
+ RTE_LOG(ERR, EAL, "Device event is not supported for FreeBSD\n");
+ return -1;
+}
+
+int __rte_experimental
+rte_dev_event_monitor_stop(void)
+{
+ RTE_LOG(ERR, EAL, "Device event is not supported for FreeBSD\n");
+ return -1;
+}
diff --git a/lib/librte_eal/bsdapp/eal/meson.build b/lib/librte_eal/bsdapp/eal/meson.build
index 4b40223..4c56118 100644
--- a/lib/librte_eal/bsdapp/eal/meson.build
+++ b/lib/librte_eal/bsdapp/eal/meson.build
@@ -13,4 +13,5 @@ env_sources = files('eal_alarm.c',
'eal_timer.c',
'eal.c',
'eal_memory.c',
+ 'eal_dev.c'
)
diff --git a/lib/librte_eal/common/eal_common_dev.c b/lib/librte_eal/common/eal_common_dev.c
index cd07144..0628b62 100644
--- a/lib/librte_eal/common/eal_common_dev.c
+++ b/lib/librte_eal/common/eal_common_dev.c
@@ -14,9 +14,34 @@
#include <rte_devargs.h>
#include <rte_debug.h>
#include <rte_log.h>
+#include <rte_spinlock.h>
+#include <rte_malloc.h>
#include "eal_private.h"
+/**
+ * The device event callback description.
+ *
+ * It contains callback address to be registered by user application,
+ * the pointer to the parameters for callback, and the device name.
+ */
+struct dev_event_callback {
+ TAILQ_ENTRY(dev_event_callback) next; /**< Callbacks list */
+ rte_dev_event_cb_fn cb_fn; /**< Callback address */
+ void *cb_arg; /**< Callback parameter */
+ char *dev_name; /**< Callback device name, NULL is for all device */
+ uint32_t active; /**< Callback is executing */
+};
+
+/** @internal Structure to keep track of registered callbacks */
+TAILQ_HEAD(dev_event_cb_list, dev_event_callback);
+
+/* The device event callback list for all registered callbacks. */
+static struct dev_event_cb_list dev_event_cbs;
+
+/* spinlock for device callbacks */
+static rte_spinlock_t dev_event_lock = RTE_SPINLOCK_INITIALIZER;
+
static int cmp_detached_dev_name(const struct rte_device *dev,
const void *_name)
{
@@ -207,3 +232,139 @@ rte_eal_hotplug_remove(const char *busname, const char *devname)
rte_eal_devargs_remove(busname, devname);
return ret;
}
+
+int __rte_experimental
+rte_dev_event_callback_register(const char *device_name,
+ rte_dev_event_cb_fn cb_fn,
+ void *cb_arg)
+{
+ struct dev_event_callback *event_cb;
+ int ret;
+
+ if (!cb_fn)
+ return -EINVAL;
+
+ rte_spinlock_lock(&dev_event_lock);
+
+ if (TAILQ_EMPTY(&dev_event_cbs))
+ TAILQ_INIT(&dev_event_cbs);
+
+ TAILQ_FOREACH(event_cb, &dev_event_cbs, next) {
+ if (event_cb->cb_fn == cb_fn && event_cb->cb_arg == cb_arg) {
+ if (device_name == NULL && event_cb->dev_name == NULL)
+ break;
+ if (device_name == NULL || event_cb->dev_name == NULL)
+ continue;
+ if (!strcmp(event_cb->dev_name, device_name))
+ break;
+ }
+ }
+
+ /* create a new callback. */
+ if (event_cb == NULL) {
+ event_cb = malloc(sizeof(struct dev_event_callback));
+ if (event_cb != NULL) {
+ event_cb->cb_fn = cb_fn;
+ event_cb->cb_arg = cb_arg;
+ event_cb->active = 0;
+ if (!device_name) {
+ event_cb->dev_name = NULL;
+ } else {
+ event_cb->dev_name = strdup(device_name);
+ if (event_cb->dev_name == NULL) {
+ ret = -ENOMEM;
+ goto error;
+ }
+ }
+ TAILQ_INSERT_TAIL(&dev_event_cbs, event_cb, next);
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Failed to allocate memory for device "
+ "event callback.");
+ ret = -ENOMEM;
+ goto error;
+ }
+ } else {
+ RTE_LOG(ERR, EAL,
+ "The callback is already exist, no need "
+ "to register again.\n");
+ ret = -EEXIST;
+ }
+
+ rte_spinlock_unlock(&dev_event_lock);
+ return 0;
+error:
+ free(event_cb);
+ rte_spinlock_unlock(&dev_event_lock);
+ return ret;
+}
+
+int __rte_experimental
+rte_dev_event_callback_unregister(const char *device_name,
+ rte_dev_event_cb_fn cb_fn,
+ void *cb_arg)
+{
+ int ret = 0;
+ struct dev_event_callback *event_cb, *next;
+
+ if (!cb_fn)
+ return -EINVAL;
+
+ rte_spinlock_lock(&dev_event_lock);
+ /*walk through the callbacks and remove all that match. */
+ for (event_cb = TAILQ_FIRST(&dev_event_cbs); event_cb != NULL;
+ event_cb = next) {
+
+ next = TAILQ_NEXT(event_cb, next);
+
+ if (device_name != NULL && event_cb->dev_name != NULL) {
+ if (!strcmp(event_cb->dev_name, device_name)) {
+ if (event_cb->cb_fn != cb_fn ||
+ (cb_arg != (void *)-1 &&
+ event_cb->cb_arg != cb_arg))
+ continue;
+ }
+ } else if (device_name != NULL) {
+ continue;
+ }
+
+ /*
+ * if this callback is not executing right now,
+ * then remove it.
+ */
+ if (event_cb->active == 0) {
+ TAILQ_REMOVE(&dev_event_cbs, event_cb, next);
+ free(event_cb);
+ ret++;
+ } else {
+ continue;
+ }
+ }
+ rte_spinlock_unlock(&dev_event_lock);
+ return ret;
+}
+
+void
+dev_callback_process(char *device_name, enum rte_dev_event_type event)
+{
+ struct dev_event_callback *cb_lst;
+
+ if (device_name == NULL)
+ return;
+
+ rte_spinlock_lock(&dev_event_lock);
+
+ TAILQ_FOREACH(cb_lst, &dev_event_cbs, next) {
+ if (cb_lst->dev_name) {
+ if (strcmp(cb_lst->dev_name, device_name))
+ continue;
+ }
+ cb_lst->active = 1;
+ rte_spinlock_unlock(&dev_event_lock);
+ cb_lst->cb_fn(device_name, event,
+ cb_lst->cb_arg);
+ rte_spinlock_lock(&dev_event_lock);
+ cb_lst->active = 0;
+ }
+ rte_spinlock_unlock(&dev_event_lock);
+}
diff --git a/lib/librte_eal/common/eal_private.h b/lib/librte_eal/common/eal_private.h
index 3fed436..c359589 100644
--- a/lib/librte_eal/common/eal_private.h
+++ b/lib/librte_eal/common/eal_private.h
@@ -9,6 +9,8 @@
#include <stdint.h>
#include <stdio.h>
+#include <rte_dev.h>
+
/**
* Initialize the memzone subsystem (private to eal).
*
@@ -238,4 +240,17 @@ struct rte_bus *rte_bus_find_by_device_name(const char *str);
int rte_mp_channel_init(void);
+/**
+ * Internal Executes all the user application registered callbacks for
+ * the specific device. It is for DPDK internal user only. User
+ * application should not call it directly.
+ *
+ * @param device_name
+ * The device name.
+ * @param event
+ * the device event type.
+ *
+ */
+void
+dev_callback_process(char *device_name, enum rte_dev_event_type event);
#endif /* _EAL_PRIVATE_H_ */
diff --git a/lib/librte_eal/common/include/rte_dev.h b/lib/librte_eal/common/include/rte_dev.h
index b688f1e..a5203e7 100644
--- a/lib/librte_eal/common/include/rte_dev.h
+++ b/lib/librte_eal/common/include/rte_dev.h
@@ -24,6 +24,25 @@ extern "C" {
#include <rte_compat.h>
#include <rte_log.h>
+/**
+ * The device event type.
+ */
+enum rte_dev_event_type {
+ RTE_DEV_EVENT_ADD, /**< device being added */
+ RTE_DEV_EVENT_REMOVE, /**< device being removed */
+ RTE_DEV_EVENT_MAX /**< max value of this enum */
+};
+
+struct rte_dev_event {
+ enum rte_dev_event_type type; /**< device event type */
+ int subsystem; /**< subsystem id */
+ char *devname; /**< device name */
+};
+
+typedef void (*rte_dev_event_cb_fn)(char *device_name,
+ enum rte_dev_event_type event,
+ void *cb_arg);
+
__attribute__((format(printf, 2, 0)))
static inline void
rte_pmd_debug_trace(const char *func_name, const char *fmt, ...)
@@ -267,4 +286,79 @@ __attribute__((used)) = str
}
#endif
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice
+ *
+ * It registers the callback for the specific device.
+ * Multiple callbacks cal be registered at the same time.
+ *
+ * @param device_name
+ * The device name, that is the param name of the struct rte_device,
+ * null value means for all devices.
+ * @param cb_fn
+ * callback address.
+ * @param cb_arg
+ * address of parameter for callback.
+ *
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int __rte_experimental
+rte_dev_event_callback_register(const char *device_name,
+ rte_dev_event_cb_fn cb_fn,
+ void *cb_arg);
+
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice
+ *
+ * It unregisters the callback according to the specified device.
+ *
+ * @param device_name
+ * The device name, that is the param name of the struct rte_device,
+ * null value means for all devices and their callbacks.
+ * @param cb_fn
+ * callback address.
+ * @param cb_arg
+ * address of parameter for callback, (void *)-1 means to remove all
+ * registered which has the same callback address.
+ *
+ * @return
+ * - On success, return the number of callback entities removed.
+ * - On failure, a negative value.
+ */
+int __rte_experimental
+rte_dev_event_callback_unregister(const char *device_name,
+ rte_dev_event_cb_fn cb_fn,
+ void *cb_arg);
+
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice
+ *
+ * Start the device event monitoring.
+ *
+ * @param none
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int __rte_experimental
+rte_dev_event_monitor_start(void);
+
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice
+ *
+ * Stop the device event monitoring .
+ *
+ * @param none
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int __rte_experimental
+rte_dev_event_monitor_stop(void);
#endif /* _RTE_DEV_H_ */
diff --git a/lib/librte_eal/linuxapp/eal/Makefile b/lib/librte_eal/linuxapp/eal/Makefile
index 542bf7e..45517a2 100644
--- a/lib/librte_eal/linuxapp/eal/Makefile
+++ b/lib/librte_eal/linuxapp/eal/Makefile
@@ -42,6 +42,7 @@ SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_lcore.c
SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_timer.c
SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_interrupts.c
SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_alarm.c
+SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_dev.c
# from common dir
SRCS-$(CONFIG_RTE_EXEC_ENV_LINUXAPP) += eal_common_lcore.c
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
new file mode 100644
index 0000000..9c8d1a0
--- /dev/null
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -0,0 +1,22 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2018 Intel Corporation
+ */
+
+#include <rte_log.h>
+#include <rte_compat.h>
+#include <rte_dev.h>
+
+
+int __rte_experimental
+rte_dev_event_monitor_start(void)
+{
+ /* TODO: start uevent monitor for linux */
+ return 0;
+}
+
+int __rte_experimental
+rte_dev_event_monitor_stop(void)
+{
+ /* TODO: stop uevent monitor for linux */
+ return 0;
+}
diff --git a/lib/librte_eal/linuxapp/eal/meson.build b/lib/librte_eal/linuxapp/eal/meson.build
index 5254c6c..9c01931 100644
--- a/lib/librte_eal/linuxapp/eal/meson.build
+++ b/lib/librte_eal/linuxapp/eal/meson.build
@@ -19,6 +19,7 @@ env_sources = files('eal_alarm.c',
'eal_vfio_mp_sync.c',
'eal.c',
'eal_memory.c',
+ 'eal_dev.c',
)
if has_libnuma == 1
diff --git a/lib/librte_eal/rte_eal_version.map b/lib/librte_eal/rte_eal_version.map
index 603c744..d02d80b 100644
--- a/lib/librte_eal/rte_eal_version.map
+++ b/lib/librte_eal/rte_eal_version.map
@@ -213,6 +213,10 @@ DPDK_18.02 {
EXPERIMENTAL {
global:
+ rte_dev_event_callback_register;
+ rte_dev_event_callback_unregister;
+ rte_dev_event_monitor_start;
+ rte_dev_event_monitor_stop;
rte_eal_cleanup;
rte_eal_devargs_insert;
rte_eal_devargs_parse;
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V22 3/4] eal/linux: uevent parse and process
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
2018-04-13 8:30 ` [PATCH V22 1/4] eal: add device event handle in interrupt thread Jeff Guo
2018-04-13 8:30 ` [PATCH V22 2/4] eal: add device event monitor framework Jeff Guo
@ 2018-04-13 8:30 ` Jeff Guo
2018-04-13 8:30 ` [PATCH V22 4/4] app/testpmd: enable device hotplug monitoring Jeff Guo
2018-04-13 10:03 ` [PATCH V22 0/4] add device event monitor framework Thomas Monjalon
4 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-13 8:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
In order to handle the uevent which has been detected from the kernel
side, add uevent parse and process function to translate the uevent into
device event, which user has subscribed to monitor.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Reviewed-by: Jianfeng Tan <jianfeng.tan@intel.com>
---
v22->v21:
correct some doc style
---
doc/guides/rel_notes/release_18_05.rst | 2 +
lib/librte_eal/linuxapp/eal/eal_dev.c | 205 ++++++++++++++++++++++++++++++++-
2 files changed, 205 insertions(+), 2 deletions(-)
diff --git a/doc/guides/rel_notes/release_18_05.rst b/doc/guides/rel_notes/release_18_05.rst
index 071ec91..a018ef5 100644
--- a/doc/guides/rel_notes/release_18_05.rst
+++ b/doc/guides/rel_notes/release_18_05.rst
@@ -68,6 +68,8 @@ New Features
* ``rte_dev_event_callback_register`` and ``rte_dev_event_callback_unregister``
are for the user's callbacks register and unregister.
+ Linux uevent is supported as backend of this device event notification framework.
+
API Changes
-----------
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 9c8d1a0..9478a39 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -2,21 +2,222 @@
* Copyright(c) 2018 Intel Corporation
*/
+#include <string.h>
+#include <unistd.h>
+#include <sys/socket.h>
+#include <linux/netlink.h>
+
#include <rte_log.h>
#include <rte_compat.h>
#include <rte_dev.h>
+#include <rte_malloc.h>
+#include <rte_interrupts.h>
+#include <rte_alarm.h>
+
+#include "eal_private.h"
+
+static struct rte_intr_handle intr_handle = {.fd = -1 };
+static bool monitor_started;
+
+#define EAL_UEV_MSG_LEN 4096
+#define EAL_UEV_MSG_ELEM_LEN 128
+
+static void dev_uev_handler(__rte_unused void *param);
+
+/* identify the system layer which reports this event. */
+enum eal_dev_event_subsystem {
+ EAL_DEV_EVENT_SUBSYSTEM_PCI, /* PCI bus device event */
+ EAL_DEV_EVENT_SUBSYSTEM_UIO, /* UIO driver device event */
+ EAL_DEV_EVENT_SUBSYSTEM_VFIO, /* VFIO driver device event */
+ EAL_DEV_EVENT_SUBSYSTEM_MAX
+};
+
+static int
+dev_uev_socket_fd_create(void)
+{
+ struct sockaddr_nl addr;
+ int ret;
+
+ intr_handle.fd = socket(PF_NETLINK, SOCK_RAW | SOCK_CLOEXEC |
+ SOCK_NONBLOCK,
+ NETLINK_KOBJECT_UEVENT);
+ if (intr_handle.fd < 0) {
+ RTE_LOG(ERR, EAL, "create uevent fd failed.\n");
+ return -1;
+ }
+
+ memset(&addr, 0, sizeof(addr));
+ addr.nl_family = AF_NETLINK;
+ addr.nl_pid = 0;
+ addr.nl_groups = 0xffffffff;
+
+ ret = bind(intr_handle.fd, (struct sockaddr *) &addr, sizeof(addr));
+ if (ret < 0) {
+ RTE_LOG(ERR, EAL, "Failed to bind uevent socket.\n");
+ goto err;
+ }
+ return 0;
+err:
+ close(intr_handle.fd);
+ intr_handle.fd = -1;
+ return ret;
+}
+
+static int
+dev_uev_parse(const char *buf, struct rte_dev_event *event, int length)
+{
+ char action[EAL_UEV_MSG_ELEM_LEN];
+ char subsystem[EAL_UEV_MSG_ELEM_LEN];
+ char pci_slot_name[EAL_UEV_MSG_ELEM_LEN];
+ int i = 0;
+
+ memset(action, 0, EAL_UEV_MSG_ELEM_LEN);
+ memset(subsystem, 0, EAL_UEV_MSG_ELEM_LEN);
+ memset(pci_slot_name, 0, EAL_UEV_MSG_ELEM_LEN);
+
+ while (i < length) {
+ for (; i < length; i++) {
+ if (*buf)
+ break;
+ buf++;
+ }
+ /**
+ * check device uevent from kernel side, no need to check
+ * uevent from udev.
+ */
+ if (!strncmp(buf, "libudev", 7)) {
+ buf += 7;
+ i += 7;
+ return -1;
+ }
+ if (!strncmp(buf, "ACTION=", 7)) {
+ buf += 7;
+ i += 7;
+ snprintf(action, sizeof(action), "%s", buf);
+ } else if (!strncmp(buf, "SUBSYSTEM=", 10)) {
+ buf += 10;
+ i += 10;
+ snprintf(subsystem, sizeof(subsystem), "%s", buf);
+ } else if (!strncmp(buf, "PCI_SLOT_NAME=", 14)) {
+ buf += 14;
+ i += 14;
+ snprintf(pci_slot_name, sizeof(subsystem), "%s", buf);
+ event->devname = strdup(pci_slot_name);
+ }
+ for (; i < length; i++) {
+ if (*buf == '\0')
+ break;
+ buf++;
+ }
+ }
+
+ /* parse the subsystem layer */
+ if (!strncmp(subsystem, "uio", 3))
+ event->subsystem = EAL_DEV_EVENT_SUBSYSTEM_UIO;
+ else if (!strncmp(subsystem, "pci", 3))
+ event->subsystem = EAL_DEV_EVENT_SUBSYSTEM_PCI;
+ else if (!strncmp(subsystem, "vfio", 4))
+ event->subsystem = EAL_DEV_EVENT_SUBSYSTEM_VFIO;
+ else
+ return -1;
+
+ /* parse the action type */
+ if (!strncmp(action, "add", 3))
+ event->type = RTE_DEV_EVENT_ADD;
+ else if (!strncmp(action, "remove", 6))
+ event->type = RTE_DEV_EVENT_REMOVE;
+ else
+ return -1;
+ return 0;
+}
+
+static void
+dev_delayed_unregister(void *param)
+{
+ rte_intr_callback_unregister(&intr_handle, dev_uev_handler, param);
+ close(intr_handle.fd);
+ intr_handle.fd = -1;
+}
+
+static void
+dev_uev_handler(__rte_unused void *param)
+{
+ struct rte_dev_event uevent;
+ int ret;
+ char buf[EAL_UEV_MSG_LEN];
+
+ memset(&uevent, 0, sizeof(struct rte_dev_event));
+ memset(buf, 0, EAL_UEV_MSG_LEN);
+
+ ret = recv(intr_handle.fd, buf, EAL_UEV_MSG_LEN, MSG_DONTWAIT);
+ if (ret < 0 && errno == EAGAIN)
+ return;
+ else if (ret <= 0) {
+ /* connection is closed or broken, can not up again. */
+ RTE_LOG(ERR, EAL, "uevent socket connection is broken.\n");
+ rte_eal_alarm_set(1, dev_delayed_unregister, NULL);
+ return;
+ }
+
+ ret = dev_uev_parse(buf, &uevent, EAL_UEV_MSG_LEN);
+ if (ret < 0) {
+ RTE_LOG(DEBUG, EAL, "It is not an valid event "
+ "that need to be handle.\n");
+ return;
+ }
+
+ RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
+ uevent.devname, uevent.type, uevent.subsystem);
+
+ if (uevent.devname)
+ dev_callback_process(uevent.devname, uevent.type);
+}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
- /* TODO: start uevent monitor for linux */
+ int ret;
+
+ if (monitor_started)
+ return 0;
+
+ ret = dev_uev_socket_fd_create();
+ if (ret) {
+ RTE_LOG(ERR, EAL, "error create device event fd.\n");
+ return -1;
+ }
+
+ intr_handle.type = RTE_INTR_HANDLE_DEV_EVENT;
+ ret = rte_intr_callback_register(&intr_handle, dev_uev_handler, NULL);
+
+ if (ret) {
+ RTE_LOG(ERR, EAL, "fail to register uevent callback.\n");
+ return -1;
+ }
+
+ monitor_started = true;
+
return 0;
}
int __rte_experimental
rte_dev_event_monitor_stop(void)
{
- /* TODO: stop uevent monitor for linux */
+ int ret;
+
+ if (!monitor_started)
+ return 0;
+
+ ret = rte_intr_callback_unregister(&intr_handle, dev_uev_handler,
+ (void *)-1);
+ if (ret < 0) {
+ RTE_LOG(ERR, EAL, "fail to unregister uevent callback.\n");
+ return ret;
+ }
+
+ close(intr_handle.fd);
+ intr_handle.fd = -1;
+ monitor_started = false;
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V22 4/4] app/testpmd: enable device hotplug monitoring
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
` (2 preceding siblings ...)
2018-04-13 8:30 ` [PATCH V22 3/4] eal/linux: uevent parse and process Jeff Guo
@ 2018-04-13 8:30 ` Jeff Guo
2018-04-13 10:03 ` [PATCH V22 0/4] add device event monitor framework Thomas Monjalon
4 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-13 8:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application uses device event
APIs to monitor the hotplug events, including both hot removal event
and hot insertion event.
The process is that, testpmd first enable hotplug by below commands,
E.g. ./build/app/testpmd -c 0x3 --n 4 -- -i --hot-plug
then testpmd starts the device event monitor by calling the new API
(rte_dev_event_monitor_start) and register the user's callback by call
the API (rte_dev_event_callback_register), when device being hotplug
insertion or hotplug removal, the device event monitor detects the event
and call user's callbacks, user could process the event in the callback
accordingly.
This patch only shows the event monitoring, device attach/detach would
not be involved here, will add from other hotplug patch set.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Reviewed-by: Jianfeng Tan <jianfeng.tan@intel.com>
---
v22->v21:
no change
---
app/test-pmd/parameters.c | 5 +-
app/test-pmd/testpmd.c | 101 +++++++++++++++++++++++++++++++++-
app/test-pmd/testpmd.h | 2 +
doc/guides/testpmd_app_ug/run_app.rst | 4 ++
4 files changed, 110 insertions(+), 2 deletions(-)
diff --git a/app/test-pmd/parameters.c b/app/test-pmd/parameters.c
index 2192bdc..1a05284 100644
--- a/app/test-pmd/parameters.c
+++ b/app/test-pmd/parameters.c
@@ -186,6 +186,7 @@ usage(char* progname)
printf(" --flow-isolate-all: "
"requests flow API isolated mode on all ports at initialization time.\n");
printf(" --tx-offloads=0xXXXXXXXX: hexadecimal bitmask of TX queue offloads\n");
+ printf(" --hot-plug: enable hot plug for device.\n");
}
#ifdef RTE_LIBRTE_CMDLINE
@@ -621,6 +622,7 @@ launch_args_parse(int argc, char** argv)
{ "print-event", 1, 0, 0 },
{ "mask-event", 1, 0, 0 },
{ "tx-offloads", 1, 0, 0 },
+ { "hot-plug", 0, 0, 0 },
{ 0, 0, 0, 0 },
};
@@ -1101,7 +1103,8 @@ launch_args_parse(int argc, char** argv)
rte_exit(EXIT_FAILURE,
"invalid mask-event argument\n");
}
-
+ if (!strcmp(lgopts[opt_idx].name, "hot-plug"))
+ hot_plug = 1;
break;
case 'h':
usage(argv[0]);
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 4c0e258..d2c122a 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -12,6 +12,7 @@
#include <sys/mman.h>
#include <sys/types.h>
#include <errno.h>
+#include <stdbool.h>
#include <sys/queue.h>
#include <sys/stat.h>
@@ -284,6 +285,8 @@ uint8_t lsc_interrupt = 1; /* enabled by default */
*/
uint8_t rmv_interrupt = 1; /* enabled by default */
+uint8_t hot_plug = 0; /**< hotplug disabled by default. */
+
/*
* Display or mask ether events
* Default to all events except VF_MBOX
@@ -391,6 +394,12 @@ static void check_all_ports_link_status(uint32_t port_mask);
static int eth_event_callback(portid_t port_id,
enum rte_eth_event_type type,
void *param, void *ret_param);
+static void eth_dev_event_callback(char *device_name,
+ enum rte_dev_event_type type,
+ void *param);
+static int eth_dev_event_callback_register(void);
+static int eth_dev_event_callback_unregister(void);
+
/*
* Check if all the ports are started.
@@ -1853,6 +1862,39 @@ reset_port(portid_t pid)
printf("Done\n");
}
+static int
+eth_dev_event_callback_register(void)
+{
+ int ret;
+
+ /* register the device event callback */
+ ret = rte_dev_event_callback_register(NULL,
+ eth_dev_event_callback, NULL);
+ if (ret) {
+ printf("Failed to register device event callback\n");
+ return -1;
+ }
+
+ return 0;
+}
+
+
+static int
+eth_dev_event_callback_unregister(void)
+{
+ int ret;
+
+ /* unregister the device event callback */
+ ret = rte_dev_event_callback_unregister(NULL,
+ eth_dev_event_callback, NULL);
+ if (ret < 0) {
+ printf("Failed to unregister device event callback\n");
+ return -1;
+ }
+
+ return 0;
+}
+
void
attach_port(char *identifier)
{
@@ -1916,6 +1958,7 @@ void
pmd_test_exit(void)
{
portid_t pt_id;
+ int ret;
if (test_done == 0)
stop_packet_forwarding();
@@ -1929,6 +1972,18 @@ pmd_test_exit(void)
close_port(pt_id);
}
}
+
+ if (hot_plug) {
+ ret = rte_dev_event_monitor_stop();
+ if (ret)
+ RTE_LOG(ERR, EAL,
+ "fail to stop device event monitor.");
+
+ ret = eth_dev_event_callback_unregister();
+ if (ret)
+ RTE_LOG(ERR, EAL,
+ "fail to unregister all event callbacks.");
+ }
printf("\nBye...\n");
}
@@ -2059,6 +2114,37 @@ eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void *param,
return 0;
}
+/* This function is used by the interrupt thread */
+static void
+eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
+ __rte_unused void *arg)
+{
+ if (type >= RTE_DEV_EVENT_MAX) {
+ fprintf(stderr, "%s called upon invalid event %d\n",
+ __func__, type);
+ fflush(stderr);
+ }
+
+ switch (type) {
+ case RTE_DEV_EVENT_REMOVE:
+ RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
+ device_name);
+ /* TODO: After finish failure handle, begin to stop
+ * packet forward, stop port, close port, detach port.
+ */
+ break;
+ case RTE_DEV_EVENT_ADD:
+ RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
+ device_name);
+ /* TODO: After finish kernel driver binding,
+ * begin to attach port.
+ */
+ break;
+ default:
+ break;
+ }
+}
+
static int
set_tx_queue_stats_mapping_registers(portid_t port_id, struct rte_port *port)
{
@@ -2474,8 +2560,9 @@ signal_handler(int signum)
int
main(int argc, char** argv)
{
- int diag;
+ int diag;
portid_t port_id;
+ int ret;
signal(SIGINT, signal_handler);
signal(SIGTERM, signal_handler);
@@ -2543,6 +2630,18 @@ main(int argc, char** argv)
nb_rxq, nb_txq);
init_config();
+
+ if (hot_plug) {
+ /* enable hot plug monitoring */
+ ret = rte_dev_event_monitor_start();
+ if (ret) {
+ rte_errno = EINVAL;
+ return -1;
+ }
+ eth_dev_event_callback_register();
+
+ }
+
if (start_port(RTE_PORT_ALL) != 0)
rte_exit(EXIT_FAILURE, "Start ports failed\n");
diff --git a/app/test-pmd/testpmd.h b/app/test-pmd/testpmd.h
index 153abea..8fde68d 100644
--- a/app/test-pmd/testpmd.h
+++ b/app/test-pmd/testpmd.h
@@ -319,6 +319,8 @@ extern volatile int test_done; /* stop packet forwarding when set to 1. */
extern uint8_t lsc_interrupt; /**< disabled by "--no-lsc-interrupt" parameter */
extern uint8_t rmv_interrupt; /**< disabled by "--no-rmv-interrupt" parameter */
extern uint32_t event_print_mask;
+extern uint8_t hot_plug; /**< enable by "--hot-plug" parameter */
+
/**< set by "--print-event xxxx" and "--mask-event xxxx parameters */
#ifdef RTE_LIBRTE_IXGBE_BYPASS
diff --git a/doc/guides/testpmd_app_ug/run_app.rst b/doc/guides/testpmd_app_ug/run_app.rst
index 1fd5395..d0ced36 100644
--- a/doc/guides/testpmd_app_ug/run_app.rst
+++ b/doc/guides/testpmd_app_ug/run_app.rst
@@ -479,3 +479,7 @@ The commandline options are:
Set the hexadecimal bitmask of TX queue offloads.
The default value is 0.
+
+* ``--hot-plug``
+
+ Enable device event monitor machenism for hotplug.
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V22 0/4] add device event monitor framework
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
` (3 preceding siblings ...)
2018-04-13 8:30 ` [PATCH V22 4/4] app/testpmd: enable device hotplug monitoring Jeff Guo
@ 2018-04-13 10:03 ` Thomas Monjalon
4 siblings, 0 replies; 494+ messages in thread
From: Thomas Monjalon @ 2018-04-13 10:03 UTC (permalink / raw)
To: Jeff Guo
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, harry.van.haaren, jianfeng.tan,
shreyansh.jain, helin.zhang
13/04/2018 10:30, Jeff Guo:
> About hot plug in dpdk, We already have proactive way to add/remove devices
> through APIs (rte_eal_hotplug_add/remove), and also have fail-safe driver
> to offload the fail-safe work from the app user. But there are still lack
> of a general mechanism to monitor hotplug event for all driver, now the
> hotplug interrupt event is diversity between each device and driver, such
> as mlx4, pci driver and others.
>
> Use the hot removal event for example, pci drivers not all exposure the
> remove interrupt, so in order to make user to easy use the hot plug
> feature for pci driver, something must be done to detect the remove event
> at the kernel level and offer a new line of interrupt to the user land.
>
> Base on the uevent of kobject mechanism in kernel, we could use it to
> benefit for monitoring the hot plug status of the device which not only
> uio/vfio of pci bus devices, but also other, such as cpu/usb/pci-express bus devices.
[...]
> Jeff Guo (4):
> eal: add device event handle in interrupt thread
> eal: add device event monitor framework
> eal/linux: uevent parse and process
> app/testpmd: enable device hotplug monitoring
Applied, thanks
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V20 0/4] add hot plug recovery mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (2 preceding siblings ...)
2018-04-13 8:30 ` [PATCH V22 0/4] add device event monitor framework Jeff Guo
@ 2018-04-18 13:38 ` Jeff Guo
2018-04-18 13:38 ` [PATCH V20 1/4] bus/pci: introduce device hot unplug handle Jeff Guo
` (3 more replies)
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
` (19 subsequent siblings)
23 siblings, 4 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-18 13:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
At the prior, device event monitor framework have been introduced,
the typical usage of it is for device hot plug. If we want application
would not be break down when device hot plug in or out, we still need some
measures to do recovery to do preparation for device detach, so that we will
not encounter any memory fault after device be hot unplug, that will let
application to keep working.
This patch set will introduces an API to implement the recovery mechanism to
handle hot plug, and also use testpmd to show example how to
use the API for process hot plug event, let the process could be
smoothly like below:
plug out->failure handle->stop forward->stop port->close port->detach port
with this mechanism, user such as fail-safe driver or testpmd could be able to
develop their own hot plug application.
patchset history:
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set
"add device event monitor framework"
Jeff Guo (4):
bus/pci: introduce device hot unplug handle
eal: add failure handler mechanism for hot plug
igb_uio: fix uio release issue when hot unplug
app/testpmd: show example to handler hot unplug
app/test-pmd/testpmd.c | 29 ++++++--
doc/guides/rel_notes/release_18_05.rst | 6 ++
drivers/bus/pci/pci_common.c | 67 +++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 32 +++++++++
drivers/bus/pci/private.h | 12 ++++
kernel/linux/igb_uio/igb_uio.c | 4 ++
lib/librte_eal/common/include/rte_bus.h | 16 +++++
lib/librte_eal/common/include/rte_dev.h | 11 +++
lib/librte_eal/linuxapp/eal/eal_dev.c | 124 +++++++++++++++++++++++++++++++-
lib/librte_eal/rte_eal_version.map | 1 +
10 files changed, 297 insertions(+), 5 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V20 1/4] bus/pci: introduce device hot unplug handle
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
@ 2018-04-18 13:38 ` Jeff Guo
2018-04-20 10:32 ` Ananyev, Konstantin
2018-04-18 13:38 ` [PATCH V20 2/4] eal: add failure handler mechanism for hot plug Jeff Guo
` (2 subsequent siblings)
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-04-18 13:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As of device hot unplug, we need some preparatory measures so that we will
not encounter memory fault after device be plug out of the system,
and also let we could recover the running data path but not been break.
This patch allows the buses to handle device hot unplug event.
The patch only enable the ops in pci bus, when handle device hot unplug
event, remap a dummy memory to avoid bus read/write error.
Other buses could accordingly implement this ops specific by themselves.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v20->19:
clean the code
---
drivers/bus/pci/pci_common.c | 67 +++++++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 32 ++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
4 files changed, 127 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 2a00f36..709eaf3 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -474,6 +474,72 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
}
static int
+pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0, i, isfound = 0;
+
+ if (failure_addr != NULL) {
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != sizeof(pdev->mem_resource) /
+ sizeof(pdev->mem_resource[0]); i++) {
+ if ((uint64_t)failure_addr >=
+ (uint64_t)pdev->mem_resource[i].addr &&
+ (uint64_t)failure_addr <=
+ (uint64_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(ERR, EAL, "Failure address "
+ "%16.16"PRIx64" is belong to "
+ "resource of device %s!\n",
+ (uint64_t)failure_addr,
+ pdev->device.name);
+ isfound = 1;
+ break;
+ }
+ }
+ if (isfound)
+ break;
+ }
+ } else if (dev != NULL) {
+ pdev = RTE_DEV_TO_PCI(dev);
+ } else {
+ return -EINVAL;
+ }
+
+ if (!pdev)
+ return -1;
+
+ /* remap resources for devices */
+ switch (pdev->kdrv) {
+ case RTE_KDRV_VFIO:
+#ifdef VFIO_PRESENT
+ /* TODO */
+#endif
+ break;
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ if (rte_eal_using_phys_addrs()) {
+ /* map resources for devices that use uio */
+ ret = pci_uio_remap_resource(pdev);
+ }
+ break;
+ case RTE_KDRV_NIC_UIO:
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ " Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ if (ret != 0)
+ RTE_LOG(ERR, EAL, "failed to handle hot unplug of %s",
+ pdev->name);
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -503,6 +569,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .handle_hot_unplug = pci_handle_hot_unplug,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..ba2c458 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,38 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ pci_unmap_resource(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len);
+ map_address = pci_map_resource(
+ dev->mem_resource[i].addr, -1, 0,
+ (size_t)dev->mem_resource[i].len,
+ MAP_ANONYMOUS | MAP_FIXED);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 88fa587..cc1668c 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * remap the pci uio resource.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index 6fb0834..d2c5778 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation specific hot unplug handler function which is responsible
+ * for handle the failure when hot unplug the device, guaranty the system
+ * would not crash in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
+ void *dev_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -209,6 +223,8 @@ struct rte_bus {
rte_bus_plug_t plug; /**< Probe single device for drivers */
rte_bus_unplug_t unplug; /**< Remove single device from driver */
rte_bus_parse_t parse; /**< Parse a device name */
+ rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
+ device event */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
};
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V20 1/4] bus/pci: introduce device hot unplug handle
2018-04-18 13:38 ` [PATCH V20 1/4] bus/pci: introduce device hot unplug handle Jeff Guo
@ 2018-04-20 10:32 ` Ananyev, Konstantin
2018-05-03 3:05 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-04-20 10:32 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
Hi Jeff,
> As of device hot unplug, we need some preparatory measures so that we will
> not encounter memory fault after device be plug out of the system,
> and also let we could recover the running data path but not been break.
> This patch allows the buses to handle device hot unplug event.
> The patch only enable the ops in pci bus, when handle device hot unplug
> event, remap a dummy memory to avoid bus read/write error.
> Other buses could accordingly implement this ops specific by themselves.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v20->19:
> clean the code
> ---
> drivers/bus/pci/pci_common.c | 67 +++++++++++++++++++++++++++++++++
> drivers/bus/pci/pci_common_uio.c | 32 ++++++++++++++++
> drivers/bus/pci/private.h | 12 ++++++
> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
> 4 files changed, 127 insertions(+)
>
> diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
> index 2a00f36..709eaf3 100644
> --- a/drivers/bus/pci/pci_common.c
> +++ b/drivers/bus/pci/pci_common.c
> @@ -474,6 +474,72 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
> }
>
> static int
> +pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
> +{
> + struct rte_pci_device *pdev = NULL;
> + int ret = 0, i, isfound = 0;
> +
> + if (failure_addr != NULL) {
> + FOREACH_DEVICE_ON_PCIBUS(pdev) {
> + for (i = 0; i != sizeof(pdev->mem_resource) /
> + sizeof(pdev->mem_resource[0]); i++) {
You can do i != RTE_DIM(pdev->mem_resource) here.
> + if ((uint64_t)failure_addr >=
> + (uint64_t)pdev->mem_resource[i].addr &&
> + (uint64_t)failure_addr <=
> + (uint64_t)pdev->mem_resource[i].addr +
> + pdev->mem_resource[i].len) {
I think it should be failure_addr < addr + len
> + RTE_LOG(ERR, EAL, "Failure address "
> + "%16.16"PRIx64" is belong to "
> + "resource of device %s!\n",
> + (uint64_t)failure_addr,
> + pdev->device.name);
> + isfound = 1;
> + break;
> + }
> + }
> + if (isfound)
> + break;
Might be it is a good thing to put the code that searches for address into a separate function.
> + }
> + } else if (dev != NULL) {
> + pdev = RTE_DEV_TO_PCI(dev);
> + } else {
> + return -EINVAL;
> + }
> +
> + if (!pdev)
> + return -1;
> +
> + /* remap resources for devices */
> + switch (pdev->kdrv) {
> + case RTE_KDRV_VFIO:
> +#ifdef VFIO_PRESENT
> + /* TODO */
> +#endif
Should set ret =-1 as not implemented now.
> + break;
> + case RTE_KDRV_IGB_UIO:
> + case RTE_KDRV_UIO_GENERIC:
> + if (rte_eal_using_phys_addrs()) {
> + /* map resources for devices that use uio */
> + ret = pci_uio_remap_resource(pdev);
> + }
> + break;
> + case RTE_KDRV_NIC_UIO:
> + ret = pci_uio_remap_resource(pdev);
> + break;
> + default:
> + RTE_LOG(DEBUG, EAL,
> + " Not managed by a supported kernel driver, skipped\n");
> + ret = -1;
> + break;
> + }
> +
> + if (ret != 0)
> + RTE_LOG(ERR, EAL, "failed to handle hot unplug of %s",
> + pdev->name);
> + return ret;
> +}
> +
> +static int
> pci_plug(struct rte_device *dev)
> {
> return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
> @@ -503,6 +569,7 @@ struct rte_pci_bus rte_pci_bus = {
> .unplug = pci_unplug,
> .parse = pci_parse,
> .get_iommu_class = rte_pci_get_iommu_class,
> + .handle_hot_unplug = pci_handle_hot_unplug,
> },
> .device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
> .driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
> diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
> index 54bc20b..ba2c458 100644
> --- a/drivers/bus/pci/pci_common_uio.c
> +++ b/drivers/bus/pci/pci_common_uio.c
> @@ -146,6 +146,38 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
> }
> }
>
> +/* remap the PCI resource of a PCI device in anonymous virtual memory */
> +int
> +pci_uio_remap_resource(struct rte_pci_device *dev)
> +{
> + int i;
> + void *map_address;
> +
> + if (dev == NULL)
> + return -1;
> +
> + /* Remap all BARs */
> + for (i = 0; i != PCI_MAX_RESOURCE; i++) {
> + /* skip empty BAR */
> + if (dev->mem_resource[i].phys_addr == 0)
> + continue;
> + pci_unmap_resource(dev->mem_resource[i].addr,
> + (size_t)dev->mem_resource[i].len);
> + map_address = pci_map_resource(
> + dev->mem_resource[i].addr, -1, 0,
> + (size_t)dev->mem_resource[i].len,
> + MAP_ANONYMOUS | MAP_FIXED);
Instead of using mumap/mmap() can we use mremap() here?
Might be a bit safer approach.
> + if (map_address == MAP_FAILED) {
> + RTE_LOG(ERR, EAL,
> + "Cannot remap resource for device %s\n",
> + dev->name);
> + return -1;
> + }
> + }
> +
> + return 0;
> +}
> +
> static struct mapped_pci_resource *
> pci_uio_find_resource(struct rte_pci_device *dev)
> {
> diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
> index 88fa587..cc1668c 100644
> --- a/drivers/bus/pci/private.h
> +++ b/drivers/bus/pci/private.h
> @@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
> struct mapped_pci_resource *uio_res);
>
> /**
> + * remap the pci uio resource.
> + *
> + * @param dev
> + * Point to the struct rte pci device.
> + * @return
> + * - On success, zero.
> + * - On failure, a negative value.
> + */
> +int
> +pci_uio_remap_resource(struct rte_pci_device *dev);
> +
> +/**
> * Map device memory to uio resource
> *
> * This function is private to EAL.
> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
> index 6fb0834..d2c5778 100644
> --- a/lib/librte_eal/common/include/rte_bus.h
> +++ b/lib/librte_eal/common/include/rte_bus.h
> @@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
> typedef int (*rte_bus_parse_t)(const char *name, void *addr);
>
> /**
> + * Implementation specific hot unplug handler function which is responsible
> + * for handle the failure when hot unplug the device, guaranty the system
> + * would not crash in the case.
> + * @param dev
> + * Pointer of the device structure.
> + *
> + * @return
> + * 0 on success.
> + * !0 on error.
> + */
> +typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
> + void *dev_addr);
> +
> +/**
> * Bus scan policies
> */
> enum rte_bus_scan_mode {
> @@ -209,6 +223,8 @@ struct rte_bus {
> rte_bus_plug_t plug; /**< Probe single device for drivers */
> rte_bus_unplug_t unplug; /**< Remove single device from driver */
> rte_bus_parse_t parse; /**< Parse a device name */
> + rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
> + device event */
> struct rte_bus_conf conf; /**< Bus configuration */
> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
> };
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 1/4] bus/pci: introduce device hot unplug handle
2018-04-20 10:32 ` Ananyev, Konstantin
@ 2018-05-03 3:05 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-05-03 3:05 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On 4/20/2018 6:32 PM, Ananyev, Konstantin wrote:
> Hi Jeff,
>
>> As of device hot unplug, we need some preparatory measures so that we will
>> not encounter memory fault after device be plug out of the system,
>> and also let we could recover the running data path but not been break.
>> This patch allows the buses to handle device hot unplug event.
>> The patch only enable the ops in pci bus, when handle device hot unplug
>> event, remap a dummy memory to avoid bus read/write error.
>> Other buses could accordingly implement this ops specific by themselves.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v20->19:
>> clean the code
>> ---
>> drivers/bus/pci/pci_common.c | 67 +++++++++++++++++++++++++++++++++
>> drivers/bus/pci/pci_common_uio.c | 32 ++++++++++++++++
>> drivers/bus/pci/private.h | 12 ++++++
>> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
>> 4 files changed, 127 insertions(+)
>>
>> diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
>> index 2a00f36..709eaf3 100644
>> --- a/drivers/bus/pci/pci_common.c
>> +++ b/drivers/bus/pci/pci_common.c
>> @@ -474,6 +474,72 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
>> }
>>
>> static int
>> +pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
>> +{
>> + struct rte_pci_device *pdev = NULL;
>> + int ret = 0, i, isfound = 0;
>> +
>> + if (failure_addr != NULL) {
>> + FOREACH_DEVICE_ON_PCIBUS(pdev) {
>> + for (i = 0; i != sizeof(pdev->mem_resource) /
>> + sizeof(pdev->mem_resource[0]); i++) {
> You can do i != RTE_DIM(pdev->mem_resource) here.
sure.
>> + if ((uint64_t)failure_addr >=
>> + (uint64_t)pdev->mem_resource[i].addr &&
>> + (uint64_t)failure_addr <=
>> + (uint64_t)pdev->mem_resource[i].addr +
>> + pdev->mem_resource[i].len) {
>
> I think it should be failure_addr < addr + len
i think you are right.
>> + RTE_LOG(ERR, EAL, "Failure address "
>> + "%16.16"PRIx64" is belong to "
>> + "resource of device %s!\n",
>> + (uint64_t)failure_addr,
>> + pdev->device.name);
>> + isfound = 1;
>> + break;
>> + }
>> + }
>> + if (isfound)
>> + break;
>
> Might be it is a good thing to put the code that searches for address into a separate function.
good idea.
>> + }
>> + } else if (dev != NULL) {
>> + pdev = RTE_DEV_TO_PCI(dev);
>> + } else {
>> + return -EINVAL;
>> + }
>> +
>> + if (!pdev)
>> + return -1;
>> +
>> + /* remap resources for devices */
>> + switch (pdev->kdrv) {
>> + case RTE_KDRV_VFIO:
>> +#ifdef VFIO_PRESENT
>> + /* TODO */
>> +#endif
> Should set ret =-1 as not implemented now.
ok.
>> + break;
>> + case RTE_KDRV_IGB_UIO:
>> + case RTE_KDRV_UIO_GENERIC:
>> + if (rte_eal_using_phys_addrs()) {
>> + /* map resources for devices that use uio */
>> + ret = pci_uio_remap_resource(pdev);
>> + }
>> + break;
>> + case RTE_KDRV_NIC_UIO:
>> + ret = pci_uio_remap_resource(pdev);
>> + break;
>> + default:
>> + RTE_LOG(DEBUG, EAL,
>> + " Not managed by a supported kernel driver, skipped\n");
>> + ret = -1;
>> + break;
>> + }
>> +
>> + if (ret != 0)
>> + RTE_LOG(ERR, EAL, "failed to handle hot unplug of %s",
>> + pdev->name);
>> + return ret;
>> +}
>> +
>> +static int
>> pci_plug(struct rte_device *dev)
>> {
>> return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
>> @@ -503,6 +569,7 @@ struct rte_pci_bus rte_pci_bus = {
>> .unplug = pci_unplug,
>> .parse = pci_parse,
>> .get_iommu_class = rte_pci_get_iommu_class,
>> + .handle_hot_unplug = pci_handle_hot_unplug,
>> },
>> .device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
>> .driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
>> diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
>> index 54bc20b..ba2c458 100644
>> --- a/drivers/bus/pci/pci_common_uio.c
>> +++ b/drivers/bus/pci/pci_common_uio.c
>> @@ -146,6 +146,38 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
>> }
>> }
>>
>> +/* remap the PCI resource of a PCI device in anonymous virtual memory */
>> +int
>> +pci_uio_remap_resource(struct rte_pci_device *dev)
>> +{
>> + int i;
>> + void *map_address;
>> +
>> + if (dev == NULL)
>> + return -1;
>> +
>> + /* Remap all BARs */
>> + for (i = 0; i != PCI_MAX_RESOURCE; i++) {
>> + /* skip empty BAR */
>> + if (dev->mem_resource[i].phys_addr == 0)
>> + continue;
>> + pci_unmap_resource(dev->mem_resource[i].addr,
>> + (size_t)dev->mem_resource[i].len);
>> + map_address = pci_map_resource(
>> + dev->mem_resource[i].addr, -1, 0,
>> + (size_t)dev->mem_resource[i].len,
>> + MAP_ANONYMOUS | MAP_FIXED);
> Instead of using mumap/mmap() can we use mremap() here?
> Might be a bit safer approach.
because of mremap not have the can not map an anonymous memory, so that
is not fit for this case, and i check and found that MAP_FIXED could
overlap the part of the existing mapping, no need to use unmap at first
before remap.
>> + if (map_address == MAP_FAILED) {
>> + RTE_LOG(ERR, EAL,
>> + "Cannot remap resource for device %s\n",
>> + dev->name);
>> + return -1;
>> + }
>> + }
>> +
>> + return 0;
>> +}
>> +
>> static struct mapped_pci_resource *
>> pci_uio_find_resource(struct rte_pci_device *dev)
>> {
>> diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
>> index 88fa587..cc1668c 100644
>> --- a/drivers/bus/pci/private.h
>> +++ b/drivers/bus/pci/private.h
>> @@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
>> struct mapped_pci_resource *uio_res);
>>
>> /**
>> + * remap the pci uio resource.
>> + *
>> + * @param dev
>> + * Point to the struct rte pci device.
>> + * @return
>> + * - On success, zero.
>> + * - On failure, a negative value.
>> + */
>> +int
>> +pci_uio_remap_resource(struct rte_pci_device *dev);
>> +
>> +/**
>> * Map device memory to uio resource
>> *
>> * This function is private to EAL.
>> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
>> index 6fb0834..d2c5778 100644
>> --- a/lib/librte_eal/common/include/rte_bus.h
>> +++ b/lib/librte_eal/common/include/rte_bus.h
>> @@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
>> typedef int (*rte_bus_parse_t)(const char *name, void *addr);
>>
>> /**
>> + * Implementation specific hot unplug handler function which is responsible
>> + * for handle the failure when hot unplug the device, guaranty the system
>> + * would not crash in the case.
>> + * @param dev
>> + * Pointer of the device structure.
>> + *
>> + * @return
>> + * 0 on success.
>> + * !0 on error.
>> + */
>> +typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
>> + void *dev_addr);
>> +
>> +/**
>> * Bus scan policies
>> */
>> enum rte_bus_scan_mode {
>> @@ -209,6 +223,8 @@ struct rte_bus {
>> rte_bus_plug_t plug; /**< Probe single device for drivers */
>> rte_bus_unplug_t unplug; /**< Remove single device from driver */
>> rte_bus_parse_t parse; /**< Parse a device name */
>> + rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
>> + device event */
>> struct rte_bus_conf conf; /**< Bus configuration */
>> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
>> };
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
2018-04-18 13:38 ` [PATCH V20 1/4] bus/pci: introduce device hot unplug handle Jeff Guo
@ 2018-04-18 13:38 ` Jeff Guo
2018-04-19 1:30 ` Zhang, Qi Z
` (2 more replies)
2018-04-18 13:38 ` [PATCH V20 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-04-18 13:38 ` [PATCH V20 4/4] app/testpmd: show example to handler " Jeff Guo
3 siblings, 3 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-18 13:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot unplug event. When device be hot plug out, the device resource
become invalid, if this resource is still be unexpected read/write,
system will crash. This patch let eal help application to handle
this fault, when sigbus error occur, check the failure address and
accordingly remap the invalid memory for the corresponding device,
that could guaranty the application not to be shut down when hot plug.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v20->v19:
refine the logic of remapping for multiple device.
---
doc/guides/rel_notes/release_18_05.rst | 6 ++
lib/librte_eal/common/include/rte_dev.h | 11 +++
lib/librte_eal/linuxapp/eal/eal_dev.c | 124 +++++++++++++++++++++++++++++++-
lib/librte_eal/rte_eal_version.map | 1 +
4 files changed, 141 insertions(+), 1 deletion(-)
diff --git a/doc/guides/rel_notes/release_18_05.rst b/doc/guides/rel_notes/release_18_05.rst
index a018ef5..a4ea9af 100644
--- a/doc/guides/rel_notes/release_18_05.rst
+++ b/doc/guides/rel_notes/release_18_05.rst
@@ -70,6 +70,12 @@ New Features
Linux uevent is supported as backend of this device event notification framework.
+* **Added hot plug failure handler.**
+
+ Added a failure handler machenism to handle hot unplug device.
+
+ * ``rte_dev_handle_hot_unplug`` for handle hot unplug device failure.
+
API Changes
-----------
diff --git a/lib/librte_eal/common/include/rte_dev.h b/lib/librte_eal/common/include/rte_dev.h
index 0955e9a..9933131 100644
--- a/lib/librte_eal/common/include/rte_dev.h
+++ b/lib/librte_eal/common/include/rte_dev.h
@@ -360,4 +360,15 @@ rte_dev_event_monitor_start(void);
int __rte_experimental
rte_dev_event_monitor_stop(void);
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice
+ *
+ * It can be used to handle the device signal bus error. when signal bus error
+ * occur, the handler would check the failure address to find the corresponding
+ * device and remap the memory resource of the device, that would guaranty
+ * the system not crash when the device be hot unplug.
+ */
+void __rte_experimental
+rte_dev_handle_hot_unplug(void);
#endif /* _RTE_DEV_H_ */
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 9478a39..33e7026 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -13,12 +15,16 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
@@ -33,6 +39,68 @@ enum eal_dev_event_subsystem {
};
static int
+dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
+{
+ struct rte_bus *bus;
+ int ret = 0;
+
+ if (!dev && !dev_addr) {
+ return -EINVAL;
+ } else if (dev) {
+ bus = rte_bus_find_by_device_name(dev->name);
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret) {
+ RTE_LOG(ERR, EAL,
+ "It cannot handle hot unplug "
+ "for device (%s) "
+ "on the bus.\n ",
+ dev->name);
+ }
+ }
+ } else {
+ TAILQ_FOREACH(bus, &rte_bus_list, next) {
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret) {
+ RTE_LOG(ERR, EAL,
+ "It cannot handle hot unplug "
+ "for the device "
+ "on the bus.\n ");
+ }
+ }
+ }
+ }
+ return ret;
+}
+
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
+ ret = dev_uev_failure_process(NULL, info->si_addr);
+ if (!ret)
+ RTE_LOG(DEBUG, EAL,
+ "SIGBUS error is because of hot unplug!\n");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
+static int
dev_uev_socket_fd_create(void)
{
struct sockaddr_nl addr;
@@ -146,6 +214,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -170,8 +241,41 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ uevent.devname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL,
+ "Cannot find unplugged device (%s)\n",
+ uevent.devname);
+ return;
+ }
+ ret = dev_uev_failure_process(dev, NULL);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Driver cannot remap the "
+ "device (%s)\n",
+ dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
@@ -216,8 +320,26 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ /* recover sigbus. */
+ sigaction(SIGBUS, NULL, NULL);
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
+
+void __rte_experimental
+rte_dev_handle_hot_unplug(void)
+{
+ struct sigaction act;
+
+ /* set sigbus handler for hotplug. */
+ memset(&act, 0x00, sizeof(struct sigaction));
+ act.sa_sigaction = sigbus_handler;
+ sigemptyset(&act.sa_mask);
+ sigaddset(&act.sa_mask, SIGBUS);
+ act.sa_flags = SA_SIGINFO;
+ sigaction(SIGBUS, &act, NULL);
+}
diff --git a/lib/librte_eal/rte_eal_version.map b/lib/librte_eal/rte_eal_version.map
index d02d80b..39a0213 100644
--- a/lib/librte_eal/rte_eal_version.map
+++ b/lib/librte_eal/rte_eal_version.map
@@ -217,6 +217,7 @@ EXPERIMENTAL {
rte_dev_event_callback_unregister;
rte_dev_event_monitor_start;
rte_dev_event_monitor_stop;
+ rte_dev_handle_hot_unplug;
rte_eal_cleanup;
rte_eal_devargs_insert;
rte_eal_devargs_parse;
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-18 13:38 ` [PATCH V20 2/4] eal: add failure handler mechanism for hot plug Jeff Guo
@ 2018-04-19 1:30 ` Zhang, Qi Z
2018-04-20 11:14 ` Ananyev, Konstantin
2018-04-20 16:16 ` Ananyev, Konstantin
2 siblings, 0 replies; 494+ messages in thread
From: Zhang, Qi Z @ 2018-04-19 1:30 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Guo, Jia, Zhang, Helin
Hi Jeff
> -----Original Message-----
> From: dev [mailto:dev-bounces@dpdk.org] On Behalf Of Jeff Guo
> Sent: Wednesday, April 18, 2018 9:38 PM
> To: stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>;
> Ananyev, Konstantin <konstantin.ananyev@intel.com>;
> gaetan.rivet@6wind.com; Wu, Jingjing <jingjing.wu@intel.com>;
> thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van
> Haaren, Harry <harry.van.haaren@intel.com>; Tan, Jianfeng
> <jianfeng.tan@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Guo,
> Jia <jia.guo@intel.com>; Zhang, Helin <helin.zhang@intel.com>
> Subject: [dpdk-dev] [PATCH V20 2/4] eal: add failure handler mechanism for
> hot plug
>
> This patch introduces a failure handler mechanism to handle device hot
> unplug event. When device be hot plug out, the device resource become
> invalid, if this resource is still be unexpected read/write, system will crash.
> This patch let eal help application to handle this fault, when sigbus error
> occur, check the failure address and accordingly remap the invalid memory
> for the corresponding device, that could guaranty the application not to be
> shut down when hot plug.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v20->v19:
> refine the logic of remapping for multiple device.
> ---
> doc/guides/rel_notes/release_18_05.rst | 6 ++
> lib/librte_eal/common/include/rte_dev.h | 11 +++
> lib/librte_eal/linuxapp/eal/eal_dev.c | 124
> +++++++++++++++++++++++++++++++-
> lib/librte_eal/rte_eal_version.map | 1 +
> 4 files changed, 141 insertions(+), 1 deletion(-)
>
> diff --git a/doc/guides/rel_notes/release_18_05.rst
> b/doc/guides/rel_notes/release_18_05.rst
> index a018ef5..a4ea9af 100644
> --- a/doc/guides/rel_notes/release_18_05.rst
> +++ b/doc/guides/rel_notes/release_18_05.rst
> @@ -70,6 +70,12 @@ New Features
>
> Linux uevent is supported as backend of this device event notification
> framework.
>
> +* **Added hot plug failure handler.**
> +
> + Added a failure handler machenism to handle hot unplug device.
> +
> + * ``rte_dev_handle_hot_unplug`` for handle hot unplug device failure.
> +
>
> API Changes
> -----------
> diff --git a/lib/librte_eal/common/include/rte_dev.h
> b/lib/librte_eal/common/include/rte_dev.h
> index 0955e9a..9933131 100644
> --- a/lib/librte_eal/common/include/rte_dev.h
> +++ b/lib/librte_eal/common/include/rte_dev.h
> @@ -360,4 +360,15 @@ rte_dev_event_monitor_start(void);
> int __rte_experimental
> rte_dev_event_monitor_stop(void);
>
> +/**
> + * @warning
> + * @b EXPERIMENTAL: this API may change without prior notice
> + *
> + * It can be used to handle the device signal bus error. when signal
> +bus error
> + * occur, the handler would check the failure address to find the
> +corresponding
> + * device and remap the memory resource of the device, that would
> +guaranty
> + * the system not crash when the device be hot unplug.
> + */
> +void __rte_experimental
> +rte_dev_handle_hot_unplug(void);
> #endif /* _RTE_DEV_H_ */
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c
> b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 9478a39..33e7026 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -13,12 +15,16 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 }; static bool
> monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
> #define EAL_UEV_MSG_LEN 4096
> #define EAL_UEV_MSG_ELEM_LEN 128
>
> @@ -33,6 +39,68 @@ enum eal_dev_event_subsystem { };
>
> static int
> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr) {
> + struct rte_bus *bus;
> + int ret = 0;
> +
> + if (!dev && !dev_addr) {
> + return -EINVAL;
> + } else if (dev) {
> + bus = rte_bus_find_by_device_name(dev->name);
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret) {
> + RTE_LOG(ERR, EAL,
> + "It cannot handle hot unplug "
> + "for device (%s) "
> + "on the bus.\n ",
> + dev->name);
> + }
> + }
> + } else {
> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret) {
> + RTE_LOG(ERR, EAL,
> + "It cannot handle hot unplug "
> + "for the device "
> + "on the bus.\n ");
> + }
> + }
> + }
> + }
> + return ret;
> +}
> +
> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> + void *ctx __rte_unused)
> +{
> + int ret;
> +
> + RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
> + ret = dev_uev_failure_process(NULL, info->si_addr);
> + if (!ret)
> + RTE_LOG(DEBUG, EAL,
> + "SIGBUS error is because of hot unplug!\n"); }
> +
> +static int cmp_dev_name(const struct rte_device *dev,
> + const void *_name)
> +{
> + const char *name = _name;
> +
> + return strcmp(dev->name, name);
> +}
> +
> +static int
> dev_uev_socket_fd_create(void)
> {
> struct sockaddr_nl addr;
> @@ -146,6 +214,9 @@ dev_uev_handler(__rte_unused void *param)
> struct rte_dev_event uevent;
> int ret;
> char buf[EAL_UEV_MSG_LEN];
> + struct rte_bus *bus;
> + struct rte_device *dev;
> + const char *busname;
>
> memset(&uevent, 0, sizeof(struct rte_dev_event));
> memset(buf, 0, EAL_UEV_MSG_LEN);
> @@ -170,8 +241,41 @@ dev_uev_handler(__rte_unused void *param)
> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d,
> subsystem:%d)\n",
> uevent.devname, uevent.type, uevent.subsystem);
>
> - if (uevent.devname)
> + switch (uevent.subsystem) {
> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> + busname = "pci";
> + break;
> + default:
> + break;
> + }
> +
> + if (uevent.devname) {
> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> + bus = rte_bus_find_by_name(busname);
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> + uevent.devname);
> + return;
> + }
> + dev = bus->find_device(NULL, cmp_dev_name,
> + uevent.devname);
> + if (dev == NULL) {
> + RTE_LOG(ERR, EAL,
> + "Cannot find unplugged device (%s)\n",
> + uevent.devname);
> + return;
> + }
> + ret = dev_uev_failure_process(dev, NULL);
> + if (ret) {
> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
> + "device (%s)\n",
> + dev->name);
> + return;
> + }
> + }
> dev_callback_process(uevent.devname, uevent.type);
> + }
> }
>
> int __rte_experimental
> @@ -216,8 +320,26 @@ rte_dev_event_monitor_stop(void)
> return ret;
> }
>
> + /* recover sigbus. */
> + sigaction(SIGBUS, NULL, NULL);
> +
> close(intr_handle.fd);
> intr_handle.fd = -1;
> monitor_started = false;
> +
> return 0;
> }
> +
> +void __rte_experimental
> +rte_dev_handle_hot_unplug(void)
> +{
> + struct sigaction act;
> +
> + /* set sigbus handler for hotplug. */
> + memset(&act, 0x00, sizeof(struct sigaction));
> + act.sa_sigaction = sigbus_handler;
> + sigemptyset(&act.sa_mask);
> + sigaddset(&act.sa_mask, SIGBUS);
> + act.sa_flags = SA_SIGINFO;
> + sigaction(SIGBUS, &act, NULL);
> +}
Not sure if it's necessary to expose this API,
it can be invoked in rte_dev_event_monitor_start,
since register a sigbus handler looks like an init step when user device to enable hotplug
Regards
Qi
> diff --git a/lib/librte_eal/rte_eal_version.map
> b/lib/librte_eal/rte_eal_version.map
> index d02d80b..39a0213 100644
> --- a/lib/librte_eal/rte_eal_version.map
> +++ b/lib/librte_eal/rte_eal_version.map
> @@ -217,6 +217,7 @@ EXPERIMENTAL {
> rte_dev_event_callback_unregister;
> rte_dev_event_monitor_start;
> rte_dev_event_monitor_stop;
> + rte_dev_handle_hot_unplug;
> rte_eal_cleanup;
> rte_eal_devargs_insert;
> rte_eal_devargs_parse;
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-18 13:38 ` [PATCH V20 2/4] eal: add failure handler mechanism for hot plug Jeff Guo
2018-04-19 1:30 ` Zhang, Qi Z
@ 2018-04-20 11:14 ` Ananyev, Konstantin
2018-05-03 3:13 ` Guo, Jia
2018-04-20 16:16 ` Ananyev, Konstantin
2 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-04-20 11:14 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> This patch introduces a failure handler mechanism to handle device
> hot unplug event. When device be hot plug out, the device resource
> become invalid, if this resource is still be unexpected read/write,
> system will crash. This patch let eal help application to handle
> this fault, when sigbus error occur, check the failure address and
> accordingly remap the invalid memory for the corresponding device,
> that could guaranty the application not to be shut down when hot plug.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v20->v19:
> refine the logic of remapping for multiple device.
> ---
> doc/guides/rel_notes/release_18_05.rst | 6 ++
> lib/librte_eal/common/include/rte_dev.h | 11 +++
> lib/librte_eal/linuxapp/eal/eal_dev.c | 124 +++++++++++++++++++++++++++++++-
> lib/librte_eal/rte_eal_version.map | 1 +
> 4 files changed, 141 insertions(+), 1 deletion(-)
>
> diff --git a/doc/guides/rel_notes/release_18_05.rst b/doc/guides/rel_notes/release_18_05.rst
> index a018ef5..a4ea9af 100644
> --- a/doc/guides/rel_notes/release_18_05.rst
> +++ b/doc/guides/rel_notes/release_18_05.rst
> @@ -70,6 +70,12 @@ New Features
>
> Linux uevent is supported as backend of this device event notification framework.
>
> +* **Added hot plug failure handler.**
> +
> + Added a failure handler machenism to handle hot unplug device.
> +
> + * ``rte_dev_handle_hot_unplug`` for handle hot unplug device failure.
> +
>
> API Changes
> -----------
> diff --git a/lib/librte_eal/common/include/rte_dev.h b/lib/librte_eal/common/include/rte_dev.h
> index 0955e9a..9933131 100644
> --- a/lib/librte_eal/common/include/rte_dev.h
> +++ b/lib/librte_eal/common/include/rte_dev.h
> @@ -360,4 +360,15 @@ rte_dev_event_monitor_start(void);
> int __rte_experimental
> rte_dev_event_monitor_stop(void);
>
> +/**
> + * @warning
> + * @b EXPERIMENTAL: this API may change without prior notice
> + *
> + * It can be used to handle the device signal bus error. when signal bus error
> + * occur, the handler would check the failure address to find the corresponding
> + * device and remap the memory resource of the device, that would guaranty
> + * the system not crash when the device be hot unplug.
> + */
> +void __rte_experimental
> +rte_dev_handle_hot_unplug(void);
> #endif /* _RTE_DEV_H_ */
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 9478a39..33e7026 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -13,12 +15,16 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 };
> static bool monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
> #define EAL_UEV_MSG_LEN 4096
> #define EAL_UEV_MSG_ELEM_LEN 128
>
> @@ -33,6 +39,68 @@ enum eal_dev_event_subsystem {
> };
>
> static int
> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
> +{
> + struct rte_bus *bus;
> + int ret = 0;
> +
> + if (!dev && !dev_addr) {
> + return -EINVAL;
> + } else if (dev) {
> + bus = rte_bus_find_by_device_name(dev->name);
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret) {
> + RTE_LOG(ERR, EAL,
> + "It cannot handle hot unplug "
> + "for device (%s) "
> + "on the bus.\n ",
> + dev->name);
> + }
> + }
You would retrun 0 if bus->handle_hot_unplug == NULL.
Is that intended?
Shouldn't be I think.
> + } else {
> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret) {
> + RTE_LOG(ERR, EAL,
> + "It cannot handle hot unplug "
> + "for the device "
> + "on the bus.\n ");
> + }
So how we would know what happened here:
That address doesn't belong to that bus or unplug_handler failed?
Should we separate search and unplug ops?
Another question - shouldn't we break out of loop if bus->handle_hot_unplug()
returns 0?
Otherwise you can return error value even when unplug handled worked correctly.
> + }
> + }
> + }
> + return ret;
> +}
> +
> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> + void *ctx __rte_unused)
> +{
> + int ret;
> +
> + RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
> + ret = dev_uev_failure_process(NULL, info->si_addr);
As now you can try to mmap/munmap same address from two or more different threads
you probably need some synchronization here.
Something simple as spinlock seems to be enough here.
We might have one per device or might be even a global one would be ok here.
> + if (!ret)
> + RTE_LOG(DEBUG, EAL,
> + "SIGBUS error is because of hot unplug!\n");
> +}
> +
> +static int cmp_dev_name(const struct rte_device *dev,
> + const void *_name)
> +{
> + const char *name = _name;
> +
> + return strcmp(dev->name, name);
> +}
Is it really worth a separate function?
> +
> +static int
> dev_uev_socket_fd_create(void)
> {
> struct sockaddr_nl addr;
> @@ -146,6 +214,9 @@ dev_uev_handler(__rte_unused void *param)
> struct rte_dev_event uevent;
> int ret;
> char buf[EAL_UEV_MSG_LEN];
> + struct rte_bus *bus;
> + struct rte_device *dev;
> + const char *busname;
>
> memset(&uevent, 0, sizeof(struct rte_dev_event));
> memset(buf, 0, EAL_UEV_MSG_LEN);
> @@ -170,8 +241,41 @@ dev_uev_handler(__rte_unused void *param)
> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> uevent.devname, uevent.type, uevent.subsystem);
>
> - if (uevent.devname)
> + switch (uevent.subsystem) {
> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> + busname = "pci";
> + break;
> + default:
> + break;
> + }
> +
> + if (uevent.devname) {
> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> + bus = rte_bus_find_by_name(busname);
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> + uevent.devname);
> + return;
> + }
> + dev = bus->find_device(NULL, cmp_dev_name,
> + uevent.devname);
> + if (dev == NULL) {
> + RTE_LOG(ERR, EAL,
> + "Cannot find unplugged device (%s)\n",
> + uevent.devname);
> + return;
> + }
> + ret = dev_uev_failure_process(dev, NULL);
> + if (ret) {
> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
> + "device (%s)\n",
> + dev->name);
> + return;
> + }
> + }
> dev_callback_process(uevent.devname, uevent.type);
> + }
> }
>
> int __rte_experimental
> @@ -216,8 +320,26 @@ rte_dev_event_monitor_stop(void)
> return ret;
> }
>
> + /* recover sigbus. */
> + sigaction(SIGBUS, NULL, NULL);
> +
Probably better to restore previous action.
> close(intr_handle.fd);
> intr_handle.fd = -1;
> monitor_started = false;
> +
> return 0;
> }
> +
> +void __rte_experimental
> +rte_dev_handle_hot_unplug(void)
> +{
> + struct sigaction act;
> +
> + /* set sigbus handler for hotplug. */
> + memset(&act, 0x00, sizeof(struct sigaction));
> + act.sa_sigaction = sigbus_handler;
> + sigemptyset(&act.sa_mask);
> + sigaddset(&act.sa_mask, SIGBUS);
> + act.sa_flags = SA_SIGINFO;
> + sigaction(SIGBUS, &act, NULL);
> +}
> diff --git a/lib/librte_eal/rte_eal_version.map b/lib/librte_eal/rte_eal_version.map
> index d02d80b..39a0213 100644
> --- a/lib/librte_eal/rte_eal_version.map
> +++ b/lib/librte_eal/rte_eal_version.map
> @@ -217,6 +217,7 @@ EXPERIMENTAL {
> rte_dev_event_callback_unregister;
> rte_dev_event_monitor_start;
> rte_dev_event_monitor_stop;
> + rte_dev_handle_hot_unplug;
> rte_eal_cleanup;
> rte_eal_devargs_insert;
> rte_eal_devargs_parse;
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-20 11:14 ` Ananyev, Konstantin
@ 2018-05-03 3:13 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-05-03 3:13 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On 4/20/2018 7:14 PM, Ananyev, Konstantin wrote:
>
>> This patch introduces a failure handler mechanism to handle device
>> hot unplug event. When device be hot plug out, the device resource
>> become invalid, if this resource is still be unexpected read/write,
>> system will crash. This patch let eal help application to handle
>> this fault, when sigbus error occur, check the failure address and
>> accordingly remap the invalid memory for the corresponding device,
>> that could guaranty the application not to be shut down when hot plug.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v20->v19:
>> refine the logic of remapping for multiple device.
>> ---
>> doc/guides/rel_notes/release_18_05.rst | 6 ++
>> lib/librte_eal/common/include/rte_dev.h | 11 +++
>> lib/librte_eal/linuxapp/eal/eal_dev.c | 124 +++++++++++++++++++++++++++++++-
>> lib/librte_eal/rte_eal_version.map | 1 +
>> 4 files changed, 141 insertions(+), 1 deletion(-)
>>
>> diff --git a/doc/guides/rel_notes/release_18_05.rst b/doc/guides/rel_notes/release_18_05.rst
>> index a018ef5..a4ea9af 100644
>> --- a/doc/guides/rel_notes/release_18_05.rst
>> +++ b/doc/guides/rel_notes/release_18_05.rst
>> @@ -70,6 +70,12 @@ New Features
>>
>> Linux uevent is supported as backend of this device event notification framework.
>>
>> +* **Added hot plug failure handler.**
>> +
>> + Added a failure handler machenism to handle hot unplug device.
>> +
>> + * ``rte_dev_handle_hot_unplug`` for handle hot unplug device failure.
>> +
>>
>> API Changes
>> -----------
>> diff --git a/lib/librte_eal/common/include/rte_dev.h b/lib/librte_eal/common/include/rte_dev.h
>> index 0955e9a..9933131 100644
>> --- a/lib/librte_eal/common/include/rte_dev.h
>> +++ b/lib/librte_eal/common/include/rte_dev.h
>> @@ -360,4 +360,15 @@ rte_dev_event_monitor_start(void);
>> int __rte_experimental
>> rte_dev_event_monitor_stop(void);
>>
>> +/**
>> + * @warning
>> + * @b EXPERIMENTAL: this API may change without prior notice
>> + *
>> + * It can be used to handle the device signal bus error. when signal bus error
>> + * occur, the handler would check the failure address to find the corresponding
>> + * device and remap the memory resource of the device, that would guaranty
>> + * the system not crash when the device be hot unplug.
>> + */
>> +void __rte_experimental
>> +rte_dev_handle_hot_unplug(void);
>> #endif /* _RTE_DEV_H_ */
>> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> index 9478a39..33e7026 100644
>> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> @@ -4,6 +4,8 @@
>>
>> #include <string.h>
>> #include <unistd.h>
>> +#include <fcntl.h>
>> +#include <signal.h>
>> #include <sys/socket.h>
>> #include <linux/netlink.h>
>>
>> @@ -13,12 +15,16 @@
>> #include <rte_malloc.h>
>> #include <rte_interrupts.h>
>> #include <rte_alarm.h>
>> +#include <rte_bus.h>
>> +#include <rte_eal.h>
>>
>> #include "eal_private.h"
>>
>> static struct rte_intr_handle intr_handle = {.fd = -1 };
>> static bool monitor_started;
>>
>> +extern struct rte_bus_list rte_bus_list;
>> +
>> #define EAL_UEV_MSG_LEN 4096
>> #define EAL_UEV_MSG_ELEM_LEN 128
>>
>> @@ -33,6 +39,68 @@ enum eal_dev_event_subsystem {
>> };
>>
>> static int
>> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
>> +{
>> + struct rte_bus *bus;
>> + int ret = 0;
>> +
>> + if (!dev && !dev_addr) {
>> + return -EINVAL;
>> + } else if (dev) {
>> + bus = rte_bus_find_by_device_name(dev->name);
>> + if (bus->handle_hot_unplug) {
>> + /**
>> + * call bus ops to handle hot unplug.
>> + */
>> + ret = bus->handle_hot_unplug(dev, dev_addr);
>> + if (ret) {
>> + RTE_LOG(ERR, EAL,
>> + "It cannot handle hot unplug "
>> + "for device (%s) "
>> + "on the bus.\n ",
>> + dev->name);
>> + }
>> + }
>
> You would retrun 0 if bus->handle_hot_unplug == NULL.
> Is that intended?
> Shouldn't be I think.
shouldn't be. will modify it.
>> + } else {
>> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
>> + if (bus->handle_hot_unplug) {
>> + /**
>> + * call bus ops to handle hot unplug.
>> + */
>> + ret = bus->handle_hot_unplug(dev, dev_addr);
>> + if (ret) {
>> + RTE_LOG(ERR, EAL,
>> + "It cannot handle hot unplug "
>> + "for the device "
>> + "on the bus.\n ");
>> + }
> So how we would know what happened here:
> That address doesn't belong to that bus or unplug_handler failed?
> Should we separate search and unplug ops?
> Another question - shouldn't we break out of loop if bus->handle_hot_unplug()
> returns 0?
> Otherwise you can return error value even when unplug handled worked correctly.
>
you are right here.
>> + }
>> + }
>> + }
>> + return ret;
>> +}
>> +
>> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
>> + void *ctx __rte_unused)
>> +{
>> + int ret;
>> +
>> + RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
>> + ret = dev_uev_failure_process(NULL, info->si_addr);
> As now you can try to mmap/munmap same address from two or more different threads
> you probably need some synchronization here.
> Something simple as spinlock seems to be enough here.
> We might have one per device or might be even a global one would be ok here.
i think global one and synchronization would be fine.
>> + if (!ret)
>> + RTE_LOG(DEBUG, EAL,
>> + "SIGBUS error is because of hot unplug!\n");
>> +}
>> +
>> +static int cmp_dev_name(const struct rte_device *dev,
>> + const void *_name)
>> +{
>> + const char *name = _name;
>> +
>> + return strcmp(dev->name, name);
>> +}
> Is it really worth a separate function?
i think that would be the bus ops struct of rte_bus_find_device_t usage
here.
>> +
>> +static int
>> dev_uev_socket_fd_create(void)
>> {
>> struct sockaddr_nl addr;
>> @@ -146,6 +214,9 @@ dev_uev_handler(__rte_unused void *param)
>> struct rte_dev_event uevent;
>> int ret;
>> char buf[EAL_UEV_MSG_LEN];
>> + struct rte_bus *bus;
>> + struct rte_device *dev;
>> + const char *busname;
>>
>> memset(&uevent, 0, sizeof(struct rte_dev_event));
>> memset(buf, 0, EAL_UEV_MSG_LEN);
>> @@ -170,8 +241,41 @@ dev_uev_handler(__rte_unused void *param)
>> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
>> uevent.devname, uevent.type, uevent.subsystem);
>>
>> - if (uevent.devname)
>> + switch (uevent.subsystem) {
>> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
>> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
>> + busname = "pci";
>> + break;
>> + default:
>> + break;
>> + }
>> +
>> + if (uevent.devname) {
>> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
>> + bus = rte_bus_find_by_name(busname);
>> + if (bus == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
>> + uevent.devname);
>> + return;
>> + }
>> + dev = bus->find_device(NULL, cmp_dev_name,
>> + uevent.devname);
>> + if (dev == NULL) {
>> + RTE_LOG(ERR, EAL,
>> + "Cannot find unplugged device (%s)\n",
>> + uevent.devname);
>> + return;
>> + }
>> + ret = dev_uev_failure_process(dev, NULL);
>> + if (ret) {
>> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
>> + "device (%s)\n",
>> + dev->name);
>> + return;
>> + }
>> + }
>> dev_callback_process(uevent.devname, uevent.type);
>> + }
>> }
>>
>> int __rte_experimental
>> @@ -216,8 +320,26 @@ rte_dev_event_monitor_stop(void)
>> return ret;
>> }
>>
>> + /* recover sigbus. */
>> + sigaction(SIGBUS, NULL, NULL);
>> +
> Probably better to restore previous action.
correct, restore the previous sigbus action so that no affect other.
>> close(intr_handle.fd);
>> intr_handle.fd = -1;
>> monitor_started = false;
>> +
>> return 0;
>> }
>> +
>> +void __rte_experimental
>> +rte_dev_handle_hot_unplug(void)
>> +{
>> + struct sigaction act;
>> +
>> + /* set sigbus handler for hotplug. */
>> + memset(&act, 0x00, sizeof(struct sigaction));
>> + act.sa_sigaction = sigbus_handler;
>> + sigemptyset(&act.sa_mask);
>> + sigaddset(&act.sa_mask, SIGBUS);
>> + act.sa_flags = SA_SIGINFO;
>> + sigaction(SIGBUS, &act, NULL);
>> +}
>> diff --git a/lib/librte_eal/rte_eal_version.map b/lib/librte_eal/rte_eal_version.map
>> index d02d80b..39a0213 100644
>> --- a/lib/librte_eal/rte_eal_version.map
>> +++ b/lib/librte_eal/rte_eal_version.map
>> @@ -217,6 +217,7 @@ EXPERIMENTAL {
>> rte_dev_event_callback_unregister;
>> rte_dev_event_monitor_start;
>> rte_dev_event_monitor_stop;
>> + rte_dev_handle_hot_unplug;
>> rte_eal_cleanup;
>> rte_eal_devargs_insert;
>> rte_eal_devargs_parse;
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-18 13:38 ` [PATCH V20 2/4] eal: add failure handler mechanism for hot plug Jeff Guo
2018-04-19 1:30 ` Zhang, Qi Z
2018-04-20 11:14 ` Ananyev, Konstantin
@ 2018-04-20 16:16 ` Ananyev, Konstantin
2018-05-03 3:17 ` Guo, Jia
2 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-04-20 16:16 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> > +
> > +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> > + void *ctx __rte_unused)
> > +{
> > + int ret;
> > +
> > + RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
> > + ret = dev_uev_failure_process(NULL, info->si_addr);
>
> As now you can try to mmap/munmap same address from two or more different threads
> you probably need some synchronization here.
> Something simple as spinlock seems to be enough here.
> We might have one per device or might be even a global one would be ok here.
>
> > + if (!ret)
> > + RTE_LOG(DEBUG, EAL,
> > + "SIGBUS error is because of hot unplug!\n");
Also if sigbus handler wasn't able to fix things - failure addr doesn't belong to
any devices, or remaping fails - we probably should invoke previously installed handler
or just apply default action.
Konstantin
> > +}
> > +
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 2/4] eal: add failure handler mechanism for hot plug
2018-04-20 16:16 ` Ananyev, Konstantin
@ 2018-05-03 3:17 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-05-03 3:17 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On 4/21/2018 12:16 AM, Ananyev, Konstantin wrote:
>>> +
>>> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
>>> + void *ctx __rte_unused)
>>> +{
>>> + int ret;
>>> +
>>> + RTE_LOG(ERR, EAL, "SIGBUS error, fault address:%p\n", info->si_addr);
>>> + ret = dev_uev_failure_process(NULL, info->si_addr);
>> As now you can try to mmap/munmap same address from two or more different threads
>> you probably need some synchronization here.
>> Something simple as spinlock seems to be enough here.
>> We might have one per device or might be even a global one would be ok here.
>>
>>> + if (!ret)
>>> + RTE_LOG(DEBUG, EAL,
>>> + "SIGBUS error is because of hot unplug!\n");
> Also if sigbus handler wasn't able to fix things - failure addr doesn't belong to
> any devices, or remaping fails - we probably should invoke previously installed handler
> or just apply default action.
> Konstantin
i think just exception here by exit for apply default action, and info
that is a normal sigbus error should be ok.
>>> +}
>>> +
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V20 3/4] igb_uio: fix uio release issue when hot unplug
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
2018-04-18 13:38 ` [PATCH V20 1/4] bus/pci: introduce device hot unplug handle Jeff Guo
2018-04-18 13:38 ` [PATCH V20 2/4] eal: add failure handler mechanism for hot plug Jeff Guo
@ 2018-04-18 13:38 ` Jeff Guo
2018-04-18 13:38 ` [PATCH V20 4/4] app/testpmd: show example to handler " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-04-18 13:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
when device being hot unplug, release a none exist uio resource will
result kernel null pointer error, so this patch will check if device
has been remove before release uio release procedure, if so just return
back.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v20->v19:
split patch independently.
---
kernel/linux/igb_uio/igb_uio.c | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index cbc5ab6..c296332 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -344,6 +344,10 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ /* check if device has been remove before release */
+ if ((&dev->dev.kobj)->state_remove_uevent_sent == 1)
+ return -1;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V20 4/4] app/testpmd: show example to handler hot unplug
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
` (2 preceding siblings ...)
2018-04-18 13:38 ` [PATCH V20 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-04-18 13:38 ` Jeff Guo
2018-05-03 7:25 ` Matan Azrad
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-04-18 13:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. Once app detect the removal event,
the callback would be called, it first stop the packet forwarding, then
stop the port, close the port and finally detach the port.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v20->v19:
remove the auto binding example.
---
app/test-pmd/testpmd.c | 29 +++++++++++++++++++++++++----
1 file changed, 25 insertions(+), 4 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 5986ff7..3751901 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -1125,6 +1125,9 @@ run_pkt_fwd_on_lcore(struct fwd_lcore *fc, packet_fwd_t pkt_fwd)
tics_datum = rte_rdtsc();
tics_per_1sec = rte_get_timer_hz();
#endif
+ if (hot_plug)
+ rte_dev_handle_hot_unplug();
+
fsm = &fwd_streams[fc->stream_idx];
nb_fs = fc->stream_nb;
do {
@@ -2069,6 +2072,26 @@ rmv_event_callback(void *arg)
dev->device->name);
}
+static void
+rmv_dev_event_callback(char *dev_name)
+{
+ uint16_t port_id;
+ int ret;
+
+ ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", dev_name);
+ return;
+ }
+
+ RTE_ETH_VALID_PORTID_OR_RET(port_id);
+ printf("removing port id:%u\n", port_id);
+ stop_packet_forwarding();
+ stop_port(port_id);
+ close_port(port_id);
+ detach_port(port_id);
+}
+
/* This function is used by the interrupt thread */
static int
eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void *param,
@@ -2130,9 +2153,7 @@ eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ rmv_dev_event_callback(device_name);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2640,7 +2661,7 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
+ rte_dev_handle_hot_unplug();
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V20 4/4] app/testpmd: show example to handler hot unplug
2018-04-18 13:38 ` [PATCH V20 4/4] app/testpmd: show example to handler " Jeff Guo
@ 2018-05-03 7:25 ` Matan Azrad
2018-05-03 9:35 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-05-03 7:25 UTC (permalink / raw)
To: Jeff Guo, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
jianfeng.tan@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Jeff
> From: Jeff Guo, Wednesday, April 18, 2018 4:38 PM
> Use testpmd for example, to show how an application smoothly handle
> failure when device being hot unplug. Once app detect the removal event,
> the callback would be called, it first stop the packet forwarding, then stop the
> port, close the port and finally detach the port.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v20->v19:
> remove the auto binding example.
> ---
> app/test-pmd/testpmd.c | 29 +++++++++++++++++++++++++----
> 1 file changed, 25 insertions(+), 4 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> 5986ff7..3751901 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -1125,6 +1125,9 @@ run_pkt_fwd_on_lcore(struct fwd_lcore *fc,
> packet_fwd_t pkt_fwd)
> tics_datum = rte_rdtsc();
> tics_per_1sec = rte_get_timer_hz();
> #endif
> + if (hot_plug)
> + rte_dev_handle_hot_unplug();
> +
Again, I don't understand why the application should configure it - it already started the hot-plug,
Can't the EAL handle this automatically when the user starts the hot-plug?
> fsm = &fwd_streams[fc->stream_idx];
> nb_fs = fc->stream_nb;
> do {
> @@ -2069,6 +2072,26 @@ rmv_event_callback(void *arg)
> dev->device->name);
> }
>
> +static void
> +rmv_dev_event_callback(char *dev_name)
> +{
> + uint16_t port_id;
> + int ret;
> +
> + ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
> + if (ret) {
> + printf("can not get port by device %s!\n", dev_name);
> + return;
> + }
> +
> + RTE_ETH_VALID_PORTID_OR_RET(port_id);
> + printf("removing port id:%u\n", port_id);
> + stop_packet_forwarding();
> + stop_port(port_id);
> + close_port(port_id);
> + detach_port(port_id);
> +}
We have also the rmv_event_callback() which is triggered by a RMV interrupt and running by the host thread.
What is the context thread of rmv_dev_event_callback()?
Shouldn't they be synchronized? Should we need both in the same time?
> +
> /* This function is used by the interrupt thread */ static int
> eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void
> *param, @@ -2130,9 +2153,7 @@ eth_dev_event_callback(char
> *device_name, enum rte_dev_event_type type,
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + rmv_dev_event_callback(device_name);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
> @@ -2640,7 +2661,7 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> + rte_dev_handle_hot_unplug();
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 4/4] app/testpmd: show example to handler hot unplug
2018-05-03 7:25 ` Matan Azrad
@ 2018-05-03 9:35 ` Guo, Jia
2018-05-03 11:27 ` Matan Azrad
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-05-03 9:35 UTC (permalink / raw)
To: Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Thomas Monjalon, Mordechay Haimovsky,
harry.van.haaren@intel.com, jianfeng.tan@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
hi, matan
On 5/3/2018 3:25 PM, Matan Azrad wrote:
> Hi Jeff
>
>> From: Jeff Guo, Wednesday, April 18, 2018 4:38 PM
>> Use testpmd for example, to show how an application smoothly handle
>> failure when device being hot unplug. Once app detect the removal event,
>> the callback would be called, it first stop the packet forwarding, then stop the
>> port, close the port and finally detach the port.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v20->v19:
>> remove the auto binding example.
>> ---
>> app/test-pmd/testpmd.c | 29 +++++++++++++++++++++++++----
>> 1 file changed, 25 insertions(+), 4 deletions(-)
>>
>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>> 5986ff7..3751901 100644
>> --- a/app/test-pmd/testpmd.c
>> +++ b/app/test-pmd/testpmd.c
>> @@ -1125,6 +1125,9 @@ run_pkt_fwd_on_lcore(struct fwd_lcore *fc,
>> packet_fwd_t pkt_fwd)
>> tics_datum = rte_rdtsc();
>> tics_per_1sec = rte_get_timer_hz();
>> #endif
>> + if (hot_plug)
>> + rte_dev_handle_hot_unplug();
>> +
> Again, I don't understand why the application should configure it - it already started the hot-plug,
> Can't the EAL handle this automatically when the user starts the hot-plug?
please check v21, agree with you and have already modify it.
>> fsm = &fwd_streams[fc->stream_idx];
>> nb_fs = fc->stream_nb;
>> do {
>> @@ -2069,6 +2072,26 @@ rmv_event_callback(void *arg)
>> dev->device->name);
>> }
>>
>> +static void
>> +rmv_dev_event_callback(char *dev_name)
>> +{
>> + uint16_t port_id;
>> + int ret;
>> +
>> + ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
>> + if (ret) {
>> + printf("can not get port by device %s!\n", dev_name);
>> + return;
>> + }
>> +
>> + RTE_ETH_VALID_PORTID_OR_RET(port_id);
>> + printf("removing port id:%u\n", port_id);
>> + stop_packet_forwarding();
>> + stop_port(port_id);
>> + close_port(port_id);
>> + detach_port(port_id);
>> +}
> We have also the rmv_event_callback() which is triggered by a RMV interrupt and running by the host thread.
> What is the context thread of rmv_dev_event_callback()?
> Shouldn't they be synchronized? Should we need both in the same time?
the context thread is interrupt thread. and we might be discuss how to
sync it. do you have comment if i combine these 2 into 1 callback?
>> +
>> /* This function is used by the interrupt thread */ static int
>> eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void
>> *param, @@ -2130,9 +2153,7 @@ eth_dev_event_callback(char
>> *device_name, enum rte_dev_event_type type,
>> case RTE_DEV_EVENT_REMOVE:
>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>> device_name);
>> - /* TODO: After finish failure handle, begin to stop
>> - * packet forward, stop port, close port, detach port.
>> - */
>> + rmv_dev_event_callback(device_name);
>> break;
>> case RTE_DEV_EVENT_ADD:
>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
>> @@ -2640,7 +2661,7 @@ main(int argc, char** argv)
>> return -1;
>> }
>> eth_dev_event_callback_register();
>> -
>> + rte_dev_handle_hot_unplug();
>> }
>>
>> if (start_port(RTE_PORT_ALL) != 0)
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V20 4/4] app/testpmd: show example to handler hot unplug
2018-05-03 9:35 ` Guo, Jia
@ 2018-05-03 11:27 ` Matan Azrad
0 siblings, 0 replies; 494+ messages in thread
From: Matan Azrad @ 2018-05-03 11:27 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
jianfeng.tan@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Guo
From: Guo, Jia, Thursday, May 3, 2018 12:36 PM
> hi, matan
>
>
> On 5/3/2018 3:25 PM, Matan Azrad wrote:
> > Hi Jeff
> >
> >> From: Jeff Guo, Wednesday, April 18, 2018 4:38 PM Use testpmd for
> >> example, to show how an application smoothly handle failure when
> >> device being hot unplug. Once app detect the removal event, the
> >> callback would be called, it first stop the packet forwarding, then
> >> stop the port, close the port and finally detach the port.
> >>
> >> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> >> ---
> >> v20->v19:
> >> remove the auto binding example.
> >> ---
> >> app/test-pmd/testpmd.c | 29 +++++++++++++++++++++++++----
> >> 1 file changed, 25 insertions(+), 4 deletions(-)
> >>
> >> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> >> 5986ff7..3751901 100644
> >> --- a/app/test-pmd/testpmd.c
> >> +++ b/app/test-pmd/testpmd.c
> >> @@ -1125,6 +1125,9 @@ run_pkt_fwd_on_lcore(struct fwd_lcore *fc,
> >> packet_fwd_t pkt_fwd)
> >> tics_datum = rte_rdtsc();
> >> tics_per_1sec = rte_get_timer_hz();
> >> #endif
> >> + if (hot_plug)
> >> + rte_dev_handle_hot_unplug();
> >> +
> > Again, I don't understand why the application should configure it - it
> > already started the hot-plug, Can't the EAL handle this automatically when
> the user starts the hot-plug?
> please check v21, agree with you and have already modify it.
Looks good, thanks.
> >> fsm = &fwd_streams[fc->stream_idx];
> >> nb_fs = fc->stream_nb;
> >> do {
> >> @@ -2069,6 +2072,26 @@ rmv_event_callback(void *arg)
> >> dev->device->name);
> >> }
> >>
> >> +static void
> >> +rmv_dev_event_callback(char *dev_name) {
> >> + uint16_t port_id;
> >> + int ret;
> >> +
> >> + ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
> >> + if (ret) {
> >> + printf("can not get port by device %s!\n", dev_name);
> >> + return;
> >> + }
> >> +
> >> + RTE_ETH_VALID_PORTID_OR_RET(port_id);
> >> + printf("removing port id:%u\n", port_id);
> >> + stop_packet_forwarding();
> >> + stop_port(port_id);
> >> + close_port(port_id);
> >> + detach_port(port_id);
> >> +}
> > We have also the rmv_event_callback() which is triggered by a RMV
> interrupt and running by the host thread.
> > What is the context thread of rmv_dev_event_callback()?
> > Shouldn't they be synchronized? Should we need both in the same time?
> the context thread is interrupt thread. and we might be discuss how to sync
> it. do you have comment if i combine these 2 into 1 callback?
Please see the patch series I sent today regarding rmv_event_callback() function:
" [PATCH 0/6] Testpmd: fix port hotplug".
Yes, I think you should use rmv_event_callback() by your function (after the port id retrieving) to do code reuse.
Regarding synchronization, let's discuss:
So, the both callbacks are running from the same thread, but by different fd.
Right?
So, they will be triggered sequentially by the kernel when a device is plugged-out and we cannot know the order.
Right?
The second one may get an "invalid port" error because the first one was detached the port - not a major issue,
but the port id can be reused between them and then the second one may detach an available port.
Right?
So, looks like it is better to choose only one of them,
If all the above conclusions are correct , I suggest to disable the ethdev mechanism when the EAL hotplug is enabled:
@@ -2152,9 +2152,10 @@ struct pmd_test_command {
switch (type) {
case RTE_ETH_EVENT_INTR_RMV:
- if (rte_eal_alarm_set(100000,
- rmv_event_callback, (void *)(intptr_t)port_id))
- fprintf(stderr, "Could not set up deferred device removal\n");
+ if (!hot_plug)
+ if (rte_eal_alarm_set(100000, rmv_event_callback,
+ (void *)(intptr_t)port_id))
+ fprintf(stderr, "Could not set up deferred device removal\n");
break;
default:
break;
What do you think?
Matan.
> >> /* This function is used by the interrupt thread */ static int
> >> eth_event_callback(portid_t port_id, enum rte_eth_event_type type,
> >> void *param, @@ -2130,9 +2153,7 @@ eth_dev_event_callback(char
> >> *device_name, enum rte_dev_event_type type,
> >> case RTE_DEV_EVENT_REMOVE:
> >> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> >> device_name);
> >> - /* TODO: After finish failure handle, begin to stop
> >> - * packet forward, stop port, close port, detach port.
> >> - */
> >> + rmv_dev_event_callback(device_name);
> >> break;
> >> case RTE_DEV_EVENT_ADD:
> >> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
> @@ -2640,7
> >> +2661,7 @@ main(int argc, char** argv)
> >> return -1;
> >> }
> >> eth_dev_event_callback_register();
> >> -
> >> + rte_dev_handle_hot_unplug();
> >> }
> >>
> >> if (start_port(RTE_PORT_ALL) != 0)
> >> --
> >> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V21 0/4] hot plug recovery mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (3 preceding siblings ...)
2018-04-18 13:38 ` [PATCH V20 0/4] add hot plug recovery mechanism Jeff Guo
@ 2018-05-03 8:57 ` Jeff Guo
2018-05-03 8:57 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
` (3 more replies)
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
` (18 subsequent siblings)
23 siblings, 4 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 8:57 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
At the prior, device event monitor framework have been introduced,
the typical usage is for device hot plug. If we want application
would not be break down when device hot plug in or out, we still need
some measures to help app to handle that, such as recovery device for
device detaching, so that app can keep running smoothly but not be
disturbed by any hotplug behaviors.
This patch set will introduces an recovery mechanism to handle hot unplug,
and also use testpmd to show example of how to use this mechanism to process
hot plug event. The process could be shown as below:
plug out->failure handle->stop forward->stop port->close port->detach port
with this mechanism, user such as fail-safe driver or testpmd could be
able to develop their own hot plug application.
patchset history:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue
fix attach port issue for multiple devices case.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (4):
bus/pci: handle device hot unplug
eal: add failure handle mechanism for hot plug
igb_uio: fix uio release issue when hot unplug
app/testpmd: show example to handle hot unplug
app/test-pmd/testpmd.c | 28 ++++--
drivers/bus/pci/pci_common.c | 65 ++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++
drivers/bus/pci/private.h | 12 +++
kernel/linux/igb_uio/igb_uio.c | 4 +
lib/librte_eal/common/include/rte_bus.h | 16 ++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++-
7 files changed, 306 insertions(+), 6 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V21 1/4] bus/pci: handle device hot unplug
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
@ 2018-05-03 8:57 ` Jeff Guo
2018-05-03 8:57 ` [PATCH V21 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
` (2 subsequent siblings)
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 8:57 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As of device hot unplug, we need some preparatory measures, so that when
we encounter memory fault (like SIGBUS error) due to the unplug action,
we can recover instead of crash.
To handle device hot unplug is bus-specific behavior, this patch introduces
a bus ops so that each kind of bus can implement its own logic. Further,
this patch implements the ops for PCI bus: remap a dummy memory to avoid
bus read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
split function in hot unplug ops
---
drivers/bus/pci/pci_common.c | 65 +++++++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
4 files changed, 126 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 7215aae..74d9aa8 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -472,6 +472,70 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+static struct rte_pci_device *
+pci_find_device_by_addr(void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(ERR, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+static int
+pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ if (dev != NULL)
+ pdev = RTE_DEV_TO_PCI(dev);
+ else
+ pdev = pci_find_device_by_addr(failure_addr);
+
+ if (!pdev)
+ return -1;
+
+ /* remap resources for devices */
+ switch (pdev->kdrv) {
+ case RTE_KDRV_VFIO:
+#ifdef VFIO_PRESENT
+ /* TODO */
+ ret = -1;
+#endif
+ break;
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ if (ret != 0)
+ RTE_LOG(ERR, EAL, "Failed to handle hot unplug of device %s",
+ pdev->name);
+ return ret;
+}
+
static int
pci_plug(struct rte_device *dev)
{
@@ -502,6 +566,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .handle_hot_unplug = pci_handle_hot_unplug,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 88fa587..5551506 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..6a5609f 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hot unplug handler, which is responsible
+ * for handle the failure when hot unplug the device, guaranty the system
+ * would not hung in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
+ void *dev_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -209,6 +223,8 @@ struct rte_bus {
rte_bus_plug_t plug; /**< Probe single device for drivers */
rte_bus_unplug_t unplug; /**< Remove single device from driver */
rte_bus_parse_t parse; /**< Parse a device name */
+ rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
+ device event */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
};
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
2018-05-03 8:57 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
@ 2018-05-03 8:57 ` Jeff Guo
2018-05-03 8:57 ` [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-05-03 8:57 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 8:57 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot unplug event. When device be hot plug out, the device resource
become invalid, if this resource is still be unexpected read/write,
system will crash. This patch let eal help application to handle
this fault, when sigbus error occur, check the failure address and
accordingly remap the invalid memory for the corresponding device,
that could guaranty the application not to be shut down when hot plug.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
sync failure hanlde to fix multiple process issue
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
1 file changed, 153 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..3067f39 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,27 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
+static struct sigaction sigbus_action_old;
+
+static int sigbus_need_recover;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
};
static int
+dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
+{
+ struct rte_bus *bus;
+ int ret = 0;
+
+ if (!dev && !dev_addr) {
+ return -EINVAL;
+ } else if (dev) {
+ bus = rte_bus_find_by_device_name(dev->name);
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret) {
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for device %s "
+ "on the bus %s.\n ",
+ dev->name, bus->name);
+ }
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug for bus %s!\n",
+ bus->name);
+ ret = -ENOTSUP;
+ }
+ } else {
+ TAILQ_FOREACH(bus, &rte_bus_list, next) {
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret)
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for the device "
+ "on the bus %s!\n", bus->name);
+ else
+ break;
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug "
+ "for bus %s!\n", bus->name);
+ ret = -ENOTSUP;
+ }
+ }
+ }
+ return ret;
+}
+
+static void
+sigbus_action_recover(void)
+{
+ if (sigbus_need_recover) {
+ sigaction(SIGBUS, &sigbus_action_old, NULL);
+ sigbus_need_recover = 0;
+ }
+}
+
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(NULL, info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (!ret)
+ RTE_LOG(DEBUG, EAL,
+ "Success to handle SIGBUS error for hot unplug!\n");
+ else
+ rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
+static int
dev_uev_socket_fd_create(void)
{
struct sockaddr_nl addr;
@@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ uevent.devname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL,
+ "Cannot find unplugged device (%s)\n",
+ uevent.devname);
+ return;
+ }
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(dev, NULL);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Driver cannot remap the "
+ "device (%s)\n",
+ dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
+
monitor_started = true;
return 0;
@@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ sigbus_action_recover();
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
2018-05-03 8:57 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
2018-05-03 8:57 ` [PATCH V21 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-05-03 8:57 ` Jeff Guo
2018-05-03 8:57 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 8:57 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
when device being hot unplug, release a none exist uio resource will
result kernel null pointer error, so this patch will check if device
has been remove before release uio release procedure, if so just return
back.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
no change
---
kernel/linux/igb_uio/igb_uio.c | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index cd9b7e7..f0b1cfe 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -344,6 +344,10 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ /* check if device has been remove before release */
+ if ((&dev->dev.kobj)->state_remove_uevent_sent == 1)
+ return -1;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V21 4/4] app/testpmd: show example to handle hot unplug
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
` (2 preceding siblings ...)
2018-05-03 8:57 ` [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-05-03 8:57 ` Jeff Guo
2018-05-16 14:30 ` Iremonger, Bernard
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 8:57 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. Once app detect the removal event,
the callback would be called, it first stop the packet forwarding, then
stop the port, close the port and finally detach the port.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
fix attach port issue, let it work for multiple device case.
---
app/test-pmd/testpmd.c | 28 +++++++++++++++++++++++-----
1 file changed, 23 insertions(+), 5 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index db23f23..81f41e3 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -1908,9 +1908,10 @@ eth_dev_event_callback_unregister(void)
void
attach_port(char *identifier)
{
- portid_t pi = 0;
unsigned int socket_id;
+ portid_t pi = rte_eth_dev_count_avail();
+
printf("Attaching a new port...\n");
if (identifier == NULL) {
@@ -2079,6 +2080,26 @@ rmv_event_callback(void *arg)
dev->device->name);
}
+static void
+rmv_dev_event_callback(char *dev_name)
+{
+ uint16_t port_id;
+ int ret;
+
+ ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", dev_name);
+ return;
+ }
+
+ RTE_ETH_VALID_PORTID_OR_RET(port_id);
+ printf("removing port id:%u\n", port_id);
+ stop_packet_forwarding();
+ stop_port(port_id);
+ close_port(port_id);
+ detach_port(port_id);
+}
+
/* This function is used by the interrupt thread */
static int
eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void *param,
@@ -2141,9 +2162,7 @@ eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ rmv_dev_event_callback(device_name);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2666,7 +2685,6 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V21 4/4] app/testpmd: show example to handle hot unplug
2018-05-03 8:57 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
@ 2018-05-16 14:30 ` Iremonger, Bernard
0 siblings, 0 replies; 494+ messages in thread
From: Iremonger, Bernard @ 2018-05-16 14:30 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Guo, Jia, Zhang, Helin
Hi Jeff
> -----Original Message-----
> From: dev [mailto:dev-bounces@dpdk.org] On Behalf Of Jeff Guo
> Sent: Thursday, May 3, 2018 9:57 AM
> To: stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>;
> Ananyev, Konstantin <konstantin.ananyev@intel.com>;
> gaetan.rivet@6wind.com; Wu, Jingjing <jingjing.wu@intel.com>;
> thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van
> Haaren, Harry <harry.van.haaren@intel.com>; Tan, Jianfeng
> <jianfeng.tan@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Guo,
> Jia <jia.guo@intel.com>; Zhang, Helin <helin.zhang@intel.com>
> Subject: [dpdk-dev] [PATCH V21 4/4] app/testpmd: show example to handle
> hot unplug
>
> Use testpmd for example, to show how an application smoothly handle
> failure when device being hot unplug. Once app detect the removal event,
> the callback would be called, it first stop the packet forwarding, then stop the
> port, close the port and finally detach the port.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v21->v20:
> fix attach port issue, let it work for multiple device case.
> ---
> app/test-pmd/testpmd.c | 28 +++++++++++++++++++++++-----
> 1 file changed, 23 insertions(+), 5 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> db23f23..81f41e3 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -1908,9 +1908,10 @@ eth_dev_event_callback_unregister(void)
> void
> attach_port(char *identifier)
> {
> - portid_t pi = 0;
> unsigned int socket_id;
>
> + portid_t pi = rte_eth_dev_count_avail();
> +
> printf("Attaching a new port...\n");
>
> if (identifier == NULL) {
> @@ -2079,6 +2080,26 @@ rmv_event_callback(void *arg)
> dev->device->name);
> }
>
> +static void
> +rmv_dev_event_callback(char *dev_name)
> +{
> + uint16_t port_id;
> + int ret;
> +
> + ret = rte_eth_dev_get_port_by_name(dev_name, &port_id);
> + if (ret) {
> + printf("can not get port by device %s!\n", dev_name);
> + return;
> + }
> +
> + RTE_ETH_VALID_PORTID_OR_RET(port_id);
> + printf("removing port id:%u\n", port_id);
> + stop_packet_forwarding();
> + stop_port(port_id);
> + close_port(port_id);
> + detach_port(port_id);
> +}
> +
> /* This function is used by the interrupt thread */ static int
> eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void
> *param, @@ -2141,9 +2162,7 @@ eth_dev_event_callback(char
> *device_name, enum rte_dev_event_type type,
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + rmv_dev_event_callback(device_name);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
> @@ -2666,7 +2685,6 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
This patch does not apply to dpdk_18_05_R4C master branch and needs to be rebased.
Regards,
Bernard.
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V21 0/4] hot plug recovery mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (4 preceding siblings ...)
2018-05-03 8:57 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
@ 2018-05-03 10:48 ` Jeff Guo
2018-05-03 10:48 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
` (3 more replies)
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
` (17 subsequent siblings)
23 siblings, 4 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 10:48 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
At the prior, device event monitor framework have been introduced,
the typical usage is for device hot plug. If we want application
would not be break down when device hot plug in or out, we still need
some measures to help app to handle that, such as recovery device for
device detaching, so that app can keep running smoothly but not be
disturbed by any hotplug behaviors.
This patch set will introduces an recovery mechanism to handle hot unplug,
and also use testpmd to show example of how to use this mechanism to process
hot plug event. The process could be shown as below:
plug out->failure handle->stop forward->stop port->close port->detach port
with this mechanism, user such as fail-safe driver or testpmd could be
able to develop their own hot plug application.
patchset history:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue
fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (4):
bus/pci: handle device hot unplug
eal: add failure handle mechanism for hot plug
igb_uio: fix uio release issue when hot unplug
app/testpmd: show example to handle hot unplug
app/test-pmd/testpmd.c | 27 ++++--
drivers/bus/pci/pci_common.c | 65 ++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++
drivers/bus/pci/private.h | 12 +++
kernel/linux/igb_uio/igb_uio.c | 4 +
lib/librte_eal/common/include/rte_bus.h | 16 ++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++-
7 files changed, 301 insertions(+), 10 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V21 1/4] bus/pci: handle device hot unplug
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
@ 2018-05-03 10:48 ` Jeff Guo
2018-05-03 10:48 ` [PATCH V21 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
` (2 subsequent siblings)
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 10:48 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As of device hot unplug, we need some preparatory measures, so that when
we encounter memory fault (like SIGBUS error) due to the unplug action,
we can recover instead of crash.
To handle device hot unplug is bus-specific behavior, this patch introduces
a bus ops so that each kind of bus can implement its own logic. Further,
this patch implements the ops for PCI bus: remap a dummy memory to avoid
bus read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
split function in hot unplug ops
---
drivers/bus/pci/pci_common.c | 65 +++++++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
4 files changed, 126 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 7215aae..74d9aa8 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -472,6 +472,70 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+static struct rte_pci_device *
+pci_find_device_by_addr(void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(ERR, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+static int
+pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ if (dev != NULL)
+ pdev = RTE_DEV_TO_PCI(dev);
+ else
+ pdev = pci_find_device_by_addr(failure_addr);
+
+ if (!pdev)
+ return -1;
+
+ /* remap resources for devices */
+ switch (pdev->kdrv) {
+ case RTE_KDRV_VFIO:
+#ifdef VFIO_PRESENT
+ /* TODO */
+ ret = -1;
+#endif
+ break;
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ if (ret != 0)
+ RTE_LOG(ERR, EAL, "Failed to handle hot unplug of device %s",
+ pdev->name);
+ return ret;
+}
+
static int
pci_plug(struct rte_device *dev)
{
@@ -502,6 +566,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .handle_hot_unplug = pci_handle_hot_unplug,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 88fa587..5551506 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..6a5609f 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hot unplug handler, which is responsible
+ * for handle the failure when hot unplug the device, guaranty the system
+ * would not hung in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
+ void *dev_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -209,6 +223,8 @@ struct rte_bus {
rte_bus_plug_t plug; /**< Probe single device for drivers */
rte_bus_unplug_t unplug; /**< Remove single device from driver */
rte_bus_parse_t parse; /**< Parse a device name */
+ rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
+ device event */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
};
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
2018-05-03 10:48 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
@ 2018-05-03 10:48 ` Jeff Guo
2018-05-04 15:56 ` Ananyev, Konstantin
2018-05-03 10:48 ` [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-05-03 10:48 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 10:48 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot unplug event. When device be hot plug out, the device resource
become invalid, if this resource is still be unexpected read/write,
system will crash. This patch let eal help application to handle
this fault, when sigbus error occur, check the failure address and
accordingly remap the invalid memory for the corresponding device,
that could guaranty the application not to be shut down when hot plug.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
sync failure hanlde to fix multiple process issue
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
1 file changed, 153 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..3067f39 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,27 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
+static struct sigaction sigbus_action_old;
+
+static int sigbus_need_recover;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
};
static int
+dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
+{
+ struct rte_bus *bus;
+ int ret = 0;
+
+ if (!dev && !dev_addr) {
+ return -EINVAL;
+ } else if (dev) {
+ bus = rte_bus_find_by_device_name(dev->name);
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret) {
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for device %s "
+ "on the bus %s.\n ",
+ dev->name, bus->name);
+ }
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug for bus %s!\n",
+ bus->name);
+ ret = -ENOTSUP;
+ }
+ } else {
+ TAILQ_FOREACH(bus, &rte_bus_list, next) {
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret)
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for the device "
+ "on the bus %s!\n", bus->name);
+ else
+ break;
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug "
+ "for bus %s!\n", bus->name);
+ ret = -ENOTSUP;
+ }
+ }
+ }
+ return ret;
+}
+
+static void
+sigbus_action_recover(void)
+{
+ if (sigbus_need_recover) {
+ sigaction(SIGBUS, &sigbus_action_old, NULL);
+ sigbus_need_recover = 0;
+ }
+}
+
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(NULL, info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (!ret)
+ RTE_LOG(DEBUG, EAL,
+ "Success to handle SIGBUS error for hot unplug!\n");
+ else
+ rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
+static int
dev_uev_socket_fd_create(void)
{
struct sockaddr_nl addr;
@@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ uevent.devname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL,
+ "Cannot find unplugged device (%s)\n",
+ uevent.devname);
+ return;
+ }
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(dev, NULL);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Driver cannot remap the "
+ "device (%s)\n",
+ dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
+
monitor_started = true;
return 0;
@@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ sigbus_action_recover();
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
2018-05-03 10:48 ` [PATCH V21 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-05-04 15:56 ` Ananyev, Konstantin
2018-05-08 14:57 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-05-04 15:56 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
Hi Jeff,
>
> This patch introduces a failure handler mechanism to handle device
> hot unplug event. When device be hot plug out, the device resource
> become invalid, if this resource is still be unexpected read/write,
> system will crash. This patch let eal help application to handle
> this fault, when sigbus error occur, check the failure address and
> accordingly remap the invalid memory for the corresponding device,
> that could guaranty the application not to be shut down when hot plug.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v21->v20:
> sync failure hanlde to fix multiple process issue
> ---
> lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
> 1 file changed, 153 insertions(+), 1 deletion(-)
>
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 1cf6aeb..3067f39 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -14,15 +16,27 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
> +#include <rte_spinlock.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 };
> static bool monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
> #define EAL_UEV_MSG_LEN 4096
> #define EAL_UEV_MSG_ELEM_LEN 128
>
> +/* spinlock for device failure process */
> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
> +
> +static struct sigaction sigbus_action_old;
> +
> +static int sigbus_need_recover;
> +
> static void dev_uev_handler(__rte_unused void *param);
>
> /* identify the system layer which reports this event. */
> @@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
> };
>
> static int
> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
> +{
> + struct rte_bus *bus;
> + int ret = 0;
> +
> + if (!dev && !dev_addr) {
> + return -EINVAL;
> + } else if (dev) {
> + bus = rte_bus_find_by_device_name(dev->name);
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret) {
> + RTE_LOG(ERR, EAL,
> + "Cannot handle hot unplug "
> + "for device %s "
> + "on the bus %s.\n ",
> + dev->name, bus->name);
> + }
> + } else {
> + RTE_LOG(ERR, EAL,
> + "Not support handle hot unplug for bus %s!\n",
> + bus->name);
> + ret = -ENOTSUP;
> + }
> + } else {
> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
> + if (bus->handle_hot_unplug) {
> + /**
> + * call bus ops to handle hot unplug.
> + */
> + ret = bus->handle_hot_unplug(dev, dev_addr);
> + if (ret)
> + RTE_LOG(ERR, EAL,
> + "Cannot handle hot unplug "
> + "for the device "
> + "on the bus %s!\n", bus->name);
> + else
> + break;
> + } else {
> + RTE_LOG(ERR, EAL,
> + "Not support handle hot unplug "
> + "for bus %s!\n", bus->name);
> + ret = -ENOTSUP;
> + }
> + }
> + }
> + return ret;
> +}
> +
> +static void
> +sigbus_action_recover(void)
> +{
> + if (sigbus_need_recover) {
> + sigaction(SIGBUS, &sigbus_action_old, NULL);
> + sigbus_need_recover = 0;
> + }
> +}
> +
> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> + void *ctx __rte_unused)
> +{
> + int ret;
> +
> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
> + (int)pthread_self(), info->si_addr);
> + rte_spinlock_lock(&dev_failure_lock);
> + ret = dev_uev_failure_process(NULL, info->si_addr);
> + rte_spinlock_unlock(&dev_failure_lock);
> + if (!ret)
> + RTE_LOG(DEBUG, EAL,
> + "Success to handle SIGBUS error for hot unplug!\n");
> + else
> + rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
I still think we have to distinguish here 2 cases:
1) failure addr is not belong to any dpdk devices
2) failure addr does belong to dpdk device, but we fail to remap it.
For 1) we probably need to call previous sigbus handler.
For 2) we probably can only do exit().
> +}
> +
> +static int cmp_dev_name(const struct rte_device *dev,
> + const void *_name)
> +{
> + const char *name = _name;
> +
> + return strcmp(dev->name, name);
> +}
> +
> +static int
> dev_uev_socket_fd_create(void)
> {
> struct sockaddr_nl addr;
> @@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
> struct rte_dev_event uevent;
> int ret;
> char buf[EAL_UEV_MSG_LEN];
> + struct rte_bus *bus;
> + struct rte_device *dev;
> + const char *busname;
>
> memset(&uevent, 0, sizeof(struct rte_dev_event));
> memset(buf, 0, EAL_UEV_MSG_LEN);
> @@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> uevent.devname, uevent.type, uevent.subsystem);
>
> - if (uevent.devname)
> + switch (uevent.subsystem) {
> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> + busname = "pci";
> + break;
> + default:
> + break;
> + }
> +
> + if (uevent.devname) {
> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> + bus = rte_bus_find_by_name(busname);
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> + uevent.devname);
> + return;
> + }
> + dev = bus->find_device(NULL, cmp_dev_name,
> + uevent.devname);
> + if (dev == NULL) {
> + RTE_LOG(ERR, EAL,
> + "Cannot find unplugged device (%s)\n",
> + uevent.devname);
> + return;
> + }
> + rte_spinlock_lock(&dev_failure_lock);
> + ret = dev_uev_failure_process(dev, NULL);
> + rte_spinlock_unlock(&dev_failure_lock);
That's interrupt thread, right?
I wonder could it happen that user will call device_detach() at the same moment?
Konstantin
> + if (ret) {
> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
> + "device (%s)\n",
> + dev->name);
> + return;
> + }
> + }
> dev_callback_process(uevent.devname, uevent.type);
> + }
> }
>
> int __rte_experimental
> rte_dev_event_monitor_start(void)
> {
> + sigset_t mask;
> + struct sigaction action;
> int ret;
>
> if (monitor_started)
> @@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
> return -1;
> }
>
> + /* register sigbus handler */
> + sigemptyset(&mask);
> + sigaddset(&mask, SIGBUS);
> + action.sa_flags = SA_SIGINFO;
> + action.sa_mask = mask;
> + action.sa_sigaction = sigbus_handler;
> + sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
> +
> monitor_started = true;
>
> return 0;
> @@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
> return ret;
> }
>
> + sigbus_action_recover();
> +
> close(intr_handle.fd);
> intr_handle.fd = -1;
> monitor_started = false;
> +
> return 0;
> }
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
2018-05-04 15:56 ` Ananyev, Konstantin
@ 2018-05-08 14:57 ` Guo, Jia
2018-05-08 15:19 ` Ananyev, Konstantin
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-05-08 14:57 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On 5/4/2018 11:56 PM, Ananyev, Konstantin wrote:
> Hi Jeff,
>
>> This patch introduces a failure handler mechanism to handle device
>> hot unplug event. When device be hot plug out, the device resource
>> become invalid, if this resource is still be unexpected read/write,
>> system will crash. This patch let eal help application to handle
>> this fault, when sigbus error occur, check the failure address and
>> accordingly remap the invalid memory for the corresponding device,
>> that could guaranty the application not to be shut down when hot plug.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v21->v20:
>> sync failure hanlde to fix multiple process issue
>> ---
>> lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
>> 1 file changed, 153 insertions(+), 1 deletion(-)
>>
>> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> index 1cf6aeb..3067f39 100644
>> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> @@ -4,6 +4,8 @@
>>
>> #include <string.h>
>> #include <unistd.h>
>> +#include <fcntl.h>
>> +#include <signal.h>
>> #include <sys/socket.h>
>> #include <linux/netlink.h>
>>
>> @@ -14,15 +16,27 @@
>> #include <rte_malloc.h>
>> #include <rte_interrupts.h>
>> #include <rte_alarm.h>
>> +#include <rte_bus.h>
>> +#include <rte_eal.h>
>> +#include <rte_spinlock.h>
>>
>> #include "eal_private.h"
>>
>> static struct rte_intr_handle intr_handle = {.fd = -1 };
>> static bool monitor_started;
>>
>> +extern struct rte_bus_list rte_bus_list;
>> +
>> #define EAL_UEV_MSG_LEN 4096
>> #define EAL_UEV_MSG_ELEM_LEN 128
>>
>> +/* spinlock for device failure process */
>> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
>> +
>> +static struct sigaction sigbus_action_old;
>> +
>> +static int sigbus_need_recover;
>> +
>> static void dev_uev_handler(__rte_unused void *param);
>>
>> /* identify the system layer which reports this event. */
>> @@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
>> };
>>
>> static int
>> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
>> +{
>> + struct rte_bus *bus;
>> + int ret = 0;
>> +
>> + if (!dev && !dev_addr) {
>> + return -EINVAL;
>> + } else if (dev) {
>> + bus = rte_bus_find_by_device_name(dev->name);
>> + if (bus->handle_hot_unplug) {
>> + /**
>> + * call bus ops to handle hot unplug.
>> + */
>> + ret = bus->handle_hot_unplug(dev, dev_addr);
>> + if (ret) {
>> + RTE_LOG(ERR, EAL,
>> + "Cannot handle hot unplug "
>> + "for device %s "
>> + "on the bus %s.\n ",
>> + dev->name, bus->name);
>> + }
>> + } else {
>> + RTE_LOG(ERR, EAL,
>> + "Not support handle hot unplug for bus %s!\n",
>> + bus->name);
>> + ret = -ENOTSUP;
>> + }
>> + } else {
>> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
>> + if (bus->handle_hot_unplug) {
>> + /**
>> + * call bus ops to handle hot unplug.
>> + */
>> + ret = bus->handle_hot_unplug(dev, dev_addr);
>> + if (ret)
>> + RTE_LOG(ERR, EAL,
>> + "Cannot handle hot unplug "
>> + "for the device "
>> + "on the bus %s!\n", bus->name);
>> + else
>> + break;
>> + } else {
>> + RTE_LOG(ERR, EAL,
>> + "Not support handle hot unplug "
>> + "for bus %s!\n", bus->name);
>> + ret = -ENOTSUP;
>> + }
>> + }
>> + }
>> + return ret;
>> +}
>> +
>> +static void
>> +sigbus_action_recover(void)
>> +{
>> + if (sigbus_need_recover) {
>> + sigaction(SIGBUS, &sigbus_action_old, NULL);
>> + sigbus_need_recover = 0;
>> + }
>> +}
>> +
>> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
>> + void *ctx __rte_unused)
>> +{
>> + int ret;
>> +
>> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
>> + (int)pthread_self(), info->si_addr);
>> + rte_spinlock_lock(&dev_failure_lock);
>> + ret = dev_uev_failure_process(NULL, info->si_addr);
>> + rte_spinlock_unlock(&dev_failure_lock);
>> + if (!ret)
>> + RTE_LOG(DEBUG, EAL,
>> + "Success to handle SIGBUS error for hot unplug!\n");
>> + else
>> + rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
> I still think we have to distinguish here 2 cases:
> 1) failure addr is not belong to any dpdk devices
> 2) failure addr does belong to dpdk device, but we fail to remap it.
>
> For 1) we probably need to call previous sigbus handler.
> For 2) we probably can only do exit().
i think the previous sigbus handler is just a exception of sigbus error
and exit out of the process, so i think should use one way to handler
1)+2) should be fine, do you agree with that? or you could find any
chance to
call any other sigbus handler at this positoin?
>> +}
>> +
>> +static int cmp_dev_name(const struct rte_device *dev,
>> + const void *_name)
>> +{
>> + const char *name = _name;
>> +
>> + return strcmp(dev->name, name);
>> +}
>> +
>> +static int
>> dev_uev_socket_fd_create(void)
>> {
>> struct sockaddr_nl addr;
>> @@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
>> struct rte_dev_event uevent;
>> int ret;
>> char buf[EAL_UEV_MSG_LEN];
>> + struct rte_bus *bus;
>> + struct rte_device *dev;
>> + const char *busname;
>>
>> memset(&uevent, 0, sizeof(struct rte_dev_event));
>> memset(buf, 0, EAL_UEV_MSG_LEN);
>> @@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
>> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
>> uevent.devname, uevent.type, uevent.subsystem);
>>
>> - if (uevent.devname)
>> + switch (uevent.subsystem) {
>> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
>> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
>> + busname = "pci";
>> + break;
>> + default:
>> + break;
>> + }
>> +
>> + if (uevent.devname) {
>> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
>> + bus = rte_bus_find_by_name(busname);
>> + if (bus == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
>> + uevent.devname);
>> + return;
>> + }
>> + dev = bus->find_device(NULL, cmp_dev_name,
>> + uevent.devname);
>> + if (dev == NULL) {
>> + RTE_LOG(ERR, EAL,
>> + "Cannot find unplugged device (%s)\n",
>> + uevent.devname);
>> + return;
>> + }
>> + rte_spinlock_lock(&dev_failure_lock);
>> + ret = dev_uev_failure_process(dev, NULL);
>> + rte_spinlock_unlock(&dev_failure_lock);
> That's interrupt thread, right?
> I wonder could it happen that user will call device_detach() at the same moment?
> Konstantin
it is in interrupt thread, and user will call device_detach after failure process, you concern about twice or more device detach? i don't think is there any problem here.
>> + if (ret) {
>> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
>> + "device (%s)\n",
>> + dev->name);
>> + return;
>> + }
>> + }
>> dev_callback_process(uevent.devname, uevent.type);
>> + }
>> }
>>
>> int __rte_experimental
>> rte_dev_event_monitor_start(void)
>> {
>> + sigset_t mask;
>> + struct sigaction action;
>> int ret;
>>
>> if (monitor_started)
>> @@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
>> return -1;
>> }
>>
>> + /* register sigbus handler */
>> + sigemptyset(&mask);
>> + sigaddset(&mask, SIGBUS);
>> + action.sa_flags = SA_SIGINFO;
>> + action.sa_mask = mask;
>> + action.sa_sigaction = sigbus_handler;
>> + sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
>> +
>> monitor_started = true;
>>
>> return 0;
>> @@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
>> return ret;
>> }
>>
>> + sigbus_action_recover();
>> +
>> close(intr_handle.fd);
>> intr_handle.fd = -1;
>> monitor_started = false;
>> +
>> return 0;
>> }
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
2018-05-08 14:57 ` Guo, Jia
@ 2018-05-08 15:19 ` Ananyev, Konstantin
0 siblings, 0 replies; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-05-08 15:19 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Tuesday, May 8, 2018 3:57 PM
> To: Ananyev, Konstantin <konstantin.ananyev@intel.com>; stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; gaetan.rivet@6wind.com; Wu, Jingjing
> <jingjing.wu@intel.com>; thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
> <harry.van.haaren@intel.com>; Tan, Jianfeng <jianfeng.tan@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Zhang, Helin <helin.zhang@intel.com>
> Subject: Re: [PATCH V21 2/4] eal: add failure handle mechanism for hot plug
>
>
>
> On 5/4/2018 11:56 PM, Ananyev, Konstantin wrote:
> > Hi Jeff,
> >
> >> This patch introduces a failure handler mechanism to handle device
> >> hot unplug event. When device be hot plug out, the device resource
> >> become invalid, if this resource is still be unexpected read/write,
> >> system will crash. This patch let eal help application to handle
> >> this fault, when sigbus error occur, check the failure address and
> >> accordingly remap the invalid memory for the corresponding device,
> >> that could guaranty the application not to be shut down when hot plug.
> >>
> >> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> >> ---
> >> v21->v20:
> >> sync failure hanlde to fix multiple process issue
> >> ---
> >> lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
> >> 1 file changed, 153 insertions(+), 1 deletion(-)
> >>
> >> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> index 1cf6aeb..3067f39 100644
> >> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> @@ -4,6 +4,8 @@
> >>
> >> #include <string.h>
> >> #include <unistd.h>
> >> +#include <fcntl.h>
> >> +#include <signal.h>
> >> #include <sys/socket.h>
> >> #include <linux/netlink.h>
> >>
> >> @@ -14,15 +16,27 @@
> >> #include <rte_malloc.h>
> >> #include <rte_interrupts.h>
> >> #include <rte_alarm.h>
> >> +#include <rte_bus.h>
> >> +#include <rte_eal.h>
> >> +#include <rte_spinlock.h>
> >>
> >> #include "eal_private.h"
> >>
> >> static struct rte_intr_handle intr_handle = {.fd = -1 };
> >> static bool monitor_started;
> >>
> >> +extern struct rte_bus_list rte_bus_list;
> >> +
> >> #define EAL_UEV_MSG_LEN 4096
> >> #define EAL_UEV_MSG_ELEM_LEN 128
> >>
> >> +/* spinlock for device failure process */
> >> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
> >> +
> >> +static struct sigaction sigbus_action_old;
> >> +
> >> +static int sigbus_need_recover;
> >> +
> >> static void dev_uev_handler(__rte_unused void *param);
> >>
> >> /* identify the system layer which reports this event. */
> >> @@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
> >> };
> >>
> >> static int
> >> +dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
> >> +{
> >> + struct rte_bus *bus;
> >> + int ret = 0;
> >> +
> >> + if (!dev && !dev_addr) {
> >> + return -EINVAL;
> >> + } else if (dev) {
> >> + bus = rte_bus_find_by_device_name(dev->name);
> >> + if (bus->handle_hot_unplug) {
> >> + /**
> >> + * call bus ops to handle hot unplug.
> >> + */
> >> + ret = bus->handle_hot_unplug(dev, dev_addr);
> >> + if (ret) {
> >> + RTE_LOG(ERR, EAL,
> >> + "Cannot handle hot unplug "
> >> + "for device %s "
> >> + "on the bus %s.\n ",
> >> + dev->name, bus->name);
> >> + }
> >> + } else {
> >> + RTE_LOG(ERR, EAL,
> >> + "Not support handle hot unplug for bus %s!\n",
> >> + bus->name);
> >> + ret = -ENOTSUP;
> >> + }
> >> + } else {
> >> + TAILQ_FOREACH(bus, &rte_bus_list, next) {
> >> + if (bus->handle_hot_unplug) {
> >> + /**
> >> + * call bus ops to handle hot unplug.
> >> + */
> >> + ret = bus->handle_hot_unplug(dev, dev_addr);
> >> + if (ret)
> >> + RTE_LOG(ERR, EAL,
> >> + "Cannot handle hot unplug "
> >> + "for the device "
> >> + "on the bus %s!\n", bus->name);
> >> + else
> >> + break;
> >> + } else {
> >> + RTE_LOG(ERR, EAL,
> >> + "Not support handle hot unplug "
> >> + "for bus %s!\n", bus->name);
> >> + ret = -ENOTSUP;
> >> + }
> >> + }
> >> + }
> >> + return ret;
> >> +}
> >> +
> >> +static void
> >> +sigbus_action_recover(void)
> >> +{
> >> + if (sigbus_need_recover) {
> >> + sigaction(SIGBUS, &sigbus_action_old, NULL);
> >> + sigbus_need_recover = 0;
> >> + }
> >> +}
> >> +
> >> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> >> + void *ctx __rte_unused)
> >> +{
> >> + int ret;
> >> +
> >> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
> >> + (int)pthread_self(), info->si_addr);
> >> + rte_spinlock_lock(&dev_failure_lock);
> >> + ret = dev_uev_failure_process(NULL, info->si_addr);
> >> + rte_spinlock_unlock(&dev_failure_lock);
> >> + if (!ret)
> >> + RTE_LOG(DEBUG, EAL,
> >> + "Success to handle SIGBUS error for hot unplug!\n");
> >> + else
> >> + rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
> > I still think we have to distinguish here 2 cases:
> > 1) failure addr is not belong to any dpdk devices
> > 2) failure addr does belong to dpdk device, but we fail to remap it.
> >
> > For 1) we probably need to call previous sigbus handler.
> > For 2) we probably can only do exit().
>
> i think the previous sigbus handler is just a exception of sigbus error
> and exit out of the process, so i think should use one way to handler
> 1)+2) should be fine, do you agree with that? or you could find any
> chance to
> call any other sigbus handler at this positoin?
I think application can have its own sigbus handler installed (same as we do).
> >> +}
> >> +
> >> +static int cmp_dev_name(const struct rte_device *dev,
> >> + const void *_name)
> >> +{
> >> + const char *name = _name;
> >> +
> >> + return strcmp(dev->name, name);
> >> +}
> >> +
> >> +static int
> >> dev_uev_socket_fd_create(void)
> >> {
> >> struct sockaddr_nl addr;
> >> @@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
> >> struct rte_dev_event uevent;
> >> int ret;
> >> char buf[EAL_UEV_MSG_LEN];
> >> + struct rte_bus *bus;
> >> + struct rte_device *dev;
> >> + const char *busname;
> >>
> >> memset(&uevent, 0, sizeof(struct rte_dev_event));
> >> memset(buf, 0, EAL_UEV_MSG_LEN);
> >> @@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
> >> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> >> uevent.devname, uevent.type, uevent.subsystem);
> >>
> >> - if (uevent.devname)
> >> + switch (uevent.subsystem) {
> >> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> >> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> >> + busname = "pci";
> >> + break;
> >> + default:
> >> + break;
> >> + }
> >> +
> >> + if (uevent.devname) {
> >> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> >> + bus = rte_bus_find_by_name(busname);
> >> + if (bus == NULL) {
> >> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> >> + uevent.devname);
> >> + return;
> >> + }
> >> + dev = bus->find_device(NULL, cmp_dev_name,
> >> + uevent.devname);
> >> + if (dev == NULL) {
> >> + RTE_LOG(ERR, EAL,
> >> + "Cannot find unplugged device (%s)\n",
> >> + uevent.devname);
> >> + return;
> >> + }
> >> + rte_spinlock_lock(&dev_failure_lock);
> >> + ret = dev_uev_failure_process(dev, NULL);
> >> + rte_spinlock_unlock(&dev_failure_lock);
> > That's interrupt thread, right?
> > I wonder could it happen that user will call device_detach() at the same moment?
> > Konstantin
>
> it is in interrupt thread, and user will call device_detach after failure process, you concern about twice or more device detach? i
> don't think is there any problem here.
Ok, but user can call device_detach() on his own, without waiting for failure to happen, right?
>
> >> + if (ret) {
> >> + RTE_LOG(ERR, EAL, "Driver cannot remap the "
> >> + "device (%s)\n",
> >> + dev->name);
> >> + return;
> >> + }
> >> + }
> >> dev_callback_process(uevent.devname, uevent.type);
> >> + }
> >> }
> >>
> >> int __rte_experimental
> >> rte_dev_event_monitor_start(void)
> >> {
> >> + sigset_t mask;
> >> + struct sigaction action;
> >> int ret;
> >>
> >> if (monitor_started)
> >> @@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
> >> return -1;
> >> }
> >>
> >> + /* register sigbus handler */
> >> + sigemptyset(&mask);
> >> + sigaddset(&mask, SIGBUS);
> >> + action.sa_flags = SA_SIGINFO;
> >> + action.sa_mask = mask;
> >> + action.sa_sigaction = sigbus_handler;
> >> + sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
> >> +
> >> monitor_started = true;
> >>
> >> return 0;
> >> @@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
> >> return ret;
> >> }
> >>
> >> + sigbus_action_recover();
> >> +
> >> close(intr_handle.fd);
> >> intr_handle.fd = -1;
> >> monitor_started = false;
> >> +
> >> return 0;
> >> }
> >> --
> >> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
2018-05-03 10:48 ` [PATCH V21 1/4] bus/pci: handle device hot unplug Jeff Guo
2018-05-03 10:48 ` [PATCH V21 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-05-03 10:48 ` Jeff Guo
2018-05-03 10:48 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 10:48 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
when device being hot unplug, release a none exist uio resource will
result kernel null pointer error, so this patch will check if device
has been remove before release uio release procedure, if so just return
back.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
no change
---
kernel/linux/igb_uio/igb_uio.c | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index cd9b7e7..f0b1cfe 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -344,6 +344,10 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ /* check if device has been remove before release */
+ if ((&dev->dev.kobj)->state_remove_uevent_sent == 1)
+ return -1;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V21 4/4] app/testpmd: show example to handle hot unplug
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
` (2 preceding siblings ...)
2018-05-03 10:48 ` [PATCH V21 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-05-03 10:48 ` Jeff Guo
2018-06-14 12:59 ` Iremonger, Bernard
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-05-03 10:48 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
jianfeng.tan
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. Once app detect the removal event,
the callback would be called, it first stop the packet forwarding, then
stop the port, close the port and finally detach the port.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v21->v20:
fix attach port issue, let it work for multiple device case.
combind rmv callback to only one.
---
app/test-pmd/testpmd.c | 27 ++++++++++++++++++---------
1 file changed, 18 insertions(+), 9 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index db23f23..a1ff8f3 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -1908,9 +1908,10 @@ eth_dev_event_callback_unregister(void)
void
attach_port(char *identifier)
{
- portid_t pi = 0;
unsigned int socket_id;
+ portid_t pi = rte_eth_dev_count_avail();
+
printf("Attaching a new port...\n");
if (identifier == NULL) {
@@ -2071,12 +2072,14 @@ rmv_event_callback(void *arg)
RTE_ETH_VALID_PORTID_OR_RET(port_id);
dev = &rte_eth_devices[port_id];
+ if (dev->state == RTE_ETH_DEV_UNUSED)
+ return;
+
+ printf("removing device %s\n", dev->device->name);
+ stop_packet_forwarding();
stop_port(port_id);
close_port(port_id);
- printf("removing device %s\n", dev->device->name);
- if (rte_eal_dev_detach(dev->device))
- TESTPMD_LOG(ERR, "Failed to detach device %s\n",
- dev->device->name);
+ detach_port(port_id);
}
/* This function is used by the interrupt thread */
@@ -2131,19 +2134,26 @@ static void
eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
__rte_unused void *arg)
{
+ uint16_t port_id;
+ int ret;
+
if (type >= RTE_DEV_EVENT_MAX) {
fprintf(stderr, "%s called upon invalid event %d\n",
__func__, type);
fflush(stderr);
}
+ ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", device_name);
+ return;
+ }
+
switch (type) {
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ rmv_event_callback((void *)(intptr_t)port_id);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2666,7 +2676,6 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V21 4/4] app/testpmd: show example to handle hot unplug
2018-05-03 10:48 ` [PATCH V21 4/4] app/testpmd: show example to handle " Jeff Guo
@ 2018-06-14 12:59 ` Iremonger, Bernard
2018-06-15 8:32 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Iremonger, Bernard @ 2018-06-14 12:59 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Guo, Jia, Zhang, Helin
Hi Jeff,
> -----Original Message-----
> From: dev [mailto:dev-bounces@dpdk.org] On Behalf Of Jeff Guo
> Sent: Thursday, May 3, 2018 11:49 AM
> To: stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>;
> Ananyev, Konstantin <konstantin.ananyev@intel.com>;
> gaetan.rivet@6wind.com; Wu, Jingjing <jingjing.wu@intel.com>;
> thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van
> Haaren, Harry <harry.van.haaren@intel.com>; Tan, Jianfeng
> <jianfeng.tan@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Guo,
> Jia <jia.guo@intel.com>; Zhang, Helin <helin.zhang@intel.com>
> Subject: [dpdk-dev] [PATCH V21 4/4] app/testpmd: show example to handle
> hot unplug
>
> Use testpmd for example, to show how an application smoothly handle
> failure when device being hot unplug. Once app detect the removal event,
> the callback would be called, it first stop the packet forwarding, then stop the
> port, close the port and finally detach the port.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v21->v20:
> fix attach port issue, let it work for multiple device case.
> combind rmv callback to only one.
> ---
> app/test-pmd/testpmd.c | 27 ++++++++++++++++++---------
> 1 file changed, 18 insertions(+), 9 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> db23f23..a1ff8f3 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -1908,9 +1908,10 @@ eth_dev_event_callback_unregister(void)
> void
> attach_port(char *identifier)
> {
> - portid_t pi = 0;
> unsigned int socket_id;
>
> + portid_t pi = rte_eth_dev_count_avail();
> +
> printf("Attaching a new port...\n");
>
> if (identifier == NULL) {
> @@ -2071,12 +2072,14 @@ rmv_event_callback(void *arg)
> RTE_ETH_VALID_PORTID_OR_RET(port_id);
> dev = &rte_eth_devices[port_id];
>
> + if (dev->state == RTE_ETH_DEV_UNUSED)
> + return;
> +
> + printf("removing device %s\n", dev->device->name);
> + stop_packet_forwarding();
> stop_port(port_id);
> close_port(port_id);
> - printf("removing device %s\n", dev->device->name);
> - if (rte_eal_dev_detach(dev->device))
> - TESTPMD_LOG(ERR, "Failed to detach device %s\n",
> - dev->device->name);
> + detach_port(port_id);
> }
>
> /* This function is used by the interrupt thread */ @@ -2131,19 +2134,26
> @@ static void eth_dev_event_callback(char *device_name, enum
> rte_dev_event_type type,
> __rte_unused void *arg)
> {
> + uint16_t port_id;
> + int ret;
> +
> if (type >= RTE_DEV_EVENT_MAX) {
> fprintf(stderr, "%s called upon invalid event %d\n",
> __func__, type);
> fflush(stderr);
> }
>
> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> + if (ret) {
> + printf("can not get port by device %s!\n", device_name);
> + return;
> + }
> +
> switch (type) {
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + rmv_event_callback((void *)(intptr_t)port_id);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
> @@ -2666,7 +2676,6 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
This patch fails to apply to dpdk 18.08-rc0 and needs to be rebased.
Regards,
Bernard.
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V21 4/4] app/testpmd: show example to handle hot unplug
2018-06-14 12:59 ` Iremonger, Bernard
@ 2018-06-15 8:32 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-06-15 8:32 UTC (permalink / raw)
To: Iremonger, Bernard, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Tan, Jianfeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On 6/14/2018 8:59 PM, Iremonger, Bernard wrote:
> Hi Jeff,
>
>> -----Original Message-----
>> From: dev [mailto:dev-bounces@dpdk.org] On Behalf Of Jeff Guo
>> Sent: Thursday, May 3, 2018 11:49 AM
>> To: stephen@networkplumber.org; Richardson, Bruce
>> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>;
>> Ananyev, Konstantin <konstantin.ananyev@intel.com>;
>> gaetan.rivet@6wind.com; Wu, Jingjing <jingjing.wu@intel.com>;
>> thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van
>> Haaren, Harry <harry.van.haaren@intel.com>; Tan, Jianfeng
>> <jianfeng.tan@intel.com>
>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Guo,
>> Jia <jia.guo@intel.com>; Zhang, Helin <helin.zhang@intel.com>
>> Subject: [dpdk-dev] [PATCH V21 4/4] app/testpmd: show example to handle
>> hot unplug
>>
>> Use testpmd for example, to show how an application smoothly handle
>> failure when device being hot unplug. Once app detect the removal event,
>> the callback would be called, it first stop the packet forwarding, then stop the
>> port, close the port and finally detach the port.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v21->v20:
>> fix attach port issue, let it work for multiple device case.
>> combind rmv callback to only one.
>> ---
>> app/test-pmd/testpmd.c | 27 ++++++++++++++++++---------
>> 1 file changed, 18 insertions(+), 9 deletions(-)
>>
>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>> db23f23..a1ff8f3 100644
>> --- a/app/test-pmd/testpmd.c
>> +++ b/app/test-pmd/testpmd.c
>> @@ -1908,9 +1908,10 @@ eth_dev_event_callback_unregister(void)
>> void
>> attach_port(char *identifier)
>> {
>> - portid_t pi = 0;
>> unsigned int socket_id;
>>
>> + portid_t pi = rte_eth_dev_count_avail();
>> +
>> printf("Attaching a new port...\n");
>>
>> if (identifier == NULL) {
>> @@ -2071,12 +2072,14 @@ rmv_event_callback(void *arg)
>> RTE_ETH_VALID_PORTID_OR_RET(port_id);
>> dev = &rte_eth_devices[port_id];
>>
>> + if (dev->state == RTE_ETH_DEV_UNUSED)
>> + return;
>> +
>> + printf("removing device %s\n", dev->device->name);
>> + stop_packet_forwarding();
>> stop_port(port_id);
>> close_port(port_id);
>> - printf("removing device %s\n", dev->device->name);
>> - if (rte_eal_dev_detach(dev->device))
>> - TESTPMD_LOG(ERR, "Failed to detach device %s\n",
>> - dev->device->name);
>> + detach_port(port_id);
>> }
>>
>> /* This function is used by the interrupt thread */ @@ -2131,19 +2134,26
>> @@ static void eth_dev_event_callback(char *device_name, enum
>> rte_dev_event_type type,
>> __rte_unused void *arg)
>> {
>> + uint16_t port_id;
>> + int ret;
>> +
>> if (type >= RTE_DEV_EVENT_MAX) {
>> fprintf(stderr, "%s called upon invalid event %d\n",
>> __func__, type);
>> fflush(stderr);
>> }
>>
>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
>> + if (ret) {
>> + printf("can not get port by device %s!\n", device_name);
>> + return;
>> + }
>> +
>> switch (type) {
>> case RTE_DEV_EVENT_REMOVE:
>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>> device_name);
>> - /* TODO: After finish failure handle, begin to stop
>> - * packet forward, stop port, close port, detach port.
>> - */
>> + rmv_event_callback((void *)(intptr_t)port_id);
>> break;
>> case RTE_DEV_EVENT_ADD:
>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
>> @@ -2666,7 +2676,6 @@ main(int argc, char** argv)
>> return -1;
>> }
>> eth_dev_event_callback_register();
>> -
>> }
>>
>> if (start_port(RTE_PORT_ALL) != 0)
>> --
>> 2.7.4
> This patch fails to apply to dpdk 18.08-rc0 and needs to be rebased.
>
> Regards,
>
> Bernard.
thanks your notify, bernard, the coming next patch set will update to
fix it.
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH v2 0/4] hot plug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (5 preceding siblings ...)
2018-05-03 10:48 ` [PATCH V21 0/4] hot plug recovery mechanism Jeff Guo
@ 2018-06-22 11:51 ` Jeff Guo
2018-06-22 11:51 ` [PATCH v2 1/4] bus/pci: handle device hot unplug Jeff Guo
` (3 more replies)
2018-06-26 15:36 ` [PATCH V3 1/4] bus/pci: handle device " Jeff Guo
` (16 subsequent siblings)
23 siblings, 4 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-22 11:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event detect mechanism, failsafe driver,
bonding driver and hot plug/unplug api in framework, app could use these to
develop their hot plug solution.
let’s see the case of hot unplug, it can happen when a hardware device is
be removed physically, or when the software disables it. App need to call
ether dev API to detach the device, to unplug the device at the bus level and
make access to the device invalid. But the problem is that, the removal of the
device from the software lists is not going to be instantaneous, at this time
if the data(fast) path still read/write the device, it will cause MMIO error
and result of the app crash out.
Seems that we have got fail-safe driver(or app) + RTE_ETH_EVENT_INTR_RMV +
kernel core driver solution to handle it, but still not have failsafe driver
(or app) + RTE_DEV_EVENT_REMOVE + PCIe pmd driver failure handle solution. So
there is an absence in dpdk hot plug solution right now.
Also, we know that kernel only guaranty hot plug on the kernel side, but not for
the user mode side. Firstly we can hardly have a gatekeeper for any MMIO for
multiple PMD driver. Secondly, no more specific 3rd tools such as udev/driverctl
have especially cover these hot plug failure processing. Third, the feasibility
of app’s implement for multiple user mode PMD driver is still a problem. Here,
a general hot plug failure handle mechanism in dpdk framework would be proposed,
it aim to guaranty that, when hot unplug occur, the system will not crash and
app will not be break out, and user space can normally stop and release any
relevant resources, then unplug of the device at the bus level cleanly.
The mechanism should be come across as bellow:
Firstly, app enabled the device event monitor and register the hot plug event’s
callback before running data path. Once the hot unplug behave occur, the
mechanism will detect the removal event and then accordingly do the failure
handle. In order to do that, below functional will be bring in.
- Add a new bus ops “handle_hot_unplug” to handle bus read/write error, it is
bus-specific and each kind of bus can implement its own logic.
- Implement pci bus specific ops “pci_handle_hot_unplug”. It will base on the
failure address to remap memory for the corresponding device that unplugged.
For the data path or other unexpected control from the control path when hot
unplug occur.
- Implement a new sigbus handler, it is registered when start device even
monitoring. The handler is per process. Base on the signal event principle,
control path thread and data path thread will randomly receive the sigbus
error, but will go to the common sigbus handler. Once the MMIO sigbus error
exposure, it will trigger the above hot unplug operation. The sigbus will be
check if it is cause of the hot unplug or not, if not will info exception as
the original sigbus handler. If yes, will do memory remapping.
For the control path and the igb uio release:
- When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this resource,
it will cause kernel crash.
On the other hand, something like interrupt disable do not automatically
process in kernel side. If not handler it, this redundancy and dirty thing
will affect the interrupt resource be used by other device.
So the igb_uio driver have to check the hot plug status and corresponding
process should be taken in igb uio deriver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such as
probed/opened/released/removed/unplug. When detect the unexpected removal
which cause of hot unplug behavior, it will corresponding disable interrupt
resource, while for the part of releasement which kernel have already handle,
just skip it to avoid double free or null pointer kernel crash issue.
The mechanism could be use for fail-safe driver and app which want to use hot
plug solution. At this stage, will only use testpmd as reference to show how to
use the mechanism.
- Enable device event monitor->device unplug->failure handle->stop forwarding->
stop port->close port->detach port.
This process will not breaking the app/fail-safe running, and will not break
other irrelevance device. And app could plug in the device and restart the date
path again by below.
- Device plug in->bind igb_uio driver ->attached device->start port->
start forwarding.
patchset history:
v2->v1(v21):
refine some doc and commit log
fix igb uio kernel issue for control path failure
rebase testpmd code
Since the hot plug solution be discussed serval around in the public, the
scope be changed and the patch set be split into many times. Coming to the
recently RFC and feature design, it just focus on the hot unplug failure
handler at this patch set, so in order let this topic more clear and focus,
summarize privours patch set in history “v1(v21)”, the v2 here go ahead
for further track.
"v1(21)" == v21 as below:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (4):
bus/pci: handle device hot unplug
eal: add failure handle mechanism for hot plug
igb_uio: fix uio release issue when hot unplug
app/testpmd: show example to handle hot unplug
app/test-pmd/testpmd.c | 25 ++++--
drivers/bus/pci/pci_common.c | 65 ++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++
drivers/bus/pci/private.h | 12 +++
kernel/linux/igb_uio/igb_uio.c | 50 +++++++++--
lib/librte_eal/common/include/rte_bus.h | 16 ++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++-
7 files changed, 344 insertions(+), 11 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH v2 1/4] bus/pci: handle device hot unplug
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
@ 2018-06-22 11:51 ` Jeff Guo
2018-06-22 12:59 ` Gaëtan Rivet
2018-06-22 11:51 ` [PATCH v2 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
` (2 subsequent siblings)
3 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-22 11:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When a hardware device is removed physically or the software disables
it, the hot unplug occur. App need to call ether dev API to detach the
device, to unplug the device at the bus level and make access to the device
invalid. But the problem is that, the removal of the device from the
software lists is not going to be instantaneous, at this time if the data
path still read/write the device, it will cause MMIO error and result of
the app crash out. So a hot unplug handle mechanism need to guaranty app
will not crash out when hot unplug device.
To handle device hot unplug is bus-specific behavior, this patch introduces
a bus ops so that each kind of bus can implement its own logic. Further,
this patch implements the ops for PCI bus: remap a dummy memory to avoid
bus read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v2->v1(v21):
refind commit log
---
drivers/bus/pci/pci_common.c | 65 +++++++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
4 files changed, 126 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 7215aae..74d9aa8 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -472,6 +472,70 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+static struct rte_pci_device *
+pci_find_device_by_addr(void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(ERR, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+static int
+pci_handle_hot_unplug(struct rte_device *dev, void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ if (dev != NULL)
+ pdev = RTE_DEV_TO_PCI(dev);
+ else
+ pdev = pci_find_device_by_addr(failure_addr);
+
+ if (!pdev)
+ return -1;
+
+ /* remap resources for devices */
+ switch (pdev->kdrv) {
+ case RTE_KDRV_VFIO:
+#ifdef VFIO_PRESENT
+ /* TODO */
+ ret = -1;
+#endif
+ break;
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ if (ret != 0)
+ RTE_LOG(ERR, EAL, "Failed to handle hot unplug of device %s",
+ pdev->name);
+ return ret;
+}
+
static int
pci_plug(struct rte_device *dev)
{
@@ -502,6 +566,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .handle_hot_unplug = pci_handle_hot_unplug,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 88fa587..5551506 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..6a5609f 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hot unplug handler, which is responsible
+ * for handle the failure when hot unplug the device, guaranty the system
+ * would not hung in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
+ void *dev_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -209,6 +223,8 @@ struct rte_bus {
rte_bus_plug_t plug; /**< Probe single device for drivers */
rte_bus_unplug_t unplug; /**< Remove single device from driver */
rte_bus_parse_t parse; /**< Parse a device name */
+ rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
+ device event */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
};
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH v2 1/4] bus/pci: handle device hot unplug
2018-06-22 11:51 ` [PATCH v2 1/4] bus/pci: handle device hot unplug Jeff Guo
@ 2018-06-22 12:59 ` Gaëtan Rivet
2018-06-26 15:30 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Gaëtan Rivet @ 2018-06-22 12:59 UTC (permalink / raw)
To: Jeff Guo
Cc: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, jblunck, shreyansh.jain, dev, helin.zhang
Hi Jeff,
Sorry, I followed this development from afar,
I have a remark regarding this API, I think it can be made simpler.
Details below.
On Fri, Jun 22, 2018 at 07:51:05PM +0800, Jeff Guo wrote:
> When a hardware device is removed physically or the software disables
> it, the hot unplug occur. App need to call ether dev API to detach the
> device, to unplug the device at the bus level and make access to the device
> invalid. But the problem is that, the removal of the device from the
> software lists is not going to be instantaneous, at this time if the data
> path still read/write the device, it will cause MMIO error and result of
> the app crash out. So a hot unplug handle mechanism need to guaranty app
> will not crash out when hot unplug device.
>
> To handle device hot unplug is bus-specific behavior, this patch introduces
> a bus ops so that each kind of bus can implement its own logic. Further,
> this patch implements the ops for PCI bus: remap a dummy memory to avoid
> bus read/write error.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v2->v1(v21):
> refind commit log
> ---
> drivers/bus/pci/pci_common.c | 65 +++++++++++++++++++++++++++++++++
> drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++
> drivers/bus/pci/private.h | 12 ++++++
> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
> 4 files changed, 126 insertions(+)
>
> diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
> index 7215aae..74d9aa8 100644
> --- a/drivers/bus/pci/pci_common.c
> +++ b/drivers/bus/pci/pci_common.c
> @@ -472,6 +472,70 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
> return NULL;
> }
>
> +static struct rte_pci_device *
> +pci_find_device_by_addr(void *failure_addr)
> +{
> + struct rte_pci_device *pdev = NULL;
> + int i;
> +
> + FOREACH_DEVICE_ON_PCIBUS(pdev) {
> + for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
> + if ((uint64_t)(uintptr_t)failure_addr >=
> + (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
> + (uint64_t)(uintptr_t)failure_addr <
> + (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
> + pdev->mem_resource[i].len) {
> + RTE_LOG(ERR, EAL, "Failure address "
> + "%16.16"PRIx64" belongs to "
> + "device %s!\n",
> + (uint64_t)(uintptr_t)failure_addr,
> + pdev->device.name);
> + return pdev;
> + }
> + }
> + }
> + return NULL;
> +}
You define here a new bus ops that takes either an rte_device or an
arbitrary address as input.
In the uev handler that would call this ops afterward, you similarly try
to find either a bus using the device name, or then iterate over all
buses and try to find one able to handle the error.
This seems redundant and prone to ambiguity: should one check that the
device address is actually linked with the provided address? If not, is
it an improper call or a special case? This is unclear.
Note: I haven't followed the previous discussion, maybe the
dual dev_addr + failure_addr is warranted in the API here,
if so why not. Otherwise it just seems redundant:
the dev addr will never be within a physical BAR mapping,
and for all buses / drivers not using physical mappings,
the addr is only meaningful as a cue to find an internal
resource.
You can use the bus->find_device() to iterate over buses, and design
your bus ops such that when provided with an addr, would do whatever it
needs internally to find a relevant resource and either handle the
error, or return that the error was not handled.
Something like that:
/* new bus ops: */
/* If generic error codes are defined as part of the API,
>0 should mean that the sigbus was not handled,
<0 that an error occured but that one should stop trying,
0 that everything is ok.
*/
int (*handle_sigbus)(void *addr);
/* new rte_bus API: */
static int
bus_handle_sigbus(const struct rte_bus *bus,
const void *addr)
{
/* If additional error codes are defined as part of the API,
negative values should stop the iteration.
In this case, rte_errno would need to be set as well.
*/
return !(bus->handle_sigbus && bus->handle_sigbus(addr) <= 0);
}
int
rte_bus_sigbus_handler(void *addr)
{
struct rte_bus *bus;
int old_errno = rte_errno;
rte_errno = 0;
bus = rte_bus_find(NULL, bus_handle_sigbus, addr);
if (bus == NULL) {
/* ERROR: no bus could handle the error. */
RTE_LOG(ERR, EAL, "No bus was able to handle the error");
return -1;
} else if {rte_errno != 0) {
/* ERROR: a generic sigbus handling error. */
RTE_LOG(ERR, EAL, "Say what the error is");
return -1;
}
rte_errno = old_errno;
return 0;
}
Which would afterward be implemented, for example in PCI bus:
static rte_pci_device *
pci_find_device_by_addr(void *addr)
{
struct rte_pci_device *pdev;
FOREACH_DEVICE_ON_PCIBUS(pdev)
if (&pdev->device == addr ||
/* addr within mappings of pdev */)
return pdev;
return NULL;
}
static int
pci_handle_sigbus(void *addr)
{
static rte_pci_device *pdev;
pdev = pci_find_device_by_addr(addr);
if (pdev == NULL)
return -1;
/* Here, handle uio remap if needed. */
}
-------------------------
- Leave the bus iteration and check within rte_bus. Centralize the call
in a tighter rte_bus API, do not use directly your OPS from your other
EAL facility.
- You have left several error messages for signaling success (!), or
even simply that a bus could not handle a specific address. This is
bad. An error should only appear on error. Otherwise, all of this can
easily be traced using a debugger, so I don't think it's necessary
to leave it at a DEBUG level.
But in any case, remove all ERR-level messages for success.
<...>
> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
> index eb9eded..6a5609f 100644
> --- a/lib/librte_eal/common/include/rte_bus.h
> +++ b/lib/librte_eal/common/include/rte_bus.h
> @@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
> typedef int (*rte_bus_parse_t)(const char *name, void *addr);
>
> /**
> + * Implementation a specific hot unplug handler, which is responsible
> + * for handle the failure when hot unplug the device, guaranty the system
> + * would not hung in the case.
> + * @param dev
> + * Pointer of the device structure.
> + *
> + * @return
> + * 0 on success.
> + * !0 on error.
> + */
> +typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
> + void *dev_addr);
> +
I don't like the name of the OPS.
The documentation evokes only "the failure".
So is it a handle for any and all error possibly happening to a device?
If so, where is the input to describe the error?
If it is only meant to handle SIGBUS, because it is a very specific
error state only meant to happen on certain parts of the bus (the queue
mappings, if relevant), then it makes sense to only have an arbitrary
address as context for handling.
But then, it needs to be called as such. The expected failure to be
handled should be explicit in the name of the ops, and the documentation
should be more precise about what a bus developper should do with the
input.
> +/**
> * Bus scan policies
> */
> enum rte_bus_scan_mode {
> @@ -209,6 +223,8 @@ struct rte_bus {
> rte_bus_plug_t plug; /**< Probe single device for drivers */
> rte_bus_unplug_t unplug; /**< Remove single device from driver */
> rte_bus_parse_t parse; /**< Parse a device name */
> + rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
> + device event */
The new ops should be added at the end of the structure.
Regards,
--
Gaëtan Rivet
6WIND
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH v2 1/4] bus/pci: handle device hot unplug
2018-06-22 12:59 ` Gaëtan Rivet
@ 2018-06-26 15:30 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-06-26 15:30 UTC (permalink / raw)
To: Gaëtan Rivet
Cc: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, jblunck, shreyansh.jain, dev, helin.zhang
hi, gaetan,
thanks for your review, see comment as bellow
On 6/22/2018 8:59 PM, Gaëtan Rivet wrote:
> Hi Jeff,
>
> Sorry, I followed this development from afar,
> I have a remark regarding this API, I think it can be made simpler.
> Details below.
>
> On Fri, Jun 22, 2018 at 07:51:05PM +0800, Jeff Guo wrote:
>> When a hardware device is removed physically or the software disables
>> it, the hot unplug occur. App need to call ether dev API to detach the
>> device, to unplug the device at the bus level and make access to the device
>> invalid. But the problem is that, the removal of the device from the
>> software lists is not going to be instantaneous, at this time if the data
>> path still read/write the device, it will cause MMIO error and result of
>> the app crash out. So a hot unplug handle mechanism need to guaranty app
>> will not crash out when hot unplug device.
>>
>> To handle device hot unplug is bus-specific behavior, this patch introduces
>> a bus ops so that each kind of bus can implement its own logic. Further,
>> this patch implements the ops for PCI bus: remap a dummy memory to avoid
>> bus read/write error.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v2->v1(v21):
>> refind commit log
>> ---
>> drivers/bus/pci/pci_common.c | 65 +++++++++++++++++++++++++++++++++
>> drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++
>> drivers/bus/pci/private.h | 12 ++++++
>> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++
>> 4 files changed, 126 insertions(+)
>>
>> diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
>> index 7215aae..74d9aa8 100644
>> --- a/drivers/bus/pci/pci_common.c
>> +++ b/drivers/bus/pci/pci_common.c
>> @@ -472,6 +472,70 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
>> return NULL;
>> }
>>
>> +static struct rte_pci_device *
>> +pci_find_device_by_addr(void *failure_addr)
>> +{
>> + struct rte_pci_device *pdev = NULL;
>> + int i;
>> +
>> + FOREACH_DEVICE_ON_PCIBUS(pdev) {
>> + for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
>> + if ((uint64_t)(uintptr_t)failure_addr >=
>> + (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
>> + (uint64_t)(uintptr_t)failure_addr <
>> + (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
>> + pdev->mem_resource[i].len) {
>> + RTE_LOG(ERR, EAL, "Failure address "
>> + "%16.16"PRIx64" belongs to "
>> + "device %s!\n",
>> + (uint64_t)(uintptr_t)failure_addr,
>> + pdev->device.name);
>> + return pdev;
>> + }
>> + }
>> + }
>> + return NULL;
>> +}
> You define here a new bus ops that takes either an rte_device or an
> arbitrary address as input.
>
> In the uev handler that would call this ops afterward, you similarly try
> to find either a bus using the device name, or then iterate over all
> buses and try to find one able to handle the error.
>
> This seems redundant and prone to ambiguity: should one check that the
> device address is actually linked with the provided address? If not, is
> it an improper call or a special case? This is unclear.
>
> Note: I haven't followed the previous discussion, maybe the
> dual dev_addr + failure_addr is warranted in the API here,
> if so why not. Otherwise it just seems redundant:
> the dev addr will never be within a physical BAR mapping,
> and for all buses / drivers not using physical mappings,
> the addr is only meaningful as a cue to find an internal
> resource.
>
> You can use the bus->find_device() to iterate over buses, and design
> your bus ops such that when provided with an addr, would do whatever it
> needs internally to find a relevant resource and either handle the
> error, or return that the error was not handled.
if bus->find_device() can make think more simpler, why not? i will check
that and modify it.
> Something like that:
>
> /* new bus ops: */
> /* If generic error codes are defined as part of the API,
> >0 should mean that the sigbus was not handled,
> <0 that an error occured but that one should stop trying,
> 0 that everything is ok.
> */
> int (*handle_sigbus)(void *addr);
>
> /* new rte_bus API: */
>
> static int
> bus_handle_sigbus(const struct rte_bus *bus,
> const void *addr)
> {
> /* If additional error codes are defined as part of the API,
> negative values should stop the iteration.
> In this case, rte_errno would need to be set as well.
> */
> return !(bus->handle_sigbus && bus->handle_sigbus(addr) <= 0);
> }
>
> int
> rte_bus_sigbus_handler(void *addr)
> {
> struct rte_bus *bus;
> int old_errno = rte_errno;
>
> rte_errno = 0;
> bus = rte_bus_find(NULL, bus_handle_sigbus, addr);
> if (bus == NULL) {
> /* ERROR: no bus could handle the error. */
> RTE_LOG(ERR, EAL, "No bus was able to handle the error");
> return -1;
> } else if {rte_errno != 0) {
> /* ERROR: a generic sigbus handling error. */
> RTE_LOG(ERR, EAL, "Say what the error is");
> return -1;
> }
> rte_errno = old_errno;
> return 0;
> }
>
> Which would afterward be implemented, for example in PCI bus:
>
> static rte_pci_device *
> pci_find_device_by_addr(void *addr)
> {
> struct rte_pci_device *pdev;
>
> FOREACH_DEVICE_ON_PCIBUS(pdev)
> if (&pdev->device == addr ||
> /* addr within mappings of pdev */)
> return pdev;
> return NULL;
> }
>
> static int
> pci_handle_sigbus(void *addr)
> {
> static rte_pci_device *pdev;
>
> pdev = pci_find_device_by_addr(addr);
> if (pdev == NULL)
> return -1;
> /* Here, handle uio remap if needed. */
> }
>
> -------------------------
>
> - Leave the bus iteration and check within rte_bus. Centralize the call
> in a tighter rte_bus API, do not use directly your OPS from your other
> EAL facility.
>
> - You have left several error messages for signaling success (!), or
> even simply that a bus could not handle a specific address. This is
> bad. An error should only appear on error. Otherwise, all of this can
> easily be traced using a debugger, so I don't think it's necessary
> to leave it at a DEBUG level.
>
> But in any case, remove all ERR-level messages for success.
make sense. i will check which debug message is no need at least.
> <...>
>
>> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
>> index eb9eded..6a5609f 100644
>> --- a/lib/librte_eal/common/include/rte_bus.h
>> +++ b/lib/librte_eal/common/include/rte_bus.h
>> @@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
>> typedef int (*rte_bus_parse_t)(const char *name, void *addr);
>>
>> /**
>> + * Implementation a specific hot unplug handler, which is responsible
>> + * for handle the failure when hot unplug the device, guaranty the system
>> + * would not hung in the case.
>> + * @param dev
>> + * Pointer of the device structure.
>> + *
>> + * @return
>> + * 0 on success.
>> + * !0 on error.
>> + */
>> +typedef int (*rte_bus_handle_hot_unplug_t)(struct rte_device *dev,
>> + void *dev_addr);
>> +
> I don't like the name of the OPS.
> The documentation evokes only "the failure".
> So is it a handle for any and all error possibly happening to a device?
> If so, where is the input to describe the error?
> If it is only meant to handle SIGBUS, because it is a very specific
> error state only meant to happen on certain parts of the bus (the queue
> mappings, if relevant), then it makes sense to only have an arbitrary
> address as context for handling.
>
> But then, it needs to be called as such. The expected failure to be
> handled should be explicit in the name of the ops, and the documentation
> should be more precise about what a bus developper should do with the
> input.
I agree with your point of let the name more explicit, but i think here
maybe we should spit it into two ops, the one is hotplug_handler, the
other is sigbus_handler, because there are
2 path that both, data path and control path, they are also need to call
remap function when detect the hot remove event,even there are no sigbus
happen.
>> +/**
>> * Bus scan policies
>> */
>> enum rte_bus_scan_mode {
>> @@ -209,6 +223,8 @@ struct rte_bus {
>> rte_bus_plug_t plug; /**< Probe single device for drivers */
>> rte_bus_unplug_t unplug; /**< Remove single device from driver */
>> rte_bus_parse_t parse; /**< Parse a device name */
>> + rte_bus_handle_hot_unplug_t handle_hot_unplug; /**< handle hot unplug
>> + device event */
> The new ops should be added at the end of the structure.
ok.
> Regards,
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH v2 2/4] eal: add failure handle mechanism for hot plug
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
2018-06-22 11:51 ` [PATCH v2 1/4] bus/pci: handle device hot unplug Jeff Guo
@ 2018-06-22 11:51 ` Jeff Guo
2018-06-22 11:51 ` [PATCH v2 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-06-22 11:51 ` [PATCH v2 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-22 11:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot unplug event. When device be hot plug out, the device resource
become invalid, if this resource is still be unexpected read/write,
system will crash.
This patch let framework help application to handle this fault. When
sigbus error occur, check the failure address and accordingly remap
the invalid memory for the corresponding device, that could guaranty
the application not to be shut down when hot unplug devices.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v2->v1(v21):
refine commit log
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 154 +++++++++++++++++++++++++++++++++-
1 file changed, 153 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..3067f39 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,27 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
+static struct sigaction sigbus_action_old;
+
+static int sigbus_need_recover;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -34,6 +48,93 @@ enum eal_dev_event_subsystem {
};
static int
+dev_uev_failure_process(struct rte_device *dev, void *dev_addr)
+{
+ struct rte_bus *bus;
+ int ret = 0;
+
+ if (!dev && !dev_addr) {
+ return -EINVAL;
+ } else if (dev) {
+ bus = rte_bus_find_by_device_name(dev->name);
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret) {
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for device %s "
+ "on the bus %s.\n ",
+ dev->name, bus->name);
+ }
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug for bus %s!\n",
+ bus->name);
+ ret = -ENOTSUP;
+ }
+ } else {
+ TAILQ_FOREACH(bus, &rte_bus_list, next) {
+ if (bus->handle_hot_unplug) {
+ /**
+ * call bus ops to handle hot unplug.
+ */
+ ret = bus->handle_hot_unplug(dev, dev_addr);
+ if (ret)
+ RTE_LOG(ERR, EAL,
+ "Cannot handle hot unplug "
+ "for the device "
+ "on the bus %s!\n", bus->name);
+ else
+ break;
+ } else {
+ RTE_LOG(ERR, EAL,
+ "Not support handle hot unplug "
+ "for bus %s!\n", bus->name);
+ ret = -ENOTSUP;
+ }
+ }
+ }
+ return ret;
+}
+
+static void
+sigbus_action_recover(void)
+{
+ if (sigbus_need_recover) {
+ sigaction(SIGBUS, &sigbus_action_old, NULL);
+ sigbus_need_recover = 0;
+ }
+}
+
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(NULL, info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (!ret)
+ RTE_LOG(DEBUG, EAL,
+ "Success to handle SIGBUS error for hot unplug!\n");
+ else
+ rte_exit(EXIT_FAILURE, "exit for SIGBUS error!");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
+static int
dev_uev_socket_fd_create(void)
{
struct sockaddr_nl addr;
@@ -147,6 +248,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +275,50 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ uevent.devname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL,
+ "Cannot find unplugged device (%s)\n",
+ uevent.devname);
+ return;
+ }
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = dev_uev_failure_process(dev, NULL);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Driver cannot remap the "
+ "device (%s)\n",
+ dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +338,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
+
monitor_started = true;
return 0;
@@ -217,8 +366,11 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ sigbus_action_recover();
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v2 3/4] igb_uio: fix uio release issue when hot unplug
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
2018-06-22 11:51 ` [PATCH v2 1/4] bus/pci: handle device hot unplug Jeff Guo
2018-06-22 11:51 ` [PATCH v2 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-06-22 11:51 ` Jeff Guo
2018-06-22 11:51 ` [PATCH v2 4/4] app/testpmd: show example to handle " Jeff Guo
3 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-22 11:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this
resource, it will cause kernel crash. On the other hand, something like
interrupt disabling do not automatically process in kernel side. If not
handler it, this redundancy and dirty thing will affect the interrupt
resource be used by other device. So the igb_uio driver have to check the
hot plug status, and the corresponding process should be taken in igb uio
driver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such
as probed/opened/released/removed/unplug. When detect the unexpected
removal which cause of hot unplug behavior, it will corresponding disable
interrupt resource, while for the part of releasement which kernel have
already handle, just skip it to avoid double free or null pointer kernel
crash issue.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v2->v1(v21):
add uio device state to check hot plug unexpected removmal.
fix igb uio kernel driver issue.
---
kernel/linux/igb_uio/igb_uio.c | 50 +++++++++++++++++++++++++++++++++++++-----
1 file changed, 45 insertions(+), 5 deletions(-)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index cd9b7e7..fdd692a 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -19,6 +19,15 @@
#include "compat.h"
+/* uio pci device state */
+enum rte_udev_state {
+ RTE_UDEV_PROBED,
+ RTE_UDEV_OPENNED,
+ RTE_UDEV_RELEASED,
+ RTE_UDEV_REMOVED,
+ RTE_UDEV_UNPLUG
+};
+
/**
* A structure describing the private information for a uio device.
*/
@@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
enum rte_intr_mode mode;
struct mutex lock;
int refcnt;
+ enum rte_udev_state state;
};
static char *intr_mode;
@@ -194,12 +204,20 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
{
struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
struct uio_info *info = &udev->info;
+ struct pci_dev *pdev = udev->pdev;
/* Legacy mode need to mask in hardware */
if (udev->mode == RTE_INTR_MODE_LEGACY &&
!pci_check_and_mask_intx(udev->pdev))
return IRQ_NONE;
+ /* check the uevent of the kobj */
+ if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
+ dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
+ (&pdev->dev.kobj)->name);
+ udev->state = RTE_UDEV_UNPLUG;
+ }
+
uio_event_notify(info);
/* Message signal mode, no share IRQ and automasked */
@@ -308,7 +326,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
#endif
}
-
/**
* This gets called while opening uio device file.
*/
@@ -330,24 +347,33 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
/* enable interrupts */
err = igbuio_pci_enable_interrupts(udev);
- mutex_unlock(&udev->lock);
if (err) {
dev_err(&dev->dev, "Enable interrupt fails\n");
+ pci_clear_master(dev);
return err;
}
+ udev->state = RTE_UDEV_OPENNED;
+ mutex_unlock(&udev->lock);
return 0;
}
+/**
+ * This gets called while closing uio device file.
+ */
static int
igbuio_pci_release(struct uio_info *info, struct inode *inode)
{
+
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ if (udev->state == RTE_UDEV_REMOVED)
+ return 0;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
- return 0;
+ return -1;
}
/* disable interrupts */
@@ -355,7 +381,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
/* stop the device from further DMA */
pci_clear_master(dev);
-
+ udev->state = RTE_UDEV_RELEASED;
mutex_unlock(&udev->lock);
return 0;
}
@@ -557,6 +583,7 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
(unsigned long long)map_dma_addr, map_addr);
}
+ udev->state = RTE_UDEV_PROBED;
return 0;
fail_remove_group:
@@ -573,11 +600,24 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
static void
igbuio_pci_remove(struct pci_dev *dev)
{
+
struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
+ int ret;
+
+ /* handler hot unplug */
+ if (udev->state == RTE_UDEV_OPENNED ||
+ udev->state == RTE_UDEV_UNPLUG) {
+ dev_notice(&dev->dev, "Unexpected removal!\n");
+ ret = igbuio_pci_release(&udev->info, NULL);
+ if (ret)
+ return;
+ udev->state = RTE_UDEV_REMOVED;
+ return;
+ }
mutex_destroy(&udev->lock);
- sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
uio_unregister_device(&udev->info);
+ sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
igbuio_pci_release_iomem(&udev->info);
pci_disable_device(dev);
pci_set_drvdata(dev, NULL);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
` (2 preceding siblings ...)
2018-06-22 11:51 ` [PATCH v2 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-06-22 11:51 ` Jeff Guo
2018-06-26 10:06 ` Iremonger, Bernard
2018-06-26 11:58 ` Matan Azrad
3 siblings, 2 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-22 11:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. If app have enabled the device event
monitor and register the hot plug event’s callback before running, once
app detect the removal event, the callback would be called. It will first
stop the packet forwarding, then stop the port, close the port, and finally
detach the port to remove the device out from the device lists.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v2->v1(v21):
rebase testpmd code
---
app/test-pmd/testpmd.c | 25 ++++++++++++++++++++-----
1 file changed, 20 insertions(+), 5 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 24c1998..286f242 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
void
attach_port(char *identifier)
{
- portid_t pi = 0;
unsigned int socket_id;
+ portid_t pi = rte_eth_dev_count_avail();
+
printf("Attaching a new port...\n");
if (identifier == NULL) {
@@ -2125,16 +2126,25 @@ check_all_ports_link_status(uint32_t port_mask)
static void
rmv_event_callback(void *arg)
{
+ struct rte_eth_dev *dev;
+
int need_to_start = 0;
int org_no_link_check = no_link_check;
portid_t port_id = (intptr_t)arg;
RTE_ETH_VALID_PORTID_OR_RET(port_id);
+ dev = &rte_eth_devices[port_id];
+
+ if (dev->state == RTE_ETH_DEV_UNUSED)
+ return;
+
+ printf("removing device %s\n", dev->device->name);
if (!test_done && port_is_forwarding(port_id)) {
need_to_start = 1;
stop_packet_forwarding();
}
+
no_link_check = 1;
stop_port(port_id);
no_link_check = org_no_link_check;
@@ -2196,6 +2206,9 @@ static void
eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
__rte_unused void *arg)
{
+ uint16_t port_id;
+ int ret;
+
if (type >= RTE_DEV_EVENT_MAX) {
fprintf(stderr, "%s called upon invalid event %d\n",
__func__, type);
@@ -2206,9 +2219,12 @@ eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", device_name);
+ return;
+ }
+ rmv_event_callback((void *)(intptr_t)port_id);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2736,7 +2752,6 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
2018-06-22 11:51 ` [PATCH v2 4/4] app/testpmd: show example to handle " Jeff Guo
@ 2018-06-26 10:06 ` Iremonger, Bernard
2018-06-26 11:58 ` Matan Azrad
1 sibling, 0 replies; 494+ messages in thread
From: Iremonger, Bernard @ 2018-06-26 10:06 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Guo, Jia, Zhang, Helin
> -----Original Message-----
> From: dev [mailto:dev-bounces@dpdk.org] On Behalf Of Jeff Guo
> Sent: Friday, June 22, 2018 12:51 PM
> To: stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; Ananyev,
> Konstantin <konstantin.ananyev@intel.com>; gaetan.rivet@6wind.com; Wu,
> Jingjing <jingjing.wu@intel.com>; thomas@monjalon.net;
> motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
> <harry.van.haaren@intel.com>; Zhang, Qi Z <qi.z.zhang@intel.com>; He,
> Shaopeng <shaopeng.he@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Guo, Jia
> <jia.guo@intel.com>; Zhang, Helin <helin.zhang@intel.com>
> Subject: [dpdk-dev] [PATCH v2 4/4] app/testpmd: show example to handle hot
> unplug
>
> Use testpmd for example, to show how an application smoothly handle failure
> when device being hot unplug. If app have enabled the device event monitor and
> register the hot plug event’s callback before running, once app detect the
> removal event, the callback would be called. It will first stop the packet
> forwarding, then stop the port, close the port, and finally detach the port to
> remove the device out from the device lists.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
Acked-by: Bernard Iremonger <bernard.iremonger@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
2018-06-22 11:51 ` [PATCH v2 4/4] app/testpmd: show example to handle " Jeff Guo
2018-06-26 10:06 ` Iremonger, Bernard
@ 2018-06-26 11:58 ` Matan Azrad
2018-06-26 15:33 ` Guo, Jia
1 sibling, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-06-26 11:58 UTC (permalink / raw)
To: Jeff Guo, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
qi.z.zhang@intel.com, shaopeng.he@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Jeff
Please see comments...
From: Jeff Guo
> Sent: Friday, June 22, 2018 2:51 PM
> To: stephen@networkplumber.org; bruce.richardson@intel.com;
> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
> qi.z.zhang@intel.com; shaopeng.he@intel.com
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
> jia.guo@intel.com; helin.zhang@intel.com
> Subject: [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
>
> Use testpmd for example, to show how an application smoothly handle failure
> when device being hot unplug. If app have enabled the device event monitor
> and register the hot plug event’s callback before running, once app detect the
> removal event, the callback would be called. It will first stop the packet
> forwarding, then stop the port, close the port, and finally detach the port to
> remove the device out from the device lists.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v2->v1(v21):
> rebase testpmd code
> ---
> app/test-pmd/testpmd.c | 25 ++++++++++++++++++++-----
> 1 file changed, 20 insertions(+), 5 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> 24c1998..286f242 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
> void
> attach_port(char *identifier)
> {
> - portid_t pi = 0;
> unsigned int socket_id;
>
> + portid_t pi = rte_eth_dev_count_avail();
> +
> printf("Attaching a new port...\n");
>
> if (identifier == NULL) {
> @@ -2125,16 +2126,25 @@ check_all_ports_link_status(uint32_t port_mask)
> static void rmv_event_callback(void *arg) {
There is a race between ethdev RMV event to the EAL remove event, I think the application must synchronize it if both are configured.
> + struct rte_eth_dev *dev;
> +
> int need_to_start = 0;
> int org_no_link_check = no_link_check;
> portid_t port_id = (intptr_t)arg;
>
> RTE_ETH_VALID_PORTID_OR_RET(port_id);
> + dev = &rte_eth_devices[port_id];
> +
> + if (dev->state == RTE_ETH_DEV_UNUSED)
> + return;
Can you explain why do you check the state?
Doesn't RTE_ETH_VALID_PORTID_OR_RET do it?
> + printf("removing device %s\n", dev->device->name);
>
> if (!test_done && port_is_forwarding(port_id)) {
> need_to_start = 1;
> stop_packet_forwarding();
> }
> +
> no_link_check = 1;
> stop_port(port_id);
> no_link_check = org_no_link_check;
> @@ -2196,6 +2206,9 @@ static void
> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
> __rte_unused void *arg)
> {
> + uint16_t port_id;
> + int ret;
> +
> if (type >= RTE_DEV_EVENT_MAX) {
> fprintf(stderr, "%s called upon invalid event %d\n",
> __func__, type);
> @@ -2206,9 +2219,12 @@ eth_dev_event_callback(char *device_name, enum
> rte_dev_event_type type,
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> + if (ret) {
> + printf("can not get port by device %s!\n",
> device_name);
> + return;
> + }
> + rmv_event_callback((void *)(intptr_t)port_id);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
> 2736,7 +2752,6 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
2018-06-26 11:58 ` Matan Azrad
@ 2018-06-26 15:33 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-06-26 15:33 UTC (permalink / raw)
To: Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Thomas Monjalon, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
hi, matan
thanks for your review, see comment.
On 6/26/2018 7:58 PM, Matan Azrad wrote:
> Hi Jeff
>
> Please see comments...
>
> From: Jeff Guo
>> Sent: Friday, June 22, 2018 2:51 PM
>> To: stephen@networkplumber.org; bruce.richardson@intel.com;
>> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
>> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
>> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
>> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
>> qi.z.zhang@intel.com; shaopeng.he@intel.com
>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
>> jia.guo@intel.com; helin.zhang@intel.com
>> Subject: [PATCH v2 4/4] app/testpmd: show example to handle hot unplug
>>
>> Use testpmd for example, to show how an application smoothly handle failure
>> when device being hot unplug. If app have enabled the device event monitor
>> and register the hot plug event’s callback before running, once app detect the
>> removal event, the callback would be called. It will first stop the packet
>> forwarding, then stop the port, close the port, and finally detach the port to
>> remove the device out from the device lists.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v2->v1(v21):
>> rebase testpmd code
>> ---
>> app/test-pmd/testpmd.c | 25 ++++++++++++++++++++-----
>> 1 file changed, 20 insertions(+), 5 deletions(-)
>>
>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>> 24c1998..286f242 100644
>> --- a/app/test-pmd/testpmd.c
>> +++ b/app/test-pmd/testpmd.c
>> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
>> void
>> attach_port(char *identifier)
>> {
>> - portid_t pi = 0;
>> unsigned int socket_id;
>>
>> + portid_t pi = rte_eth_dev_count_avail();
>> +
>> printf("Attaching a new port...\n");
>>
>> if (identifier == NULL) {
>> @@ -2125,16 +2126,25 @@ check_all_ports_link_status(uint32_t port_mask)
>> static void rmv_event_callback(void *arg) {
> There is a race between ethdev RMV event to the EAL remove event, I think the application must synchronize it if both are configured.
Is this race will affect the device detaching? what is the side effect
and what is your propose to synchronize it, and i still think about
that.....
>> + struct rte_eth_dev *dev;
>> +
>> int need_to_start = 0;
>> int org_no_link_check = no_link_check;
>> portid_t port_id = (intptr_t)arg;
>>
>> RTE_ETH_VALID_PORTID_OR_RET(port_id);
>> + dev = &rte_eth_devices[port_id];
>> +
>> + if (dev->state == RTE_ETH_DEV_UNUSED)
>> + return;
> Can you explain why do you check the state?
> Doesn't RTE_ETH_VALID_PORTID_OR_RET do it?
correct, i check that it is no used here. thank info.
>> + printf("removing device %s\n", dev->device->name);
>>
>> if (!test_done && port_is_forwarding(port_id)) {
>> need_to_start = 1;
>> stop_packet_forwarding();
>> }
>> +
>> no_link_check = 1;
>> stop_port(port_id);
>> no_link_check = org_no_link_check;
>> @@ -2196,6 +2206,9 @@ static void
>> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
>> __rte_unused void *arg)
>> {
>> + uint16_t port_id;
>> + int ret;
>> +
>> if (type >= RTE_DEV_EVENT_MAX) {
>> fprintf(stderr, "%s called upon invalid event %d\n",
>> __func__, type);
>> @@ -2206,9 +2219,12 @@ eth_dev_event_callback(char *device_name, enum
>> rte_dev_event_type type,
>> case RTE_DEV_EVENT_REMOVE:
>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>> device_name);
>> - /* TODO: After finish failure handle, begin to stop
>> - * packet forward, stop port, close port, detach port.
>> - */
>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
>> + if (ret) {
>> + printf("can not get port by device %s!\n",
>> device_name);
>> + return;
>> + }
>> + rmv_event_callback((void *)(intptr_t)port_id);
>> break;
>> case RTE_DEV_EVENT_ADD:
>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
>> 2736,7 +2752,6 @@ main(int argc, char** argv)
>> return -1;
>> }
>> eth_dev_event_callback_register();
>> -
>> }
>>
>> if (start_port(RTE_PORT_ALL) != 0)
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V3 1/4] bus/pci: handle device hot unplug
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (6 preceding siblings ...)
2018-06-22 11:51 ` [PATCH v2 0/4] hot plug failure handle mechanism Jeff Guo
@ 2018-06-26 15:36 ` Jeff Guo
2018-06-26 15:36 ` [PATCH V3 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
` (2 more replies)
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (15 subsequent siblings)
23 siblings, 3 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-26 15:36 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When a hardware device is removed physically or the software disables
it, the hot unplug occur. App need to call ether dev API to detach the
device, to unplug the device at the bus level and make access to the device
invalid. But the problem is that, the removal of the device from the
software lists is not going to be instantaneous, at this time if the data
path still read/write the device, it will cause MMIO error and result of
the app crash out. So a hot unplug handle mechanism need to guaranty app
will not crash out when hot unplug device.
To handle device hot unplug is bus-specific behavior, this patch introduces
a bus ops so that each kind of bus can implement its own logic. Further,
this patch implements the ops for PCI bus: remap a dummy memory to avoid
bus read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v3->v2:
change bus ops name to bus_hotplug_handler.
---
drivers/bus/pci/pci_common.c | 34 +++++++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 ++++++++++++++++++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++++++++
lib/librte_eal/common/include/rte_bus.h | 14 ++++++++++++++
4 files changed, 93 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 7215aae..e607d08 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -473,6 +473,39 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
}
static int
+pci_hotplug_handler(struct rte_device *dev)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = RTE_DEV_TO_PCI(dev);
+ if (!pdev)
+ return -1;
+
+ /* remap resources for devices */
+ switch (pdev->kdrv) {
+ case RTE_KDRV_VFIO:
+#ifdef VFIO_PRESENT
+ /* TODO */
+ ret = -1;
+#endif
+ break;
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -502,6 +535,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .hotplug_handler = pci_hotplug_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 88fa587..5551506 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -173,6 +173,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..6507f24 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,19 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hot plug handler, which is responsible
+ * for handle the failure when hot remove the device, guaranty the system
+ * would not crash in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -211,6 +224,7 @@ struct rte_bus {
rte_bus_parse_t parse; /**< Parse a device name */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
+ rte_bus_hotplug_handler_t hotplug_handler; /**< handle hot plug on bus */
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V3 2/4] eal: add failure handle mechanism for hot plug
2018-06-26 15:36 ` [PATCH V3 1/4] bus/pci: handle device " Jeff Guo
@ 2018-06-26 15:36 ` Jeff Guo
2018-06-26 15:36 ` [PATCH V3 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-06-26 15:36 ` [PATCH V3 4/4] app/testpmd: show example to handle " Jeff Guo
2 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-26 15:36 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot unplug event. When device be hot plug out, the device resource
become invalid, if this resource is still be unexpected read/write,
system will crash.
This patch let framework help application to handle this fault. When
sigbus error occur, check the failure address and accordingly remap
the invalid memory for the corresponding device, that could guaranty
the application not to be shut down when hot unplug devices.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v3->v2:
add new API and bus ops of bus_signal_handler
distingush handle generic sigbus and hotplug sigbus
---
drivers/bus/pci/pci_common.c | 53 ++++++++++++++++++++
lib/librte_eal/common/eal_common_bus.c | 34 ++++++++++++-
lib/librte_eal/common/include/rte_bus.h | 19 +++++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++-
4 files changed, 192 insertions(+), 2 deletions(-)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index e607d08..4c0ac98 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -472,6 +472,32 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+/* check the failure address belongs to which device. */
+static struct rte_pci_device *
+pci_find_device_by_addr(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(INFO, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+
static int
pci_hotplug_handler(struct rte_device *dev)
{
@@ -506,6 +532,32 @@ pci_hotplug_handler(struct rte_device *dev)
}
static int
+pci_sigbus_handler(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = pci_find_device_by_addr(failure_addr);
+ if (!pdev) {
+ /* not found the device which is illegal access in MMIO,
+ * so it is a generic sigbus error.
+ */
+ ret = 1;
+ }
+
+ /* handle hotplug when sigbus error is caused of hot removal */
+ ret = pci_hotplug_handler(&pdev->device);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Failed to handle hot plug for device %s",
+ pdev->name);
+ ret = -1;
+ rte_errno = -1;
+ }
+
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -536,6 +588,7 @@ struct rte_pci_bus rte_pci_bus = {
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
.hotplug_handler = pci_hotplug_handler,
+ .sigbus_handler = pci_sigbus_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/lib/librte_eal/common/eal_common_bus.c b/lib/librte_eal/common/eal_common_bus.c
index 0943851..b505b9b 100644
--- a/lib/librte_eal/common/eal_common_bus.c
+++ b/lib/librte_eal/common/eal_common_bus.c
@@ -37,6 +37,7 @@
#include <rte_bus.h>
#include <rte_debug.h>
#include <rte_string_fns.h>
+#include <rte_errno.h>
#include "eal_private.h"
@@ -220,7 +221,6 @@ rte_bus_find_by_device_name(const char *str)
return rte_bus_find(NULL, bus_can_parse, name);
}
-
/*
* Get iommu class of devices on the bus.
*/
@@ -242,3 +242,35 @@ rte_bus_get_iommu_class(void)
}
return mode;
}
+
+static int
+bus_handle_sigbus(const struct rte_bus *bus,
+ const void *failure_addr)
+{
+ return !(bus->sigbus_handler && bus->sigbus_handler(failure_addr) <= 0);
+}
+
+int
+rte_bus_sigbus_handler(const void *failure_addr)
+{
+ struct rte_bus *bus;
+ int old_errno = rte_errno;
+ int no_handle = 0;
+
+ rte_errno = 0;
+
+ bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
+ no_handle = 1;
+ } else if (rte_errno != 0) {
+ RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
+ no_handle = 1;
+ }
+
+ /* if sigbus not be handled, return back old errno. */
+ if (no_handle)
+ rte_errno = old_errno;
+
+ return no_handle;
+}
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index 6507f24..4389c42 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -181,6 +181,19 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
/**
+ * Implementation a specific sigbus handler, which is responsible
+ * for handle the sigbus error which is original memory error, or specific
+ * memory error that caused of hot unplug.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -225,6 +238,7 @@ struct rte_bus {
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
rte_bus_hotplug_handler_t hotplug_handler; /**< handle hot plug on bus */
+ rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
};
/**
@@ -335,6 +349,11 @@ struct rte_bus *rte_bus_find_by_name(const char *busname);
enum rte_iova_mode rte_bus_get_iommu_class(void);
/**
+ * Handle the sigbus error on corresponding bus.
+ */
+int rte_bus_sigbus_handler(const void* failure_addr);
+
+/**
* Helper for Bus registration.
* The constructor has higher priority than PMD constructors.
*/
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..c9dddab 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,24 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
+#include <rte_errno.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -33,6 +44,34 @@ enum eal_dev_event_subsystem {
EAL_DEV_EVENT_SUBSYSTEM_MAX
};
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = rte_bus_sigbus_handler(info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (!ret)
+ RTE_LOG(INFO, EAL,
+ "Success to handle SIGBUS error for hotplug!\n");
+ else
+ rte_exit(EXIT_FAILURE,
+ "A generic SIGBUS error, (rte_errno: %s)!",
+ strerror(rte_errno));
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
static int
dev_uev_socket_fd_create(void)
{
@@ -147,6 +186,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +213,48 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ busname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
+ "bus (%s)\n", uevent.devname, busname);
+ return;
+ }
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = bus->hotplug_handler(dev);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Can not handle hotplug for "
+ "device (%s)\n", dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +274,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigaction(SIGBUS, &action, NULL);
+
monitor_started = true;
return 0;
@@ -220,5 +305,6 @@ rte_dev_event_monitor_stop(void)
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V3 3/4] igb_uio: fix uio release issue when hot unplug
2018-06-26 15:36 ` [PATCH V3 1/4] bus/pci: handle device " Jeff Guo
2018-06-26 15:36 ` [PATCH V3 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-06-26 15:36 ` Jeff Guo
2018-06-26 15:36 ` [PATCH V3 4/4] app/testpmd: show example to handle " Jeff Guo
2 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-26 15:36 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this
resource, it will cause kernel crash. On the other hand, something like
interrupt disabling do not automatically process in kernel side. If not
handler it, this redundancy and dirty thing will affect the interrupt
resource be used by other device. So the igb_uio driver have to check the
hot plug status, and the corresponding process should be taken in igb uio
driver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such
as probed/opened/released/removed/unplug. When detect the unexpected
removal which cause of hot unplug behavior, it will corresponding disable
interrupt resource, while for the part of releasement which kernel have
already handle, just skip it to avoid double free or null pointer kernel
crash issue.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v3->v2:
no change.
---
kernel/linux/igb_uio/igb_uio.c | 50 +++++++++++++++++++++++++++++++++++++-----
1 file changed, 45 insertions(+), 5 deletions(-)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index cd9b7e7..fdd692a 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -19,6 +19,15 @@
#include "compat.h"
+/* uio pci device state */
+enum rte_udev_state {
+ RTE_UDEV_PROBED,
+ RTE_UDEV_OPENNED,
+ RTE_UDEV_RELEASED,
+ RTE_UDEV_REMOVED,
+ RTE_UDEV_UNPLUG
+};
+
/**
* A structure describing the private information for a uio device.
*/
@@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
enum rte_intr_mode mode;
struct mutex lock;
int refcnt;
+ enum rte_udev_state state;
};
static char *intr_mode;
@@ -194,12 +204,20 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
{
struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
struct uio_info *info = &udev->info;
+ struct pci_dev *pdev = udev->pdev;
/* Legacy mode need to mask in hardware */
if (udev->mode == RTE_INTR_MODE_LEGACY &&
!pci_check_and_mask_intx(udev->pdev))
return IRQ_NONE;
+ /* check the uevent of the kobj */
+ if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
+ dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
+ (&pdev->dev.kobj)->name);
+ udev->state = RTE_UDEV_UNPLUG;
+ }
+
uio_event_notify(info);
/* Message signal mode, no share IRQ and automasked */
@@ -308,7 +326,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
#endif
}
-
/**
* This gets called while opening uio device file.
*/
@@ -330,24 +347,33 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
/* enable interrupts */
err = igbuio_pci_enable_interrupts(udev);
- mutex_unlock(&udev->lock);
if (err) {
dev_err(&dev->dev, "Enable interrupt fails\n");
+ pci_clear_master(dev);
return err;
}
+ udev->state = RTE_UDEV_OPENNED;
+ mutex_unlock(&udev->lock);
return 0;
}
+/**
+ * This gets called while closing uio device file.
+ */
static int
igbuio_pci_release(struct uio_info *info, struct inode *inode)
{
+
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ if (udev->state == RTE_UDEV_REMOVED)
+ return 0;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
- return 0;
+ return -1;
}
/* disable interrupts */
@@ -355,7 +381,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
/* stop the device from further DMA */
pci_clear_master(dev);
-
+ udev->state = RTE_UDEV_RELEASED;
mutex_unlock(&udev->lock);
return 0;
}
@@ -557,6 +583,7 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
(unsigned long long)map_dma_addr, map_addr);
}
+ udev->state = RTE_UDEV_PROBED;
return 0;
fail_remove_group:
@@ -573,11 +600,24 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
static void
igbuio_pci_remove(struct pci_dev *dev)
{
+
struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
+ int ret;
+
+ /* handler hot unplug */
+ if (udev->state == RTE_UDEV_OPENNED ||
+ udev->state == RTE_UDEV_UNPLUG) {
+ dev_notice(&dev->dev, "Unexpected removal!\n");
+ ret = igbuio_pci_release(&udev->info, NULL);
+ if (ret)
+ return;
+ udev->state = RTE_UDEV_REMOVED;
+ return;
+ }
mutex_destroy(&udev->lock);
- sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
uio_unregister_device(&udev->info);
+ sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
igbuio_pci_release_iomem(&udev->info);
pci_disable_device(dev);
pci_set_drvdata(dev, NULL);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
2018-06-26 15:36 ` [PATCH V3 1/4] bus/pci: handle device " Jeff Guo
2018-06-26 15:36 ` [PATCH V3 2/4] eal: add failure handle mechanism for hot plug Jeff Guo
2018-06-26 15:36 ` [PATCH V3 3/4] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-06-26 15:36 ` Jeff Guo
2018-06-26 17:07 ` Matan Azrad
2 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-26 15:36 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. If app have enabled the device event
monitor and register the hot plug event’s callback before running, once
app detect the removal event, the callback would be called. It will first
stop the packet forwarding, then stop the port, close the port, and finally
detach the port to remove the device out from the device lists.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v3->v2:
delete some unused check
---
app/test-pmd/testpmd.c | 22 +++++++++++++++++-----
1 file changed, 17 insertions(+), 5 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 24c1998..2ee5621 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
void
attach_port(char *identifier)
{
- portid_t pi = 0;
unsigned int socket_id;
+ portid_t pi = rte_eth_dev_count_avail();
+
printf("Attaching a new port...\n");
if (identifier == NULL) {
@@ -2125,16 +2126,22 @@ check_all_ports_link_status(uint32_t port_mask)
static void
rmv_event_callback(void *arg)
{
+ struct rte_eth_dev *dev;
+
int need_to_start = 0;
int org_no_link_check = no_link_check;
portid_t port_id = (intptr_t)arg;
RTE_ETH_VALID_PORTID_OR_RET(port_id);
+ dev = &rte_eth_devices[port_id];
+
+ printf("removing device %s\n", dev->device->name);
if (!test_done && port_is_forwarding(port_id)) {
need_to_start = 1;
stop_packet_forwarding();
}
+
no_link_check = 1;
stop_port(port_id);
no_link_check = org_no_link_check;
@@ -2196,6 +2203,9 @@ static void
eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
__rte_unused void *arg)
{
+ uint16_t port_id;
+ int ret;
+
if (type >= RTE_DEV_EVENT_MAX) {
fprintf(stderr, "%s called upon invalid event %d\n",
__func__, type);
@@ -2206,9 +2216,12 @@ eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", device_name);
+ return;
+ }
+ rmv_event_callback((void *)(intptr_t)port_id);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2736,7 +2749,6 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
2018-06-26 15:36 ` [PATCH V3 4/4] app/testpmd: show example to handle " Jeff Guo
@ 2018-06-26 17:07 ` Matan Azrad
2018-06-27 3:56 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-06-26 17:07 UTC (permalink / raw)
To: Jeff Guo, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
qi.z.zhang@intel.com, shaopeng.he@intel.com,
bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Jeff
Continue session from last version + more comments\question.
From: Jeff Guo
> Sent: Tuesday, June 26, 2018 6:36 PM
> To: stephen@networkplumber.org; bruce.richardson@intel.com;
> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
> qi.z.zhang@intel.com; shaopeng.he@intel.com; bernard.iremonger@intel.com
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
> jia.guo@intel.com; helin.zhang@intel.com
> Subject: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
>
> Use testpmd for example, to show how an application smoothly handle failure
> when device being hot unplug. If app have enabled the device event monitor
> and register the hot plug event’s callback before running, once app detect the
> removal event, the callback would be called. It will first stop the packet
> forwarding, then stop the port, close the port, and finally detach the port to
> remove the device out from the device lists.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v3->v2:
> delete some unused check
> ---
> app/test-pmd/testpmd.c | 22 +++++++++++++++++-----
> 1 file changed, 17 insertions(+), 5 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> 24c1998..2ee5621 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
> void
> attach_port(char *identifier)
> {
> - portid_t pi = 0;
> unsigned int socket_id;
>
> + portid_t pi = rte_eth_dev_count_avail();
> +
I don't understand this change... can you explain?
> printf("Attaching a new port...\n");
>
> if (identifier == NULL) {
> @@ -2125,16 +2126,22 @@ check_all_ports_link_status(uint32_t port_mask)
> static void rmv_event_callback(void *arg) {
> + struct rte_eth_dev *dev;
> +
> int need_to_start = 0;
> int org_no_link_check = no_link_check;
> portid_t port_id = (intptr_t)arg;
>
> RTE_ETH_VALID_PORTID_OR_RET(port_id);
> + dev = &rte_eth_devices[port_id];
> +
> + printf("removing device %s\n", dev->device->name);
>
> if (!test_done && port_is_forwarding(port_id)) {
> need_to_start = 1;
> stop_packet_forwarding();
> }
> +
I don't think you need to change anything in this function.
You can add the print in the caller code.
> no_link_check = 1;
> stop_port(port_id);
> no_link_check = org_no_link_check;
Suggestion for synchronization:
Don't register to ethdev RMV event if EAL hotplug is enabled.
> @@ -2196,6 +2203,9 @@ static void
> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
> __rte_unused void *arg)
> {
> + uint16_t port_id;
> + int ret;
> +
> if (type >= RTE_DEV_EVENT_MAX) {
> fprintf(stderr, "%s called upon invalid event %d\n",
> __func__, type);
> @@ -2206,9 +2216,12 @@ eth_dev_event_callback(char *device_name, enum
> rte_dev_event_type type,
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> + if (ret) {
> + printf("can not get port by device %s!\n",
> device_name);
> + return;
> + }
> + rmv_event_callback((void *)(intptr_t)port_id);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
> 2736,7 +2749,6 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
2018-06-26 17:07 ` Matan Azrad
@ 2018-06-27 3:56 ` Guo, Jia
2018-06-27 6:05 ` Matan Azrad
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-06-27 3:56 UTC (permalink / raw)
To: Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Thomas Monjalon, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
hi, mantan
On 6/27/2018 1:07 AM, Matan Azrad wrote:
> Hi Jeff
>
> Continue session from last version + more comments\question.
>
> From: Jeff Guo
>> Sent: Tuesday, June 26, 2018 6:36 PM
>> To: stephen@networkplumber.org; bruce.richardson@intel.com;
>> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
>> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
>> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
>> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
>> qi.z.zhang@intel.com; shaopeng.he@intel.com; bernard.iremonger@intel.com
>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
>> jia.guo@intel.com; helin.zhang@intel.com
>> Subject: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
>>
>> Use testpmd for example, to show how an application smoothly handle failure
>> when device being hot unplug. If app have enabled the device event monitor
>> and register the hot plug event’s callback before running, once app detect the
>> removal event, the callback would be called. It will first stop the packet
>> forwarding, then stop the port, close the port, and finally detach the port to
>> remove the device out from the device lists.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v3->v2:
>> delete some unused check
>> ---
>> app/test-pmd/testpmd.c | 22 +++++++++++++++++-----
>> 1 file changed, 17 insertions(+), 5 deletions(-)
>>
>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>> 24c1998..2ee5621 100644
>> --- a/app/test-pmd/testpmd.c
>> +++ b/app/test-pmd/testpmd.c
>> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
>> void
>> attach_port(char *identifier)
>> {
>> - portid_t pi = 0;
>> unsigned int socket_id;
>>
>> + portid_t pi = rte_eth_dev_count_avail();
>> +
> I don't understand this change... can you explain?
think about if there are 2 or more device have been attached? The new
device should not always add into port 0, right?
>> printf("Attaching a new port...\n");
>>
>> if (identifier == NULL) {
>> @@ -2125,16 +2126,22 @@ check_all_ports_link_status(uint32_t port_mask)
>> static void rmv_event_callback(void *arg) {
>> + struct rte_eth_dev *dev;
>> +
>> int need_to_start = 0;
>> int org_no_link_check = no_link_check;
>> portid_t port_id = (intptr_t)arg;
>>
>> RTE_ETH_VALID_PORTID_OR_RET(port_id);
>> + dev = &rte_eth_devices[port_id];
>> +
>> + printf("removing device %s\n", dev->device->name);
>>
>> if (!test_done && port_is_forwarding(port_id)) {
>> need_to_start = 1;
>> stop_packet_forwarding();
>> }
>> +
> I don't think you need to change anything in this function.
> You can add the print in the caller code.
ok, i am fine for your point.
>> no_link_check = 1;
>> stop_port(port_id);
>> no_link_check = org_no_link_check;
> Suggestion for synchronization:
> Don't register to ethdev RMV event if EAL hotplug is enabled.
i think that what you propose might be a chose right now, and might need
we think more about the better for all,
but do you agree it is better split it in another fix patch, to let it
patch focus on the feature propose and implement?
>> @@ -2196,6 +2203,9 @@ static void
>> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
>> __rte_unused void *arg)
>> {
>> + uint16_t port_id;
>> + int ret;
>> +
>> if (type >= RTE_DEV_EVENT_MAX) {
>> fprintf(stderr, "%s called upon invalid event %d\n",
>> __func__, type);
>> @@ -2206,9 +2216,12 @@ eth_dev_event_callback(char *device_name, enum
>> rte_dev_event_type type,
>> case RTE_DEV_EVENT_REMOVE:
>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>> device_name);
>> - /* TODO: After finish failure handle, begin to stop
>> - * packet forward, stop port, close port, detach port.
>> - */
>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
>> + if (ret) {
>> + printf("can not get port by device %s!\n",
>> device_name);
>> + return;
>> + }
>> + rmv_event_callback((void *)(intptr_t)port_id);
>> break;
>> case RTE_DEV_EVENT_ADD:
>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
>> 2736,7 +2749,6 @@ main(int argc, char** argv)
>> return -1;
>> }
>> eth_dev_event_callback_register();
>> -
>> }
>>
>> if (start_port(RTE_PORT_ALL) != 0)
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
2018-06-27 3:56 ` Guo, Jia
@ 2018-06-27 6:05 ` Matan Azrad
2018-06-29 10:26 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-06-27 6:05 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
qi.z.zhang@intel.com, shaopeng.he@intel.com,
bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Guo
From: Guo, Jia
> Sent: Wednesday, June 27, 2018 6:56 AM
> To: Matan Azrad <matan@mellanox.com>; stephen@networkplumber.org;
> bruce.richardson@intel.com; ferruh.yigit@intel.com;
> konstantin.ananyev@intel.com; gaetan.rivet@6wind.com;
> jingjing.wu@intel.com; Thomas Monjalon <thomas@monjalon.net>;
> Mordechay Haimovsky <motih@mellanox.com>; harry.van.haaren@intel.com;
> qi.z.zhang@intel.com; shaopeng.he@intel.com; bernard.iremonger@intel.com
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
> helin.zhang@intel.com
> Subject: Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
>
> hi, mantan
>
>
> On 6/27/2018 1:07 AM, Matan Azrad wrote:
> > Hi Jeff
> >
> > Continue session from last version + more comments\question.
> >
> > From: Jeff Guo
> >> Sent: Tuesday, June 26, 2018 6:36 PM
> >> To: stephen@networkplumber.org; bruce.richardson@intel.com;
> >> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
> >> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
> >> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
> >> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
> >> qi.z.zhang@intel.com; shaopeng.he@intel.com;
> >> bernard.iremonger@intel.com
> >> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
> >> jia.guo@intel.com; helin.zhang@intel.com
> >> Subject: [PATCH V3 4/4] app/testpmd: show example to handle hot
> >> unplug
> >>
> >> Use testpmd for example, to show how an application smoothly handle
> >> failure when device being hot unplug. If app have enabled the device
> >> event monitor and register the hot plug event’s callback before
> >> running, once app detect the removal event, the callback would be
> >> called. It will first stop the packet forwarding, then stop the port,
> >> close the port, and finally detach the port to remove the device out from the
> device lists.
> >>
> >> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> >> ---
> >> v3->v2:
> >> delete some unused check
> >> ---
> >> app/test-pmd/testpmd.c | 22 +++++++++++++++++-----
> >> 1 file changed, 17 insertions(+), 5 deletions(-)
> >>
> >> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> >> 24c1998..2ee5621 100644
> >> --- a/app/test-pmd/testpmd.c
> >> +++ b/app/test-pmd/testpmd.c
> >> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
> >> void
> >> attach_port(char *identifier)
> >> {
> >> - portid_t pi = 0;
> >> unsigned int socket_id;
> >>
> >> + portid_t pi = rte_eth_dev_count_avail();
> >> +
> > I don't understand this change... can you explain?
>
> think about if there are 2 or more device have been attached? The new device
> should not always add into port 0, right?
I think you miss here something, you are getting the port id from ethdev, you are just passing a pointer to get it.
I think you should remove this change too.
>
> >> printf("Attaching a new port...\n");
> >>
> >> if (identifier == NULL) {
> >> @@ -2125,16 +2126,22 @@ check_all_ports_link_status(uint32_t
> >> port_mask) static void rmv_event_callback(void *arg) {
> >> + struct rte_eth_dev *dev;
> >> +
> >> int need_to_start = 0;
> >> int org_no_link_check = no_link_check;
> >> portid_t port_id = (intptr_t)arg;
> >>
> >> RTE_ETH_VALID_PORTID_OR_RET(port_id);
> >> + dev = &rte_eth_devices[port_id];
> >> +
> >> + printf("removing device %s\n", dev->device->name);
> >>
> >> if (!test_done && port_is_forwarding(port_id)) {
> >> need_to_start = 1;
> >> stop_packet_forwarding();
> >> }
> >> +
> > I don't think you need to change anything in this function.
> > You can add the print in the caller code.
>
> ok, i am fine for your point.
>
> >> no_link_check = 1;
> >> stop_port(port_id);
> >> no_link_check = org_no_link_check;
> > Suggestion for synchronization:
> > Don't register to ethdev RMV event if EAL hotplug is enabled.
>
> i think that what you propose might be a chose right now, and might need we
> think more about the better for all, but do you agree it is better split it in
> another fix patch, to let it patch focus on the feature propose and implement?
So, Are you suggesting to insert a bug and then to fix it ?:)
My suggestion:
Add a prior patch to depend the ethdev RMV event by a user parameter (can be your hotplug parameter and should be true by default).
In this patch add one more mode to the parameter to enable hotplug by the EAL.
So finally the options of hotplug parameter can be:
0 - for no hotplug handle.
1 - ethdev hotplug (should be the default)
2 - EAL hotplug
What do you think?
> >> @@ -2196,6 +2203,9 @@ static void
> >> eth_dev_event_callback(char *device_name, enum rte_dev_event_type
> type,
> >> __rte_unused void *arg)
> >> {
> >> + uint16_t port_id;
> >> + int ret;
> >> +
> >> if (type >= RTE_DEV_EVENT_MAX) {
> >> fprintf(stderr, "%s called upon invalid event %d\n",
> >> __func__, type);
> >> @@ -2206,9 +2216,12 @@ eth_dev_event_callback(char *device_name,
> enum
> >> rte_dev_event_type type,
> >> case RTE_DEV_EVENT_REMOVE:
> >> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> >> device_name);
> >> - /* TODO: After finish failure handle, begin to stop
> >> - * packet forward, stop port, close port, detach port.
> >> - */
> >> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> >> + if (ret) {
> >> + printf("can not get port by device %s!\n",
> >> device_name);
> >> + return;
> >> + }
> >> + rmv_event_callback((void *)(intptr_t)port_id);
> >> break;
> >> case RTE_DEV_EVENT_ADD:
> >> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
> >> 2736,7 +2749,6 @@ main(int argc, char** argv)
> >> return -1;
> >> }
> >> eth_dev_event_callback_register();
> >> -
> >> }
> >>
> >> if (start_port(RTE_PORT_ALL) != 0)
> >> --
> >> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
2018-06-27 6:05 ` Matan Azrad
@ 2018-06-29 10:26 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-06-29 10:26 UTC (permalink / raw)
To: Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Thomas Monjalon, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
matan
On 6/27/2018 2:05 PM, Matan Azrad wrote:
> Hi Guo
>
> From: Guo, Jia
>> Sent: Wednesday, June 27, 2018 6:56 AM
>> To: Matan Azrad <matan@mellanox.com>; stephen@networkplumber.org;
>> bruce.richardson@intel.com; ferruh.yigit@intel.com;
>> konstantin.ananyev@intel.com; gaetan.rivet@6wind.com;
>> jingjing.wu@intel.com; Thomas Monjalon <thomas@monjalon.net>;
>> Mordechay Haimovsky <motih@mellanox.com>; harry.van.haaren@intel.com;
>> qi.z.zhang@intel.com; shaopeng.he@intel.com; bernard.iremonger@intel.com
>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
>> helin.zhang@intel.com
>> Subject: Re: [PATCH V3 4/4] app/testpmd: show example to handle hot unplug
>>
>> hi, mantan
>>
>>
>> On 6/27/2018 1:07 AM, Matan Azrad wrote:
>>> Hi Jeff
>>>
>>> Continue session from last version + more comments\question.
>>>
>>> From: Jeff Guo
>>>> Sent: Tuesday, June 26, 2018 6:36 PM
>>>> To: stephen@networkplumber.org; bruce.richardson@intel.com;
>>>> ferruh.yigit@intel.com; konstantin.ananyev@intel.com;
>>>> gaetan.rivet@6wind.com; jingjing.wu@intel.com; Thomas Monjalon
>>>> <thomas@monjalon.net>; Mordechay Haimovsky <motih@mellanox.com>;
>>>> Matan Azrad <matan@mellanox.com>; harry.van.haaren@intel.com;
>>>> qi.z.zhang@intel.com; shaopeng.he@intel.com;
>>>> bernard.iremonger@intel.com
>>>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org;
>>>> jia.guo@intel.com; helin.zhang@intel.com
>>>> Subject: [PATCH V3 4/4] app/testpmd: show example to handle hot
>>>> unplug
>>>>
>>>> Use testpmd for example, to show how an application smoothly handle
>>>> failure when device being hot unplug. If app have enabled the device
>>>> event monitor and register the hot plug event’s callback before
>>>> running, once app detect the removal event, the callback would be
>>>> called. It will first stop the packet forwarding, then stop the port,
>>>> close the port, and finally detach the port to remove the device out from the
>> device lists.
>>>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>>>> ---
>>>> v3->v2:
>>>> delete some unused check
>>>> ---
>>>> app/test-pmd/testpmd.c | 22 +++++++++++++++++-----
>>>> 1 file changed, 17 insertions(+), 5 deletions(-)
>>>>
>>>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>>>> 24c1998..2ee5621 100644
>>>> --- a/app/test-pmd/testpmd.c
>>>> +++ b/app/test-pmd/testpmd.c
>>>> @@ -1951,9 +1951,10 @@ eth_dev_event_callback_unregister(void)
>>>> void
>>>> attach_port(char *identifier)
>>>> {
>>>> - portid_t pi = 0;
>>>> unsigned int socket_id;
>>>>
>>>> + portid_t pi = rte_eth_dev_count_avail();
>>>> +
>>> I don't understand this change... can you explain?
>> think about if there are 2 or more device have been attached? The new device
>> should not always add into port 0, right?
> I think you miss here something, you are getting the port id from ethdev, you are just passing a pointer to get it.
> I think you should remove this change too.
ok, seems i am missing something, let me check.
>>>> printf("Attaching a new port...\n");
>>>>
>>>> if (identifier == NULL) {
>>>> @@ -2125,16 +2126,22 @@ check_all_ports_link_status(uint32_t
>>>> port_mask) static void rmv_event_callback(void *arg) {
>>>> + struct rte_eth_dev *dev;
>>>> +
>>>> int need_to_start = 0;
>>>> int org_no_link_check = no_link_check;
>>>> portid_t port_id = (intptr_t)arg;
>>>>
>>>> RTE_ETH_VALID_PORTID_OR_RET(port_id);
>>>> + dev = &rte_eth_devices[port_id];
>>>> +
>>>> + printf("removing device %s\n", dev->device->name);
>>>>
>>>> if (!test_done && port_is_forwarding(port_id)) {
>>>> need_to_start = 1;
>>>> stop_packet_forwarding();
>>>> }
>>>> +
>>> I don't think you need to change anything in this function.
>>> You can add the print in the caller code.
>> ok, i am fine for your point.
>>
>>>> no_link_check = 1;
>>>> stop_port(port_id);
>>>> no_link_check = org_no_link_check;
>>> Suggestion for synchronization:
>>> Don't register to ethdev RMV event if EAL hotplug is enabled.
>> i think that what you propose might be a chose right now, and might need we
>> think more about the better for all, but do you agree it is better split it in
>> another fix patch, to let it patch focus on the feature propose and implement?
> So, Are you suggesting to insert a bug and then to fix it ?:)
>
> My suggestion:
> Add a prior patch to depend the ethdev RMV event by a user parameter (can be your hotplug parameter and should be true by default).
> In this patch add one more mode to the parameter to enable hotplug by the EAL.
>
> So finally the options of hotplug parameter can be:
> 0 - for no hotplug handle.
> 1 - ethdev hotplug (should be the default)
> 2 - EAL hotplug
>
> What do you think?
sure, i think you absolutely know i don't want to add any bug here :)
just want to make it more focus and clear.
your propose looks fine by me. good idea, thanks.
please check my v4 patch set.
>>>> @@ -2196,6 +2203,9 @@ static void
>>>> eth_dev_event_callback(char *device_name, enum rte_dev_event_type
>> type,
>>>> __rte_unused void *arg)
>>>> {
>>>> + uint16_t port_id;
>>>> + int ret;
>>>> +
>>>> if (type >= RTE_DEV_EVENT_MAX) {
>>>> fprintf(stderr, "%s called upon invalid event %d\n",
>>>> __func__, type);
>>>> @@ -2206,9 +2216,12 @@ eth_dev_event_callback(char *device_name,
>> enum
>>>> rte_dev_event_type type,
>>>> case RTE_DEV_EVENT_REMOVE:
>>>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>>>> device_name);
>>>> - /* TODO: After finish failure handle, begin to stop
>>>> - * packet forward, stop port, close port, detach port.
>>>> - */
>>>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
>>>> + if (ret) {
>>>> + printf("can not get port by device %s!\n",
>>>> device_name);
>>>> + return;
>>>> + }
>>>> + rmv_event_callback((void *)(intptr_t)port_id);
>>>> break;
>>>> case RTE_DEV_EVENT_ADD:
>>>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
>>>> 2736,7 +2749,6 @@ main(int argc, char** argv)
>>>> return -1;
>>>> }
>>>> eth_dev_event_callback_register();
>>>> -
>>>> }
>>>>
>>>> if (start_port(RTE_PORT_ALL) != 0)
>>>> --
>>>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 0/9] hot plug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (7 preceding siblings ...)
2018-06-26 15:36 ` [PATCH V3 1/4] bus/pci: handle device " Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-06-29 10:30 ` [PATCH V4 1/9] bus: introduce hotplug failure handler Jeff Guo
` (8 more replies)
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
` (14 subsequent siblings)
23 siblings, 9 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event detect mechanism, failsafe driver,
bonding driver and hot plug/unplug api in framework, app could use these to
develop their hot plug solution.
let’s see the case of hot unplug, it can happen when a hardware device is
be removed physically, or when the software disables it. App need to call
ether dev API to detach the device, to unplug the device at the bus level and
make access to the device invalid. But the problem is that, the removal of the
device from the software lists is not going to be instantaneous, at this time
if the data(fast) path still read/write the device, it will cause MMIO error
and result of the app crash out.
Seems that we have got fail-safe driver(or app) + RTE_ETH_EVENT_INTR_RMV +
kernel core driver solution to handle it, but still not have failsafe driver
(or app) + RTE_DEV_EVENT_REMOVE + PCIe pmd driver failure handle solution. So
there is an absence in dpdk hot plug solution right now.
Also, we know that kernel only guaranty hot plug on the kernel side, but not for
the user mode side. Firstly we can hardly have a gatekeeper for any MMIO for
multiple PMD driver. Secondly, no more specific 3rd tools such as udev/driverctl
have especially cover these hot plug failure processing. Third, the feasibility
of app’s implement for multiple user mode PMD driver is still a problem. Here,
a general hot plug failure handle mechanism in dpdk framework would be proposed,
it aim to guaranty that, when hot unplug occur, the system will not crash and
app will not be break out, and user space can normally stop and release any
relevant resources, then unplug of the device at the bus level cleanly.
The mechanism should be come across as bellow:
Firstly, app enabled the device event monitor and register the hot plug event’s
callback before running data path. Once the hot unplug behave occur, the
mechanism will detect the removal event and then accordingly do the failure
handle. In order to do that, below functional will be bring in.
- Add a new bus ops “handle_hot_unplug” to handle bus read/write error, it is
bus-specific and each kind of bus can implement its own logic.
- Implement pci bus specific ops “pci_handle_hot_unplug”. It will base on the
failure address to remap memory for the corresponding device that unplugged.
For the data path or other unexpected control from the control path when hot
unplug occur.
- Implement a new sigbus handler, it is registered when start device even
monitoring. The handler is per process. Base on the signal event principle,
control path thread and data path thread will randomly receive the sigbus
error, but will go to the common sigbus handler. Once the MMIO sigbus error
exposure, it will trigger the above hot unplug operation. The sigbus will be
check if it is cause of the hot unplug or not, if not will info exception as
the original sigbus handler. If yes, will do memory remapping.
For the control path and the igb uio release:
- When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this resource,
it will cause kernel crash.
On the other hand, something like interrupt disable do not automatically
process in kernel side. If not handler it, this redundancy and dirty thing
will affect the interrupt resource be used by other device.
So the igb_uio driver have to check the hot plug status and corresponding
process should be taken in igb uio deriver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such as
probed/opened/released/removed/unplug. When detect the unexpected removal
which cause of hot unplug behavior, it will corresponding disable interrupt
resource, while for the part of releasement which kernel have already handle,
just skip it to avoid double free or null pointer kernel crash issue.
The mechanism could be use for fail-safe driver and app which want to use hot
plug solution. At this stage, will only use testpmd as reference to show how to
use the mechanism.
- Enable device event monitor->device unplug->failure handle->stop forwarding->
stop port->close port->detach port.
This process will not breaking the app/fail-safe running, and will not break
other irrelevance device. And app could plug in the device and restart the date
path again by below.
- Device plug in->bind igb_uio driver ->attached device->start port->
start forwarding.
patchset history:
v4->v3:
split patches to be small and clear
change to use new parameter "--hotplug-mode" in testpmd
to identify the eal hotplug and ethdev hotplug
v3->v2:
change bus ops name to bus_hotplug_handler.
add new API and bus ops of bus_signal_handler
distingush handle generic sigbus and hotplug sigbus
v2->v1(v21):
refine some doc and commit log
fix igb uio kernel issue for control path failure
rebase testpmd code
Since the hot plug solution be discussed serval around in the public, the
scope be changed and the patch set be split into many times. Coming to the
recently RFC and feature design, it just focus on the hot unplug failure
handler at this patch set, so in order let this topic more clear and focus,
summarize privours patch set in history “v1(v21)”, the v2 here go ahead
for further track.
"v1(21)" == v21 as below:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (9):
bus: introduce hotplug failure handler
bus/pci: implement hotplug handler operation
bus: introduce sigbus handler
bus/pci: implement sigbus handler operation
bus: add helper to handle sigbus
eal: add failure handle mechanism for hot plug
igb_uio: fix uio release issue when hot unplug
app/testpmd: show example to handle hot unplug
app/testpmd: enable device hotplug monitoring
app/test-pmd/parameters.c | 20 ++++++--
app/test-pmd/testpmd.c | 31 +++++++-----
app/test-pmd/testpmd.h | 8 ++-
doc/guides/testpmd_app_ug/run_app.rst | 10 +++-
drivers/bus/pci/pci_common.c | 78 +++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++
drivers/bus/pci/private.h | 12 +++++
kernel/linux/igb_uio/igb_uio.c | 50 +++++++++++++++++--
lib/librte_eal/common/eal_common_bus.c | 34 ++++++++++++-
lib/librte_eal/common/eal_private.h | 11 +++++
lib/librte_eal/common/include/rte_bus.h | 31 ++++++++++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++-
12 files changed, 381 insertions(+), 25 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-07-03 22:21 ` Thomas Monjalon
2018-06-29 10:30 ` [PATCH V4 2/9] bus/pci: implement hotplug handler operation Jeff Guo
` (7 subsequent siblings)
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When a hardware device is removed physically or the software disables
it, the hotplug occur. App need to call ether dev API to detach the device,
to unplug the device at the bus level and make access to the device
invalid. But the removal of the device from the software lists is not going
to be instantaneous, at this time if the data path still read/write the
device, it will cause MMIO error and result of the app crash out. So a
hotplug failure handle mechanism need to be used to guaranty app will not
crash out when hot unplug device.
To handle device hot plug failure is a bus-specific behavior, this patch
introduces a bus ops so that each kind of bus can implement its own logic.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
lib/librte_eal/common/include/rte_bus.h | 15 +++++++++++++++
1 file changed, 15 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..3642aeb 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,19 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hot plug handler, which is responsible
+ * for handle the failure when hot remove the device, guaranty the system
+ * would not crash in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -211,6 +224,8 @@ struct rte_bus {
rte_bus_parse_t parse; /**< Parse a device name */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
+ rte_bus_hotplug_handler_t hotplug_handler;
+ /**< handle hot plug on bus */
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-06-29 10:30 ` [PATCH V4 1/9] bus: introduce hotplug failure handler Jeff Guo
@ 2018-07-03 22:21 ` Thomas Monjalon
2018-07-04 7:16 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Thomas Monjalon @ 2018-07-03 22:21 UTC (permalink / raw)
To: Jeff Guo
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, jblunck,
shreyansh.jain, helin.zhang
29/06/2018 12:30, Jeff Guo:
> /**
> + * Implementation a specific hot plug handler, which is responsible
> + * for handle the failure when hot remove the device, guaranty the system
> + * would not crash in the case.
> + * @param dev
> + * Pointer of the device structure.
> + *
> + * @return
> + * 0 on success.
> + * !0 on error.
> + */
> +typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
[...]
> @@ -211,6 +224,8 @@ struct rte_bus {
> rte_bus_parse_t parse; /**< Parse a device name */
> struct rte_bus_conf conf; /**< Bus configuration */
> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
> + rte_bus_hotplug_handler_t hotplug_handler;
> + /**< handle hot plug on bus */
The name is misleading.
It is to handle unplugging but is called "hotplug".
In order to demonstrate how the handler is used, you should
introduce the code using this handler in the same patch.
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-07-03 22:21 ` Thomas Monjalon
@ 2018-07-04 7:16 ` Guo, Jia
2018-07-04 7:55 ` Thomas Monjalon
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-07-04 7:16 UTC (permalink / raw)
To: Thomas Monjalon
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, jblunck,
shreyansh.jain, helin.zhang
On 7/4/2018 6:21 AM, Thomas Monjalon wrote:
> 29/06/2018 12:30, Jeff Guo:
>> /**
>> + * Implementation a specific hot plug handler, which is responsible
>> + * for handle the failure when hot remove the device, guaranty the system
>> + * would not crash in the case.
>> + * @param dev
>> + * Pointer of the device structure.
>> + *
>> + * @return
>> + * 0 on success.
>> + * !0 on error.
>> + */
>> +typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
> [...]
>> @@ -211,6 +224,8 @@ struct rte_bus {
>> rte_bus_parse_t parse; /**< Parse a device name */
>> struct rte_bus_conf conf; /**< Bus configuration */
>> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
>> + rte_bus_hotplug_handler_t hotplug_handler;
>> + /**< handle hot plug on bus */
> The name is misleading.
> It is to handle unplugging but is called "hotplug".
ok, so i prefer hotplug_failure_handler than hot_unplug_handler, since
it is more explicit for failure handle, and more clearly.
> In order to demonstrate how the handler is used, you should
> introduce the code using this handler in the same patch.
>
sorry, i check the history of rte_bus.h, and the way is introduce ops at
first, second implement in specific bus, then come across the usage.
I think that way clear and make sense. what do you think?
Anyway, i will check the commit log if is there any misleading.
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-07-04 7:16 ` Guo, Jia
@ 2018-07-04 7:55 ` Thomas Monjalon
2018-07-05 6:23 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Thomas Monjalon @ 2018-07-04 7:55 UTC (permalink / raw)
To: Guo, Jia
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, jblunck,
shreyansh.jain, helin.zhang
04/07/2018 09:16, Guo, Jia:
>
> On 7/4/2018 6:21 AM, Thomas Monjalon wrote:
> > 29/06/2018 12:30, Jeff Guo:
> >> /**
> >> + * Implementation a specific hot plug handler, which is responsible
> >> + * for handle the failure when hot remove the device, guaranty the system
> >> + * would not crash in the case.
> >> + * @param dev
> >> + * Pointer of the device structure.
> >> + *
> >> + * @return
> >> + * 0 on success.
> >> + * !0 on error.
> >> + */
> >> +typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
> > [...]
> >> @@ -211,6 +224,8 @@ struct rte_bus {
> >> rte_bus_parse_t parse; /**< Parse a device name */
> >> struct rte_bus_conf conf; /**< Bus configuration */
> >> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
> >> + rte_bus_hotplug_handler_t hotplug_handler;
> >> + /**< handle hot plug on bus */
> > The name is misleading.
> > It is to handle unplugging but is called "hotplug".
>
> ok, so i prefer hotplug_failure_handler than hot_unplug_handler, since
> it is more explicit for failure handle, and more clearly.
>
> > In order to demonstrate how the handler is used, you should
> > introduce the code using this handler in the same patch.
> >
>
> sorry, i check the history of rte_bus.h, and the way is introduce ops at
> first, second implement in specific bus, then come across the usage.
> I think that way clear and make sense. what do you think?
> Anyway, i will check the commit log if is there any misleading.
I think it is better to call ops when they are introduced,
and implement the ops in second step.
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-07-04 7:55 ` Thomas Monjalon
@ 2018-07-05 6:23 ` Guo, Jia
2018-07-05 8:30 ` Thomas Monjalon
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-07-05 6:23 UTC (permalink / raw)
To: Thomas Monjalon
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, jblunck,
shreyansh.jain, helin.zhang
On 7/4/2018 3:55 PM, Thomas Monjalon wrote:
> 04/07/2018 09:16, Guo, Jia:
>> On 7/4/2018 6:21 AM, Thomas Monjalon wrote:
>>> 29/06/2018 12:30, Jeff Guo:
>>>> /**
>>>> + * Implementation a specific hot plug handler, which is responsible
>>>> + * for handle the failure when hot remove the device, guaranty the system
>>>> + * would not crash in the case.
>>>> + * @param dev
>>>> + * Pointer of the device structure.
>>>> + *
>>>> + * @return
>>>> + * 0 on success.
>>>> + * !0 on error.
>>>> + */
>>>> +typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
>>> [...]
>>>> @@ -211,6 +224,8 @@ struct rte_bus {
>>>> rte_bus_parse_t parse; /**< Parse a device name */
>>>> struct rte_bus_conf conf; /**< Bus configuration */
>>>> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
>>>> + rte_bus_hotplug_handler_t hotplug_handler;
>>>> + /**< handle hot plug on bus */
>>> The name is misleading.
>>> It is to handle unplugging but is called "hotplug".
>> ok, so i prefer hotplug_failure_handler than hot_unplug_handler, since
>> it is more explicit for failure handle, and more clearly.
>>
>>> In order to demonstrate how the handler is used, you should
>>> introduce the code using this handler in the same patch.
>>>
>> sorry, i check the history of rte_bus.h, and the way is introduce ops at
>> first, second implement in specific bus, then come across the usage.
>> I think that way clear and make sense. what do you think?
>> Anyway, i will check the commit log if is there any misleading.
> I think it is better to call ops when they are introduced,
> and implement the ops in second step.
>
Hi, Thomas
sorry but i want to detail the relationship of the ops and api as bellow
to try if we can get the better sequence.
Patch num:
1: introduce ops hotplug_failure_handler
2: implement ops hotplug_failure_handler
3:introduce ops sigbus_handler.
4:implement ops sigbus_handler
5: introduce helper rte_bus_sigbus_handler to call the ops sigbus_handler
6: introduce the mechanism to call helper rte_bus_sigbus_handler and
call hotplug_failure_handler.
If per you said , could I modify the sequence like 6->5->3->4->1->2? I
don't think it will make sense, and might be more confused.
And I think should be better that introduce each ops just say item, then
when introduce the caller patch, the functional is ready to use by the
patch.
if i did not got your point and you have other better sequence about
that please explicit to let me know. Thanks.
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 1/9] bus: introduce hotplug failure handler
2018-07-05 6:23 ` Guo, Jia
@ 2018-07-05 8:30 ` Thomas Monjalon
0 siblings, 0 replies; 494+ messages in thread
From: Thomas Monjalon @ 2018-07-05 8:30 UTC (permalink / raw)
To: Guo, Jia
Cc: dev, stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, jblunck,
shreyansh.jain, helin.zhang
05/07/2018 08:23, Guo, Jia:
>
> On 7/4/2018 3:55 PM, Thomas Monjalon wrote:
> > 04/07/2018 09:16, Guo, Jia:
> >> On 7/4/2018 6:21 AM, Thomas Monjalon wrote:
> >>> 29/06/2018 12:30, Jeff Guo:
> >>>> /**
> >>>> + * Implementation a specific hot plug handler, which is responsible
> >>>> + * for handle the failure when hot remove the device, guaranty the system
> >>>> + * would not crash in the case.
> >>>> + * @param dev
> >>>> + * Pointer of the device structure.
> >>>> + *
> >>>> + * @return
> >>>> + * 0 on success.
> >>>> + * !0 on error.
> >>>> + */
> >>>> +typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
> >>> [...]
> >>>> @@ -211,6 +224,8 @@ struct rte_bus {
> >>>> rte_bus_parse_t parse; /**< Parse a device name */
> >>>> struct rte_bus_conf conf; /**< Bus configuration */
> >>>> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
> >>>> + rte_bus_hotplug_handler_t hotplug_handler;
> >>>> + /**< handle hot plug on bus */
> >>> The name is misleading.
> >>> It is to handle unplugging but is called "hotplug".
> >> ok, so i prefer hotplug_failure_handler than hot_unplug_handler, since
> >> it is more explicit for failure handle, and more clearly.
> >>
> >>> In order to demonstrate how the handler is used, you should
> >>> introduce the code using this handler in the same patch.
> >>>
> >> sorry, i check the history of rte_bus.h, and the way is introduce ops at
> >> first, second implement in specific bus, then come across the usage.
> >> I think that way clear and make sense. what do you think?
> >> Anyway, i will check the commit log if is there any misleading.
> > I think it is better to call ops when they are introduced,
> > and implement the ops in second step.
> >
>
> Hi, Thomas
>
> sorry but i want to detail the relationship of the ops and api as bellow
> to try if we can get the better sequence.
>
> Patch num:
>
> 1: introduce ops hotplug_failure_handler
>
> 2: implement ops hotplug_failure_handler
>
> 3:introduce ops sigbus_handler.
>
> 4:implement ops sigbus_handler
>
> 5: introduce helper rte_bus_sigbus_handler to call the ops sigbus_handler
>
> 6: introduce the mechanism to call helper rte_bus_sigbus_handler and
> call hotplug_failure_handler.
>
> If per you said , could I modify the sequence like 6->5->3->4->1->2? I
> don't think it will make sense, and might be more confused.
>
> And I think should be better that introduce each ops just say item, then
> when introduce the caller patch, the functional is ready to use by the
> patch.
>
>
> if i did not got your point and you have other better sequence about
> that please explicit to let me know. Thanks.
The main concern is to be able to understand each patch separately.
When introducing a new op, we need to understand how it will be used.
But actually, no need to change patch organization,
you just need to provide a clear doxygen documentation,
and introduce the context in the commit log.
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 2/9] bus/pci: implement hotplug handler operation
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
2018-06-29 10:30 ` [PATCH V4 1/9] bus: introduce hotplug failure handler Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-06-29 10:30 ` [PATCH V4 3/9] bus: introduce sigbus handler Jeff Guo
` (6 subsequent siblings)
8 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of hotplug handler for PCI bus, it is
functional to remap a new dummy memory which overlap to the failure
memory to avoid MMIO read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
drivers/bus/pci/pci_common.c | 28 ++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++++++++
3 files changed, 73 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index d8151b0..095cd4e 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -401,6 +401,33 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
}
static int
+pci_hotplug_handler(struct rte_device *dev)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = RTE_DEV_TO_PCI(dev);
+ if (!pdev)
+ return -1;
+
+ switch (pdev->kdrv) {
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ /* mmio resources is invalid, remap it to be safe. */
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -430,6 +457,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .hotplug_handler = pci_hotplug_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 8ddd03e..6b312e5 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -123,6 +123,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V4 3/9] bus: introduce sigbus handler
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
2018-06-29 10:30 ` [PATCH V4 1/9] bus: introduce hotplug failure handler Jeff Guo
2018-06-29 10:30 ` [PATCH V4 2/9] bus/pci: implement hotplug handler operation Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-07-10 21:55 ` Stephen Hemminger
2018-06-29 10:30 ` [PATCH V4 4/9] bus/pci: implement sigbus handler operation Jeff Guo
` (5 subsequent siblings)
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When device be hotplug, if data path still read/write device, the sigbus
error will occur, this error need to be handled. So a handler need to be
here to capture the signal and handle it correspondingly.
To handle sigbus error is a bus-specific behavior, this patch introduces
a bus ops so that each kind of bus can implement its own logic.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index 3642aeb..231bd3d 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -181,6 +181,20 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
/**
+ * Implementation a specific sigbus handler, which is responsible
+ * for handle the sigbus error which is original memory error, or specific
+ * memory error that caused of hot unplug.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 for success handle the sigbus.
+ * 1 for no handle the sigbus.
+ * -1 for failed to handle the sigbus
+ */
+typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -226,6 +240,8 @@ struct rte_bus {
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
rte_bus_hotplug_handler_t hotplug_handler;
/**< handle hot plug on bus */
+ rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
+
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 3/9] bus: introduce sigbus handler
2018-06-29 10:30 ` [PATCH V4 3/9] bus: introduce sigbus handler Jeff Guo
@ 2018-07-10 21:55 ` Stephen Hemminger
2018-07-11 2:15 ` Jeff Guo
0 siblings, 1 reply; 494+ messages in thread
From: Stephen Hemminger @ 2018-07-10 21:55 UTC (permalink / raw)
To: Jeff Guo
Cc: bruce.richardson, ferruh.yigit, konstantin.ananyev, gaetan.rivet,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, bernard.iremonger, jblunck, shreyansh.jain, dev,
helin.zhang
On Fri, 29 Jun 2018 18:30:42 +0800
Jeff Guo <jia.guo@intel.com> wrote:
> When device be hotplug, if data path still read/write device, the sigbus
> error will occur, this error need to be handled. So a handler need to be
> here to capture the signal and handle it correspondingly.
>
> To handle sigbus error is a bus-specific behavior, this patch introduces
> a bus ops so that each kind of bus can implement its own logic.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v4->v3:
> split patches to be small and clear.
> ---
> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++++++++++
> 1 file changed, 16 insertions(+)
>
> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
> index 3642aeb..231bd3d 100644
> --- a/lib/librte_eal/common/include/rte_bus.h
> +++ b/lib/librte_eal/common/include/rte_bus.h
> @@ -181,6 +181,20 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
> typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
>
> /**
> + * Implementation a specific sigbus handler, which is responsible
> + * for handle the sigbus error which is original memory error, or specific
> + * memory error that caused of hot unplug.
> + * @param failure_addr
> + * Pointer of the fault address of the sigbus error.
> + *
> + * @return
> + * 0 for success handle the sigbus.
> + * 1 for no handle the sigbus.
> + * -1 for failed to handle the sigbus
> + */
> +typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
> +
> +/**
> * Bus scan policies
> */
> enum rte_bus_scan_mode {
> @@ -226,6 +240,8 @@ struct rte_bus {
> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
> rte_bus_hotplug_handler_t hotplug_handler;
> /**< handle hot plug on bus */
> + rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
> +
> };
>
> /**
One issue with handling sigbus is that you are going to trap program errors
as well as hotplug. How can you distinguish between removed device and a
buggy userspace program (or worse comprimised program)?
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 3/9] bus: introduce sigbus handler
2018-07-10 21:55 ` Stephen Hemminger
@ 2018-07-11 2:15 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-11 2:15 UTC (permalink / raw)
To: Stephen Hemminger
Cc: bruce.richardson, ferruh.yigit, konstantin.ananyev, gaetan.rivet,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, bernard.iremonger, jblunck, shreyansh.jain, dev,
helin.zhang
On 7/11/2018 5:55 AM, Stephen Hemminger wrote:
> On Fri, 29 Jun 2018 18:30:42 +0800
> Jeff Guo <jia.guo@intel.com> wrote:
>
>> When device be hotplug, if data path still read/write device, the sigbus
>> error will occur, this error need to be handled. So a handler need to be
>> here to capture the signal and handle it correspondingly.
>>
>> To handle sigbus error is a bus-specific behavior, this patch introduces
>> a bus ops so that each kind of bus can implement its own logic.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v4->v3:
>> split patches to be small and clear.
>> ---
>> lib/librte_eal/common/include/rte_bus.h | 16 ++++++++++++++++
>> 1 file changed, 16 insertions(+)
>>
>> diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
>> index 3642aeb..231bd3d 100644
>> --- a/lib/librte_eal/common/include/rte_bus.h
>> +++ b/lib/librte_eal/common/include/rte_bus.h
>> @@ -181,6 +181,20 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
>> typedef int (*rte_bus_hotplug_handler_t)(struct rte_device *dev);
>>
>> /**
>> + * Implementation a specific sigbus handler, which is responsible
>> + * for handle the sigbus error which is original memory error, or specific
>> + * memory error that caused of hot unplug.
>> + * @param failure_addr
>> + * Pointer of the fault address of the sigbus error.
>> + *
>> + * @return
>> + * 0 for success handle the sigbus.
>> + * 1 for no handle the sigbus.
>> + * -1 for failed to handle the sigbus
>> + */
>> +typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
>> +
>> +/**
>> * Bus scan policies
>> */
>> enum rte_bus_scan_mode {
>> @@ -226,6 +240,8 @@ struct rte_bus {
>> rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
>> rte_bus_hotplug_handler_t hotplug_handler;
>> /**< handle hot plug on bus */
>> + rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
>> +
>> };
>>
>> /**
> One issue with handling sigbus is that you are going to trap program errors
> as well as hotplug. How can you distinguish between removed device and a
> buggy userspace program (or worse comprimised program)?
That is a problem which i have been considerate in this mechanism and do
it in other patch, the way is that first check if the error domain is
belong to the mmio device resource or not,
if it is will do new sigbus handler for hotplug, if not will mean that
it is buggy user space program, will use generic sigbus handler to
handler it.
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 4/9] bus/pci: implement sigbus handler operation
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (2 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 3/9] bus: introduce sigbus handler Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-06-29 10:30 ` [PATCH V4 5/9] bus: add helper to handle sigbus Jeff Guo
` (4 subsequent siblings)
8 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of sigbus handler for PCI bus, it is
functional to find the corresponding pci device which is be hot removal.
and then handle the hot plug failure for this device.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
drivers/bus/pci/pci_common.c | 50 ++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 50 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 095cd4e..0f5b4af 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -400,6 +400,32 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+/* check the failure address belongs to which device. */
+static struct rte_pci_device *
+pci_find_device_by_addr(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(INFO, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+
static int
pci_hotplug_handler(struct rte_device *dev)
{
@@ -428,6 +454,29 @@ pci_hotplug_handler(struct rte_device *dev)
}
static int
+pci_sigbus_handler(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = pci_find_device_by_addr(failure_addr);
+ if (!pdev) {
+ /* It is a generic sigbus error. */
+ ret = 1;
+ } else {
+ /* The sigbus error is caused of hot removal. */
+ ret = pci_hotplug_handler(&pdev->device);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Failed to handle hot plug for "
+ "device %s", pdev->name);
+ ret = -1;
+ rte_errno = -1;
+ }
+ }
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -458,6 +507,7 @@ struct rte_pci_bus rte_pci_bus = {
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
.hotplug_handler = pci_hotplug_handler,
+ .sigbus_handler = pci_sigbus_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (3 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 4/9] bus/pci: implement sigbus handler operation Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-06-29 10:51 ` Ananyev, Konstantin
2018-06-29 10:30 ` [PATCH V4 6/9] eal: add failure handle mechanism for hot plug Jeff Guo
` (3 subsequent siblings)
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch aim to add a helper to iterate all buses to find the
corresponding bus to handle the sigbus error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
lib/librte_eal/common/eal_common_bus.c | 34 +++++++++++++++++++++++++++++++++-
lib/librte_eal/common/eal_private.h | 11 +++++++++++
2 files changed, 44 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/common/eal_common_bus.c b/lib/librte_eal/common/eal_common_bus.c
index 0943851..34c4f2d 100644
--- a/lib/librte_eal/common/eal_common_bus.c
+++ b/lib/librte_eal/common/eal_common_bus.c
@@ -37,6 +37,7 @@
#include <rte_bus.h>
#include <rte_debug.h>
#include <rte_string_fns.h>
+#include <rte_errno.h>
#include "eal_private.h"
@@ -220,7 +221,6 @@ rte_bus_find_by_device_name(const char *str)
return rte_bus_find(NULL, bus_can_parse, name);
}
-
/*
* Get iommu class of devices on the bus.
*/
@@ -242,3 +242,35 @@ rte_bus_get_iommu_class(void)
}
return mode;
}
+
+static int
+bus_handle_sigbus(const struct rte_bus *bus,
+ const void *failure_addr)
+{
+ return !(bus->sigbus_handler && bus->sigbus_handler(failure_addr) <= 0);
+}
+
+int
+rte_bus_sigbus_handler(const void *failure_addr)
+{
+ struct rte_bus *bus;
+ int old_errno = rte_errno;
+ int ret = 0;
+
+ rte_errno = 0;
+
+ bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
+ ret = -1;
+ } else if (rte_errno != 0) {
+ RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
+ ret = -1;
+ }
+
+ /* if sigbus not be handled, return back old errno. */
+ if (ret)
+ rte_errno = old_errno;
+
+ return ret;
+}
diff --git a/lib/librte_eal/common/eal_private.h b/lib/librte_eal/common/eal_private.h
index bdadc4d..9517f2b 100644
--- a/lib/librte_eal/common/eal_private.h
+++ b/lib/librte_eal/common/eal_private.h
@@ -258,4 +258,15 @@ int rte_mp_channel_init(void);
*/
void dev_callback_process(char *device_name, enum rte_dev_event_type event);
+
+/**
+ * Iterate all buses to find the corresponding bus, to handle the sigbus error.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 on success.
+ * -1 on error
+ */
+int rte_bus_sigbus_handler(const void *failure_addr);
#endif /* _EAL_PRIVATE_H_ */
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 10:30 ` [PATCH V4 5/9] bus: add helper to handle sigbus Jeff Guo
@ 2018-06-29 10:51 ` Ananyev, Konstantin
2018-06-29 11:23 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-06-29 10:51 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng, Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> +int
> +rte_bus_sigbus_handler(const void *failure_addr)
> +{
> + struct rte_bus *bus;
> + int old_errno = rte_errno;
> + int ret = 0;
> +
> + rte_errno = 0;
> +
> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
> + ret = -1;
> + } else if (rte_errno != 0) {
> + RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
> + ret = -1;
> + }
> +
> + /* if sigbus not be handled, return back old errno. */
> + if (ret)
> + rte_errno = old_errno;
Hmm, not sure why we need to set/restore rte_errno here?
> +
> + return ret;
> +}
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 10:51 ` Ananyev, Konstantin
@ 2018-06-29 11:23 ` Guo, Jia
2018-06-29 12:21 ` Ananyev, Konstantin
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-06-29 11:23 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
hi, konstantin
On 6/29/2018 6:51 PM, Ananyev, Konstantin wrote:
>> +int
>> +rte_bus_sigbus_handler(const void *failure_addr)
>> +{
>> + struct rte_bus *bus;
>> + int old_errno = rte_errno;
>> + int ret = 0;
>> +
>> + rte_errno = 0;
>> +
>> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
>> + if (bus == NULL) {
>> + RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
>> + ret = -1;
>> + } else if (rte_errno != 0) {
>> + RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
>> + ret = -1;
>> + }
>> +
>> + /* if sigbus not be handled, return back old errno. */
>> + if (ret)
>> + rte_errno = old_errno;
> Hmm, not sure why we need to set/restore rte_errno here?
restore old_errno just use to let caller know that the generic sigbus
still not handler by bus hotplug handler, that involve find a bus
handle but failed and can not find a hander, and can corresponding use
the previous sigbus handler to process it.
that is also unwser your question in other patch. do you think that make
sense?
>> +
>> + return ret;
>> +}
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 11:23 ` Guo, Jia
@ 2018-06-29 12:21 ` Ananyev, Konstantin
2018-06-29 12:52 ` Gaëtan Rivet
0 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-06-29 12:21 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng, Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Friday, June 29, 2018 12:23 PM
> To: Ananyev, Konstantin <konstantin.ananyev@intel.com>; stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; gaetan.rivet@6wind.com; Wu, Jingjing
> <jingjing.wu@intel.com>; thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
> <harry.van.haaren@intel.com>; Zhang, Qi Z <qi.z.zhang@intel.com>; He, Shaopeng <shaopeng.he@intel.com>; Iremonger, Bernard
> <bernard.iremonger@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Zhang, Helin <helin.zhang@intel.com>
> Subject: Re: [PATCH V4 5/9] bus: add helper to handle sigbus
>
> hi, konstantin
>
>
> On 6/29/2018 6:51 PM, Ananyev, Konstantin wrote:
> >> +int
> >> +rte_bus_sigbus_handler(const void *failure_addr)
> >> +{
> >> + struct rte_bus *bus;
> >> + int old_errno = rte_errno;
> >> + int ret = 0;
> >> +
> >> + rte_errno = 0;
> >> +
> >> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
> >> + if (bus == NULL) {
> >> + RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
> >> + ret = -1;
> >> + } else if (rte_errno != 0) {
> >> + RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
> >> + ret = -1;
> >> + }
> >> +
> >> + /* if sigbus not be handled, return back old errno. */
> >> + if (ret)
> >> + rte_errno = old_errno;
> > Hmm, not sure why we need to set/restore rte_errno here?
>
> restore old_errno just use to let caller know that the generic sigbus
> still not handler by bus hotplug handler, that involve find a bus
> handle but failed and can not find a hander, and can corresponding use
> the previous sigbus handler to process it.
> that is also unwser your question in other patch. do you think that make
> sense?
Sorry, still don't understand the intention.
Suppose rte_bus_find() will return NULL, in that case you'll setup rte_errno
to what it was before calling that function.
If the returned bus is not NULL, but bus_find() set's an rte_errno,
you still would restore rte_ernno?
What is the prupose?
Why do you need to touch rte_errno at all in that function?
Konstantin
>
> >> +
> >> + return ret;
> >> +}
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 12:21 ` Ananyev, Konstantin
@ 2018-06-29 12:52 ` Gaëtan Rivet
2018-07-03 11:24 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Gaëtan Rivet @ 2018-06-29 12:52 UTC (permalink / raw)
To: Ananyev, Konstantin
Cc: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Wu, Jingjing, thomas@monjalon.net,
motih@mellanox.com, matan@mellanox.com, Van Haaren, Harry,
Zhang, Qi Z, He, Shaopeng, Iremonger, Bernard,
jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
On Fri, Jun 29, 2018 at 12:21:39PM +0000, Ananyev, Konstantin wrote:
>
>
> > -----Original Message-----
> > From: Guo, Jia
> > Sent: Friday, June 29, 2018 12:23 PM
> > To: Ananyev, Konstantin <konstantin.ananyev@intel.com>; stephen@networkplumber.org; Richardson, Bruce
> > <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; gaetan.rivet@6wind.com; Wu, Jingjing
> > <jingjing.wu@intel.com>; thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
> > <harry.van.haaren@intel.com>; Zhang, Qi Z <qi.z.zhang@intel.com>; He, Shaopeng <shaopeng.he@intel.com>; Iremonger, Bernard
> > <bernard.iremonger@intel.com>
> > Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Zhang, Helin <helin.zhang@intel.com>
> > Subject: Re: [PATCH V4 5/9] bus: add helper to handle sigbus
> >
> > hi, konstantin
> >
> >
> > On 6/29/2018 6:51 PM, Ananyev, Konstantin wrote:
> > >> +int
> > >> +rte_bus_sigbus_handler(const void *failure_addr)
> > >> +{
> > >> + struct rte_bus *bus;
> > >> + int old_errno = rte_errno;
> > >> + int ret = 0;
> > >> +
> > >> + rte_errno = 0;
> > >> +
> > >> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
> > >> + if (bus == NULL) {
> > >> + RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
> > >> + ret = -1;
> > >> + } else if (rte_errno != 0) {
> > >> + RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
> > >> + ret = -1;
> > >> + }
> > >> +
> > >> + /* if sigbus not be handled, return back old errno. */
> > >> + if (ret)
> > >> + rte_errno = old_errno;
> > > Hmm, not sure why we need to set/restore rte_errno here?
> >
> > restore old_errno just use to let caller know that the generic sigbus
> > still not handler by bus hotplug handler, that involve find a bus
> > handle but failed and can not find a hander, and can corresponding use
> > the previous sigbus handler to process it.
> > that is also unwser your question in other patch. do you think that make
> > sense?
>
> Sorry, still don't understand the intention.
> Suppose rte_bus_find() will return NULL, in that case you'll setup rte_errno
> to what it was before calling that function.
> If the returned bus is not NULL, but bus_find() set's an rte_errno,
> you still would restore rte_ernno?
> What is the prupose?
> Why do you need to touch rte_errno at all in that function?
> Konstantin
>
The way it is written here does not work, but the intention is
to make sure that a previous error is still catched. Something like
that:
int old_errno = rte_errno;
rte_errno = 0;
rte_eal_call();
if (rte_errno)
return -1;
else {
rte_errno = old_errno;
return 0;
}
If someone calls the function while rte_errno is already set, then an
earlier error would be hidden by setting rte_errno to 0 within the
function.
I'm not sure this is useful, but sometimes when using errno within a
library call I'm bothered that I am masking previous issues.
Should it be avoided?
--
Gaëtan Rivet
6WIND
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 5/9] bus: add helper to handle sigbus
2018-06-29 12:52 ` Gaëtan Rivet
@ 2018-07-03 11:24 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-07-03 11:24 UTC (permalink / raw)
To: Gaëtan Rivet, Ananyev, Konstantin
Cc: stephen@networkplumber.org, Richardson, Bruce, Yigit, Ferruh,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng,
Iremonger, Bernard, jblunck@infradead.org, shreyansh.jain@nxp.com,
dev@dpdk.org, Zhang, Helin
hi, gaetan and konstantin
answer both of your questions here as below.
On 6/29/2018 8:52 PM, Gaëtan Rivet wrote:
> On Fri, Jun 29, 2018 at 12:21:39PM +0000, Ananyev, Konstantin wrote:
>>
>>> -----Original Message-----
>>> From: Guo, Jia
>>> Sent: Friday, June 29, 2018 12:23 PM
>>> To: Ananyev, Konstantin <konstantin.ananyev@intel.com>; stephen@networkplumber.org; Richardson, Bruce
>>> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; gaetan.rivet@6wind.com; Wu, Jingjing
>>> <jingjing.wu@intel.com>; thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
>>> <harry.van.haaren@intel.com>; Zhang, Qi Z <qi.z.zhang@intel.com>; He, Shaopeng <shaopeng.he@intel.com>; Iremonger, Bernard
>>> <bernard.iremonger@intel.com>
>>> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Zhang, Helin <helin.zhang@intel.com>
>>> Subject: Re: [PATCH V4 5/9] bus: add helper to handle sigbus
>>>
>>> hi, konstantin
>>>
>>>
>>> On 6/29/2018 6:51 PM, Ananyev, Konstantin wrote:
>>>>> +int
>>>>> +rte_bus_sigbus_handler(const void *failure_addr)
>>>>> +{
>>>>> + struct rte_bus *bus;
>>>>> + int old_errno = rte_errno;
>>>>> + int ret = 0;
>>>>> +
>>>>> + rte_errno = 0;
>>>>> +
>>>>> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
>>>>> + if (bus == NULL) {
>>>>> + RTE_LOG(ERR, EAL, "No bus can handle the sigbus error!");
>>>>> + ret = -1;
>>>>> + } else if (rte_errno != 0) {
>>>>> + RTE_LOG(ERR, EAL, "Failed to handle the sigbus error!");
>>>>> + ret = -1;
>>>>> + }
>>>>> +
>>>>> + /* if sigbus not be handled, return back old errno. */
>>>>> + if (ret)
>>>>> + rte_errno = old_errno;
>>>> Hmm, not sure why we need to set/restore rte_errno here?
>>> restore old_errno just use to let caller know that the generic sigbus
>>> still not handler by bus hotplug handler, that involve find a bus
>>> handle but failed and can not find a hander, and can corresponding use
>>> the previous sigbus handler to process it.
>>> that is also unwser your question in other patch. do you think that make
>>> sense?
>> Sorry, still don't understand the intention.
>> Suppose rte_bus_find() will return NULL, in that case you'll setup rte_errno
>> to what it was before calling that function.
>> If the returned bus is not NULL, but bus_find() set's an rte_errno,
>> you still would restore rte_ernno?
>> What is the prupose?
>> Why do you need to touch rte_errno at all in that function?
>> Konstantin
>>
> The way it is written here does not work, but the intention is
> to make sure that a previous error is still catched. Something like
> that:
>
> int old_errno = rte_errno;
>
> rte_errno = 0;
> rte_eal_call();
>
> if (rte_errno)
> return -1;
> else {
> rte_errno = old_errno;
> return 0;
> }
>
> If someone calls the function while rte_errno is already set, then an
> earlier error would be hidden by setting rte_errno to 0 within the
> function.
>
> I'm not sure this is useful, but sometimes when using errno within a
> library call I'm bothered that I am masking previous issues.
>
> Should it be avoided?
i agree with konstantin about distinguish to process the handle failed
or no handle,
and agree with gaetan about restore the errno if it is not belong to the
sigbus handler.
Could you check if it is fulfill that as bellow,
-1 means find bus but handle failed, use rte_exit.
1 means can no find bus, use older handler to handle.
0 means find bus and success handle. the handler is the new handler.
static int
bus_handle_sigbus(const struct rte_bus *bus,
const void *failure_addr)
{
int ret;
ret = bus->sigbus_handler(failure_addr);
rte_errno = ret;
return !(bus->sigbus_handler && ret <= 0);
}
int
rte_bus_sigbus_handler(const void *failure_addr)
{
struct rte_bus *bus;
int ret = 0;
int old_errno = rte_errno;
rte_errno = 0;
bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
/* failed to handle the sigbus, pass the new errno. */
if (bus && rte_errno == -1)
return -1;
else if (!bus)
ret =1;
/* otherwise restore the old errno. */
rte_errno = old_errno;
return ret;
}
static void sigbus_handler(int signum, siginfo_t *info,
void *ctx __rte_unused)
{
int ret;
rte_spinlock_lock(&dev_failure_lock);
ret = rte_bus_sigbus_handler(info->si_addr);
rte_spinlock_unlock(&dev_failure_lock);
if (ret == -1) {
rte_exit(EXIT_FAILURE,
"Failed to handle SIGBUS for hotplug, "
"(rte_errno: %s)!", strerror(rte_errno));
} else if (ret == 1) {
if (sigbus_action_old.sa_handler)
(*(sigbus_action_old.sa_handler))(signum);
else
rte_exit(EXIT_FAILURE,
"Failed to handle generic SIGBUS!");
}
}
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 6/9] eal: add failure handle mechanism for hot plug
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (4 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 5/9] bus: add helper to handle sigbus Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-06-29 10:49 ` Ananyev, Konstantin
2018-06-29 10:30 ` [PATCH V4 7/9] igb_uio: fix uio release issue when hot unplug Jeff Guo
` (2 subsequent siblings)
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot plug removal event.
First register sigbus handler, once sigbus error be captured, will
check the failure address and accordingly remap the invalid memory
for the corresponding device. Bese on this mechanism, it could
guaranty the application not to be crash when hot unplug devices.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
split patches to be small and clear.
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++++-
1 file changed, 87 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..c9dddab 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,24 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
+#include <rte_errno.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -33,6 +44,34 @@ enum eal_dev_event_subsystem {
EAL_DEV_EVENT_SUBSYSTEM_MAX
};
+static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = rte_bus_sigbus_handler(info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (!ret)
+ RTE_LOG(INFO, EAL,
+ "Success to handle SIGBUS error for hotplug!\n");
+ else
+ rte_exit(EXIT_FAILURE,
+ "A generic SIGBUS error, (rte_errno: %s)!",
+ strerror(rte_errno));
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
static int
dev_uev_socket_fd_create(void)
{
@@ -147,6 +186,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +213,48 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ busname);
+ return;
+ }
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
+ "bus (%s)\n", uevent.devname, busname);
+ return;
+ }
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = bus->hotplug_handler(dev);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Can not handle hotplug for "
+ "device (%s)\n", dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +274,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigaction(SIGBUS, &action, NULL);
+
monitor_started = true;
return 0;
@@ -220,5 +305,6 @@ rte_dev_event_monitor_stop(void)
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 6/9] eal: add failure handle mechanism for hot plug
2018-06-29 10:30 ` [PATCH V4 6/9] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-06-29 10:49 ` Ananyev, Konstantin
2018-06-29 11:15 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-06-29 10:49 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng, Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
Hi Jeff,
>
> This patch introduces a failure handler mechanism to handle device
> hot plug removal event.
>
> First register sigbus handler, once sigbus error be captured, will
> check the failure address and accordingly remap the invalid memory
> for the corresponding device. Bese on this mechanism, it could
> guaranty the application not to be crash when hot unplug devices.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v4->v3:
> split patches to be small and clear.
> ---
> lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++++-
> 1 file changed, 87 insertions(+), 1 deletion(-)
>
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 1cf6aeb..c9dddab 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -14,15 +16,24 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
> +#include <rte_spinlock.h>
> +#include <rte_errno.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 };
> static bool monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
> #define EAL_UEV_MSG_LEN 4096
> #define EAL_UEV_MSG_ELEM_LEN 128
>
> +/* spinlock for device failure process */
> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
> +
> static void dev_uev_handler(__rte_unused void *param);
>
> /* identify the system layer which reports this event. */
> @@ -33,6 +44,34 @@ enum eal_dev_event_subsystem {
> EAL_DEV_EVENT_SUBSYSTEM_MAX
> };
>
> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> + void *ctx __rte_unused)
> +{
> + int ret;
> +
> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
> + (int)pthread_self(), info->si_addr);
> +
> + rte_spinlock_lock(&dev_failure_lock);
> + ret = rte_bus_sigbus_handler(info->si_addr);
> + rte_spinlock_unlock(&dev_failure_lock);
> + if (!ret)
> + RTE_LOG(INFO, EAL,
> + "Success to handle SIGBUS error for hotplug!\n");
> + else
> + rte_exit(EXIT_FAILURE,
> + "A generic SIGBUS error, (rte_errno: %s)!",
> + strerror(rte_errno));
> +}
As I said in comments for previous versions:
I think we need to distinguish why do we fail -
1) address doesn't belong to any device,
2) we failed to remap
For 1) we probably need to call previous sigbus handler.
> +
> +static int cmp_dev_name(const struct rte_device *dev,
> + const void *_name)
> +{
> + const char *name = _name;
> +
> + return strcmp(dev->name, name);
> +}
> +
> static int
> dev_uev_socket_fd_create(void)
> {
> @@ -147,6 +186,9 @@ dev_uev_handler(__rte_unused void *param)
> struct rte_dev_event uevent;
> int ret;
> char buf[EAL_UEV_MSG_LEN];
> + struct rte_bus *bus;
> + struct rte_device *dev;
> + const char *busname;
>
> memset(&uevent, 0, sizeof(struct rte_dev_event));
> memset(buf, 0, EAL_UEV_MSG_LEN);
> @@ -171,13 +213,48 @@ dev_uev_handler(__rte_unused void *param)
> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> uevent.devname, uevent.type, uevent.subsystem);
>
> - if (uevent.devname)
> + switch (uevent.subsystem) {
> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> + busname = "pci";
> + break;
> + default:
> + break;
> + }
> +
> + if (uevent.devname) {
> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> + bus = rte_bus_find_by_name(busname);
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> + busname);
> + return;
> + }
> + dev = bus->find_device(NULL, cmp_dev_name,
> + uevent.devname);
> + if (dev == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
> + "bus (%s)\n", uevent.devname, busname);
> + return;
> + }
> + rte_spinlock_lock(&dev_failure_lock);
> + ret = bus->hotplug_handler(dev);
> + rte_spinlock_unlock(&dev_failure_lock);
Ok, but this function is executed from interrupt thread, correct?
What would happen if user would do dev-detach() at the same time and dev would not be valid anymore?
Shouldn't we have a lock (per bus?) that we would grab before find_device() and release after hotplug_handler?
Though in that case we probably need to revisit other bus ops too.
> + if (ret) {
> + RTE_LOG(ERR, EAL, "Can not handle hotplug for "
> + "device (%s)\n", dev->name);
> + return;
> + }
> + }
> dev_callback_process(uevent.devname, uevent.type);
> + }
> }
>
> int __rte_experimental
> rte_dev_event_monitor_start(void)
> {
> + sigset_t mask;
> + struct sigaction action;
> int ret;
>
> if (monitor_started)
> @@ -197,6 +274,14 @@ rte_dev_event_monitor_start(void)
> return -1;
> }
>
> + /* register sigbus handler */
> + sigemptyset(&mask);
> + sigaddset(&mask, SIGBUS);
> + action.sa_flags = SA_SIGINFO;
> + action.sa_mask = mask;
> + action.sa_sigaction = sigbus_handler;
> + sigaction(SIGBUS, &action, NULL);
> +
I still think we have to save (and restore at monitor_stop) previous sigbus handler.
> monitor_started = true;
>
> return 0;
> @@ -220,5 +305,6 @@ rte_dev_event_monitor_stop(void)
> close(intr_handle.fd);
> intr_handle.fd = -1;
> monitor_started = false;
> +
> return 0;
> }
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 6/9] eal: add failure handle mechanism for hot plug
2018-06-29 10:49 ` Ananyev, Konstantin
@ 2018-06-29 11:15 ` Guo, Jia
2018-06-29 12:06 ` Ananyev, Konstantin
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-06-29 11:15 UTC (permalink / raw)
To: Ananyev, Konstantin, stephen@networkplumber.org,
Richardson, Bruce, Yigit, Ferruh, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
hi,konstantin
On 6/29/2018 6:49 PM, Ananyev, Konstantin wrote:
> Hi Jeff,
>
>> This patch introduces a failure handler mechanism to handle device
>> hot plug removal event.
>>
>> First register sigbus handler, once sigbus error be captured, will
>> check the failure address and accordingly remap the invalid memory
>> for the corresponding device. Bese on this mechanism, it could
>> guaranty the application not to be crash when hot unplug devices.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v4->v3:
>> split patches to be small and clear.
>> ---
>> lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++++-
>> 1 file changed, 87 insertions(+), 1 deletion(-)
>>
>> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> index 1cf6aeb..c9dddab 100644
>> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> @@ -4,6 +4,8 @@
>>
>> #include <string.h>
>> #include <unistd.h>
>> +#include <fcntl.h>
>> +#include <signal.h>
>> #include <sys/socket.h>
>> #include <linux/netlink.h>
>>
>> @@ -14,15 +16,24 @@
>> #include <rte_malloc.h>
>> #include <rte_interrupts.h>
>> #include <rte_alarm.h>
>> +#include <rte_bus.h>
>> +#include <rte_eal.h>
>> +#include <rte_spinlock.h>
>> +#include <rte_errno.h>
>>
>> #include "eal_private.h"
>>
>> static struct rte_intr_handle intr_handle = {.fd = -1 };
>> static bool monitor_started;
>>
>> +extern struct rte_bus_list rte_bus_list;
>> +
>> #define EAL_UEV_MSG_LEN 4096
>> #define EAL_UEV_MSG_ELEM_LEN 128
>>
>> +/* spinlock for device failure process */
>> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
>> +
>> static void dev_uev_handler(__rte_unused void *param);
>>
>> /* identify the system layer which reports this event. */
>> @@ -33,6 +44,34 @@ enum eal_dev_event_subsystem {
>> EAL_DEV_EVENT_SUBSYSTEM_MAX
>> };
>>
>> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
>> + void *ctx __rte_unused)
>> +{
>> + int ret;
>> +
>> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
>> + (int)pthread_self(), info->si_addr);
>> +
>> + rte_spinlock_lock(&dev_failure_lock);
>> + ret = rte_bus_sigbus_handler(info->si_addr);
>> + rte_spinlock_unlock(&dev_failure_lock);
>> + if (!ret)
>> + RTE_LOG(INFO, EAL,
>> + "Success to handle SIGBUS error for hotplug!\n");
>> + else
>> + rte_exit(EXIT_FAILURE,
>> + "A generic SIGBUS error, (rte_errno: %s)!",
>> + strerror(rte_errno));
>> +}
> As I said in comments for previous versions:
> I think we need to distinguish why do we fail -
> 1) address doesn't belong to any device,
> 2) we failed to remap
> For 1) we probably need to call previous sigbus handler.
i know your point, but i think what ever 1) or 2), we should also need
to call previous sigbus handler to show exception of the memory error,
to cut down any other after try to run.
and for the previous sigbus handler, i still not find a explicit call to
use it, i think it the sigbus handler could be restore but only will use
when next error occur, right?
if so, do you think i just use a rte_exit to replace this origin handler
is make sense?
>> +
>> +static int cmp_dev_name(const struct rte_device *dev,
>> + const void *_name)
>> +{
>> + const char *name = _name;
>> +
>> + return strcmp(dev->name, name);
>> +}
>> +
>> static int
>> dev_uev_socket_fd_create(void)
>> {
>> @@ -147,6 +186,9 @@ dev_uev_handler(__rte_unused void *param)
>> struct rte_dev_event uevent;
>> int ret;
>> char buf[EAL_UEV_MSG_LEN];
>> + struct rte_bus *bus;
>> + struct rte_device *dev;
>> + const char *busname;
>>
>> memset(&uevent, 0, sizeof(struct rte_dev_event));
>> memset(buf, 0, EAL_UEV_MSG_LEN);
>> @@ -171,13 +213,48 @@ dev_uev_handler(__rte_unused void *param)
>> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
>> uevent.devname, uevent.type, uevent.subsystem);
>>
>> - if (uevent.devname)
>> + switch (uevent.subsystem) {
>> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
>> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
>> + busname = "pci";
>> + break;
>> + default:
>> + break;
>> + }
>> +
>> + if (uevent.devname) {
>> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
>> + bus = rte_bus_find_by_name(busname);
>> + if (bus == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
>> + busname);
>> + return;
>> + }
>> + dev = bus->find_device(NULL, cmp_dev_name,
>> + uevent.devname);
>> + if (dev == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
>> + "bus (%s)\n", uevent.devname, busname);
>> + return;
>> + }
>> + rte_spinlock_lock(&dev_failure_lock);
>> + ret = bus->hotplug_handler(dev);
>> + rte_spinlock_unlock(&dev_failure_lock);
> Ok, but this function is executed from interrupt thread, correct?
yes.
> What would happen if user would do dev-detach() at the same time and dev would not be valid anymore?
> Shouldn't we have a lock (per bus?) that we would grab before find_device() and release after hotplug_handler?
> Though in that case we probably need to revisit other bus ops too.
make sense, i think should be the case and need lock any bus ops here to
sync.
>> + if (ret) {
>> + RTE_LOG(ERR, EAL, "Can not handle hotplug for "
>> + "device (%s)\n", dev->name);
>> + return;
>> + }
>> + }
>> dev_callback_process(uevent.devname, uevent.type);
>> + }
>> }
>>
>> int __rte_experimental
>> rte_dev_event_monitor_start(void)
>> {
>> + sigset_t mask;
>> + struct sigaction action;
>> int ret;
>>
>> if (monitor_started)
>> @@ -197,6 +274,14 @@ rte_dev_event_monitor_start(void)
>> return -1;
>> }
>>
>> + /* register sigbus handler */
>> + sigemptyset(&mask);
>> + sigaddset(&mask, SIGBUS);
>> + action.sa_flags = SA_SIGINFO;
>> + action.sa_mask = mask;
>> + action.sa_sigaction = sigbus_handler;
>> + sigaction(SIGBUS, &action, NULL);
>> +
> I still think we have to save (and restore at monitor_stop) previous sigbus handler.
ok, i think i missing here, if monitor_stop, definitely should restore
the previous sigbus handler.
>> monitor_started = true;
>>
>> return 0;
>> @@ -220,5 +305,6 @@ rte_dev_event_monitor_stop(void)
>> close(intr_handle.fd);
>> intr_handle.fd = -1;
>> monitor_started = false;
>> +
>> return 0;
>> }
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 6/9] eal: add failure handle mechanism for hot plug
2018-06-29 11:15 ` Guo, Jia
@ 2018-06-29 12:06 ` Ananyev, Konstantin
0 siblings, 0 replies; 494+ messages in thread
From: Ananyev, Konstantin @ 2018-06-29 12:06 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, gaetan.rivet@6wind.com, Wu, Jingjing,
thomas@monjalon.net, motih@mellanox.com, matan@mellanox.com,
Van Haaren, Harry, Zhang, Qi Z, He, Shaopeng, Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Friday, June 29, 2018 12:15 PM
> To: Ananyev, Konstantin <konstantin.ananyev@intel.com>; stephen@networkplumber.org; Richardson, Bruce
> <bruce.richardson@intel.com>; Yigit, Ferruh <ferruh.yigit@intel.com>; gaetan.rivet@6wind.com; Wu, Jingjing
> <jingjing.wu@intel.com>; thomas@monjalon.net; motih@mellanox.com; matan@mellanox.com; Van Haaren, Harry
> <harry.van.haaren@intel.com>; Zhang, Qi Z <qi.z.zhang@intel.com>; He, Shaopeng <shaopeng.he@intel.com>; Iremonger, Bernard
> <bernard.iremonger@intel.com>
> Cc: jblunck@infradead.org; shreyansh.jain@nxp.com; dev@dpdk.org; Zhang, Helin <helin.zhang@intel.com>
> Subject: Re: [PATCH V4 6/9] eal: add failure handle mechanism for hot plug
>
> hi,konstantin
>
>
> On 6/29/2018 6:49 PM, Ananyev, Konstantin wrote:
> > Hi Jeff,
> >
> >> This patch introduces a failure handler mechanism to handle device
> >> hot plug removal event.
> >>
> >> First register sigbus handler, once sigbus error be captured, will
> >> check the failure address and accordingly remap the invalid memory
> >> for the corresponding device. Bese on this mechanism, it could
> >> guaranty the application not to be crash when hot unplug devices.
> >>
> >> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> >> ---
> >> v4->v3:
> >> split patches to be small and clear.
> >> ---
> >> lib/librte_eal/linuxapp/eal/eal_dev.c | 88 ++++++++++++++++++++++++++++++++++-
> >> 1 file changed, 87 insertions(+), 1 deletion(-)
> >>
> >> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> index 1cf6aeb..c9dddab 100644
> >> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> >> @@ -4,6 +4,8 @@
> >>
> >> #include <string.h>
> >> #include <unistd.h>
> >> +#include <fcntl.h>
> >> +#include <signal.h>
> >> #include <sys/socket.h>
> >> #include <linux/netlink.h>
> >>
> >> @@ -14,15 +16,24 @@
> >> #include <rte_malloc.h>
> >> #include <rte_interrupts.h>
> >> #include <rte_alarm.h>
> >> +#include <rte_bus.h>
> >> +#include <rte_eal.h>
> >> +#include <rte_spinlock.h>
> >> +#include <rte_errno.h>
> >>
> >> #include "eal_private.h"
> >>
> >> static struct rte_intr_handle intr_handle = {.fd = -1 };
> >> static bool monitor_started;
> >>
> >> +extern struct rte_bus_list rte_bus_list;
> >> +
> >> #define EAL_UEV_MSG_LEN 4096
> >> #define EAL_UEV_MSG_ELEM_LEN 128
> >>
> >> +/* spinlock for device failure process */
> >> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
> >> +
> >> static void dev_uev_handler(__rte_unused void *param);
> >>
> >> /* identify the system layer which reports this event. */
> >> @@ -33,6 +44,34 @@ enum eal_dev_event_subsystem {
> >> EAL_DEV_EVENT_SUBSYSTEM_MAX
> >> };
> >>
> >> +static void sigbus_handler(int signum __rte_unused, siginfo_t *info,
> >> + void *ctx __rte_unused)
> >> +{
> >> + int ret;
> >> +
> >> + RTE_LOG(DEBUG, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
> >> + (int)pthread_self(), info->si_addr);
> >> +
> >> + rte_spinlock_lock(&dev_failure_lock);
> >> + ret = rte_bus_sigbus_handler(info->si_addr);
> >> + rte_spinlock_unlock(&dev_failure_lock);
> >> + if (!ret)
> >> + RTE_LOG(INFO, EAL,
> >> + "Success to handle SIGBUS error for hotplug!\n");
> >> + else
> >> + rte_exit(EXIT_FAILURE,
> >> + "A generic SIGBUS error, (rte_errno: %s)!",
> >> + strerror(rte_errno));
> >> +}
> > As I said in comments for previous versions:
> > I think we need to distinguish why do we fail -
> > 1) address doesn't belong to any device,
> > 2) we failed to remap
> > For 1) we probably need to call previous sigbus handler.
>
> i know your point, but i think what ever 1) or 2), we should also need
> to call previous sigbus handler to show exception of the memory error,
> to cut down any other after try to run.
I don't agree.
If 1) - that error doesn't belong to us (DPDK), but user app might know how to handle it.
So we just invoke previously saved previous (if any) sigbus handler.
for 2) - there is not much we can do but rte_exit().
> and for the previous sigbus handler, i still not find a explicit call to
> use it, i think it the sigbus handler could be restore but only will use
> when next error occur, right?
> if so, do you think i just use a rte_exit to replace this origin handler
> is make sense?
>
> >> +
> >> +static int cmp_dev_name(const struct rte_device *dev,
> >> + const void *_name)
> >> +{
> >> + const char *name = _name;
> >> +
> >> + return strcmp(dev->name, name);
> >> +}
> >> +
> >> static int
> >> dev_uev_socket_fd_create(void)
> >> {
> >> @@ -147,6 +186,9 @@ dev_uev_handler(__rte_unused void *param)
> >> struct rte_dev_event uevent;
> >> int ret;
> >> char buf[EAL_UEV_MSG_LEN];
> >> + struct rte_bus *bus;
> >> + struct rte_device *dev;
> >> + const char *busname;
> >>
> >> memset(&uevent, 0, sizeof(struct rte_dev_event));
> >> memset(buf, 0, EAL_UEV_MSG_LEN);
> >> @@ -171,13 +213,48 @@ dev_uev_handler(__rte_unused void *param)
> >> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> >> uevent.devname, uevent.type, uevent.subsystem);
> >>
> >> - if (uevent.devname)
> >> + switch (uevent.subsystem) {
> >> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> >> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> >> + busname = "pci";
> >> + break;
> >> + default:
> >> + break;
> >> + }
> >> +
> >> + if (uevent.devname) {
> >> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> >> + bus = rte_bus_find_by_name(busname);
> >> + if (bus == NULL) {
> >> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> >> + busname);
> >> + return;
> >> + }
> >> + dev = bus->find_device(NULL, cmp_dev_name,
> >> + uevent.devname);
> >> + if (dev == NULL) {
> >> + RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
> >> + "bus (%s)\n", uevent.devname, busname);
> >> + return;
> >> + }
> >> + rte_spinlock_lock(&dev_failure_lock);
> >> + ret = bus->hotplug_handler(dev);
> >> + rte_spinlock_unlock(&dev_failure_lock);
> > Ok, but this function is executed from interrupt thread, correct?
>
> yes.
>
> > What would happen if user would do dev-detach() at the same time and dev would not be valid anymore?
> > Shouldn't we have a lock (per bus?) that we would grab before find_device() and release after hotplug_handler?
> > Though in that case we probably need to revisit other bus ops too.
>
> make sense, i think should be the case and need lock any bus ops here to
> sync.
>
> >> + if (ret) {
> >> + RTE_LOG(ERR, EAL, "Can not handle hotplug for "
> >> + "device (%s)\n", dev->name);
> >> + return;
> >> + }
> >> + }
> >> dev_callback_process(uevent.devname, uevent.type);
> >> + }
> >> }
> >>
> >> int __rte_experimental
> >> rte_dev_event_monitor_start(void)
> >> {
> >> + sigset_t mask;
> >> + struct sigaction action;
> >> int ret;
> >>
> >> if (monitor_started)
> >> @@ -197,6 +274,14 @@ rte_dev_event_monitor_start(void)
> >> return -1;
> >> }
> >>
> >> + /* register sigbus handler */
> >> + sigemptyset(&mask);
> >> + sigaddset(&mask, SIGBUS);
> >> + action.sa_flags = SA_SIGINFO;
> >> + action.sa_mask = mask;
> >> + action.sa_sigaction = sigbus_handler;
> >> + sigaction(SIGBUS, &action, NULL);
> >> +
> > I still think we have to save (and restore at monitor_stop) previous sigbus handler.
>
> ok, i think i missing here, if monitor_stop, definitely should restore
> the previous sigbus handler.
>
> >> monitor_started = true;
> >>
> >> return 0;
> >> @@ -220,5 +305,6 @@ rte_dev_event_monitor_stop(void)
> >> close(intr_handle.fd);
> >> intr_handle.fd = -1;
> >> monitor_started = false;
> >> +
> >> return 0;
> >> }
> >> --
> >> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 7/9] igb_uio: fix uio release issue when hot unplug
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (5 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 6/9] eal: add failure handle mechanism for hot plug Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-07-03 12:12 ` Ferruh Yigit
2018-06-29 10:30 ` [PATCH V4 8/9] app/testpmd: show example to handle " Jeff Guo
2018-06-29 10:30 ` [PATCH V4 9/9] app/testpmd: enable device hotplug monitoring Jeff Guo
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this
resource, it will cause kernel crash. On the other hand, something like
interrupt disabling do not automatically process in kernel side. If not
handler it, this redundancy and dirty thing will affect the interrupt
resource be used by other device. So the igb_uio driver have to check the
hot plug status, and the corresponding process should be taken in igb uio
driver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such
as probed/opened/released/removed/unplug. When detect the unexpected
removal which cause of hot unplug behavior, it will corresponding disable
interrupt resource, while for the part of releasement which kernel have
already handle, just skip it to avoid double free or null pointer kernel
crash issue.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
no change
---
kernel/linux/igb_uio/igb_uio.c | 50 +++++++++++++++++++++++++++++++++++++-----
1 file changed, 45 insertions(+), 5 deletions(-)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index b3233f1..d301302 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -19,6 +19,15 @@
#include "compat.h"
+/* uio pci device state */
+enum rte_udev_state {
+ RTE_UDEV_PROBED,
+ RTE_UDEV_OPENNED,
+ RTE_UDEV_RELEASED,
+ RTE_UDEV_REMOVED,
+ RTE_UDEV_UNPLUG
+};
+
/**
* A structure describing the private information for a uio device.
*/
@@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
enum rte_intr_mode mode;
struct mutex lock;
int refcnt;
+ enum rte_udev_state state;
};
static char *intr_mode;
@@ -194,12 +204,20 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
{
struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
struct uio_info *info = &udev->info;
+ struct pci_dev *pdev = udev->pdev;
/* Legacy mode need to mask in hardware */
if (udev->mode == RTE_INTR_MODE_LEGACY &&
!pci_check_and_mask_intx(udev->pdev))
return IRQ_NONE;
+ /* check the uevent of the kobj */
+ if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
+ dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
+ (&pdev->dev.kobj)->name);
+ udev->state = RTE_UDEV_UNPLUG;
+ }
+
uio_event_notify(info);
/* Message signal mode, no share IRQ and automasked */
@@ -308,7 +326,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
#endif
}
-
/**
* This gets called while opening uio device file.
*/
@@ -330,24 +347,33 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
/* enable interrupts */
err = igbuio_pci_enable_interrupts(udev);
- mutex_unlock(&udev->lock);
if (err) {
dev_err(&dev->dev, "Enable interrupt fails\n");
+ pci_clear_master(dev);
return err;
}
+ udev->state = RTE_UDEV_OPENNED;
+ mutex_unlock(&udev->lock);
return 0;
}
+/**
+ * This gets called while closing uio device file.
+ */
static int
igbuio_pci_release(struct uio_info *info, struct inode *inode)
{
+
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ if (udev->state == RTE_UDEV_REMOVED)
+ return 0;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
- return 0;
+ return -1;
}
/* disable interrupts */
@@ -355,7 +381,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
/* stop the device from further DMA */
pci_clear_master(dev);
-
+ udev->state = RTE_UDEV_RELEASED;
mutex_unlock(&udev->lock);
return 0;
}
@@ -557,6 +583,7 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
(unsigned long long)map_dma_addr, map_addr);
}
+ udev->state = RTE_UDEV_PROBED;
return 0;
fail_remove_group:
@@ -573,11 +600,24 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
static void
igbuio_pci_remove(struct pci_dev *dev)
{
+
struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
+ int ret;
+
+ /* handler hot unplug */
+ if (udev->state == RTE_UDEV_OPENNED ||
+ udev->state == RTE_UDEV_UNPLUG) {
+ dev_notice(&dev->dev, "Unexpected removal!\n");
+ ret = igbuio_pci_release(&udev->info, NULL);
+ if (ret)
+ return;
+ udev->state = RTE_UDEV_REMOVED;
+ return;
+ }
mutex_destroy(&udev->lock);
- sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
uio_unregister_device(&udev->info);
+ sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
igbuio_pci_release_iomem(&udev->info);
pci_disable_device(dev);
pci_set_drvdata(dev, NULL);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 7/9] igb_uio: fix uio release issue when hot unplug
2018-06-29 10:30 ` [PATCH V4 7/9] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-07-03 12:12 ` Ferruh Yigit
0 siblings, 0 replies; 494+ messages in thread
From: Ferruh Yigit @ 2018-07-03 12:12 UTC (permalink / raw)
To: Jeff Guo, stephen, bruce.richardson, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, helin.zhang
On 6/29/2018 11:30 AM, Jeff Guo wrote:
> When hot unplug device, the kernel will release the device resource in the
> kernel side, such as the fd sys file will disappear, and the irq will be
> released. At this time, if igb uio driver still try to release this
> resource, it will cause kernel crash. On the other hand, something like
> interrupt disabling do not automatically process in kernel side. If not
> handler it, this redundancy and dirty thing will affect the interrupt
> resource be used by other device. So the igb_uio driver have to check the
> hot plug status, and the corresponding process should be taken in igb uio
> driver.
>
> This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
> of igb_uio kernel driver, which will record the state of uio device, such
> as probed/opened/released/removed/unplug. When detect the unexpected
> removal which cause of hot unplug behavior, it will corresponding disable
> interrupt resource, while for the part of releasement which kernel have
> already handle, just skip it to avoid double free or null pointer kernel
> crash issue.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v4->v3:
> no change
> ---
> kernel/linux/igb_uio/igb_uio.c | 50 +++++++++++++++++++++++++++++++++++++-----
> 1 file changed, 45 insertions(+), 5 deletions(-)
>
> diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
> index b3233f1..d301302 100644
> --- a/kernel/linux/igb_uio/igb_uio.c
> +++ b/kernel/linux/igb_uio/igb_uio.c
> @@ -19,6 +19,15 @@
>
> #include "compat.h"
>
> +/* uio pci device state */
> +enum rte_udev_state {
> + RTE_UDEV_PROBED,
> + RTE_UDEV_OPENNED,
> + RTE_UDEV_RELEASED,
> + RTE_UDEV_REMOVED,
> + RTE_UDEV_UNPLUG
> +};
> +
> /**
> * A structure describing the private information for a uio device.
> */
> @@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
> enum rte_intr_mode mode;
> struct mutex lock;
> int refcnt;
> + enum rte_udev_state state;
> };
>
> static char *intr_mode;
> @@ -194,12 +204,20 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
> {
> struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
> struct uio_info *info = &udev->info;
> + struct pci_dev *pdev = udev->pdev;
>
> /* Legacy mode need to mask in hardware */
> if (udev->mode == RTE_INTR_MODE_LEGACY &&
> !pci_check_and_mask_intx(udev->pdev))
> return IRQ_NONE;
>
> + /* check the uevent of the kobj */
> + if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
> + dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
> + (&pdev->dev.kobj)->name);
> + udev->state = RTE_UDEV_UNPLUG;
> + }
I guess commit log says kernel can remove device, if so do we need any locking
before accessing dev?
> +
> uio_event_notify(info);
>
> /* Message signal mode, no share IRQ and automasked */
> @@ -308,7 +326,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
> #endif
> }
>
> -
> /**
> * This gets called while opening uio device file.
> */
> @@ -330,24 +347,33 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
>
> /* enable interrupts */
> err = igbuio_pci_enable_interrupts(udev);
> - mutex_unlock(&udev->lock);
> if (err) {
> dev_err(&dev->dev, "Enable interrupt fails\n");
> + pci_clear_master(dev);
> return err;
> }
> + udev->state = RTE_UDEV_OPENNED;
> + mutex_unlock(&udev->lock);
> return 0;
> }
>
> +/**
> + * This gets called while closing uio device file.
> + */
> static int
> igbuio_pci_release(struct uio_info *info, struct inode *inode)
> {
> +
> struct rte_uio_pci_dev *udev = info->priv;
> struct pci_dev *dev = udev->pdev;
>
> + if (udev->state == RTE_UDEV_REMOVED)
> + return 0;
> +
> mutex_lock(&udev->lock);
> if (--udev->refcnt > 0) {
> mutex_unlock(&udev->lock);
> - return 0;
> + return -1;
> }
>
> /* disable interrupts */
> @@ -355,7 +381,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
>
> /* stop the device from further DMA */
> pci_clear_master(dev);
> -
> + udev->state = RTE_UDEV_RELEASED;
> mutex_unlock(&udev->lock);
> return 0;
> }
> @@ -557,6 +583,7 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
> (unsigned long long)map_dma_addr, map_addr);
> }
>
> + udev->state = RTE_UDEV_PROBED;
> return 0;
>
> fail_remove_group:
> @@ -573,11 +600,24 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
> static void
> igbuio_pci_remove(struct pci_dev *dev)
> {
> +
> struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
> + int ret;
> +
> + /* handler hot unplug */
> + if (udev->state == RTE_UDEV_OPENNED ||
> + udev->state == RTE_UDEV_UNPLUG) {
> + dev_notice(&dev->dev, "Unexpected removal!\n");
> + ret = igbuio_pci_release(&udev->info, NULL);
> + if (ret)
> + return;
> + udev->state = RTE_UDEV_REMOVED;
> + return;
> + }
>
> mutex_destroy(&udev->lock);
> - sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
> uio_unregister_device(&udev->info);
> + sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
> igbuio_pci_release_iomem(&udev->info);
> pci_disable_device(dev);
> pci_set_drvdata(dev, NULL);
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (6 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 7/9] igb_uio: fix uio release issue when hot unplug Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
2018-07-01 7:46 ` Matan Azrad
2018-06-29 10:30 ` [PATCH V4 9/9] app/testpmd: enable device hotplug monitoring Jeff Guo
8 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
Use testpmd for example, to show how an application smoothly handle
failure when device being hot unplug. If app have enabled the device event
monitor and register the hot plug event’s callback before running, once
app detect the removal event, the callback would be called. It will first
stop the packet forwarding, then stop the port, close the port, and finally
detach the port to remove the device out from the device lists.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
remove some unused code
---
app/test-pmd/testpmd.c | 13 +++++++++----
1 file changed, 9 insertions(+), 4 deletions(-)
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 24c1998..42ed196 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -2196,6 +2196,9 @@ static void
eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
__rte_unused void *arg)
{
+ uint16_t port_id;
+ int ret;
+
if (type >= RTE_DEV_EVENT_MAX) {
fprintf(stderr, "%s called upon invalid event %d\n",
__func__, type);
@@ -2206,9 +2209,12 @@ eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
case RTE_DEV_EVENT_REMOVE:
RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
device_name);
- /* TODO: After finish failure handle, begin to stop
- * packet forward, stop port, close port, detach port.
- */
+ ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
+ if (ret) {
+ printf("can not get port by device %s!\n", device_name);
+ return;
+ }
+ rmv_event_callback((void *)(intptr_t)port_id);
break;
case RTE_DEV_EVENT_ADD:
RTE_LOG(ERR, EAL, "The device: %s has been added!\n",
@@ -2736,7 +2742,6 @@ main(int argc, char** argv)
return -1;
}
eth_dev_event_callback_register();
-
}
if (start_port(RTE_PORT_ALL) != 0)
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-06-29 10:30 ` [PATCH V4 8/9] app/testpmd: show example to handle " Jeff Guo
@ 2018-07-01 7:46 ` Matan Azrad
2018-07-03 9:35 ` Guo, Jia
0 siblings, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-07-01 7:46 UTC (permalink / raw)
To: Jeff Guo, stephen@networkplumber.org, bruce.richardson@intel.com,
ferruh.yigit@intel.com, konstantin.ananyev@intel.com,
gaetan.rivet@6wind.com, jingjing.wu@intel.com, Thomas Monjalon,
Mordechay Haimovsky, harry.van.haaren@intel.com,
qi.z.zhang@intel.com, shaopeng.he@intel.com,
bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
Hi Jeff
A good advance, thank you, but as I said in previous version, this patch inserts a bug and the next one fixes it.
Patch 9 should be before patch 8 while this patch just add 1 more option for EAL hotplug.
Please see 1 more comment below.
From: Jeff Guo
> Use testpmd for example, to show how an application smoothly handle failure
> when device being hot unplug. If app have enabled the device event monitor
> and register the hot plug event’s callback before running, once app detect the
> removal event, the callback would be called. It will first stop the packet
> forwarding, then stop the port, close the port, and finally detach the port to
> remove the device out from the device lists.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v4->v3:
> remove some unused code
> ---
> app/test-pmd/testpmd.c | 13 +++++++++----
> 1 file changed, 9 insertions(+), 4 deletions(-)
>
> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
> 24c1998..42ed196 100644
> --- a/app/test-pmd/testpmd.c
> +++ b/app/test-pmd/testpmd.c
> @@ -2196,6 +2196,9 @@ static void
> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
> __rte_unused void *arg)
> {
> + uint16_t port_id;
> + int ret;
> +
> if (type >= RTE_DEV_EVENT_MAX) {
> fprintf(stderr, "%s called upon invalid event %d\n",
> __func__, type);
> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char *device_name, enum
> rte_dev_event_type type,
> case RTE_DEV_EVENT_REMOVE:
> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> device_name);
> - /* TODO: After finish failure handle, begin to stop
> - * packet forward, stop port, close port, detach port.
> - */
> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
As you probably know, 1 rte_device may be associated to more than one ethdev ports, so the ethdev port name can be different from rte_device name.
Looks like we need a new ethdev API to get all the ports associated to one rte_device.
> + if (ret) {
> + printf("can not get port by device %s!\n",
> device_name);
> + return;
> + }
> + rmv_event_callback((void *)(intptr_t)port_id);
> break;
> case RTE_DEV_EVENT_ADD:
> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
> 2736,7 +2742,6 @@ main(int argc, char** argv)
> return -1;
> }
> eth_dev_event_callback_register();
> -
> }
>
> if (start_port(RTE_PORT_ALL) != 0)
> --
> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-07-01 7:46 ` Matan Azrad
@ 2018-07-03 9:35 ` Guo, Jia
2018-07-03 22:44 ` Thomas Monjalon
0 siblings, 1 reply; 494+ messages in thread
From: Guo, Jia @ 2018-07-03 9:35 UTC (permalink / raw)
To: Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Thomas Monjalon, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
helin.zhang@intel.com
mantan,
On 7/1/2018 3:46 PM, Matan Azrad wrote:
> Hi Jeff
>
> A good advance, thank you, but as I said in previous version, this patch inserts a bug and the next one fixes it.
> Patch 9 should be before patch 8 while this patch just add 1 more option for EAL hotplug.
i agree that patch 9 before patch 8 could be better. thank.
> Please see 1 more comment below.
>
> From: Jeff Guo
>> Use testpmd for example, to show how an application smoothly handle failure
>> when device being hot unplug. If app have enabled the device event monitor
>> and register the hot plug event’s callback before running, once app detect the
>> removal event, the callback would be called. It will first stop the packet
>> forwarding, then stop the port, close the port, and finally detach the port to
>> remove the device out from the device lists.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v4->v3:
>> remove some unused code
>> ---
>> app/test-pmd/testpmd.c | 13 +++++++++----
>> 1 file changed, 9 insertions(+), 4 deletions(-)
>>
>> diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c index
>> 24c1998..42ed196 100644
>> --- a/app/test-pmd/testpmd.c
>> +++ b/app/test-pmd/testpmd.c
>> @@ -2196,6 +2196,9 @@ static void
>> eth_dev_event_callback(char *device_name, enum rte_dev_event_type type,
>> __rte_unused void *arg)
>> {
>> + uint16_t port_id;
>> + int ret;
>> +
>> if (type >= RTE_DEV_EVENT_MAX) {
>> fprintf(stderr, "%s called upon invalid event %d\n",
>> __func__, type);
>> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char *device_name, enum
>> rte_dev_event_type type,
>> case RTE_DEV_EVENT_REMOVE:
>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>> device_name);
>> - /* TODO: After finish failure handle, begin to stop
>> - * packet forward, stop port, close port, detach port.
>> - */
>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> As you probably know, 1 rte_device may be associated to more than one ethdev ports, so the ethdev port name can be different from rte_device name.
> Looks like we need a new ethdev API to get all the ports associated to one rte_device.
agree, seems that the the old ethdev API have some issue when got all
port by device name. we could check with ethdev maintainer and fix it by
specific ethdev patch later.
>> + if (ret) {
>> + printf("can not get port by device %s!\n",
>> device_name);
>> + return;
>> + }
>> + rmv_event_callback((void *)(intptr_t)port_id);
>> break;
>> case RTE_DEV_EVENT_ADD:
>> RTE_LOG(ERR, EAL, "The device: %s has been added!\n", @@ -
>> 2736,7 +2742,6 @@ main(int argc, char** argv)
>> return -1;
>> }
>> eth_dev_event_callback_register();
>> -
>> }
>>
>> if (start_port(RTE_PORT_ALL) != 0)
>> --
>> 2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-07-03 9:35 ` Guo, Jia
@ 2018-07-03 22:44 ` Thomas Monjalon
2018-07-04 3:48 ` Guo, Jia
2018-07-04 7:06 ` Matan Azrad
0 siblings, 2 replies; 494+ messages in thread
From: Thomas Monjalon @ 2018-07-03 22:44 UTC (permalink / raw)
To: Guo, Jia
Cc: dev, Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com,
shreyansh.jain@nxp.com, helin.zhang@intel.com
03/07/2018 11:35, Guo, Jia:
> On 7/1/2018 3:46 PM, Matan Azrad wrote:
> > From: Jeff Guo
> >> --- a/app/test-pmd/testpmd.c
> >> +++ b/app/test-pmd/testpmd.c
> >> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char *device_name, enum
> >> rte_dev_event_type type,
> >> case RTE_DEV_EVENT_REMOVE:
> >> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> >> device_name);
> >> - /* TODO: After finish failure handle, begin to stop
> >> - * packet forward, stop port, close port, detach port.
> >> - */
> >> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
> > As you probably know, 1 rte_device may be associated to more than one ethdev ports, so the ethdev port name can be different from rte_device name.
> > Looks like we need a new ethdev API to get all the ports associated to one rte_device.
>
> agree, seems that the the old ethdev API have some issue when got all
> port by device name. we could check with ethdev maintainer and fix it by
> specific ethdev patch later.
This ethdev function could return an error if several ports match.
Ideally, we should not use this function at all.
If you want to manage an ethdev port, why are you using an EAL event?
There is an ethdev callback mechanism for port removal.
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-07-03 22:44 ` Thomas Monjalon
@ 2018-07-04 3:48 ` Guo, Jia
2018-07-04 7:06 ` Matan Azrad
1 sibling, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-07-04 3:48 UTC (permalink / raw)
To: Thomas Monjalon
Cc: dev, Matan Azrad, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com,
shreyansh.jain@nxp.com, helin.zhang@intel.com
hi, thomas
On 7/4/2018 6:44 AM, Thomas Monjalon wrote:
> 03/07/2018 11:35, Guo, Jia:
>> On 7/1/2018 3:46 PM, Matan Azrad wrote:
>>> From: Jeff Guo
>>>> --- a/app/test-pmd/testpmd.c
>>>> +++ b/app/test-pmd/testpmd.c
>>>> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char *device_name, enum
>>>> rte_dev_event_type type,
>>>> case RTE_DEV_EVENT_REMOVE:
>>>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>>>> device_name);
>>>> - /* TODO: After finish failure handle, begin to stop
>>>> - * packet forward, stop port, close port, detach port.
>>>> - */
>>>> + ret = rte_eth_dev_get_port_by_name(device_name, &port_id);
>>> As you probably know, 1 rte_device may be associated to more than one ethdev ports, so the ethdev port name can be different from rte_device name.
>>> Looks like we need a new ethdev API to get all the ports associated to one rte_device.
>> agree, seems that the the old ethdev API have some issue when got all
>> port by device name. we could check with ethdev maintainer and fix it by
>> specific ethdev patch later.
> This ethdev function could return an error if several ports match.
>
> Ideally, we should not use this function at all.
> If you want to manage an ethdev port, why are you using an EAL event?
> There is an ethdev callback mechanism for port removal.
>
>
i think the problem is that how to manage all ethdev port associated
with one rte_device. So the easy way is let device event callback to
check these ports. I will modify it in next version.
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-07-03 22:44 ` Thomas Monjalon
2018-07-04 3:48 ` Guo, Jia
@ 2018-07-04 7:06 ` Matan Azrad
2018-07-05 7:54 ` Guo, Jia
1 sibling, 1 reply; 494+ messages in thread
From: Matan Azrad @ 2018-07-04 7:06 UTC (permalink / raw)
To: Thomas Monjalon, Guo, Jia
Cc: dev@dpdk.org, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com,
shreyansh.jain@nxp.com, helin.zhang@intel.com
Hi Thomas, Guo
From: Thomas Monjalon
> 03/07/2018 11:35, Guo, Jia:
> > On 7/1/2018 3:46 PM, Matan Azrad wrote:
> > > From: Jeff Guo
> > >> --- a/app/test-pmd/testpmd.c
> > >> +++ b/app/test-pmd/testpmd.c
> > >> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char
> *device_name,
> > >> enum rte_dev_event_type type,
> > >> case RTE_DEV_EVENT_REMOVE:
> > >> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
> > >> device_name);
> > >> - /* TODO: After finish failure handle, begin to stop
> > >> - * packet forward, stop port, close port, detach port.
> > >> - */
> > >> + ret = rte_eth_dev_get_port_by_name(device_name,
> &port_id);
> > > As you probably know, 1 rte_device may be associated to more than one
> ethdev ports, so the ethdev port name can be different from rte_device
> name.
> > > Looks like we need a new ethdev API to get all the ports associated to
> one rte_device.
> >
> > agree, seems that the the old ethdev API have some issue when got all
> > port by device name. we could check with ethdev maintainer and fix it
> > by specific ethdev patch later.
>
> This ethdev function could return an error if several ports match.
>
Just to clarify:
The ethdev name may be different from the rte_device name of a port,
The rte_eth_dev_get_port_by_name() searches the ethdev name and not the rte_device name.
> Ideally, we should not use this function at all.
> If you want to manage an ethdev port, why are you using an EAL event?
> There is an ethdev callback mechanism for port removal.
So, looks like the EAL event should trigger an ethdev event for all the ports associated to this rte_device.
I think that the best one to do it is the PMD, so maybe the PMD(which wants to support hot unplug) should register to the EAL event and to trigger an ethdev RMV event from the EAL callback.
What do you think?
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V4 8/9] app/testpmd: show example to handle hot unplug
2018-07-04 7:06 ` Matan Azrad
@ 2018-07-05 7:54 ` Guo, Jia
0 siblings, 0 replies; 494+ messages in thread
From: Guo, Jia @ 2018-07-05 7:54 UTC (permalink / raw)
To: Matan Azrad, Thomas Monjalon
Cc: dev@dpdk.org, stephen@networkplumber.org,
bruce.richardson@intel.com, ferruh.yigit@intel.com,
konstantin.ananyev@intel.com, gaetan.rivet@6wind.com,
jingjing.wu@intel.com, Mordechay Haimovsky,
harry.van.haaren@intel.com, qi.z.zhang@intel.com,
shaopeng.he@intel.com, bernard.iremonger@intel.com,
shreyansh.jain@nxp.com, helin.zhang@intel.com
On 7/4/2018 3:06 PM, Matan Azrad wrote:
> Hi Thomas, Guo
>
> From: Thomas Monjalon
>> 03/07/2018 11:35, Guo, Jia:
>>> On 7/1/2018 3:46 PM, Matan Azrad wrote:
>>>> From: Jeff Guo
>>>>> --- a/app/test-pmd/testpmd.c
>>>>> +++ b/app/test-pmd/testpmd.c
>>>>> @@ -2206,9 +2209,12 @@ eth_dev_event_callback(char
>> *device_name,
>>>>> enum rte_dev_event_type type,
>>>>> case RTE_DEV_EVENT_REMOVE:
>>>>> RTE_LOG(ERR, EAL, "The device: %s has been removed!\n",
>>>>> device_name);
>>>>> - /* TODO: After finish failure handle, begin to stop
>>>>> - * packet forward, stop port, close port, detach port.
>>>>> - */
>>>>> + ret = rte_eth_dev_get_port_by_name(device_name,
>> &port_id);
>>>> As you probably know, 1 rte_device may be associated to more than one
>> ethdev ports, so the ethdev port name can be different from rte_device
>> name.
>>>> Looks like we need a new ethdev API to get all the ports associated to
>> one rte_device.
>>> agree, seems that the the old ethdev API have some issue when got all
>>> port by device name. we could check with ethdev maintainer and fix it
>>> by specific ethdev patch later.
>> This ethdev function could return an error if several ports match.
>>
> Just to clarify:
>
> The ethdev name may be different from the rte_device name of a port,
> The rte_eth_dev_get_port_by_name() searches the ethdev name and not the rte_device name.
>
>> Ideally, we should not use this function at all.
>> If you want to manage an ethdev port, why are you using an EAL event?
>> There is an ethdev callback mechanism for port removal.
> So, looks like the EAL event should trigger an ethdev event for all the ports associated to this rte_device.
> I think that the best one to do it is the PMD, so maybe the PMD(which wants to support hot unplug) should register to the EAL event and to trigger an ethdev RMV event from the EAL callback.
>
> What do you think?
>
i think matan give an constructive option to combine the usage of eal
event and ethdev event, but i am not sure which is the best one.
So let this discuss on going, i will remove this 8/9 and 9/9 patches,
let the patch set focus on the hotplug failure mechanism ,
and will use another patch set to cover the event management example in
testpmd.
>
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V4 9/9] app/testpmd: enable device hotplug monitoring
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
` (7 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 8/9] app/testpmd: show example to handle " Jeff Guo
@ 2018-06-29 10:30 ` Jeff Guo
8 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-06-29 10:30 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, there 2 different hotplug mechanisms in dpdk, the one is
ethdev event + kernel driver hotplug solution, while the other one is
eal device event + pci uio driver hotplug solution, each of them have
different configure and callback process in testpmd. In oder to avoid
the race between them, this patch aim to use a new parameter
"--hotplug-mode" to replace the previous "--hot-plug" command parameter,
to identify these different mode.
There are 3 modes on hotplug mode: disable, eal, or ethdev(default).
If user want to use eal device event monitor mode, could use below
command when start testpmd. If not set this parameter, ethdev hotplug
mode is default to be used.
E.g. ./build/app/testpmd -c 0x3 --n 4 -- -i --hotplug-mode=eal
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v4->v3:
change to use new parameter "--hotplug-mode" in testpmd
to identify the eal hotplug and ethdev hotplug
---
app/test-pmd/parameters.c | 20 ++++++++++++++++----
app/test-pmd/testpmd.c | 18 +++++++++++-------
app/test-pmd/testpmd.h | 8 +++++++-
doc/guides/testpmd_app_ug/run_app.rst | 10 ++++++++--
4 files changed, 42 insertions(+), 14 deletions(-)
diff --git a/app/test-pmd/parameters.c b/app/test-pmd/parameters.c
index 7580762..601e13e 100644
--- a/app/test-pmd/parameters.c
+++ b/app/test-pmd/parameters.c
@@ -186,7 +186,8 @@ usage(char* progname)
printf(" --flow-isolate-all: "
"requests flow API isolated mode on all ports at initialization time.\n");
printf(" --tx-offloads=0xXXXXXXXX: hexadecimal bitmask of TX queue offloads\n");
- printf(" --hot-plug: enable hot plug for device.\n");
+ printf(" --hotplug-mode=N: set hotplug mode for device "
+ "(N: disable (default) or eal or ethdev.\n");
printf(" --vxlan-gpe-port=N: UPD port of tunnel VXLAN-GPE\n");
printf(" --mlockall: lock all memory\n");
printf(" --no-mlockall: do not lock all memory\n");
@@ -621,7 +622,7 @@ launch_args_parse(int argc, char** argv)
{ "print-event", 1, 0, 0 },
{ "mask-event", 1, 0, 0 },
{ "tx-offloads", 1, 0, 0 },
- { "hot-plug", 0, 0, 0 },
+ { "hotplug-mode", 1, 0, 0 },
{ "vxlan-gpe-port", 1, 0, 0 },
{ "mlockall", 0, 0, 0 },
{ "no-mlockall", 0, 0, 0 },
@@ -1139,8 +1140,19 @@ launch_args_parse(int argc, char** argv)
rte_exit(EXIT_FAILURE,
"invalid mask-event argument\n");
}
- if (!strcmp(lgopts[opt_idx].name, "hot-plug"))
- hot_plug = 1;
+ if (!strcmp(lgopts[opt_idx].name, "hotplug-mode")) {
+ if (!strcmp(optarg, "disable"))
+ hotplug_mode = HOTPLUG_MODE_DISABLE;
+ else if (!strcmp(optarg, "eal"))
+ hotplug_mode = HOTPLUG_MODE_EAL;
+ else if (!strcmp(optarg, "ethdev"))
+ hotplug_mode = HOTPLUG_MODE_ETHDEV;
+ else
+ rte_exit(EXIT_FAILURE,
+ "hotplug-mode %s invalid - must be: "
+ "disable, eal, ethdev.\n",
+ optarg);
+ }
if (!strcmp(lgopts[opt_idx].name, "mlockall"))
do_mlockall = 1;
if (!strcmp(lgopts[opt_idx].name, "no-mlockall"))
diff --git a/app/test-pmd/testpmd.c b/app/test-pmd/testpmd.c
index 42ed196..9269400 100644
--- a/app/test-pmd/testpmd.c
+++ b/app/test-pmd/testpmd.c
@@ -286,7 +286,7 @@ uint8_t lsc_interrupt = 1; /* enabled by default */
*/
uint8_t rmv_interrupt = 1; /* enabled by default */
-uint8_t hot_plug = 0; /**< hotplug disabled by default. */
+uint8_t hotplug_mode = HOTPLUG_MODE_ETHDEV; /**< hotplug disabled by default. */
/*
* Display or mask ether events
@@ -2043,7 +2043,7 @@ pmd_test_exit(void)
}
}
- if (hot_plug) {
+ if (hotplug_mode == HOTPLUG_MODE_EAL) {
ret = rte_dev_event_monitor_stop();
if (ret)
RTE_LOG(ERR, EAL,
@@ -2181,9 +2181,13 @@ eth_event_callback(portid_t port_id, enum rte_eth_event_type type, void *param,
switch (type) {
case RTE_ETH_EVENT_INTR_RMV:
- if (rte_eal_alarm_set(100000,
- rmv_event_callback, (void *)(intptr_t)port_id))
- fprintf(stderr, "Could not set up deferred device removal\n");
+ if (hotplug_mode == HOTPLUG_MODE_ETHDEV) {
+ if (rte_eal_alarm_set(100000,
+ rmv_event_callback,
+ (void *)(intptr_t)port_id))
+ fprintf(stderr, "Could not set up deferred "
+ "device removal\n");
+ }
break;
default:
break;
@@ -2734,8 +2738,8 @@ main(int argc, char** argv)
init_config();
- if (hot_plug) {
- /* enable hot plug monitoring */
+ if (hotplug_mode == HOTPLUG_MODE_EAL) {
+ /* enable hotplug event monitoring */
ret = rte_dev_event_monitor_start();
if (ret) {
rte_errno = EINVAL;
diff --git a/app/test-pmd/testpmd.h b/app/test-pmd/testpmd.h
index f51cd9d..e29ee2a 100644
--- a/app/test-pmd/testpmd.h
+++ b/app/test-pmd/testpmd.h
@@ -69,6 +69,12 @@ enum {
PORT_TOPOLOGY_LOOP,
};
+enum {
+ HOTPLUG_MODE_DISABLE,
+ HOTPLUG_MODE_EAL,
+ HOTPLUG_MODE_ETHDEV,
+};
+
#ifdef RTE_TEST_PMD_RECORD_BURST_STATS
/**
* The data structure associated with RX and TX packet burst statistics
@@ -335,7 +341,7 @@ extern uint8_t lsc_interrupt; /**< disabled by "--no-lsc-interrupt" parameter */
extern uint8_t rmv_interrupt; /**< disabled by "--no-rmv-interrupt" parameter */
extern uint32_t event_print_mask;
/**< set by "--print-event xxxx" and "--mask-event xxxx parameters */
-extern uint8_t hot_plug; /**< enable by "--hot-plug" parameter */
+extern uint8_t hotplug_mode; /**< set by "--hotplug-mode" parameter */
extern int do_mlockall; /**< set by "--mlockall" or "--no-mlockall" parameter */
#ifdef RTE_LIBRTE_IXGBE_BYPASS
diff --git a/doc/guides/testpmd_app_ug/run_app.rst b/doc/guides/testpmd_app_ug/run_app.rst
index f301c2b..09e2716 100644
--- a/doc/guides/testpmd_app_ug/run_app.rst
+++ b/doc/guides/testpmd_app_ug/run_app.rst
@@ -482,9 +482,15 @@ The commandline options are:
Set the hexadecimal bitmask of TX queue offloads.
The default value is 0.
-* ``--hot-plug``
+* ``--hotplug-mode``
- Enable device event monitor machenism for hotplug.
+ Set the hotplug handle mode, that is ``disable`` or ``eal`` or ``ethdev`` (the default).
+
+ In ``disable`` mode, it will not handle the hotplug for device.
+
+ In ``eal`` mode, it will start device event monitor and register eth_dev_event_callback for hotplug process.
+
+ In ``ethdev`` mode, it will process RTE_ETH_EVENT_INTR_RMV event which is detected from ethdev.
* ``--vxlan-gpe-port=N``
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread
* [PATCH V5 0/7] hot plug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (8 preceding siblings ...)
2018-06-29 10:30 ` [PATCH V4 0/9] hot plug failure handle mechanism Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-05 7:38 ` [PATCH V5 1/7] bus: add hotplug failure handler Jeff Guo
` (5 more replies)
2018-07-05 8:21 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
` (13 subsequent siblings)
23 siblings, 6 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event detect mechanism, failsafe driver,
bonding driver and hot plug/unplug api in framework, app could use these to
develop their hot plug solution.
let’s see the case of hot unplug, it can happen when a hardware device is
be removed physically, or when the software disables it. App need to call
ether dev API to detach the device, to unplug the device at the bus level and
make access to the device invalid. But the problem is that, the removal of the
device from the software lists is not going to be instantaneous, at this time
if the data(fast) path still read/write the device, it will cause MMIO error
and result of the app crash out.
Seems that we have got fail-safe driver(or app) + RTE_ETH_EVENT_INTR_RMV +
kernel core driver solution to handle it, but still not have failsafe driver
(or app) + RTE_DEV_EVENT_REMOVE + PCIe pmd driver failure handle solution. So
there is an absence in dpdk hot plug solution right now.
Also, we know that kernel only guaranty hot plug on the kernel side, but not for
the user mode side. Firstly we can hardly have a gatekeeper for any MMIO for
multiple PMD driver. Secondly, no more specific 3rd tools such as udev/driverctl
have especially cover these hot plug failure processing. Third, the feasibility
of app’s implement for multiple user mode PMD driver is still a problem. Here,
a general hot plug failure handle mechanism in dpdk framework would be proposed,
it aim to guaranty that, when hot unplug occur, the system will not crash and
app will not be break out, and user space can normally stop and release any
relevant resources, then unplug of the device at the bus level cleanly.
The mechanism should be come across as bellow:
Firstly, app enabled the device event monitor and register the hot plug event’s
callback before running data path. Once the hot unplug behave occur, the
mechanism will detect the removal event and then accordingly do the failure
handle. In order to do that, below functional will be bring in.
- Add a new bus ops “handle_hot_unplug” to handle bus read/write error, it is
bus-specific and each kind of bus can implement its own logic.
- Implement pci bus specific ops “pci_handle_hot_unplug”. It will base on the
failure address to remap memory for the corresponding device that unplugged.
For the data path or other unexpected control from the control path when hot
unplug occur.
- Implement a new sigbus handler, it is registered when start device even
monitoring. The handler is per process. Base on the signal event principle,
control path thread and data path thread will randomly receive the sigbus
error, but will go to the common sigbus handler. Once the MMIO sigbus error
exposure, it will trigger the above hot unplug operation. The sigbus will be
check if it is cause of the hot unplug or not, if not will info exception as
the original sigbus handler. If yes, will do memory remapping.
For the control path and the igb uio release:
- When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this resource,
it will cause kernel crash.
On the other hand, something like interrupt disable do not automatically
process in kernel side. If not handler it, this redundancy and dirty thing
will affect the interrupt resource be used by other device.
So the igb_uio driver have to check the hot plug status and corresponding
process should be taken in igb uio deriver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such as
probed/opened/released/removed/unplug. When detect the unexpected removal
which cause of hot unplug behavior, it will corresponding disable interrupt
resource, while for the part of releasement which kernel have already handle,
just skip it to avoid double free or null pointer kernel crash issue.
The mechanism could be use for fail-safe driver and app which want to use hot
plug solution. let testpmd for example:
- Enable device event monitor->device unplug->failure handle->stop forwarding->
stop port->close port->detach port.
This process will not breaking the app/fail-safe running, and will not break
other irrelevance device. And app could plug in the device and restart the date
path again by below.
- Device plug in->bind igb_uio driver ->attached device->start port->
start forwarding.
patchset history:
v5->v4:
split patches to focus on the failure handle, remove the event usage by testpmd
to another patch.
change the hotplug failure handler name
refine the sigbus handle logic
add lock for udev state in igb uio driver
v4->v3:
split patches to be small and clear
change to use new parameter "--hotplug-mode" in testpmd
to identify the eal hotplug and ethdev hotplug
v3->v2:
change bus ops name to bus_hotplug_handler.
add new API and bus ops of bus_signal_handler
distingush handle generic sigbus and hotplug sigbus
v2->v1(v21):
refine some doc and commit log
fix igb uio kernel issue for control path failure
rebase testpmd code
Since the hot plug solution be discussed serval around in the public, the
scope be changed and the patch set be split into many times. Coming to the
recently RFC and feature design, it just focus on the hot unplug failure
handler at this patch set, so in order let this topic more clear and focus,
summarize privours patch set in history “v1(v21)”, the v2 here go ahead
for further track.
"v1(21)" == v21 as below:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (7):
bus: add hotplug failure handler
bus/pci: implement hotplug failure handler ops
bus: add sigbus handler
bus/pci: implement sigbus handler operation
bus: add helper to handle sigbus
eal: add failure handle mechanism for hotplug
igb_uio: fix uio release issue when hot unplug
drivers/bus/pci/pci_common.c | 77 ++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 ++++++++++
drivers/bus/pci/private.h | 12 ++++
kernel/linux/igb_uio/igb_uio.c | 51 ++++++++++++++-
lib/librte_eal/common/eal_common_bus.c | 36 ++++++++++-
lib/librte_eal/common/eal_private.h | 12 ++++
lib/librte_eal/common/include/rte_bus.h | 31 +++++++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 111 +++++++++++++++++++++++++++++++-
8 files changed, 358 insertions(+), 5 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V5 1/7] bus: add hotplug failure handler
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:17 ` He, Shaopeng
2018-07-05 7:38 ` [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops Jeff Guo
` (4 subsequent siblings)
5 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When device be hotplug out, if app still continue to access device by mmio,
it will cause of memory failure and result the system crash.
This patch introduces a bus ops to handle device hotplug failure, it is a
bus specific behavior,so that each kind of bus can implement its own logic
case by case.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
change ops name to be more clear
refine doc and commit log
---
lib/librte_eal/common/include/rte_bus.h | 15 +++++++++++++++
1 file changed, 15 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..8a993cf 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,19 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hotplug failure handler, which is responsible
+ * for handle the failure when hot remove the device, guaranty the system
+ * would not crash in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_hotplug_failure_handler_t)(struct rte_device *dev);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -211,6 +224,8 @@ struct rte_bus {
rte_bus_parse_t parse; /**< Parse a device name */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
+ rte_bus_hotplug_failure_handler_t hotplug_failure_handler;
+ /**< handle hotplug failure on bus */
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 1/7] bus: add hotplug failure handler
2018-07-05 7:38 ` [PATCH V5 1/7] bus: add hotplug failure handler Jeff Guo
@ 2018-07-06 15:17 ` He, Shaopeng
0 siblings, 0 replies; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:17 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, July 5, 2018 3:39 PM
>
> When device be hotplug out, if app still continue to access device by mmio,
> it will cause of memory failure and result the system crash.
>
> This patch introduces a bus ops to handle device hotplug failure, it is a
> bus specific behavior,so that each kind of bus can implement its own logic
> case by case.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
Minor comment: there should be a space after the "behavior,"
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
2018-07-05 7:38 ` [PATCH V5 1/7] bus: add hotplug failure handler Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:17 ` He, Shaopeng
2018-07-05 7:38 ` [PATCH V5 3/7] bus: add sigbus handler Jeff Guo
` (3 subsequent siblings)
5 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of hotplug failure handler for PCI bus,
it is functional to remap a new dummy memory which overlap to the
failure memory to avoid MMIO read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
refine log and commit log
---
drivers/bus/pci/pci_common.c | 28 ++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++++++++
3 files changed, 73 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 94b0f41..bc3bcac 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -408,6 +408,33 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
}
static int
+pci_hotplug_failure_handler(struct rte_device *dev)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = RTE_DEV_TO_PCI(dev);
+ if (!pdev)
+ return -1;
+
+ switch (pdev->kdrv) {
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ /* mmio resources is invalid, remap it to be safe. */
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -437,6 +464,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .hotplug_failure_handler = pci_hotplug_failure_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 8ddd03e..6b312e5 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -123,6 +123,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops
2018-07-05 7:38 ` [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops Jeff Guo
@ 2018-07-06 15:17 ` He, Shaopeng
2018-07-09 5:29 ` Jeff Guo
0 siblings, 1 reply; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:17 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, July 5, 2018 3:39 PM
>
[...]
> + switch (pdev->kdrv) {
> + case RTE_KDRV_IGB_UIO:
> + case RTE_KDRV_UIO_GENERIC:
> + case RTE_KDRV_NIC_UIO:
> + /* mmio resources is invalid, remap it to be safe. */
Better to keep consistent as: mmio resource is
[...]
Is it helpful that pci_uio_remap_resource could also remap UIO event and control fd?
So, up-layer application will be easier to deal with the un-plug event.
> +/* remap the PCI resource of a PCI device in anonymous virtual memory */
> +int
> +pci_uio_remap_resource(struct rte_pci_device *dev)
> +{
> + int i;
> + void *map_address;
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops
2018-07-06 15:17 ` He, Shaopeng
@ 2018-07-09 5:29 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 5:29 UTC (permalink / raw)
To: He, Shaopeng, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
hi, shaopeng
On 7/6/2018 11:17 PM, He, Shaopeng wrote:
>> -----Original Message-----
>> From: Guo, Jia
>> Sent: Thursday, July 5, 2018 3:39 PM
>>
> [...]
>> + switch (pdev->kdrv) {
>> + case RTE_KDRV_IGB_UIO:
>> + case RTE_KDRV_UIO_GENERIC:
>> + case RTE_KDRV_NIC_UIO:
>> + /* mmio resources is invalid, remap it to be safe. */
> Better to keep consistent as: mmio resource is
ok.
> [...]
>
> Is it helpful that pci_uio_remap_resource could also remap UIO event and control fd?
> So, up-layer application will be easier to deal with the un-plug event.
The fd remove should be after the device be closed, since it will still
use the fd to close the interrupt when uninitialized driver,
and removing fd is go on to let the pci_uio_unmap_resource to do it when
device detach.
>> +/* remap the PCI resource of a PCI device in anonymous virtual memory */
>> +int
>> +pci_uio_remap_resource(struct rte_pci_device *dev)
>> +{
>> + int i;
>> + void *map_address;
> Acked-by: Shaopeng He <shaopeng.he@intel.com>
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 3/7] bus: add sigbus handler
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
2018-07-05 7:38 ` [PATCH V5 1/7] bus: add hotplug failure handler Jeff Guo
2018-07-05 7:38 ` [PATCH V5 2/7] bus/pci: implement hotplug failure handler ops Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:17 ` He, Shaopeng
2018-07-05 7:38 ` [PATCH V5 4/7] bus/pci: implement sigbus handler operation Jeff Guo
` (2 subsequent siblings)
5 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When device be hotplug out, if data path still read/write device, the
sigbus error will occur, this error need to be handled. So a handler
need to be here to capture the signal and handle it correspondingly.
This patch introduces a bus ops to handle sigbus error, it is a bus
specific behavior,so that each kind of bus can implement its own logic
case by case.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
refine log and commit log
---
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index 8a993cf..d753575 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -181,6 +181,20 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
typedef int (*rte_bus_hotplug_failure_handler_t)(struct rte_device *dev);
/**
+ * Implementation a specific sigbus handler, which is responsible
+ * for handle the sigbus error which is original memory error, or specific
+ * memory error that caused of hot unplug.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 for success handle the sigbus.
+ * 1 for no bus handle the sigbus.
+ * -1 for failed to handle the sigbus
+ */
+typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -226,6 +240,8 @@ struct rte_bus {
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
rte_bus_hotplug_failure_handler_t hotplug_failure_handler;
/**< handle hotplug failure on bus */
+ rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
+
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 3/7] bus: add sigbus handler
2018-07-05 7:38 ` [PATCH V5 3/7] bus: add sigbus handler Jeff Guo
@ 2018-07-06 15:17 ` He, Shaopeng
0 siblings, 0 replies; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:17 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
>
> When device be hotplug out, if data path still read/write device, the
> sigbus error will occur, this error need to be handled. So a handler
> need to be here to capture the signal and handle it correspondingly.
>
> This patch introduces a bus ops to handle sigbus error, it is a bus
> specific behavior,so that each kind of bus can implement its own logic
> case by case.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 4/7] bus/pci: implement sigbus handler operation
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
` (2 preceding siblings ...)
2018-07-05 7:38 ` [PATCH V5 3/7] bus: add sigbus handler Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:18 ` He, Shaopeng
2018-07-05 7:38 ` [PATCH V5 5/7] bus: add helper to handle sigbus Jeff Guo
2018-07-05 7:38 ` [PATCH V5 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
5 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of sigbus handler for PCI bus, it is
functional to find the corresponding pci device which is be hotplug out.
and then handle the hotplug failure for this device.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
no change
---
drivers/bus/pci/pci_common.c | 49 ++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 49 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index bc3bcac..f065271 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -407,6 +407,32 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+/* check the failure address belongs to which device. */
+static struct rte_pci_device *
+pci_find_device_by_addr(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(INFO, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+
static int
pci_hotplug_failure_handler(struct rte_device *dev)
{
@@ -435,6 +461,28 @@ pci_hotplug_failure_handler(struct rte_device *dev)
}
static int
+pci_sigbus_handler(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = pci_find_device_by_addr(failure_addr);
+ if (!pdev) {
+ /* It is a generic sigbus error, no bus would handle it. */
+ ret = 1;
+ } else {
+ /* The sigbus error is caused of hot removal. */
+ ret = pci_hotplug_failure_handler(&pdev->device);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Failed to handle hot plug for "
+ "device %s", pdev->name);
+ ret = -1;
+ }
+ }
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -465,6 +513,7 @@ struct rte_pci_bus rte_pci_bus = {
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
.hotplug_failure_handler = pci_hotplug_failure_handler,
+ .sigbus_handler = pci_sigbus_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 4/7] bus/pci: implement sigbus handler operation
2018-07-05 7:38 ` [PATCH V5 4/7] bus/pci: implement sigbus handler operation Jeff Guo
@ 2018-07-06 15:18 ` He, Shaopeng
0 siblings, 0 replies; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:18 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, July 5, 2018 3:39 PM
>
> This patch implements the ops of sigbus handler for PCI bus, it is
> functional to find the corresponding pci device which is be hotplug out.
" which is been hotplug out "?
> and then handle the hotplug failure for this device.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 5/7] bus: add helper to handle sigbus
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
` (3 preceding siblings ...)
2018-07-05 7:38 ` [PATCH V5 4/7] bus/pci: implement sigbus handler operation Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:22 ` He, Shaopeng
2018-07-08 13:30 ` Andrew Rybchenko
2018-07-05 7:38 ` [PATCH V5 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
5 siblings, 2 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch aim to add a helper to iterate all buses to find the
corresponding bus to handle the sigbus error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
refine the errno restore logic
---
lib/librte_eal/common/eal_common_bus.c | 36 +++++++++++++++++++++++++++++++++-
lib/librte_eal/common/eal_private.h | 12 ++++++++++++
2 files changed, 47 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/common/eal_common_bus.c b/lib/librte_eal/common/eal_common_bus.c
index 0943851..c9f3566 100644
--- a/lib/librte_eal/common/eal_common_bus.c
+++ b/lib/librte_eal/common/eal_common_bus.c
@@ -37,6 +37,7 @@
#include <rte_bus.h>
#include <rte_debug.h>
#include <rte_string_fns.h>
+#include <rte_errno.h>
#include "eal_private.h"
@@ -220,7 +221,6 @@ rte_bus_find_by_device_name(const char *str)
return rte_bus_find(NULL, bus_can_parse, name);
}
-
/*
* Get iommu class of devices on the bus.
*/
@@ -242,3 +242,37 @@ rte_bus_get_iommu_class(void)
}
return mode;
}
+
+static int
+bus_handle_sigbus(const struct rte_bus *bus,
+ const void *failure_addr)
+{
+ int ret;
+
+ ret = bus->sigbus_handler(failure_addr);
+ rte_errno = ret;
+
+ return !(bus->sigbus_handler && ret <= 0);
+}
+
+int
+rte_bus_sigbus_handler(const void *failure_addr)
+{
+ struct rte_bus *bus;
+
+ int ret = 0;
+ int old_errno = rte_errno;
+ rte_errno = 0;
+
+ bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
+ /* failed to handle the sigbus, pass the new errno. */
+ if (bus && rte_errno == -1)
+ return -1;
+ else if (!bus)
+ ret = 1;
+
+ /* otherwise restore the old errno. */
+ rte_errno = old_errno;
+
+ return ret;
+}
diff --git a/lib/librte_eal/common/eal_private.h b/lib/librte_eal/common/eal_private.h
index bdadc4d..a91c4b5 100644
--- a/lib/librte_eal/common/eal_private.h
+++ b/lib/librte_eal/common/eal_private.h
@@ -258,4 +258,16 @@ int rte_mp_channel_init(void);
*/
void dev_callback_process(char *device_name, enum rte_dev_event_type event);
+
+/**
+ * Iterate all buses to find the corresponding bus, to handle the sigbus error.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 success to handle the sigbus.
+ * -1 failed to handle the sigbus
+ * 1 no bus can handler the sigbus
+ */
+int rte_bus_sigbus_handler(const void *failure_addr);
#endif /* _EAL_PRIVATE_H_ */
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 5/7] bus: add helper to handle sigbus
2018-07-05 7:38 ` [PATCH V5 5/7] bus: add helper to handle sigbus Jeff Guo
@ 2018-07-06 15:22 ` He, Shaopeng
2018-07-09 5:31 ` Jeff Guo
2018-07-08 13:30 ` Andrew Rybchenko
1 sibling, 1 reply; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:22 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, July 5, 2018 3:39 PM
>
> This patch aim to add a helper to iterate all buses to find the
> corresponding bus to handle the sigbus error.
>
[...]
> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
> + /* failed to handle the sigbus, pass the new errno. */
> + if (bus && rte_errno == -1)
> + return -1;
> + else if (!bus)
> + ret = 1;
Change the compare order, code will be a little bit shorter?
if (!bus)
ret = 1
else if (rte_errno == -1)
return -1;
[...]
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V5 5/7] bus: add helper to handle sigbus
2018-07-06 15:22 ` He, Shaopeng
@ 2018-07-09 5:31 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 5:31 UTC (permalink / raw)
To: He, Shaopeng, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
hi, shaopeng
thanks for your review.
On 7/6/2018 11:22 PM, He, Shaopeng wrote:
>> -----Original Message-----
>> From: Guo, Jia
>> Sent: Thursday, July 5, 2018 3:39 PM
>>
>> This patch aim to add a helper to iterate all buses to find the
>> corresponding bus to handle the sigbus error.
>>
> [...]
>
>> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
>> + /* failed to handle the sigbus, pass the new errno. */
>> + if (bus && rte_errno == -1)
>> + return -1;
>> + else if (!bus)
>> + ret = 1;
> Change the compare order, code will be a little bit shorter?
> if (!bus)
> ret = 1
> else if (rte_errno == -1)
> return -1;
>
> [...]
make sense.
> Acked-by: Shaopeng He <shaopeng.he@intel.com>
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V5 5/7] bus: add helper to handle sigbus
2018-07-05 7:38 ` [PATCH V5 5/7] bus: add helper to handle sigbus Jeff Guo
2018-07-06 15:22 ` He, Shaopeng
@ 2018-07-08 13:30 ` Andrew Rybchenko
2018-07-09 5:33 ` Jeff Guo
1 sibling, 1 reply; 494+ messages in thread
From: Andrew Rybchenko @ 2018-07-08 13:30 UTC (permalink / raw)
To: Jeff Guo, stephen, bruce.richardson, ferruh.yigit,
konstantin.ananyev, gaetan.rivet, jingjing.wu, thomas, motih,
matan, harry.van.haaren, qi.z.zhang, shaopeng.he,
bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, helin.zhang
On 05.07.2018 10:38, Jeff Guo wrote:
> This patch aim to add a helper to iterate all buses to find the
> corresponding bus to handle the sigbus error.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v5->v4:
> refine the errno restore logic
> ---
> lib/librte_eal/common/eal_common_bus.c | 36 +++++++++++++++++++++++++++++++++-
> lib/librte_eal/common/eal_private.h | 12 ++++++++++++
> 2 files changed, 47 insertions(+), 1 deletion(-)
>
> diff --git a/lib/librte_eal/common/eal_common_bus.c b/lib/librte_eal/common/eal_common_bus.c
> index 0943851..c9f3566 100644
> --- a/lib/librte_eal/common/eal_common_bus.c
> +++ b/lib/librte_eal/common/eal_common_bus.c
> @@ -37,6 +37,7 @@
> #include <rte_bus.h>
> #include <rte_debug.h>
> #include <rte_string_fns.h>
> +#include <rte_errno.h>
>
> #include "eal_private.h"
>
> @@ -220,7 +221,6 @@ rte_bus_find_by_device_name(const char *str)
> return rte_bus_find(NULL, bus_can_parse, name);
> }
>
> -
Unrelated change.
> /*
> * Get iommu class of devices on the bus.
> */
> @@ -242,3 +242,37 @@ rte_bus_get_iommu_class(void)
> }
> return mode;
> }
> +
> +static int
> +bus_handle_sigbus(const struct rte_bus *bus,
> + const void *failure_addr)
> +{
> + int ret;
> +
> + ret = bus->sigbus_handler(failure_addr);
Shouldn't bus->sigbus_handler be checked here against NULL?
It looks like not all buses implement it.
> + rte_errno = ret;
> +
> + return !(bus->sigbus_handler && ret <= 0);
> +}
> +
> +int
> +rte_bus_sigbus_handler(const void *failure_addr)
> +{
> + struct rte_bus *bus;
> +
> + int ret = 0;
> + int old_errno = rte_errno;
> + rte_errno = 0;
> +
> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
> + /* failed to handle the sigbus, pass the new errno. */
> + if (bus && rte_errno == -1)
> + return -1;
> + else if (!bus)
> + ret = 1;
> +
> + /* otherwise restore the old errno. */
> + rte_errno = old_errno;
> +
> + return ret;
> +}
> diff --git a/lib/librte_eal/common/eal_private.h b/lib/librte_eal/common/eal_private.h
> index bdadc4d..a91c4b5 100644
> --- a/lib/librte_eal/common/eal_private.h
> +++ b/lib/librte_eal/common/eal_private.h
> @@ -258,4 +258,16 @@ int rte_mp_channel_init(void);
> */
> void dev_callback_process(char *device_name, enum rte_dev_event_type event);
>
> +
> +/**
> + * Iterate all buses to find the corresponding bus, to handle the sigbus error.
> + * @param failure_addr
> + * Pointer of the fault address of the sigbus error.
> + *
> + * @return
> + * 0 success to handle the sigbus.
> + * -1 failed to handle the sigbus
> + * 1 no bus can handler the sigbus
> + */
> +int rte_bus_sigbus_handler(const void *failure_addr);
Empty line is missing after the function.
> #endif /* _EAL_PRIVATE_H_ */
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V5 5/7] bus: add helper to handle sigbus
2018-07-08 13:30 ` Andrew Rybchenko
@ 2018-07-09 5:33 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 5:33 UTC (permalink / raw)
To: Andrew Rybchenko, stephen, bruce.richardson, ferruh.yigit,
konstantin.ananyev, gaetan.rivet, jingjing.wu, thomas, motih,
matan, harry.van.haaren, qi.z.zhang, shaopeng.he,
bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, helin.zhang
hi, andrew
Thanks for your reviewing.
On 7/8/2018 9:30 PM, Andrew Rybchenko wrote:
> On 05.07.2018 10:38, Jeff Guo wrote:
>> This patch aim to add a helper to iterate all buses to find the
>> corresponding bus to handle the sigbus error.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v5->v4:
>> refine the errno restore logic
>> ---
>> lib/librte_eal/common/eal_common_bus.c | 36
>> +++++++++++++++++++++++++++++++++-
>> lib/librte_eal/common/eal_private.h | 12 ++++++++++++
>> 2 files changed, 47 insertions(+), 1 deletion(-)
>>
>> diff --git a/lib/librte_eal/common/eal_common_bus.c
>> b/lib/librte_eal/common/eal_common_bus.c
>> index 0943851..c9f3566 100644
>> --- a/lib/librte_eal/common/eal_common_bus.c
>> +++ b/lib/librte_eal/common/eal_common_bus.c
>> @@ -37,6 +37,7 @@
>> #include <rte_bus.h>
>> #include <rte_debug.h>
>> #include <rte_string_fns.h>
>> +#include <rte_errno.h>
>> #include "eal_private.h"
>> @@ -220,7 +221,6 @@ rte_bus_find_by_device_name(const char *str)
>> return rte_bus_find(NULL, bus_can_parse, name);
>> }
>> -
>
> Unrelated change.
>
ok. I am fine to let it left to other specific patch.
>> /*
>> * Get iommu class of devices on the bus.
>> */
>> @@ -242,3 +242,37 @@ rte_bus_get_iommu_class(void)
>> }
>> return mode;
>> }
>> +
>> +static int
>> +bus_handle_sigbus(const struct rte_bus *bus,
>> + const void *failure_addr)
>> +{
>> + int ret;
>> +
>> + ret = bus->sigbus_handler(failure_addr);
>
> Shouldn't bus->sigbus_handler be checked here against NULL?
> It looks like not all buses implement it.
>
should be like what you said.
>> + rte_errno = ret;
>> +
>> + return !(bus->sigbus_handler && ret <= 0);
>> +}
>> +
>> +int
>> +rte_bus_sigbus_handler(const void *failure_addr)
>> +{
>> + struct rte_bus *bus;
>> +
>> + int ret = 0;
>> + int old_errno = rte_errno;
>> + rte_errno = 0;
>> +
>> + bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
>> + /* failed to handle the sigbus, pass the new errno. */
>> + if (bus && rte_errno == -1)
>> + return -1;
>> + else if (!bus)
>> + ret = 1;
>> +
>> + /* otherwise restore the old errno. */
>> + rte_errno = old_errno;
>> +
>> + return ret;
>> +}
>> diff --git a/lib/librte_eal/common/eal_private.h
>> b/lib/librte_eal/common/eal_private.h
>> index bdadc4d..a91c4b5 100644
>> --- a/lib/librte_eal/common/eal_private.h
>> +++ b/lib/librte_eal/common/eal_private.h
>> @@ -258,4 +258,16 @@ int rte_mp_channel_init(void);
>> */
>> void dev_callback_process(char *device_name, enum
>> rte_dev_event_type event);
>> +
>> +/**
>> + * Iterate all buses to find the corresponding bus, to handle the
>> sigbus error.
>> + * @param failure_addr
>> + * Pointer of the fault address of the sigbus error.
>> + *
>> + * @return
>> + * 0 success to handle the sigbus.
>> + * -1 failed to handle the sigbus
>> + * 1 no bus can handler the sigbus
>> + */
>> +int rte_bus_sigbus_handler(const void *failure_addr);
>
> Empty line is missing after the function.
>
ok.
>> #endif /* _EAL_PRIVATE_H_ */
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 6/7] eal: add failure handle mechanism for hotplug
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
` (4 preceding siblings ...)
2018-07-05 7:38 ` [PATCH V5 5/7] bus: add helper to handle sigbus Jeff Guo
@ 2018-07-05 7:38 ` Jeff Guo
2018-07-06 15:22 ` He, Shaopeng
2018-07-08 13:46 ` Andrew Rybchenko
5 siblings, 2 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 7:38 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handler mechanism to handle device
hot plug removal event.
First register sigbus handler, once sigbus error be captured, will
check the failure address and accordingly remap the invalid memory
for the corresponding device. Bese on this mechanism, it could
guaranty the application not to be crash when hotplug out device.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
add sigbus old handler recover.
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 111 +++++++++++++++++++++++++++++++++-
1 file changed, 110 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..a22cb9a 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,28 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
+#include <rte_errno.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/* spinlock for device failure process */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
+static struct sigaction sigbus_action_old;
+
+static int sigbus_need_recover;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -33,6 +48,49 @@ enum eal_dev_event_subsystem {
EAL_DEV_EVENT_SUBSYSTEM_MAX
};
+static void
+sigbus_action_recover(void)
+{
+ if (sigbus_need_recover) {
+ sigaction(SIGBUS, &sigbus_action_old, NULL);
+ sigbus_need_recover = 0;
+ }
+}
+
+static void sigbus_handler(int signum, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(INFO, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = rte_bus_sigbus_handler(info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret == -1) {
+ rte_exit(EXIT_FAILURE,
+ "Failed to handle SIGBUS for hotplug, "
+ "(rte_errno: %s)!", strerror(rte_errno));
+ } else if (ret == 1) {
+ if (sigbus_action_old.sa_handler)
+ (*(sigbus_action_old.sa_handler))(signum);
+ else
+ rte_exit(EXIT_FAILURE,
+ "Failed to handle generic SIGBUS!");
+ }
+
+ RTE_LOG(INFO, EAL, "Success to handle SIGBUS for hotplug!\n");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
static int
dev_uev_socket_fd_create(void)
{
@@ -147,6 +205,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname;
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +232,50 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ rte_spinlock_lock(&dev_failure_lock);
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ busname);
+ return;
+ }
+
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
+ "bus (%s)\n", uevent.devname, busname);
+ return;
+ }
+
+ ret = bus->hotplug_failure_handler(dev);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Can not handle hotplug for "
+ "device (%s)\n", dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +295,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
+
monitor_started = true;
return 0;
@@ -217,8 +323,11 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ sigbus_action_recover();
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH V5 6/7] eal: add failure handle mechanism for hotplug
2018-07-05 7:38 ` [PATCH V5 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
@ 2018-07-06 15:22 ` He, Shaopeng
2018-07-08 13:46 ` Andrew Rybchenko
1 sibling, 0 replies; 494+ messages in thread
From: He, Shaopeng @ 2018-07-06 15:22 UTC (permalink / raw)
To: Guo, Jia, stephen@networkplumber.org, Richardson, Bruce,
Yigit, Ferruh, Ananyev, Konstantin, gaetan.rivet@6wind.com,
Wu, Jingjing, thomas@monjalon.net, motih@mellanox.com,
matan@mellanox.com, Van Haaren, Harry, Zhang, Qi Z,
Iremonger, Bernard
Cc: jblunck@infradead.org, shreyansh.jain@nxp.com, dev@dpdk.org,
Zhang, Helin
> -----Original Message-----
> From: Guo, Jia
> Sent: Thursday, July 5, 2018 3:39 PM
>
> This patch introduces a failure handler mechanism to handle device
> hot plug removal event.
>
> First register sigbus handler, once sigbus error be captured, will
> check the failure address and accordingly remap the invalid memory
> for the corresponding device. Bese on this mechanism, it could
" Besed on this mechanism "?
> guaranty the application not to be crash when hotplug out device.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
^ permalink raw reply [flat|nested] 494+ messages in thread
* Re: [PATCH V5 6/7] eal: add failure handle mechanism for hotplug
2018-07-05 7:38 ` [PATCH V5 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
2018-07-06 15:22 ` He, Shaopeng
@ 2018-07-08 13:46 ` Andrew Rybchenko
2018-07-09 5:40 ` Jeff Guo
1 sibling, 1 reply; 494+ messages in thread
From: Andrew Rybchenko @ 2018-07-08 13:46 UTC (permalink / raw)
To: Jeff Guo, stephen, bruce.richardson, ferruh.yigit,
konstantin.ananyev, gaetan.rivet, jingjing.wu, thomas, motih,
matan, harry.van.haaren, qi.z.zhang, shaopeng.he,
bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, helin.zhang
On 05.07.2018 10:38, Jeff Guo wrote:
> This patch introduces a failure handler mechanism to handle device
> hot plug removal event.
>
> First register sigbus handler, once sigbus error be captured, will
> check the failure address and accordingly remap the invalid memory
> for the corresponding device. Bese on this mechanism, it could
> guaranty the application not to be crash when hotplug out device.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> ---
> v5->v4:
> add sigbus old handler recover.
> ---
> lib/librte_eal/linuxapp/eal/eal_dev.c | 111 +++++++++++++++++++++++++++++++++-
> 1 file changed, 110 insertions(+), 1 deletion(-)
>
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 1cf6aeb..a22cb9a 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -14,15 +16,28 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
> +#include <rte_spinlock.h>
> +#include <rte_errno.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 };
> static bool monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
Shouldn't rte_bus.h provide it?
> #define EAL_UEV_MSG_LEN 4096
> #define EAL_UEV_MSG_ELEM_LEN 128
>
> +/* spinlock for device failure process */
> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
It would be useful to explain why the lock is needed and when
it should be obtained/released. Which resources are protected
by the lock?
> +
> +static struct sigaction sigbus_action_old;
> +
> +static int sigbus_need_recover;
> +
> static void dev_uev_handler(__rte_unused void *param);
>
> /* identify the system layer which reports this event. */
> @@ -33,6 +48,49 @@ enum eal_dev_event_subsystem {
> EAL_DEV_EVENT_SUBSYSTEM_MAX
> };
>
> +static void
> +sigbus_action_recover(void)
> +{
> + if (sigbus_need_recover) {
> + sigaction(SIGBUS, &sigbus_action_old, NULL);
> + sigbus_need_recover = 0;
> + }
> +}
> +
> +static void sigbus_handler(int signum, siginfo_t *info,
> + void *ctx __rte_unused)
> +{
> + int ret;
> +
> + RTE_LOG(INFO, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
> + (int)pthread_self(), info->si_addr);
> +
> + rte_spinlock_lock(&dev_failure_lock);
> + ret = rte_bus_sigbus_handler(info->si_addr);
> + rte_spinlock_unlock(&dev_failure_lock);
> + if (ret == -1) {
> + rte_exit(EXIT_FAILURE,
> + "Failed to handle SIGBUS for hotplug, "
> + "(rte_errno: %s)!", strerror(rte_errno));
> + } else if (ret == 1) {
> + if (sigbus_action_old.sa_handler)
> + (*(sigbus_action_old.sa_handler))(signum);
> + else
> + rte_exit(EXIT_FAILURE,
> + "Failed to handle generic SIGBUS!");
> + }
> +
> + RTE_LOG(INFO, EAL, "Success to handle SIGBUS for hotplug!\n");
> +}
> +
> +static int cmp_dev_name(const struct rte_device *dev,
> + const void *_name)
> +{
> + const char *name = _name;
> +
> + return strcmp(dev->name, name);
> +}
> +
> static int
> dev_uev_socket_fd_create(void)
> {
> @@ -147,6 +205,9 @@ dev_uev_handler(__rte_unused void *param)
> struct rte_dev_event uevent;
> int ret;
> char buf[EAL_UEV_MSG_LEN];
> + struct rte_bus *bus;
> + struct rte_device *dev;
> + const char *busname;
>
> memset(&uevent, 0, sizeof(struct rte_dev_event));
> memset(buf, 0, EAL_UEV_MSG_LEN);
> @@ -171,13 +232,50 @@ dev_uev_handler(__rte_unused void *param)
> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
> uevent.devname, uevent.type, uevent.subsystem);
>
> - if (uevent.devname)
> + switch (uevent.subsystem) {
> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
> + busname = "pci";
> + break;
> + default:
> + break;
> + }
> +
> + if (uevent.devname) {
> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
> + rte_spinlock_lock(&dev_failure_lock);
> + bus = rte_bus_find_by_name(busname);
It looks like busname could be uninitialized here.
> + if (bus == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
> + busname);
> + return;
> + }
> +
> + dev = bus->find_device(NULL, cmp_dev_name,
> + uevent.devname);
> + if (dev == NULL) {
> + RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
> + "bus (%s)\n", uevent.devname, busname);
> + return;
> + }
> +
> + ret = bus->hotplug_failure_handler(dev);
> + rte_spinlock_unlock(&dev_failure_lock);
> + if (ret) {
> + RTE_LOG(ERR, EAL, "Can not handle hotplug for "
> + "device (%s)\n", dev->name);
> + return;
> + }
> + }
> dev_callback_process(uevent.devname, uevent.type);
> + }
> }
>
> int __rte_experimental
> rte_dev_event_monitor_start(void)
> {
> + sigset_t mask;
> + struct sigaction action;
> int ret;
>
> if (monitor_started)
> @@ -197,6 +295,14 @@ rte_dev_event_monitor_start(void)
> return -1;
> }
>
> + /* register sigbus handler */
> + sigemptyset(&mask);
> + sigaddset(&mask, SIGBUS);
> + action.sa_flags = SA_SIGINFO;
> + action.sa_mask = mask;
> + action.sa_sigaction = sigbus_handler;
> + sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
> +
> monitor_started = true;
>
> return 0;
> @@ -217,8 +323,11 @@ rte_dev_event_monitor_stop(void)
> return ret;
> }
>
> + sigbus_action_recover();
> +
> close(intr_handle.fd);
> intr_handle.fd = -1;
> monitor_started = false;
> +
> return 0;
> }
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH V5 6/7] eal: add failure handle mechanism for hotplug
2018-07-08 13:46 ` Andrew Rybchenko
@ 2018-07-09 5:40 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 5:40 UTC (permalink / raw)
To: Andrew Rybchenko, stephen, bruce.richardson, ferruh.yigit,
konstantin.ananyev, gaetan.rivet, jingjing.wu, thomas, motih,
matan, harry.van.haaren, qi.z.zhang, shaopeng.he,
bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, helin.zhang
On 7/8/2018 9:46 PM, Andrew Rybchenko wrote:
> On 05.07.2018 10:38, Jeff Guo wrote:
>> This patch introduces a failure handler mechanism to handle device
>> hot plug removal event.
>>
>> First register sigbus handler, once sigbus error be captured, will
>> check the failure address and accordingly remap the invalid memory
>> for the corresponding device. Bese on this mechanism, it could
>> guaranty the application not to be crash when hotplug out device.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> ---
>> v5->v4:
>> add sigbus old handler recover.
>> ---
>> lib/librte_eal/linuxapp/eal/eal_dev.c | 111
>> +++++++++++++++++++++++++++++++++-
>> 1 file changed, 110 insertions(+), 1 deletion(-)
>>
>> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> index 1cf6aeb..a22cb9a 100644
>> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> @@ -4,6 +4,8 @@
>> #include <string.h>
>> #include <unistd.h>
>> +#include <fcntl.h>
>> +#include <signal.h>
>> #include <sys/socket.h>
>> #include <linux/netlink.h>
>> @@ -14,15 +16,28 @@
>> #include <rte_malloc.h>
>> #include <rte_interrupts.h>
>> #include <rte_alarm.h>
>> +#include <rte_bus.h>
>> +#include <rte_eal.h>
>> +#include <rte_spinlock.h>
>> +#include <rte_errno.h>
>> #include "eal_private.h"
>> static struct rte_intr_handle intr_handle = {.fd = -1 };
>> static bool monitor_started;
>> +extern struct rte_bus_list rte_bus_list;
>> +
>
> Shouldn't rte_bus.h provide it?
>
I think rte_bus.h provide the rte_bus_list structure, and then
announcement a variable in eal_common_bus.c, then i use it by extern in
eal_dev.c.
>> #define EAL_UEV_MSG_LEN 4096
>> #define EAL_UEV_MSG_ELEM_LEN 128
>> +/* spinlock for device failure process */
>> +static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
>
> It would be useful to explain why the lock is needed and when
> it should be obtained/released. Which resources are protected
> by the lock?
>
make sense, this locker should be use both bus and device access
protection. Will explicit to let it to be more readable.
>> +
>> +static struct sigaction sigbus_action_old;
>> +
>> +static int sigbus_need_recover;
>> +
>> static void dev_uev_handler(__rte_unused void *param);
>> /* identify the system layer which reports this event. */
>> @@ -33,6 +48,49 @@ enum eal_dev_event_subsystem {
>> EAL_DEV_EVENT_SUBSYSTEM_MAX
>> };
>> +static void
>> +sigbus_action_recover(void)
>> +{
>> + if (sigbus_need_recover) {
>> + sigaction(SIGBUS, &sigbus_action_old, NULL);
>> + sigbus_need_recover = 0;
>> + }
>> +}
>> +
>> +static void sigbus_handler(int signum, siginfo_t *info,
>> + void *ctx __rte_unused)
>> +{
>> + int ret;
>> +
>> + RTE_LOG(INFO, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
>> + (int)pthread_self(), info->si_addr);
>> +
>> + rte_spinlock_lock(&dev_failure_lock);
>> + ret = rte_bus_sigbus_handler(info->si_addr);
>> + rte_spinlock_unlock(&dev_failure_lock);
>> + if (ret == -1) {
>> + rte_exit(EXIT_FAILURE,
>> + "Failed to handle SIGBUS for hotplug, "
>> + "(rte_errno: %s)!", strerror(rte_errno));
>> + } else if (ret == 1) {
>> + if (sigbus_action_old.sa_handler)
>> + (*(sigbus_action_old.sa_handler))(signum);
>> + else
>> + rte_exit(EXIT_FAILURE,
>> + "Failed to handle generic SIGBUS!");
>> + }
>> +
>> + RTE_LOG(INFO, EAL, "Success to handle SIGBUS for hotplug!\n");
>> +}
>> +
>> +static int cmp_dev_name(const struct rte_device *dev,
>> + const void *_name)
>> +{
>> + const char *name = _name;
>> +
>> + return strcmp(dev->name, name);
>> +}
>> +
>> static int
>> dev_uev_socket_fd_create(void)
>> {
>> @@ -147,6 +205,9 @@ dev_uev_handler(__rte_unused void *param)
>> struct rte_dev_event uevent;
>> int ret;
>> char buf[EAL_UEV_MSG_LEN];
>> + struct rte_bus *bus;
>> + struct rte_device *dev;
>> + const char *busname;
>> memset(&uevent, 0, sizeof(struct rte_dev_event));
>> memset(buf, 0, EAL_UEV_MSG_LEN);
>> @@ -171,13 +232,50 @@ dev_uev_handler(__rte_unused void *param)
>> RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d,
>> subsystem:%d)\n",
>> uevent.devname, uevent.type, uevent.subsystem);
>> - if (uevent.devname)
>> + switch (uevent.subsystem) {
>> + case EAL_DEV_EVENT_SUBSYSTEM_PCI:
>> + case EAL_DEV_EVENT_SUBSYSTEM_UIO:
>> + busname = "pci";
>> + break;
>> + default:
>> + break;
>> + }
>> +
>> + if (uevent.devname) {
>> + if (uevent.type == RTE_DEV_EVENT_REMOVE) {
>> + rte_spinlock_lock(&dev_failure_lock);
>> + bus = rte_bus_find_by_name(busname);
>
> It looks like busname could be uninitialized here.
>
you are correct i think.
>> + if (bus == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
>> + busname);
>> + return;
>> + }
>> +
>> + dev = bus->find_device(NULL, cmp_dev_name,
>> + uevent.devname);
>> + if (dev == NULL) {
>> + RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
>> + "bus (%s)\n", uevent.devname, busname);
>> + return;
>> + }
>> +
>> + ret = bus->hotplug_failure_handler(dev);
>> + rte_spinlock_unlock(&dev_failure_lock);
>> + if (ret) {
>> + RTE_LOG(ERR, EAL, "Can not handle hotplug for "
>> + "device (%s)\n", dev->name);
>> + return;
>> + }
>> + }
>> dev_callback_process(uevent.devname, uevent.type);
>> + }
>> }
>> int __rte_experimental
>> rte_dev_event_monitor_start(void)
>> {
>> + sigset_t mask;
>> + struct sigaction action;
>> int ret;
>> if (monitor_started)
>> @@ -197,6 +295,14 @@ rte_dev_event_monitor_start(void)
>> return -1;
>> }
>> + /* register sigbus handler */
>> + sigemptyset(&mask);
>> + sigaddset(&mask, SIGBUS);
>> + action.sa_flags = SA_SIGINFO;
>> + action.sa_mask = mask;
>> + action.sa_sigaction = sigbus_handler;
>> + sigbus_need_recover = !sigaction(SIGBUS, &action,
>> &sigbus_action_old);
>> +
>> monitor_started = true;
>> return 0;
>> @@ -217,8 +323,11 @@ rte_dev_event_monitor_stop(void)
>> return ret;
>> }
>> + sigbus_action_recover();
>> +
>> close(intr_handle.fd);
>> intr_handle.fd = -1;
>> monitor_started = false;
>> +
>> return 0;
>> }
>
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH V5 0/7] hot plug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (9 preceding siblings ...)
2018-07-05 7:38 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
@ 2018-07-05 8:21 ` Jeff Guo
2018-07-05 8:21 ` [PATCH V5 7/7] igb_uio: fix uio release issue when hot unplug Jeff Guo
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
` (12 subsequent siblings)
23 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 8:21 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event detect mechanism, failsafe driver,
bonding driver and hot plug/unplug api in framework, app could use these to
develop their hot plug solution.
let’s see the case of hot unplug, it can happen when a hardware device is
be removed physically, or when the software disables it. App need to call
ether dev API to detach the device, to unplug the device at the bus level and
make access to the device invalid. But the problem is that, the removal of the
device from the software lists is not going to be instantaneous, at this time
if the data(fast) path still read/write the device, it will cause MMIO error
and result of the app crash out.
Seems that we have got fail-safe driver(or app) + RTE_ETH_EVENT_INTR_RMV +
kernel core driver solution to handle it, but still not have failsafe driver
(or app) + RTE_DEV_EVENT_REMOVE + PCIe pmd driver failure handle solution. So
there is an absence in dpdk hot plug solution right now.
Also, we know that kernel only guaranty hot plug on the kernel side, but not for
the user mode side. Firstly we can hardly have a gatekeeper for any MMIO for
multiple PMD driver. Secondly, no more specific 3rd tools such as udev/driverctl
have especially cover these hot plug failure processing. Third, the feasibility
of app’s implement for multiple user mode PMD driver is still a problem. Here,
a general hot plug failure handle mechanism in dpdk framework would be proposed,
it aim to guaranty that, when hot unplug occur, the system will not crash and
app will not be break out, and user space can normally stop and release any
relevant resources, then unplug of the device at the bus level cleanly.
The mechanism should be come across as bellow:
Firstly, app enabled the device event monitor and register the hot plug event’s
callback before running data path. Once the hot unplug behave occur, the
mechanism will detect the removal event and then accordingly do the failure
handle. In order to do that, below functional will be bring in.
- Add a new bus ops “handle_hot_unplug” to handle bus read/write error, it is
bus-specific and each kind of bus can implement its own logic.
- Implement pci bus specific ops “pci_handle_hot_unplug”. It will base on the
failure address to remap memory for the corresponding device that unplugged.
For the data path or other unexpected control from the control path when hot
unplug occur.
- Implement a new sigbus handler, it is registered when start device even
monitoring. The handler is per process. Base on the signal event principle,
control path thread and data path thread will randomly receive the sigbus
error, but will go to the common sigbus handler. Once the MMIO sigbus error
exposure, it will trigger the above hot unplug operation. The sigbus will be
check if it is cause of the hot unplug or not, if not will info exception as
the original sigbus handler. If yes, will do memory remapping.
For the control path and the igb uio release:
- When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this resource,
it will cause kernel crash.
On the other hand, something like interrupt disable do not automatically
process in kernel side. If not handler it, this redundancy and dirty thing
will affect the interrupt resource be used by other device.
So the igb_uio driver have to check the hot plug status and corresponding
process should be taken in igb uio deriver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such as
probed/opened/released/removed/unplug. When detect the unexpected removal
which cause of hot unplug behavior, it will corresponding disable interrupt
resource, while for the part of releasement which kernel have already handle,
just skip it to avoid double free or null pointer kernel crash issue.
The mechanism could be use for fail-safe driver and app which want to use hot
plug solution. let testpmd for example:
- Enable device event monitor->device unplug->failure handle->stop forwarding->
stop port->close port->detach port.
This process will not breaking the app/fail-safe running, and will not break
other irrelevance device. And app could plug in the device and restart the date
path again by below.
- Device plug in->bind igb_uio driver ->attached device->start port->
start forwarding.
patchset history:
v5->v4:
split patches to focus on the failure handle, remove the event usage by testpmd
to another patch.
change the hotplug failure handler name
refine the sigbus handle logic
add lock for udev state in igb uio driver
v4->v3:
split patches to be small and clear
change to use new parameter "--hotplug-mode" in testpmd
to identify the eal hotplug and ethdev hotplug
v3->v2:
change bus ops name to bus_hotplug_handler.
add new API and bus ops of bus_signal_handler
distingush handle generic sigbus and hotplug sigbus
v2->v1(v21):
refine some doc and commit log
fix igb uio kernel issue for control path failure
rebase testpmd code
Since the hot plug solution be discussed serval around in the public, the
scope be changed and the patch set be split into many times. Coming to the
recently RFC and feature design, it just focus on the hot unplug failure
handler at this patch set, so in order let this topic more clear and focus,
summarize privours patch set in history “v1(v21)”, the v2 here go ahead
for further track.
"v1(21)" == v21 as below:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (7):
bus: add hotplug failure handler
bus/pci: implement hotplug failure handler ops
bus: add sigbus handler
bus/pci: implement sigbus handler operation
bus: add helper to handle sigbus
eal: add failure handle mechanism for hotplug
igb_uio: fix uio release issue when hot unplug
drivers/bus/pci/pci_common.c | 77 ++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 ++++++++++
drivers/bus/pci/private.h | 12 ++++
kernel/linux/igb_uio/igb_uio.c | 51 ++++++++++++++-
lib/librte_eal/common/eal_common_bus.c | 36 ++++++++++-
lib/librte_eal/common/eal_private.h | 12 ++++
lib/librte_eal/common/include/rte_bus.h | 31 +++++++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 111 +++++++++++++++++++++++++++++++-
8 files changed, 358 insertions(+), 5 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH V5 7/7] igb_uio: fix uio release issue when hot unplug
2018-07-05 8:21 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
@ 2018-07-05 8:21 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-05 8:21 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When hotplug out device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this
resource, it will cause kernel crash. On the other hand, something like
interrupt disabling do not automatically process in kernel side. If not
handler it, this redundancy and dirty thing will affect the interrupt
resource be used by other device. So the igb_uio driver have to check the
hotplug status, and the corresponding process should be taken in igb uio
driver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such
as probed/opened/released/removed/unplug. When detect the unexpected
removal which cause of hotplug out behavior, it will corresponding disable
interrupt resource, while for the part of releasement which kernel have
already handle, just skip it to avoid double free or null pointer kernel
crash issue.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v5->v4:
add lock for udev state
---
kernel/linux/igb_uio/igb_uio.c | 51 +++++++++++++++++++++++++++++++++++++++---
1 file changed, 48 insertions(+), 3 deletions(-)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index 3398eac..adc8cea 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -19,6 +19,15 @@
#include "compat.h"
+/* uio pci device state */
+enum rte_udev_state {
+ RTE_UDEV_PROBED,
+ RTE_UDEV_OPENNED,
+ RTE_UDEV_RELEASED,
+ RTE_UDEV_REMOVED,
+ RTE_UDEV_UNPLUG
+};
+
/**
* A structure describing the private information for a uio device.
*/
@@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
enum rte_intr_mode mode;
struct mutex lock;
int refcnt;
+ enum rte_udev_state state;
};
static int wc_activate;
@@ -195,12 +205,22 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
{
struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
struct uio_info *info = &udev->info;
+ struct pci_dev *pdev = udev->pdev;
/* Legacy mode need to mask in hardware */
if (udev->mode == RTE_INTR_MODE_LEGACY &&
!pci_check_and_mask_intx(udev->pdev))
return IRQ_NONE;
+ mutex_lock(&udev->lock);
+ /* check the uevent of the kobj */
+ if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
+ dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
+ (&pdev->dev.kobj)->name);
+ udev->state = RTE_UDEV_UNPLUG;
+ }
+ mutex_unlock(&udev->lock);
+
uio_event_notify(info);
/* Message signal mode, no share IRQ and automasked */
@@ -309,7 +329,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
#endif
}
-
/**
* This gets called while opening uio device file.
*/
@@ -331,20 +350,29 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
/* enable interrupts */
err = igbuio_pci_enable_interrupts(udev);
- mutex_unlock(&udev->lock);
if (err) {
dev_err(&dev->dev, "Enable interrupt fails\n");
+ pci_clear_master(dev);
+ mutex_unlock(&udev->lock);
return err;
}
+ udev->state = RTE_UDEV_OPENNED;
+ mutex_unlock(&udev->lock);
return 0;
}
+/**
+ * This gets called while closing uio device file.
+ */
static int
igbuio_pci_release(struct uio_info *info, struct inode *inode)
{
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ if (udev->state == RTE_UDEV_REMOVED)
+ return 0;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
@@ -356,7 +384,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
/* stop the device from further DMA */
pci_clear_master(dev);
-
+ udev->state = RTE_UDEV_RELEASED;
mutex_unlock(&udev->lock);
return 0;
}
@@ -562,6 +590,9 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
(unsigned long long)map_dma_addr, map_addr);
}
+ mutex_lock(&udev->lock);
+ udev->state = RTE_UDEV_PROBED;
+ mutex_unlock(&udev->lock);
return 0;
fail_remove_group:
@@ -579,6 +610,20 @@ static void
igbuio_pci_remove(struct pci_dev *dev)
{
struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
+ int ret;
+
+ /* handler hot unplug */
+ if (udev->state == RTE_UDEV_OPENNED ||
+ udev->state == RTE_UDEV_UNPLUG) {
+ dev_notice(&dev->dev, "Unexpected removal!\n");
+ ret = igbuio_pci_release(&udev->info, NULL);
+ if (ret)
+ return;
+ mutex_lock(&udev->lock);
+ udev->state = RTE_UDEV_REMOVED;
+ mutex_unlock(&udev->lock);
+ return;
+ }
mutex_destroy(&udev->lock);
sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread
* [PATCH v6 0/7] hotplug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (10 preceding siblings ...)
2018-07-05 8:21 ` [PATCH V5 0/7] hot plug failure handle mechanism Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 1/7] bus: add hotplug failure handler Jeff Guo
` (6 more replies)
2018-07-09 11:56 ` [PATCH v7 0/7] hotplug failure handle mechanism Jeff Guo
` (11 subsequent siblings)
23 siblings, 7 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event detect mechanism, failsafe driver,
bonding driver and hot plug/unplug api in framework, app could use these to
develop their hot plug solution.
let’s see the case of hot unplug, it can happen when a hardware device is
be removed physically, or when the software disables it. App need to call
ether dev API to detach the device, to unplug the device at the bus level and
make access to the device invalid. But the problem is that, the removal of the
device from the software lists is not going to be instantaneous, at this time
if the data(fast) path still read/write the device, it will cause MMIO error
and result of the app crash out.
Seems that we have got fail-safe driver(or app) + RTE_ETH_EVENT_INTR_RMV +
kernel core driver solution to handle it, but still not have failsafe driver
(or app) + RTE_DEV_EVENT_REMOVE + PCIe pmd driver failure handle solution. So
there is an absence in dpdk hot plug solution right now.
Also, we know that kernel only guaranty hot plug on the kernel side, but not for
the user mode side. Firstly we can hardly have a gatekeeper for any MMIO for
multiple PMD driver. Secondly, no more specific 3rd tools such as udev/driverctl
have especially cover these hot plug failure processing. Third, the feasibility
of app’s implement for multiple user mode PMD driver is still a problem. Here,
a general hot plug failure handle mechanism in dpdk framework would be proposed,
it aim to guaranty that, when hot unplug occur, the system will not crash and
app will not be break out, and user space can normally stop and release any
relevant resources, then unplug of the device at the bus level cleanly.
The mechanism should be come across as bellow:
Firstly, app enabled the device event monitor and register the hot plug event’s
callback before running data path. Once the hot unplug behave occur, the
mechanism will detect the removal event and then accordingly do the failure
handle. In order to do that, below functional will be bring in.
- Add a new bus ops “handle_hot_unplug” to handle bus read/write error, it is
bus-specific and each kind of bus can implement its own logic.
- Implement pci bus specific ops “pci_handle_hot_unplug”. It will base on the
failure address to remap memory for the corresponding device that unplugged.
For the data path or other unexpected control from the control path when hot
unplug occur.
- Implement a new sigbus handler, it is registered when start device even
monitoring. The handler is per process. Base on the signal event principle,
control path thread and data path thread will randomly receive the sigbus
error, but will go to the common sigbus handler. Once the MMIO sigbus error
exposure, it will trigger the above hot unplug operation. The sigbus will be
check if it is cause of the hot unplug or not, if not will info exception as
the original sigbus handler. If yes, will do memory remapping.
For the control path and the igb uio release:
- When hot unplug device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this resource,
it will cause kernel crash.
On the other hand, something like interrupt disable do not automatically
process in kernel side. If not handler it, this redundancy and dirty thing
will affect the interrupt resource be used by other device.
So the igb_uio driver have to check the hot plug status and corresponding
process should be taken in igb uio deriver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such as
probed/opened/released/removed/unplug. When detect the unexpected removal
which cause of hot unplug behavior, it will corresponding disable interrupt
resource, while for the part of releasement which kernel have already handle,
just skip it to avoid double free or null pointer kernel crash issue.
The mechanism could be use for fail-safe driver and app which want to use hot
plug solution. let testpmd for example:
- Enable device event monitor->device unplug->failure handle->stop forwarding->
stop port->close port->detach port.
This process will not breaking the app/fail-safe running, and will not break
other irrelevance device. And app could plug in the device and restart the date
path again by below.
- Device plug in->bind igb_uio driver ->attached device->start port->
start forwarding.
patchset history:
v6->v5:
refine some description about bus ops
refine commit log
add some entry check.
v5->v4:
split patches to focus on the failure handle, remove the event usage by testpmd
to another patch.
change the hotplug failure handler name
refine the sigbus handle logic
add lock for udev state in igb uio driver
v4->v3:
split patches to be small and clear
change to use new parameter "--hotplug-mode" in testpmd
to identify the eal hotplug and ethdev hotplug
v3->v2:
change bus ops name to bus_hotplug_handler.
add new API and bus ops of bus_signal_handler
distingush handle generic sigbus and hotplug sigbus
v2->v1(v21):
refine some doc and commit log
fix igb uio kernel issue for control path failure
rebase testpmd code
Since the hot plug solution be discussed serval around in the public, the
scope be changed and the patch set be split into many times. Coming to the
recently RFC and feature design, it just focus on the hot unplug failure
handler at this patch set, so in order let this topic more clear and focus,
summarize privours patch set in history “v1(v21)”, the v2 here go ahead
for further track.
"v1(21)" == v21 as below:
v21->v20:
split function in hot unplug ops
sync failure hanlde to fix multiple process issue fix attach port issue for multiple devices case.
combind rmv callback function to be only one.
v20->v19:
clean the code
refine the remap logic for multiple device.
remove the auto binding
v19->18:
note for limitation of multiple hotplug,fix some typo, sqeeze patch.
v18->v15:
add document, add signal bus handler, refine the code to be more clear.
the prior patch history please check the patch set "add device event monitor framework"
Jeff Guo (7):
bus: add hotplug failure handler
bus/pci: implement hotplug failure handler ops
bus: add sigbus handler
bus/pci: implement sigbus handler operation
bus: add helper to handle sigbus
eal: add failure handle mechanism for hotplug
igb_uio: fix uio release issue when hot unplug
drivers/bus/pci/pci_common.c | 77 +++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++
drivers/bus/pci/private.h | 12 ++++
kernel/linux/igb_uio/igb_uio.c | 51 +++++++++++++-
lib/librte_eal/common/eal_common_bus.c | 42 ++++++++++++
lib/librte_eal/common/eal_private.h | 12 ++++
lib/librte_eal/common/include/rte_bus.h | 33 +++++++++
lib/librte_eal/linuxapp/eal/eal_dev.c | 114 +++++++++++++++++++++++++++++++-
8 files changed, 370 insertions(+), 4 deletions(-)
--
2.7.4
^ permalink raw reply [flat|nested] 494+ messages in thread* [PATCH v6 1/7] bus: add hotplug failure handler
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 2/7] bus/pci: implement hotplug failure handler ops Jeff Guo
` (5 subsequent siblings)
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When device be hotplug out, if app still continue to access device by mmio,
it will cause of memory failure and result the system crash.
This patch introduces a bus ops to handle device hotplug failure, it is a
bus specific behavior, so each kind of bus can implement its own logic case
by case.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some description of bus ops
---
lib/librte_eal/common/include/rte_bus.h | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index eb9eded..e3a55a8 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -168,6 +168,20 @@ typedef int (*rte_bus_unplug_t)(struct rte_device *dev);
typedef int (*rte_bus_parse_t)(const char *name, void *addr);
/**
+ * Implementation a specific hotplug failure handler, which is responsible
+ * for handle the failure when the device be hotplug out from the bus. When
+ * hotplug removal event be detected, it could call this function to handle
+ * failure and guaranty the system would not crash in the case.
+ * @param dev
+ * Pointer of the device structure.
+ *
+ * @return
+ * 0 on success.
+ * !0 on error.
+ */
+typedef int (*rte_bus_hotplug_failure_handler_t)(struct rte_device *dev);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -211,6 +225,8 @@ struct rte_bus {
rte_bus_parse_t parse; /**< Parse a device name */
struct rte_bus_conf conf; /**< Bus configuration */
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
+ rte_bus_hotplug_failure_handler_t hotplug_failure_handler;
+ /**< handle hotplug failure on bus */
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v6 2/7] bus/pci: implement hotplug failure handler ops
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
2018-07-09 6:51 ` [PATCH v6 1/7] bus: add hotplug failure handler Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 3/7] bus: add sigbus handler Jeff Guo
` (4 subsequent siblings)
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of hotplug failure handler for PCI bus,
it is functional to remap a new dummy memory which overlap to the
failure memory to avoid MMIO read/write error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some typo
---
drivers/bus/pci/pci_common.c | 28 ++++++++++++++++++++++++++++
drivers/bus/pci/pci_common_uio.c | 33 +++++++++++++++++++++++++++++++++
drivers/bus/pci/private.h | 12 ++++++++++++
3 files changed, 73 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index 94b0f41..d7abe6c 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -408,6 +408,33 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
}
static int
+pci_hotplug_failure_handler(struct rte_device *dev)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = RTE_DEV_TO_PCI(dev);
+ if (!pdev)
+ return -1;
+
+ switch (pdev->kdrv) {
+ case RTE_KDRV_IGB_UIO:
+ case RTE_KDRV_UIO_GENERIC:
+ case RTE_KDRV_NIC_UIO:
+ /* mmio resource is invalid, remap it to be safe. */
+ ret = pci_uio_remap_resource(pdev);
+ break;
+ default:
+ RTE_LOG(DEBUG, EAL,
+ "Not managed by a supported kernel driver, skipped\n");
+ ret = -1;
+ break;
+ }
+
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -437,6 +464,7 @@ struct rte_pci_bus rte_pci_bus = {
.unplug = pci_unplug,
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
+ .hotplug_failure_handler = pci_hotplug_failure_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
diff --git a/drivers/bus/pci/pci_common_uio.c b/drivers/bus/pci/pci_common_uio.c
index 54bc20b..7ea73db 100644
--- a/drivers/bus/pci/pci_common_uio.c
+++ b/drivers/bus/pci/pci_common_uio.c
@@ -146,6 +146,39 @@ pci_uio_unmap(struct mapped_pci_resource *uio_res)
}
}
+/* remap the PCI resource of a PCI device in anonymous virtual memory */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev)
+{
+ int i;
+ void *map_address;
+
+ if (dev == NULL)
+ return -1;
+
+ /* Remap all BARs */
+ for (i = 0; i != PCI_MAX_RESOURCE; i++) {
+ /* skip empty BAR */
+ if (dev->mem_resource[i].phys_addr == 0)
+ continue;
+ map_address = mmap(dev->mem_resource[i].addr,
+ (size_t)dev->mem_resource[i].len,
+ PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+ if (map_address == MAP_FAILED) {
+ RTE_LOG(ERR, EAL,
+ "Cannot remap resource for device %s\n",
+ dev->name);
+ return -1;
+ }
+ RTE_LOG(INFO, EAL,
+ "Successful remap resource for device %s\n",
+ dev->name);
+ }
+
+ return 0;
+}
+
static struct mapped_pci_resource *
pci_uio_find_resource(struct rte_pci_device *dev)
{
diff --git a/drivers/bus/pci/private.h b/drivers/bus/pci/private.h
index 8ddd03e..6b312e5 100644
--- a/drivers/bus/pci/private.h
+++ b/drivers/bus/pci/private.h
@@ -123,6 +123,18 @@ void pci_uio_free_resource(struct rte_pci_device *dev,
struct mapped_pci_resource *uio_res);
/**
+ * Remap the PCI resource of a PCI device in anonymous virtual memory.
+ *
+ * @param dev
+ * Point to the struct rte pci device.
+ * @return
+ * - On success, zero.
+ * - On failure, a negative value.
+ */
+int
+pci_uio_remap_resource(struct rte_pci_device *dev);
+
+/**
* Map device memory to uio resource
*
* This function is private to EAL.
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v6 3/7] bus: add sigbus handler
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
2018-07-09 6:51 ` [PATCH v6 1/7] bus: add hotplug failure handler Jeff Guo
2018-07-09 6:51 ` [PATCH v6 2/7] bus/pci: implement hotplug failure handler ops Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 4/7] bus/pci: implement sigbus handler operation Jeff Guo
` (3 subsequent siblings)
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When device be hotplug out, if data path still read/write device, the
sigbus error will occur, this error need to be handled. So a handler
need to be here to capture the signal and handle it correspondingly.
This patch introduces a bus ops to handle sigbus error, it is a bus
specific behavior, so that each kind of bus can implement its own logic
case by case.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some description of bus ops
---
lib/librte_eal/common/include/rte_bus.h | 17 +++++++++++++++++
1 file changed, 17 insertions(+)
diff --git a/lib/librte_eal/common/include/rte_bus.h b/lib/librte_eal/common/include/rte_bus.h
index e3a55a8..216ad1e 100644
--- a/lib/librte_eal/common/include/rte_bus.h
+++ b/lib/librte_eal/common/include/rte_bus.h
@@ -182,6 +182,21 @@ typedef int (*rte_bus_parse_t)(const char *name, void *addr);
typedef int (*rte_bus_hotplug_failure_handler_t)(struct rte_device *dev);
/**
+ * Implementation a specific sigbus handler, which is responsible for handle
+ * the sigbus error which is either original memory error, or specific memory
+ * error that caused of hot unplug. When sigbus error be captured, it could
+ * call this function to handle sigbus error.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 for success handle the sigbus.
+ * 1 for no bus handle the sigbus.
+ * -1 for failed to handle the sigbus
+ */
+typedef int (*rte_bus_sigbus_handler_t)(const void *failure_addr);
+
+/**
* Bus scan policies
*/
enum rte_bus_scan_mode {
@@ -227,6 +242,8 @@ struct rte_bus {
rte_bus_get_iommu_class_t get_iommu_class; /**< Get iommu class */
rte_bus_hotplug_failure_handler_t hotplug_failure_handler;
/**< handle hotplug failure on bus */
+ rte_bus_sigbus_handler_t sigbus_handler; /**< handle sigbus error */
+
};
/**
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v6 4/7] bus/pci: implement sigbus handler operation
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
` (2 preceding siblings ...)
2018-07-09 6:51 ` [PATCH v6 3/7] bus: add sigbus handler Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 5/7] bus: add helper to handle sigbus Jeff Guo
` (2 subsequent siblings)
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch implements the ops of sigbus handler for PCI bus, it is
functional to find the corresponding pci device which is been hotplug
out, and then call the bus ops of hotplug failure handler to handle
the failure for the device.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some typo
---
drivers/bus/pci/pci_common.c | 49 ++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 49 insertions(+)
diff --git a/drivers/bus/pci/pci_common.c b/drivers/bus/pci/pci_common.c
index d7abe6c..37ad266 100644
--- a/drivers/bus/pci/pci_common.c
+++ b/drivers/bus/pci/pci_common.c
@@ -407,6 +407,32 @@ pci_find_device(const struct rte_device *start, rte_dev_cmp_t cmp,
return NULL;
}
+/* check the failure address belongs to which device. */
+static struct rte_pci_device *
+pci_find_device_by_addr(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int i;
+
+ FOREACH_DEVICE_ON_PCIBUS(pdev) {
+ for (i = 0; i != RTE_DIM(pdev->mem_resource); i++) {
+ if ((uint64_t)(uintptr_t)failure_addr >=
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr &&
+ (uint64_t)(uintptr_t)failure_addr <
+ (uint64_t)(uintptr_t)pdev->mem_resource[i].addr +
+ pdev->mem_resource[i].len) {
+ RTE_LOG(INFO, EAL, "Failure address "
+ "%16.16"PRIx64" belongs to "
+ "device %s!\n",
+ (uint64_t)(uintptr_t)failure_addr,
+ pdev->device.name);
+ return pdev;
+ }
+ }
+ }
+ return NULL;
+}
+
static int
pci_hotplug_failure_handler(struct rte_device *dev)
{
@@ -435,6 +461,28 @@ pci_hotplug_failure_handler(struct rte_device *dev)
}
static int
+pci_sigbus_handler(const void *failure_addr)
+{
+ struct rte_pci_device *pdev = NULL;
+ int ret = 0;
+
+ pdev = pci_find_device_by_addr(failure_addr);
+ if (!pdev) {
+ /* It is a generic sigbus error, no bus would handle it. */
+ ret = 1;
+ } else {
+ /* The sigbus error is caused of hot removal. */
+ ret = pci_hotplug_failure_handler(&pdev->device);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Failed to handle hot plug for "
+ "device %s", pdev->name);
+ ret = -1;
+ }
+ }
+ return ret;
+}
+
+static int
pci_plug(struct rte_device *dev)
{
return pci_probe_all_drivers(RTE_DEV_TO_PCI(dev));
@@ -465,6 +513,7 @@ struct rte_pci_bus rte_pci_bus = {
.parse = pci_parse,
.get_iommu_class = rte_pci_get_iommu_class,
.hotplug_failure_handler = pci_hotplug_failure_handler,
+ .sigbus_handler = pci_sigbus_handler,
},
.device_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.device_list),
.driver_list = TAILQ_HEAD_INITIALIZER(rte_pci_bus.driver_list),
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v6 5/7] bus: add helper to handle sigbus
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
` (3 preceding siblings ...)
2018-07-09 6:51 ` [PATCH v6 4/7] bus/pci: implement sigbus handler operation Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 6:51 ` [PATCH v6 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
2018-07-09 6:51 ` [PATCH v6 7/7] igb_uio: fix uio release issue when hot unplug Jeff Guo
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch aim to add a helper to iterate all buses to find the
corresponding bus to handle the sigbus error.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some coding style.
---
lib/librte_eal/common/eal_common_bus.c | 42 ++++++++++++++++++++++++++++++++++
lib/librte_eal/common/eal_private.h | 12 ++++++++++
2 files changed, 54 insertions(+)
diff --git a/lib/librte_eal/common/eal_common_bus.c b/lib/librte_eal/common/eal_common_bus.c
index 0943851..8856adc 100644
--- a/lib/librte_eal/common/eal_common_bus.c
+++ b/lib/librte_eal/common/eal_common_bus.c
@@ -37,6 +37,7 @@
#include <rte_bus.h>
#include <rte_debug.h>
#include <rte_string_fns.h>
+#include <rte_errno.h>
#include "eal_private.h"
@@ -242,3 +243,44 @@ rte_bus_get_iommu_class(void)
}
return mode;
}
+
+static int
+bus_handle_sigbus(const struct rte_bus *bus,
+ const void *failure_addr)
+{
+ int ret;
+
+ if (!bus->sigbus_handler) {
+ RTE_LOG(ERR, EAL, "Function sigbus_handler not supported by "
+ "bus (%s)\n", bus->name);
+ return -1;
+ }
+
+ ret = bus->sigbus_handler(failure_addr);
+ rte_errno = ret;
+
+ return !(bus->sigbus_handler && ret <= 0);
+}
+
+int
+rte_bus_sigbus_handler(const void *failure_addr)
+{
+ struct rte_bus *bus;
+
+ int ret = 0;
+ int old_errno = rte_errno;
+
+ rte_errno = 0;
+
+ bus = rte_bus_find(NULL, bus_handle_sigbus, failure_addr);
+ /* failed to handle the sigbus, pass the new errno. */
+ if (!bus)
+ ret = 1;
+ else if (rte_errno == -1)
+ return -1;
+
+ /* otherwise restore the old errno. */
+ rte_errno = old_errno;
+
+ return ret;
+}
diff --git a/lib/librte_eal/common/eal_private.h b/lib/librte_eal/common/eal_private.h
index bdadc4d..2337e71 100644
--- a/lib/librte_eal/common/eal_private.h
+++ b/lib/librte_eal/common/eal_private.h
@@ -258,4 +258,16 @@ int rte_mp_channel_init(void);
*/
void dev_callback_process(char *device_name, enum rte_dev_event_type event);
+/**
+ * Iterate all buses to find the corresponding bus, to handle the sigbus error.
+ * @param failure_addr
+ * Pointer of the fault address of the sigbus error.
+ *
+ * @return
+ * 0 success to handle the sigbus.
+ * -1 failed to handle the sigbus
+ * 1 no bus can handler the sigbus
+ */
+int rte_bus_sigbus_handler(const void *failure_addr);
+
#endif /* _EAL_PRIVATE_H_ */
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* [PATCH v6 6/7] eal: add failure handle mechanism for hotplug
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
` (4 preceding siblings ...)
2018-07-09 6:51 ` [PATCH v6 5/7] bus: add helper to handle sigbus Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
2018-07-09 7:42 ` Gaëtan Rivet
2018-07-09 6:51 ` [PATCH v6 7/7] igb_uio: fix uio release issue when hot unplug Jeff Guo
6 siblings, 1 reply; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
This patch introduces a failure handle mechanism to handle device
hotplug removal event.
First it can register sigbus handler when enable device event monitor. Once
sigbus error be captured, it will check the failure address and accordingly
remap the invalid memory for the corresponding device. Besed on this
mechanism, it could guaranty the application not crash when the device be
hotplug out.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
Acked-by: Shaopeng He <shaopeng.he@intel.com>
---
v6->v5:
refine some doc and coding style
---
lib/librte_eal/linuxapp/eal/eal_dev.c | 114 +++++++++++++++++++++++++++++++++-
1 file changed, 113 insertions(+), 1 deletion(-)
diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
index 1cf6aeb..cb30729 100644
--- a/lib/librte_eal/linuxapp/eal/eal_dev.c
+++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
@@ -4,6 +4,8 @@
#include <string.h>
#include <unistd.h>
+#include <fcntl.h>
+#include <signal.h>
#include <sys/socket.h>
#include <linux/netlink.h>
@@ -14,15 +16,31 @@
#include <rte_malloc.h>
#include <rte_interrupts.h>
#include <rte_alarm.h>
+#include <rte_bus.h>
+#include <rte_eal.h>
+#include <rte_spinlock.h>
+#include <rte_errno.h>
#include "eal_private.h"
static struct rte_intr_handle intr_handle = {.fd = -1 };
static bool monitor_started;
+extern struct rte_bus_list rte_bus_list;
+
#define EAL_UEV_MSG_LEN 4096
#define EAL_UEV_MSG_ELEM_LEN 128
+/*
+ * spinlock for device failure process, protect the bus and the device
+ * to avoid race condition.
+ */
+static rte_spinlock_t dev_failure_lock = RTE_SPINLOCK_INITIALIZER;
+
+static struct sigaction sigbus_action_old;
+
+static int sigbus_need_recover;
+
static void dev_uev_handler(__rte_unused void *param);
/* identify the system layer which reports this event. */
@@ -33,6 +51,49 @@ enum eal_dev_event_subsystem {
EAL_DEV_EVENT_SUBSYSTEM_MAX
};
+static void
+sigbus_action_recover(void)
+{
+ if (sigbus_need_recover) {
+ sigaction(SIGBUS, &sigbus_action_old, NULL);
+ sigbus_need_recover = 0;
+ }
+}
+
+static void sigbus_handler(int signum, siginfo_t *info,
+ void *ctx __rte_unused)
+{
+ int ret;
+
+ RTE_LOG(INFO, EAL, "Thread[%d] catch SIGBUS, fault address:%p\n",
+ (int)pthread_self(), info->si_addr);
+
+ rte_spinlock_lock(&dev_failure_lock);
+ ret = rte_bus_sigbus_handler(info->si_addr);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret == -1) {
+ rte_exit(EXIT_FAILURE,
+ "Failed to handle SIGBUS for hotplug, "
+ "(rte_errno: %s)!", strerror(rte_errno));
+ } else if (ret == 1) {
+ if (sigbus_action_old.sa_handler)
+ (*(sigbus_action_old.sa_handler))(signum);
+ else
+ rte_exit(EXIT_FAILURE,
+ "Failed to handle generic SIGBUS!");
+ }
+
+ RTE_LOG(INFO, EAL, "Success to handle SIGBUS for hotplug!\n");
+}
+
+static int cmp_dev_name(const struct rte_device *dev,
+ const void *_name)
+{
+ const char *name = _name;
+
+ return strcmp(dev->name, name);
+}
+
static int
dev_uev_socket_fd_create(void)
{
@@ -147,6 +208,9 @@ dev_uev_handler(__rte_unused void *param)
struct rte_dev_event uevent;
int ret;
char buf[EAL_UEV_MSG_LEN];
+ struct rte_bus *bus;
+ struct rte_device *dev;
+ const char *busname = "";
memset(&uevent, 0, sizeof(struct rte_dev_event));
memset(buf, 0, EAL_UEV_MSG_LEN);
@@ -171,13 +235,50 @@ dev_uev_handler(__rte_unused void *param)
RTE_LOG(DEBUG, EAL, "receive uevent(name:%s, type:%d, subsystem:%d)\n",
uevent.devname, uevent.type, uevent.subsystem);
- if (uevent.devname)
+ switch (uevent.subsystem) {
+ case EAL_DEV_EVENT_SUBSYSTEM_PCI:
+ case EAL_DEV_EVENT_SUBSYSTEM_UIO:
+ busname = "pci";
+ break;
+ default:
+ break;
+ }
+
+ if (uevent.devname) {
+ if (uevent.type == RTE_DEV_EVENT_REMOVE) {
+ rte_spinlock_lock(&dev_failure_lock);
+ bus = rte_bus_find_by_name(busname);
+ if (bus == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find bus (%s)\n",
+ busname);
+ return;
+ }
+
+ dev = bus->find_device(NULL, cmp_dev_name,
+ uevent.devname);
+ if (dev == NULL) {
+ RTE_LOG(ERR, EAL, "Cannot find device (%s) on "
+ "bus (%s)\n", uevent.devname, busname);
+ return;
+ }
+
+ ret = bus->hotplug_failure_handler(dev);
+ rte_spinlock_unlock(&dev_failure_lock);
+ if (ret) {
+ RTE_LOG(ERR, EAL, "Can not handle hotplug for "
+ "device (%s)\n", dev->name);
+ return;
+ }
+ }
dev_callback_process(uevent.devname, uevent.type);
+ }
}
int __rte_experimental
rte_dev_event_monitor_start(void)
{
+ sigset_t mask;
+ struct sigaction action;
int ret;
if (monitor_started)
@@ -197,6 +298,14 @@ rte_dev_event_monitor_start(void)
return -1;
}
+ /* register sigbus handler */
+ sigemptyset(&mask);
+ sigaddset(&mask, SIGBUS);
+ action.sa_flags = SA_SIGINFO;
+ action.sa_mask = mask;
+ action.sa_sigaction = sigbus_handler;
+ sigbus_need_recover = !sigaction(SIGBUS, &action, &sigbus_action_old);
+
monitor_started = true;
return 0;
@@ -217,8 +326,11 @@ rte_dev_event_monitor_stop(void)
return ret;
}
+ sigbus_action_recover();
+
close(intr_handle.fd);
intr_handle.fd = -1;
monitor_started = false;
+
return 0;
}
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread* Re: [PATCH v6 6/7] eal: add failure handle mechanism for hotplug
2018-07-09 6:51 ` [PATCH v6 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
@ 2018-07-09 7:42 ` Gaëtan Rivet
2018-07-09 8:12 ` Jeff Guo
0 siblings, 1 reply; 494+ messages in thread
From: Gaëtan Rivet @ 2018-07-09 7:42 UTC (permalink / raw)
To: Jeff Guo
Cc: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, bernard.iremonger, arybchenko, jblunck,
shreyansh.jain, dev, helin.zhang
Hi Jeff,
On Mon, Jul 09, 2018 at 02:51:21PM +0800, Jeff Guo wrote:
> This patch introduces a failure handle mechanism to handle device
> hotplug removal event.
>
> First it can register sigbus handler when enable device event monitor. Once
> sigbus error be captured, it will check the failure address and accordingly
> remap the invalid memory for the corresponding device. Besed on this
> mechanism, it could guaranty the application not crash when the device be
> hotplug out.
>
> Signed-off-by: Jeff Guo <jia.guo@intel.com>
> Acked-by: Shaopeng He <shaopeng.he@intel.com>
> ---
> v6->v5:
> refine some doc and coding style
> ---
> lib/librte_eal/linuxapp/eal/eal_dev.c | 114 +++++++++++++++++++++++++++++++++-
> 1 file changed, 113 insertions(+), 1 deletion(-)
>
> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
> index 1cf6aeb..cb30729 100644
> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
> @@ -4,6 +4,8 @@
>
> #include <string.h>
> #include <unistd.h>
> +#include <fcntl.h>
> +#include <signal.h>
> #include <sys/socket.h>
> #include <linux/netlink.h>
>
> @@ -14,15 +16,31 @@
> #include <rte_malloc.h>
> #include <rte_interrupts.h>
> #include <rte_alarm.h>
> +#include <rte_bus.h>
> +#include <rte_eal.h>
> +#include <rte_spinlock.h>
> +#include <rte_errno.h>
>
> #include "eal_private.h"
>
> static struct rte_intr_handle intr_handle = {.fd = -1 };
> static bool monitor_started;
>
> +extern struct rte_bus_list rte_bus_list;
> +
Where do you use the rte_bus_list? It seems the reference is a remnant
from a previous version.
You do not seem to need a direct access on rte_bus_list,
as you call rte_bus_find instead.
Why do you need this extern? I think its absence is motivated: to keep the
bus list private and force users to access it through standard exposed ways.
Regards,
--
Gaëtan Rivet
6WIND
^ permalink raw reply [flat|nested] 494+ messages in thread* Re: [PATCH v6 6/7] eal: add failure handle mechanism for hotplug
2018-07-09 7:42 ` Gaëtan Rivet
@ 2018-07-09 8:12 ` Jeff Guo
0 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 8:12 UTC (permalink / raw)
To: Gaëtan Rivet
Cc: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
jingjing.wu, thomas, motih, matan, harry.van.haaren, qi.z.zhang,
shaopeng.he, bernard.iremonger, arybchenko, jblunck,
shreyansh.jain, dev, helin.zhang
hi, gaetan
On 7/9/2018 3:42 PM, Gaëtan Rivet wrote:
> Hi Jeff,
>
> On Mon, Jul 09, 2018 at 02:51:21PM +0800, Jeff Guo wrote:
>> This patch introduces a failure handle mechanism to handle device
>> hotplug removal event.
>>
>> First it can register sigbus handler when enable device event monitor. Once
>> sigbus error be captured, it will check the failure address and accordingly
>> remap the invalid memory for the corresponding device. Besed on this
>> mechanism, it could guaranty the application not crash when the device be
>> hotplug out.
>>
>> Signed-off-by: Jeff Guo <jia.guo@intel.com>
>> Acked-by: Shaopeng He <shaopeng.he@intel.com>
>> ---
>> v6->v5:
>> refine some doc and coding style
>> ---
>> lib/librte_eal/linuxapp/eal/eal_dev.c | 114 +++++++++++++++++++++++++++++++++-
>> 1 file changed, 113 insertions(+), 1 deletion(-)
>>
>> diff --git a/lib/librte_eal/linuxapp/eal/eal_dev.c b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> index 1cf6aeb..cb30729 100644
>> --- a/lib/librte_eal/linuxapp/eal/eal_dev.c
>> +++ b/lib/librte_eal/linuxapp/eal/eal_dev.c
>> @@ -4,6 +4,8 @@
>>
>> #include <string.h>
>> #include <unistd.h>
>> +#include <fcntl.h>
>> +#include <signal.h>
>> #include <sys/socket.h>
>> #include <linux/netlink.h>
>>
>> @@ -14,15 +16,31 @@
>> #include <rte_malloc.h>
>> #include <rte_interrupts.h>
>> #include <rte_alarm.h>
>> +#include <rte_bus.h>
>> +#include <rte_eal.h>
>> +#include <rte_spinlock.h>
>> +#include <rte_errno.h>
>>
>> #include "eal_private.h"
>>
>> static struct rte_intr_handle intr_handle = {.fd = -1 };
>> static bool monitor_started;
>>
>> +extern struct rte_bus_list rte_bus_list;
>> +
> Where do you use the rte_bus_list? It seems the reference is a remnant
> from a previous version.
>
> You do not seem to need a direct access on rte_bus_list,
> as you call rte_bus_find instead.
>
> Why do you need this extern? I think its absence is motivated: to keep the
> bus list private and force users to access it through standard exposed ways.
>
> Regards,
i think that is my missing here. Will delete it. Thanks for your info.
^ permalink raw reply [flat|nested] 494+ messages in thread
* [PATCH v6 7/7] igb_uio: fix uio release issue when hot unplug
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
` (5 preceding siblings ...)
2018-07-09 6:51 ` [PATCH v6 6/7] eal: add failure handle mechanism for hotplug Jeff Guo
@ 2018-07-09 6:51 ` Jeff Guo
6 siblings, 0 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 6:51 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
When hotplug out device, the kernel will release the device resource in the
kernel side, such as the fd sys file will disappear, and the irq will be
released. At this time, if igb uio driver still try to release this
resource, it will cause kernel crash. On the other hand, something like
interrupt disabling do not automatically process in kernel side. If not
handler it, this redundancy and dirty thing will affect the interrupt
resource be used by other device. So the igb_uio driver have to check the
hotplug status, and the corresponding process should be taken in igb uio
driver.
This patch propose to add structure of rte_udev_state into rte_uio_pci_dev
of igb_uio kernel driver, which will record the state of uio device, such
as probed/opened/released/removed/unplug. When detect the unexpected
removal which cause of hotplug out behavior, it will corresponding disable
interrupt resource, while for the part of releasement which kernel have
already handle, just skip it to avoid double free or null pointer kernel
crash issue.
Signed-off-by: Jeff Guo <jia.guo@intel.com>
---
v6->v5:
no change
---
kernel/linux/igb_uio/igb_uio.c | 51 +++++++++++++++++++++++++++++++++++++++---
1 file changed, 48 insertions(+), 3 deletions(-)
diff --git a/kernel/linux/igb_uio/igb_uio.c b/kernel/linux/igb_uio/igb_uio.c
index 3398eac..adc8cea 100644
--- a/kernel/linux/igb_uio/igb_uio.c
+++ b/kernel/linux/igb_uio/igb_uio.c
@@ -19,6 +19,15 @@
#include "compat.h"
+/* uio pci device state */
+enum rte_udev_state {
+ RTE_UDEV_PROBED,
+ RTE_UDEV_OPENNED,
+ RTE_UDEV_RELEASED,
+ RTE_UDEV_REMOVED,
+ RTE_UDEV_UNPLUG
+};
+
/**
* A structure describing the private information for a uio device.
*/
@@ -28,6 +37,7 @@ struct rte_uio_pci_dev {
enum rte_intr_mode mode;
struct mutex lock;
int refcnt;
+ enum rte_udev_state state;
};
static int wc_activate;
@@ -195,12 +205,22 @@ igbuio_pci_irqhandler(int irq, void *dev_id)
{
struct rte_uio_pci_dev *udev = (struct rte_uio_pci_dev *)dev_id;
struct uio_info *info = &udev->info;
+ struct pci_dev *pdev = udev->pdev;
/* Legacy mode need to mask in hardware */
if (udev->mode == RTE_INTR_MODE_LEGACY &&
!pci_check_and_mask_intx(udev->pdev))
return IRQ_NONE;
+ mutex_lock(&udev->lock);
+ /* check the uevent of the kobj */
+ if ((&pdev->dev.kobj)->state_remove_uevent_sent == 1) {
+ dev_notice(&pdev->dev, "device:%s, sent remove uevent!\n",
+ (&pdev->dev.kobj)->name);
+ udev->state = RTE_UDEV_UNPLUG;
+ }
+ mutex_unlock(&udev->lock);
+
uio_event_notify(info);
/* Message signal mode, no share IRQ and automasked */
@@ -309,7 +329,6 @@ igbuio_pci_disable_interrupts(struct rte_uio_pci_dev *udev)
#endif
}
-
/**
* This gets called while opening uio device file.
*/
@@ -331,20 +350,29 @@ igbuio_pci_open(struct uio_info *info, struct inode *inode)
/* enable interrupts */
err = igbuio_pci_enable_interrupts(udev);
- mutex_unlock(&udev->lock);
if (err) {
dev_err(&dev->dev, "Enable interrupt fails\n");
+ pci_clear_master(dev);
+ mutex_unlock(&udev->lock);
return err;
}
+ udev->state = RTE_UDEV_OPENNED;
+ mutex_unlock(&udev->lock);
return 0;
}
+/**
+ * This gets called while closing uio device file.
+ */
static int
igbuio_pci_release(struct uio_info *info, struct inode *inode)
{
struct rte_uio_pci_dev *udev = info->priv;
struct pci_dev *dev = udev->pdev;
+ if (udev->state == RTE_UDEV_REMOVED)
+ return 0;
+
mutex_lock(&udev->lock);
if (--udev->refcnt > 0) {
mutex_unlock(&udev->lock);
@@ -356,7 +384,7 @@ igbuio_pci_release(struct uio_info *info, struct inode *inode)
/* stop the device from further DMA */
pci_clear_master(dev);
-
+ udev->state = RTE_UDEV_RELEASED;
mutex_unlock(&udev->lock);
return 0;
}
@@ -562,6 +590,9 @@ igbuio_pci_probe(struct pci_dev *dev, const struct pci_device_id *id)
(unsigned long long)map_dma_addr, map_addr);
}
+ mutex_lock(&udev->lock);
+ udev->state = RTE_UDEV_PROBED;
+ mutex_unlock(&udev->lock);
return 0;
fail_remove_group:
@@ -579,6 +610,20 @@ static void
igbuio_pci_remove(struct pci_dev *dev)
{
struct rte_uio_pci_dev *udev = pci_get_drvdata(dev);
+ int ret;
+
+ /* handler hot unplug */
+ if (udev->state == RTE_UDEV_OPENNED ||
+ udev->state == RTE_UDEV_UNPLUG) {
+ dev_notice(&dev->dev, "Unexpected removal!\n");
+ ret = igbuio_pci_release(&udev->info, NULL);
+ if (ret)
+ return;
+ mutex_lock(&udev->lock);
+ udev->state = RTE_UDEV_REMOVED;
+ mutex_unlock(&udev->lock);
+ return;
+ }
mutex_destroy(&udev->lock);
sysfs_remove_group(&dev->dev.kobj, &dev_attr_grp);
--
2.7.4
^ permalink raw reply related [flat|nested] 494+ messages in thread
* [PATCH v7 0/7] hotplug failure handle mechanism
2017-06-29 4:37 ` [PATCH v3 0/2] add uevent api for hot plug Jeff Guo
` (11 preceding siblings ...)
2018-07-09 6:51 ` [PATCH v6 0/7] hotplug failure handle mechanism Jeff Guo
@ 2018-07-09 11:56 ` Jeff Guo
2018-07-09 11:56 ` [PATCH v7 1/7] bus: add hotplug failure handler Jeff Guo
` (6 more replies)
2018-07-09 12:00 ` [PATCH v7 0/7] hotplug failure handle mechanism Jeff Guo
` (10 subsequent siblings)
23 siblings, 7 replies; 494+ messages in thread
From: Jeff Guo @ 2018-07-09 11:56 UTC (permalink / raw)
To: stephen, bruce.richardson, ferruh.yigit, konstantin.ananyev,
gaetan.rivet, jingjing.wu, thomas, motih, matan, harry.van.haaren,
qi.z.zhang, shaopeng.he, bernard.iremonger, arybchenko,
wenzhuo.lu
Cc: jblunck, shreyansh.jain, dev, jia.guo, helin.zhang
As we know, hot plug is an importance feature, either use for the datacenter
device’s fail-safe, or use for SRIOV Live Migration in SDN/NFV. It could bring
the higher flexibility and continuality to the networking services in multiple
use cases in industry. So let we see, dpdk as an importance networking
framework, what can it help to implement hot plug solution for users.
We already have a general device event