From: Nigel Kirkland <nkirkland2304@gmail.com>
To: linux-scsi@vger.kernel.org, nigel.kirkland@broadcom.com
Cc: paul.ely@broadcom.com, nkirkland2304@gmail.com
Subject: [PATCH v4 11/14] lpfc: Put iocbq on phba->txq when ELS WQ is full or ELS SGL unavailable
Date: Thu, 17 Sep 2026 15:20:12 -0700 [thread overview]
Message-ID: <20260917222015.61053-12-nkirkland2304@gmail.com> (raw)
In-Reply-To: <20260917222015.61053-1-nkirkland2304@gmail.com>
When ELS/CT commands can't be sent due to ELS WQ full, queue the iocbq on
the phba->txq tail for retrying submission later in lpfc_drain_txq.
lpfc_drain_txq is flagged to be called by the worker thread when an ELS CQE
completes lpfc_sli4_sp_handle_els_wcqe through the HBA_SP_QUEUE_EVT flag.
lpfc_drain_txq is also updated to queue an iocbq back on to the head of
phba->txq when lpfc_drain_txq itself still observes ELS WQ full or SGL
unavailable events.
Signed-off-by: Nigel Kirkland <nkirkland2304@gmail.com>
---
drivers/scsi/lpfc/lpfc_bsg.c | 2 +-
drivers/scsi/lpfc/lpfc_crtn.h | 7 +-
drivers/scsi/lpfc/lpfc_ct.c | 21 ++++-
drivers/scsi/lpfc/lpfc_els.c | 110 +++++++++++++++++---------
drivers/scsi/lpfc/lpfc_init.c | 5 +-
drivers/scsi/lpfc/lpfc_nportdisc.c | 9 ++-
drivers/scsi/lpfc/lpfc_sli.c | 121 ++++++++++++++++++++++++-----
drivers/scsi/lpfc/lpfc_sli.h | 18 +++++
8 files changed, 227 insertions(+), 66 deletions(-)
diff --git a/drivers/scsi/lpfc/lpfc_bsg.c b/drivers/scsi/lpfc/lpfc_bsg.c
index 63b6839230df..7919d3bf2cce 100644
--- a/drivers/scsi/lpfc/lpfc_bsg.c
+++ b/drivers/scsi/lpfc/lpfc_bsg.c
@@ -2968,7 +2968,7 @@ static int lpfcdiag_sli3_loop_post_rxbufs(struct lpfc_hba *phba, uint16_t rxxri,
iocb_stat = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, cmdiocbq,
0);
- if (iocb_stat == IOCB_ERROR) {
+ if (lpfc_iocb_failed(iocb_stat)) {
diag_cmd_data_free(phba,
(struct lpfc_dmabufext *)mp[0]);
if (mp[1])
diff --git a/drivers/scsi/lpfc/lpfc_crtn.h b/drivers/scsi/lpfc/lpfc_crtn.h
index 2ca5f0229ca1..47a593fe5e69 100644
--- a/drivers/scsi/lpfc/lpfc_crtn.h
+++ b/drivers/scsi/lpfc/lpfc_crtn.h
@@ -536,8 +536,11 @@ int lpfc_bsg_timeout(struct bsg_job *);
int lpfc_bsg_ct_unsol_event(struct lpfc_hba *, struct lpfc_sli_ring *,
struct lpfc_iocbq *);
int lpfc_bsg_ct_unsol_abort(struct lpfc_hba *, struct hbq_dmabuf *);
-void __lpfc_sli_ringtx_put(struct lpfc_hba *, struct lpfc_sli_ring *,
- struct lpfc_iocbq *);
+void lpfc_sli4_queue_io_for_retry(struct lpfc_hba *phba,
+ struct lpfc_iocbq *iocb,
+ bool head);
+void __lpfc_sli_ringtx_put(struct lpfc_hba *phba, struct lpfc_sli_ring *ring,
+ struct lpfc_iocbq *iocb, bool head);
struct lpfc_iocbq *lpfc_sli_ringtx_get(struct lpfc_hba *,
struct lpfc_sli_ring *);
int __lpfc_sli_issue_iocb(struct lpfc_hba *, uint32_t,
diff --git a/drivers/scsi/lpfc/lpfc_ct.c b/drivers/scsi/lpfc/lpfc_ct.c
index f709e5087577..372fa5718f3a 100644
--- a/drivers/scsi/lpfc/lpfc_ct.c
+++ b/drivers/scsi/lpfc/lpfc_ct.c
@@ -246,7 +246,7 @@ lpfc_ct_reject_event(struct lpfc_nodelist *ndlp,
goto ct_no_ndlp;
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, cmdiocbq, 0);
- if (rc) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_nlp_put(ndlp);
goto ct_no_ndlp;
}
@@ -639,11 +639,24 @@ lpfc_gen_req(struct lpfc_vport *vport, struct lpfc_dmabuf *bmp,
goto out;
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, geniocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
+ lpfc_vlog_msg(vport, KERN_NOTICE,
+ LOG_ELS | LOG_DISCOVERY | LOG_NODE,
+ "0156 %ps WQE Put returned %d\n",
+ cmpl, rc);
+
+ /* Under heavy vpi counts, the driver's host_index can catch up
+ * to the hba_index causing a put error. Catch this case and
+ * put the IO on phba->txq.
+ */
+ if (rc == IOCB_FAILED_PUT && phba->sli_rev == LPFC_SLI_REV4) {
+ lpfc_sli4_queue_io_for_retry(phba, geniocb, false);
+ return 0;
+ }
+
lpfc_nlp_put(ndlp);
goto out;
}
-
return 0;
out:
lpfc_sli_release_iocbq(phba, geniocb);
@@ -2258,7 +2271,7 @@ lpfc_cmpl_ct_disc_fdmi(struct lpfc_hba *phba, struct lpfc_iocbq *cmdiocb,
/* Retry the same FDMI command */
err = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING,
cmdiocb, 0);
- if (err == IOCB_ERROR)
+ if (lpfc_iocb_failed(err))
break;
return;
default:
diff --git a/drivers/scsi/lpfc/lpfc_els.c b/drivers/scsi/lpfc/lpfc_els.c
index 7309fb948bea..bf71b5a3e55e 100644
--- a/drivers/scsi/lpfc/lpfc_els.c
+++ b/drivers/scsi/lpfc/lpfc_els.c
@@ -1431,7 +1431,7 @@ lpfc_issue_els_flogi(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
set_bit(HBA_FLOGI_OUTSTANDING, &phba->hba_flag);
rc = lpfc_issue_fabric_iocb(phba, elsiocb);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
clear_bit(HBA_FLOGI_ISSUED, &phba->hba_flag);
clear_bit(HBA_FLOGI_OUTSTANDING, &phba->hba_flag);
lpfc_nlp_put(ndlp);
@@ -2439,7 +2439,7 @@ lpfc_issue_els_plogi(struct lpfc_vport *vport, uint32_t did, uint8_t retry)
}
ret = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (ret) {
+ if (lpfc_iocb_failed(ret)) {
lpfc_vlog_msg(vport, KERN_NOTICE,
LOG_ELS | LOG_DISCOVERY | LOG_NODE,
"0157 PLOGI WQE Put returned %d\n",
@@ -2450,8 +2450,15 @@ lpfc_issue_els_plogi(struct lpfc_vport *vport, uint32_t did, uint8_t retry)
* put the IO on phba->txq.
*/
if (ret == IOCB_FAILED_PUT && phba->sli_rev == LPFC_SLI_REV4) {
+ /* The iocb will always successfully get queued
+ * for a retry
+ */
lpfc_sli4_queue_io_for_retry(phba, elsiocb, false);
- return 0;
+
+ /* Goto the out label to mark the nlp_flag correctly.
+ * The IO will eventually go out.
+ */
+ goto out;
}
lpfc_els_free_iocb(phba, elsiocb);
@@ -2459,6 +2466,7 @@ lpfc_issue_els_plogi(struct lpfc_vport *vport, uint32_t did, uint8_t retry)
return 1;
}
+out:
set_bit(NLP_PLOGI_SND, &ndlp->nlp_flag);
return 0;
}
@@ -2780,7 +2788,7 @@ lpfc_issue_els_prli(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_vlog_msg(vport, KERN_NOTICE,
LOG_ELS | LOG_DISCOVERY | LOG_NODE,
"0155 PRLI WQE Put returned %d\n",
@@ -2791,8 +2799,15 @@ lpfc_issue_els_prli(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
* put the IO on phba->txq.
*/
if (rc == IOCB_FAILED_PUT && phba->sli_rev == LPFC_SLI_REV4) {
+ /* The iocb will always successfully get queued for a
+ * retry
+ */
lpfc_sli4_queue_io_for_retry(phba, elsiocb, false);
- return 0;
+
+ /* Jump to out_next because the prli counters and an NVME
+ * PRLI need to be exercised.
+ */
+ goto out_next;
}
lpfc_els_free_iocb(phba, elsiocb);
@@ -2800,6 +2815,7 @@ lpfc_issue_els_prli(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
return 1;
}
+out_next:
/* The vport counters are used for lpfc_scan_finished, but
* the ndlp is used to track outstanding PRLIs for different
* FC4 types.
@@ -3121,7 +3137,7 @@ lpfc_issue_els_adisc(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
goto err;
@@ -3331,7 +3347,7 @@ lpfc_issue_els_logo(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
goto err;
@@ -3701,7 +3717,7 @@ lpfc_issue_els_scr(struct lpfc_vport *vport, uint8_t retry)
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR)
+ if (lpfc_iocb_failed(rc))
goto out_iocb_error;
return 0;
@@ -3804,7 +3820,7 @@ lpfc_issue_els_rscn(struct lpfc_vport *vport, uint8_t retry)
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -3899,7 +3915,7 @@ lpfc_issue_els_farpr(struct lpfc_vport *vport, uint32_t nportid, uint8_t retry)
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
/* The additional lpfc_nlp_put will cause the following
* lpfc_els_free_iocb routine to trigger the release of
* the node.
@@ -4000,7 +4016,7 @@ lpfc_issue_els_rdf(struct lpfc_vport *vport, uint8_t retry)
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
err = -EIO;
goto out_iocb_error;
}
@@ -4529,7 +4545,7 @@ lpfc_issue_els_edc(struct lpfc_vport *vport, uint8_t retry)
"Issue EDC: did:x%x refcnt %d",
ndlp->nlp_DID, kref_read(&ndlp->kref), 0);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
/* The additional lpfc_nlp_put will cause the following
* lpfc_els_free_iocb routine to trigger the release of
* the node.
@@ -6010,7 +6026,21 @@ lpfc_els_rsp_acc(struct lpfc_vport *vport, uint32_t flag,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
+ lpfc_vlog_msg(vport, KERN_NOTICE,
+ LOG_ELS | LOG_DISCOVERY | LOG_NODE,
+ "0154 LS_ACC WQE Put returned %d\n",
+ rc);
+
+ /* Under heavy vpi counts, the driver's host_index can catch up
+ * to the hba_index causing a put error. Catch this case and
+ * put the IO on phba->txq.
+ */
+ if (rc == IOCB_FAILED_PUT && phba->sli_rev == LPFC_SLI_REV4) {
+ lpfc_sli4_queue_io_for_retry(phba, elsiocb, false);
+ return 0;
+ }
+
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6112,7 +6142,7 @@ lpfc_els_rsp_reject(struct lpfc_vport *vport, uint32_t rejectError,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6204,7 +6234,7 @@ lpfc_issue_els_edc_rsp(struct lpfc_vport *vport, struct lpfc_iocbq *cmdiocb,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6310,7 +6340,7 @@ lpfc_els_rsp_adisc_acc(struct lpfc_vport *vport, struct lpfc_iocbq *oldiocb,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6502,7 +6532,7 @@ lpfc_els_rsp_prli_acc(struct lpfc_vport *vport, struct lpfc_iocbq *oldiocb,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6616,7 +6646,7 @@ lpfc_els_rsp_rnid_acc(struct lpfc_vport *vport, uint8_t format,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -6750,7 +6780,7 @@ lpfc_els_rsp_echo_acc(struct lpfc_vport *vport, uint8_t *data,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -7418,7 +7448,7 @@ lpfc_els_rdp_cmpl(struct lpfc_hba *phba, struct lpfc_rdp_context *rdp_context,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -7461,7 +7491,7 @@ lpfc_els_rdp_cmpl(struct lpfc_hba *phba, struct lpfc_rdp_context *rdp_context,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -7824,7 +7854,7 @@ lpfc_els_lcb_rsp(struct lpfc_hba *phba, LPFC_MBOXQ_t *pmb)
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -7870,7 +7900,7 @@ lpfc_els_lcb_rsp(struct lpfc_hba *phba, LPFC_MBOXQ_t *pmb)
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -8972,7 +9002,7 @@ lpfc_els_rsp_rls_acc(struct lpfc_hba *phba, LPFC_MBOXQ_t *pmb)
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -9136,7 +9166,7 @@ lpfc_els_rcv_rtv(struct lpfc_vport *vport, struct lpfc_iocbq *cmdiocb,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
}
@@ -9211,7 +9241,7 @@ lpfc_issue_els_rrq(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
goto io_err;
ret = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (ret == IOCB_ERROR) {
+ if (lpfc_iocb_failed(ret)) {
lpfc_nlp_put(ndlp);
goto io_err;
}
@@ -9332,7 +9362,7 @@ lpfc_els_rsp_rpl_acc(struct lpfc_vport *vport, uint16_t cmdsize,
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return 1;
@@ -10611,6 +10641,7 @@ lpfc_els_unsol_buffer(struct lpfc_hba *phba, struct lpfc_sli_ring *pring,
uint8_t rjt_exp, rjt_err = 0, init_link = 0;
struct lpfc_wcqe_complete *wcqe_cmpl = NULL;
LPFC_MBOXQ_t *mbox;
+ u16 rcv_oxid;
if (!vport || !elsiocb->cmd_dmabuf)
goto dropit;
@@ -10680,11 +10711,18 @@ lpfc_els_unsol_buffer(struct lpfc_hba *phba, struct lpfc_sli_ring *pring,
if ((cmd & ELS_CMD_MASK) == ELS_CMD_RSCN) {
cmd &= ELS_CMD_MASK;
}
+
+ /* Fetch the incoming ox_id for SLI4 only - SLI3 can't do this. */
+ rcv_oxid = 0xffff;
+ if (phba->sli_rev == LPFC_SLI_REV4)
+ rcv_oxid = get_job_rcvoxid(phba, elsiocb);
+
/* ELS command <elsCmd> received from NPORT <did> */
lpfc_printf_vlog(vport, KERN_INFO, LOG_ELS,
"0112 ELS command x%x received from NPORT x%x "
- "refcnt %d Data: x%x x%lx x%x x%x\n",
- cmd, did, kref_read(&ndlp->kref), vport->port_state,
+ "refcnt %d rcvoxid x%x Data: x%x x%lx x%x x%x\n",
+ cmd, did, kref_read(&ndlp->kref),
+ rcv_oxid, vport->port_state,
vport->fc_flag, vport->fc_myDID, vport->fc_prevDID);
/* reject till our FLOGI completes or PLOGI assigned DID via PT2PT */
@@ -11773,7 +11811,7 @@ lpfc_issue_els_fdisc(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp,
goto err_out;
rc = lpfc_issue_fabric_iocb(phba, elsiocb);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_nlp_put(ndlp);
goto err_out;
}
@@ -11911,7 +11949,7 @@ lpfc_issue_els_npiv_logo(struct lpfc_vport *vport, struct lpfc_nodelist *ndlp)
}
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
goto err;
@@ -11992,8 +12030,7 @@ lpfc_resume_fabric_iocbs(struct lpfc_hba *phba)
iocb->vport->port_state, 0, 0);
ret = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, iocb, 0);
-
- if (ret == IOCB_ERROR) {
+ if (lpfc_iocb_failed(ret)) {
iocb->cmd_cmpl = iocb->fabric_cmd_cmpl;
iocb->fabric_cmd_cmpl = NULL;
iocb->cmd_flag &= ~LPFC_IO_FABRIC;
@@ -12157,8 +12194,7 @@ lpfc_issue_fabric_iocb(struct lpfc_hba *phba, struct lpfc_iocbq *iocb)
iocb->vport->port_state, 0, 0);
ret = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, iocb, 0);
-
- if (ret == IOCB_ERROR) {
+ if (lpfc_iocb_failed(ret)) {
iocb->cmd_cmpl = iocb->fabric_cmd_cmpl;
iocb->fabric_cmd_cmpl = NULL;
iocb->cmd_flag &= ~LPFC_IO_FABRIC;
@@ -12580,7 +12616,7 @@ int lpfc_issue_els_qfpa(struct lpfc_vport *vport)
}
ret = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, elsiocb, 2);
- if (ret != IOCB_SUCCESS) {
+ if (lpfc_iocb_failed(ret)) {
lpfc_els_free_iocb(phba, elsiocb);
lpfc_nlp_put(ndlp);
return -EIO;
@@ -12664,7 +12700,7 @@ lpfc_vmid_uvem(struct lpfc_vport *vport,
}
ret = lpfc_sli_issue_iocb(vport->phba, LPFC_ELS_RING, elsiocb, 0);
- if (ret != IOCB_SUCCESS) {
+ if (lpfc_iocb_failed(ret)) {
lpfc_els_free_iocb(vport->phba, elsiocb);
lpfc_nlp_put(ndlp);
goto out;
diff --git a/drivers/scsi/lpfc/lpfc_init.c b/drivers/scsi/lpfc/lpfc_init.c
index 7352cb6e584b..a423bde283c9 100644
--- a/drivers/scsi/lpfc/lpfc_init.c
+++ b/drivers/scsi/lpfc/lpfc_init.c
@@ -2813,6 +2813,7 @@ lpfc_sli3_post_buffer(struct lpfc_hba *phba, struct lpfc_sli_ring *pring, int cn
IOCB_t *icmd;
struct lpfc_iocbq *iocb;
struct lpfc_dmabuf *mp1, *mp2;
+ int rc;
cnt += pring->missbufcnt;
@@ -2875,8 +2876,8 @@ lpfc_sli3_post_buffer(struct lpfc_hba *phba, struct lpfc_sli_ring *pring, int cn
icmd->ulpCommand = CMD_QUE_RING_BUF64_CN;
icmd->ulpLe = 1;
- if (lpfc_sli_issue_iocb(phba, pring->ringno, iocb, 0) ==
- IOCB_ERROR) {
+ rc = lpfc_sli_issue_iocb(phba, pring->ringno, iocb, 0);
+ if (lpfc_iocb_failed(rc)) {
lpfc_mbuf_free(phba, mp1->virt, mp1->phys);
kfree(mp1);
cnt++;
diff --git a/drivers/scsi/lpfc/lpfc_nportdisc.c b/drivers/scsi/lpfc/lpfc_nportdisc.c
index 27a891f7fcf4..49653ff35c56 100644
--- a/drivers/scsi/lpfc/lpfc_nportdisc.c
+++ b/drivers/scsi/lpfc/lpfc_nportdisc.c
@@ -1998,6 +1998,7 @@ lpfc_cmpl_reglogin_reglogin_issue(struct lpfc_vport *vport,
LPFC_MBOXQ_t *pmb = (LPFC_MBOXQ_t *) arg;
MAILBOX_t *mb = &pmb->u.mb;
uint32_t did = mb->un.varWords[1];
+ int rc;
if (mb->mbxStatus) {
/* RegLogin failed */
@@ -2076,7 +2077,13 @@ lpfc_cmpl_reglogin_reglogin_issue(struct lpfc_vport *vport,
ndlp->nlp_prev_state = NLP_STE_REG_LOGIN_ISSUE;
lpfc_nlp_set_state(vport, ndlp, NLP_STE_PRLI_ISSUE);
- if (lpfc_issue_els_prli(vport, ndlp, 0)) {
+ rc = lpfc_issue_els_prli(vport, ndlp, 0);
+ if (rc) {
+ lpfc_vlog_msg(vport, KERN_NOTICE,
+ LOG_ELS | LOG_DISCOVERY | LOG_NODE,
+ "3015 PRLI Issue returning %d to DID "
+ "x%06x, Send LOGO\n",
+ rc, ndlp->nlp_DID);
lpfc_issue_els_logo(vport, ndlp, 0);
ndlp->nlp_prev_state = NLP_STE_REG_LOGIN_ISSUE;
lpfc_nlp_set_state(vport, ndlp, NLP_STE_NPR_NODE);
diff --git a/drivers/scsi/lpfc/lpfc_sli.c b/drivers/scsi/lpfc/lpfc_sli.c
index 06dc01c9539e..bc46b14eca41 100644
--- a/drivers/scsi/lpfc/lpfc_sli.c
+++ b/drivers/scsi/lpfc/lpfc_sli.c
@@ -244,6 +244,38 @@ lpfc_sli4_pcimem_bcopy(void *srcp, void *destp, uint32_t cnt)
#define lpfc_sli4_pcimem_bcopy(a, b, c) lpfc_sli_pcimem_bcopy(a, b, c)
#endif
+/**
+ * lpfc_sli4_queue_io_for_retry - Put an ELS/CT IO request on the txq.
+ * @phba - pointer to the hba instance.
+ * @iocb - the IO request to put on the txq.
+ * @head - A boolean to request a put to the head (true) or tail
+ * (false) of the txq.
+ *
+ * This routine is for SLI4 interfaces only.
+ * When ELS/CT commands can't be sent because of ELS WQ full put status,
+ * queue the iocbq on the phba->txq, at the head or tail as requested by
+ * the caller, for a retry later in lpfc_drain_txq. The lpfc_drain_txq
+ * routine is called per ELS CQE completion.
+ */
+void lpfc_sli4_queue_io_for_retry(struct lpfc_hba *phba,
+ struct lpfc_iocbq *iocb,
+ bool head)
+{
+ unsigned long iflags;
+ struct lpfc_sli_ring *pring;
+
+ /* Mark the IO as in RETRY to stop a full reprep of the SGL/XRI. */
+ iocb->cmd_flag |= LPFC_IO_IN_RETRY;
+ pring = phba->sli4_hba.els_wq->pring;
+
+ /* Insert to the head or tail of the txq ring as indicated
+ * by the caller's head argument.
+ */
+ spin_lock_irqsave(&pring->ring_lock, iflags);
+ __lpfc_sli_ringtx_put(phba, pring, iocb, head);
+ spin_unlock_irqrestore(&pring->ring_lock, iflags);
+}
+
/**
* lpfc_sli4_wq_put - Put a Work Queue Entry on an Work Queue
* @q: The Work Queue to operate on.
@@ -276,6 +308,10 @@ lpfc_sli4_wq_put(struct lpfc_queue *q, union lpfc_wqe128 *wqe)
/* If the host has not yet processed the next entry then we are done */
idx = ((q->host_index + 1) % q->entry_count);
if (idx == q->hba_index) {
+ lpfc_printf_log(q->phba, KERN_WARNING, LOG_SLI,
+ "9998 No available WQ Slots on "
+ "q_id x%x, host x%x, hba x%x\n",
+ q->queue_id, idx, q->hba_index);
q->WQ_overflow++;
return -EBUSY;
}
@@ -10445,6 +10481,7 @@ lpfc_mbox_api_table_setup(struct lpfc_hba *phba, uint8_t dev_grp)
* @phba: Pointer to HBA context object.
* @pring: Pointer to driver SLI ring object.
* @piocb: Pointer to address of newly added command iocb.
+ * @head: put at head (true) or tail (false)
*
* This function is called with hbalock held for SLI3 ports or
* the ring lock held for SLI4 ports to add a command
@@ -10453,14 +10490,20 @@ lpfc_mbox_api_table_setup(struct lpfc_hba *phba, uint8_t dev_grp)
**/
void
__lpfc_sli_ringtx_put(struct lpfc_hba *phba, struct lpfc_sli_ring *pring,
- struct lpfc_iocbq *piocb)
+ struct lpfc_iocbq *piocb, bool head)
{
if (phba->sli_rev == LPFC_SLI_REV4)
lockdep_assert_held(&pring->ring_lock);
else
lockdep_assert_held(&phba->hbalock);
- /* Insert the caller's iocb in the txq tail for later processing. */
- list_add_tail(&piocb->list, &pring->txq);
+
+ /* Insert the caller's iocb in the txq head or tail for later
+ * processing.
+ */
+ if (head)
+ list_add(&piocb->list, &pring->txq);
+ else
+ list_add_tail(&piocb->list, &pring->txq);
}
/**
@@ -10468,6 +10511,7 @@ __lpfc_sli_ringtx_put(struct lpfc_hba *phba, struct lpfc_sli_ring *pring,
* @phba: Pointer to HBA context object.
* @pring: Pointer to driver SLI ring object.
* @piocb: Pointer to address of newly added command iocb.
+ * @head: put at head (true) or tail (false)
*
* This function is called with hbalock held before a new
* iocb is submitted to the firmware. This function checks
@@ -10613,7 +10657,7 @@ __lpfc_sli_issue_iocb_s3(struct lpfc_hba *phba, uint32_t ring_number,
out_busy:
if (!(flag & SLI_IOCB_RET_IOCB)) {
- __lpfc_sli_ringtx_put(phba, pring, piocb);
+ __lpfc_sli_ringtx_put(phba, pring, piocb, false);
return IOCB_SUCCESS;
}
@@ -10751,6 +10795,7 @@ __lpfc_sli_issue_iocb_s4(struct lpfc_hba *phba, uint32_t ring_number,
struct lpfc_queue *wq;
struct lpfc_sli_ring *pring;
u32 ulp_command = get_job_cmnd(phba, piocb);
+ int rc = IOCB_SUCCESS;
/* Get the WQ */
if ((piocb->cmd_flag & LPFC_IO_FCP) ||
@@ -10766,9 +10811,11 @@ __lpfc_sli_issue_iocb_s4(struct lpfc_hba *phba, uint32_t ring_number,
/*
* The WQE can be either 64 or 128 bytes,
*/
-
lockdep_assert_held(&pring->ring_lock);
wqe = &piocb->wqe;
+ if (piocb->cmd_flag & LPFC_IO_IN_RETRY)
+ goto retry_io;
+
if (piocb->sli4_xritag == NO_XRI) {
if (ulp_command == CMD_ABORT_XRI_CX)
sglq = NULL;
@@ -10778,7 +10825,7 @@ __lpfc_sli_issue_iocb_s4(struct lpfc_hba *phba, uint32_t ring_number,
if (!(flag & SLI_IOCB_RET_IOCB)) {
__lpfc_sli_ringtx_put(phba,
pring,
- piocb);
+ piocb, false);
return IOCB_SUCCESS;
} else {
return IOCB_BUSY;
@@ -10819,12 +10866,19 @@ __lpfc_sli_issue_iocb_s4(struct lpfc_hba *phba, uint32_t ring_number,
return IOCB_ERROR;
}
- if (lpfc_sli4_wq_put(wq, wqe))
- return IOCB_ERROR;
-
- lpfc_sli_ringtxcmpl_put(phba, pring, piocb);
+ retry_io:
+ piocb->cmd_flag &= ~LPFC_IO_IN_RETRY;
- return 0;
+ /* Push the wqe to the wq and if successful, push to the txcmplq.
+ * A return of -EBUSY the WQ is currently full and a retry possible.
+ * Otherwise the put failed for another, nonretryable reason.
+ */
+ rc = lpfc_sli4_wq_put(wq, wqe);
+ if (!rc)
+ lpfc_sli_ringtxcmpl_put(phba, pring, piocb);
+ else if (rc == -EBUSY)
+ return IOCB_FAILED_PUT;
+ return rc;
}
/*
@@ -12608,7 +12662,7 @@ lpfc_sli_issue_abort_iotag(struct lpfc_hba *phba, struct lpfc_sli_ring *pring,
ulp_context, (phba->sli_rev == LPFC_SLI_REV4) ?
cmdiocb->iotag : iotag, iotag, cmdiocb, abtsiocbp,
retval, ia, abtsiocbp->cmd_cmpl);
- if (retval) {
+ if (lpfc_iocb_failed(retval)) {
cmdiocb->cmd_flag &= ~LPFC_DRIVER_ABORTED;
__lpfc_sli_release_iocbq(phba, abtsiocbp);
}
@@ -13060,7 +13114,7 @@ lpfc_sli_abort_taskmgmt(struct lpfc_vport *vport, struct lpfc_sli_ring *pring,
spin_unlock(&lpfc_cmd->buf_lock);
- if (ret_val == IOCB_ERROR)
+ if (lpfc_iocb_failed(ret_val))
__lpfc_sli_release_iocbq(phba, abtsiocbq);
else
sum++;
@@ -19200,7 +19254,7 @@ lpfc_sli4_seq_abort_rsp(struct lpfc_vport *vport,
ctiocb->abort_rctl, oxid, phba->link_state);
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, ctiocb, 0);
- if (rc == IOCB_ERROR) {
+ if (lpfc_iocb_failed(rc)) {
lpfc_printf_vlog(vport, KERN_ERR, LOG_TRACE_EVENT,
"2925 Failed to issue CT ABTS RSP x%x on "
"xri x%x, Data x%x\n",
@@ -19581,7 +19635,7 @@ lpfc_sli4_handle_mds_loopback(struct lpfc_vport *vport,
iocbq->cmd_cmpl = lpfc_sli4_mds_loopback_cmpl;
rc = lpfc_sli_issue_iocb(phba, LPFC_ELS_RING, iocbq, 0);
- if (rc == IOCB_ERROR)
+ if (lpfc_iocb_failed(rc))
goto exit;
lpfc_in_buf_free(phba, &dmabuf->dbuf);
@@ -21301,12 +21355,15 @@ lpfc_drain_txq(struct lpfc_hba *phba)
}
txq_cnt--;
+ /* Capture the return values that indicate an error during IO
+ * submit and cannot be retried. Prefix message 2822.
+ */
ret = __lpfc_sli_issue_iocb(phba, pring->ringno, piocbq, 0);
-
- if (ret && ret != IOCB_BUSY) {
+ if (ret && ret != IOCB_BUSY && ret != IOCB_FAILED_PUT) {
fail_msg = " - Cannot send IO ";
piocbq->cmd_flag &= ~LPFC_DRIVER_ABORTED;
}
+
if (fail_msg) {
piocbq->cmd_flag |= LPFC_DRIVER_ABORTED;
/* Failed means we can't issue and need to cancel */
@@ -21319,9 +21376,35 @@ lpfc_drain_txq(struct lpfc_hba *phba)
list_add_tail(&piocbq->list, &completions);
fail_msg = NULL;
}
- spin_unlock_irqrestore(&pring->ring_lock, iflags);
- if (txq_cnt == 0 || ret == IOCB_BUSY)
+
+ if (txq_cnt == 0 || ret == IOCB_BUSY ||
+ ret == IOCB_FAILED_PUT) {
+ /* IOCB_FAILED_PUT is unique to SLI4 and means SGL/XRI
+ * resources are allocated. For this case, set the
+ * in retry flag. For SLI3 and 4, push the IO to the
+ * txq for retry.
+ */
+ lpfc_printf_log(phba, KERN_INFO, LOG_SLI,
+ "2820 IOCB iotag x%x xri x%x ret %d "
+ "cmd_flg x%x txq_cnt x%x\n",
+ piocbq->iotag, piocbq->sli4_xritag, ret,
+ piocbq->cmd_flag, txq_cnt);
+ switch (ret) {
+ case IOCB_FAILED_PUT:
+ piocbq->cmd_flag |= LPFC_IO_IN_RETRY;
+ __lpfc_sli_ringtx_put(phba, pring, piocbq,
+ true);
+ break;
+ case IOCB_BUSY:
+ __lpfc_sli_ringtx_put(phba, pring, piocbq,
+ true);
+ break;
+ }
+ spin_unlock_irqrestore(&pring->ring_lock, iflags);
break;
+ }
+
+ spin_unlock_irqrestore(&pring->ring_lock, iflags);
}
/* Cancel all the IOCBs that cannot be issued */
lpfc_sli_cancel_iocbs(phba, &completions, IOSTAT_LOCAL_REJECT,
diff --git a/drivers/scsi/lpfc/lpfc_sli.h b/drivers/scsi/lpfc/lpfc_sli.h
index cf7c42ec0306..37559efd0032 100644
--- a/drivers/scsi/lpfc/lpfc_sli.h
+++ b/drivers/scsi/lpfc/lpfc_sli.h
@@ -123,6 +123,7 @@ struct lpfc_iocbq {
#define LPFC_IO_NVMET 0x800000 /* NVMET command */
#define LPFC_IO_VMID 0x1000000 /* VMID tagged IO */
#define LPFC_IO_CMF 0x4000000 /* CMF command */
+#define LPFC_IO_IN_RETRY 0x8000000 /* Caller is retrying IO. */
uint32_t drvrTimeout; /* driver timeout in seconds */
struct lpfc_vport *vport;/* virtual port pointer */
@@ -160,6 +161,23 @@ struct lpfc_iocbq {
#define IOCB_ABORTED 4
#define IOCB_ABORTING 5
#define IOCB_NORESOURCE 6
+#define IOCB_FAILED_PUT 7
+
+/**
+ * lpfc_iocb_failed - test an IOCB issue return status for failure
+ * @rc: return status from an IOCB issue routine (e.g. lpfc_sli_issue_iocb).
+ *
+ * IOCB_SUCCESS is the only non-failure return value. Callers must test
+ * rc != IOCB_SUCCESS rather than rc == IOCB_ERROR, since that misses
+ * IOCB_FAILED_PUT (SLI-4 WQ full) and treats it as success.
+ *
+ * Return: true if the IOCB issue did not succeed.
+ */
+static inline bool
+lpfc_iocb_failed(int rc)
+{
+ return rc != IOCB_SUCCESS;
+}
#define SLI_WQE_RET_WQE 1 /* Return WQE if cmd ring full */
--
2.38.0
next prev parent reply other threads:[~2026-09-17 21:57 UTC|newest]
Thread overview: 23+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-17 22:20 [PATCH v4 00/14] Update lpfc to revision 15.0.0.1 Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 01/14] lpfc: Fix use-after-free in lpfc_cmpl_ct_cmd_vmid Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 02/14] lpfc: Early return out of lpfc_els_abort when HBA_SETUP flag is not set Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 03/14] lpfc: Fix kernel oops when unmapping scsi dma buffers for an aborted cmd Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 04/14] lpfc: Check fc4_xpt_flags before decrementing ndlp kref on FDISC error Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 05/14] lpfc: Add handling for when PLOGI or PRLI is dropped during link failure Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 06/14] lpfc: Fix ndlp use-after-free during repeated RSCN and rediscovery sequence Nigel Kirkland
2026-09-17 22:10 ` sashiko-bot
2026-09-17 22:20 ` [PATCH v4 07/14] lpfc: Rework I/O flush ordering when unloading driver Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 08/14] lpfc: Improve PLOGI retry handling for large SAN configurations Nigel Kirkland
2026-09-17 22:12 ` sashiko-bot
2026-09-17 22:20 ` [PATCH v4 09/14] lpfc: Send inhibited ABORT_WQE when PLOGI CQE SEQUENCE_TMO is received Nigel Kirkland
2026-09-17 22:15 ` sashiko-bot
2026-09-17 22:20 ` [PATCH v4 10/14] lpfc: Remove slowpath cqe process limiter in slow ring event handler Nigel Kirkland
2026-09-17 22:20 ` sashiko-bot
2026-09-17 22:20 ` Nigel Kirkland [this message]
2026-09-17 22:20 ` [PATCH v4 11/14] lpfc: Put iocbq on phba->txq when ELS WQ is full or ELS SGL unavailable sashiko-bot
2026-09-17 22:20 ` [PATCH v4 12/14] lpfc: Update ELS ACC logging for diagnostic troubleshooting Nigel Kirkland
2026-09-17 22:22 ` sashiko-bot
2026-09-17 22:20 ` [PATCH v4 13/14] lpfc: Refactor calls on fc_disctmo to lpfc_set_disctmo in RSCN handler Nigel Kirkland
2026-09-17 22:20 ` [PATCH v4 14/14] lpfc: Update lpfc version to 15.0.0.1 Nigel Kirkland
2026-09-19 7:45 ` [PATCH v4 00/14] Update lpfc to revision 15.0.0.1 Nigel Kirkland
2026-09-28 17:13 ` Nigel Kirkland
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260917222015.61053-12-nkirkland2304@gmail.com \
--to=nkirkland2304@gmail.com \
--cc=linux-scsi@vger.kernel.org \
--cc=nigel.kirkland@broadcom.com \
--cc=paul.ely@broadcom.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox