NVIDIA GPU driver infrastructure
 help / color / mirror / Atom feed
From: John Hubbard <jhubbard@nvidia.com>
To: Danilo Krummrich <dakr@kernel.org>,
	Alexandre Courbot <acourbot@nvidia.com>
Cc: "Timur Tabi" <ttabi@nvidia.com>,
	"Alistair Popple" <apopple@nvidia.com>,
	"Eliot Courtney" <ecourtney@nvidia.com>,
	"Zhi Wang" <zhiw@nvidia.com>, "David Airlie" <airlied@gmail.com>,
	"Simona Vetter" <simona@ffwll.ch>,
	"Bjorn Helgaas" <bhelgaas@google.com>,
	"Miguel Ojeda" <ojeda@kernel.org>,
	"Alex Gaynor" <alex.gaynor@gmail.com>,
	"Boqun Feng" <boqun.feng@gmail.com>,
	"Gary Guo" <gary@garyguo.net>,
	"Björn Roy Baron" <bjorn3_gh@protonmail.com>,
	"Benno Lossin" <lossin@kernel.org>,
	"Andreas Hindborg" <a.hindborg@kernel.org>,
	"Alice Ryhl" <aliceryhl@google.com>,
	"Trevor Gross" <tmgross@umich.edu>,
	nova-gpu@lists.linux.dev, LKML <linux-kernel@vger.kernel.org>,
	"John Hubbard" <jhubbard@nvidia.com>
Subject: [PATCH v3 09/14] gpu: nova-core: recover the GSP receive path from corrupt framing
Date: Wed,  2 Sep 2026 20:15:08 -0700	[thread overview]
Message-ID: <20260903031514.1515905-10-jhubbard@nvidia.com> (raw)
In-Reply-To: <20260903031514.1515905-1-jhubbard@nvidia.com>

A GSP message carries its length inside the checksummed region, so once
the framing or the checksum fails, the length cannot be trusted to skip
the message.

Two paths left a bad message at the queue head. A framing or checksum
failure returned without advancing the read pointer, so every later
receive re-parsed the same message. A validly framed message whose typed
payload failed to decode returned early and did the same.

Poison the queue on a framing or checksum failure, log what was
inconsistent, and fail every later receive, so the bad head is parsed
once and recovery requires a reset. Advance the read pointer past a
validly framed message whether or not its payload decodes.

Assisted-by: Cursor:claude-opus-5
Signed-off-by: John Hubbard <jhubbard@nvidia.com>
---
 drivers/gpu/nova-core/gsp/cmdq.rs | 96 +++++++++++++++++++++----------
 1 file changed, 66 insertions(+), 30 deletions(-)

diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs
index 5572224db233..ce4d6a111e68 100644
--- a/drivers/gpu/nova-core/gsp/cmdq.rs
+++ b/drivers/gpu/nova-core/gsp/cmdq.rs
@@ -2,7 +2,10 @@
 
 mod continuation;
 
-use core::mem;
+use core::{
+    cell::Cell,
+    mem, //
+};
 
 use kernel::{
     device,
@@ -11,6 +14,7 @@
         CoherentBox,
         DmaAddress, //
     },
+    fmt,
     io::{
         io_project,
         poll::read_poll_timeout,
@@ -531,6 +535,7 @@ pub(crate) fn new(dev: &device::Device<device::Bound>) -> impl PinInit<Self, Err
                     dev: dev.into(),
                     gsp_mem,
                     seq: 0,
+                    poisoned: Cell::new(false),
                 }),
             }))
         })
@@ -624,6 +629,12 @@ struct CmdqInner {
     dev: ARef<device::Device>,
     /// Current command sequence number.
     seq: u32,
+    /// Set once a message with corrupt framing or a bad checksum is seen. Such a message has an
+    /// untrusted length, so the queue cannot be advanced past it, and every later receive fails
+    /// until the queue is torn down and reset.
+    ///
+    /// A [`Cell`], so the shared-borrow read path [`Self::wait_for_msg`] can set it.
+    poisoned: Cell<bool>,
     /// Memory area shared with the GSP for communicating commands and messages.
     gsp_mem: DmaGspMem,
 }
@@ -732,6 +743,19 @@ fn send_command<M>(&mut self, bar: Bar0<'_>, command: M) -> Result
         }
     }
 
+    /// Marks the queue unusable and returns the error every later receive fails with.
+    ///
+    /// `reason` names the inconsistency. Without it the failure is invisible, because the queue
+    /// just stops producing messages.
+    ///
+    /// Takes `&self` so the shared-borrow read path [`Self::wait_for_msg`] can call it.
+    fn poison(&self, reason: fmt::Arguments<'_>) -> Error {
+        dev_err!(&self.dev, "GSP RPC: receive: queue poisoned: {}\n", reason);
+        self.poisoned.set(true);
+
+        EIO
+    }
+
     /// Wait for a message to become available on the message queue.
     ///
     /// This works purely at the transport layer and does not interpret or validate the message
@@ -746,11 +770,13 @@ fn send_command<M>(&mut self, bar: Bar0<'_>, command: M) -> Result
     /// # Errors
     ///
     /// - `ETIMEDOUT` if `timeout` has elapsed before any message becomes available.
-    /// - `EIO` if there was some inconsistency (e.g. message shorter than advertised) on the
-    ///   message queue.
-    ///
-    /// Error codes returned by the message constructor are propagated as-is.
+    /// - `EIO` if the framing or the checksum is invalid, or the queue was already poisoned by an
+    ///   earlier such failure. Either failure poisons the queue, so recovery requires a reset.
     fn wait_for_msg(&self, timeout: Delta) -> Result<GspMessage<'_>> {
+        if self.poisoned.get() {
+            return Err(EIO);
+        }
+
         // Wait for a message to arrive from the GSP.
         let (slice_1, slice_2) = read_poll_timeout(
             || Ok(self.gsp_mem.driver_read_area()),
@@ -761,7 +787,12 @@ fn wait_for_msg(&self, timeout: Delta) -> Result<GspMessage<'_>> {
         .map(|(slice_1, slice_2)| (slice_1.as_flattened(), slice_2.as_flattened()))?;
 
         // Extract the `GspMsgElement`.
-        let (header, slice_1) = GspMsgElement::from_bytes_prefix(slice_1).ok_or(EIO)?;
+        let Some((header, slice_1)) = GspMsgElement::from_bytes_prefix(slice_1) else {
+            return Err(self.poison(fmt!(
+                "read area of {} bytes is shorter than a message header",
+                slice_1.len()
+            )));
+        };
 
         dev_dbg!(
             &self.dev,
@@ -775,7 +806,11 @@ fn wait_for_msg(&self, timeout: Delta) -> Result<GspMessage<'_>> {
 
         // Check that the driver read area is large enough for the message.
         if slice_1.len() + slice_2.len() < payload_length {
-            return Err(EIO);
+            return Err(self.poison(fmt!(
+                "message advertises {} payload bytes but only {} are readable",
+                payload_length,
+                slice_1.len() + slice_2.len()
+            )));
         }
 
         // Cut the message slices down to the actual length of the message.
@@ -798,12 +833,10 @@ fn wait_for_msg(&self, timeout: Delta) -> Result<GspMessage<'_>> {
             slice_2,
         ])) != 0
         {
-            dev_err!(
-                &self.dev,
-                "GSP RPC: receive: Call {} - bad checksum\n",
+            return Err(self.poison(fmt!(
+                "message with sequence {} has a bad checksum",
                 header.sequence()
-            );
-            return Err(EIO);
+            )));
         }
 
         Ok(GspMessage {
@@ -824,8 +857,8 @@ fn wait_for_msg(&self, timeout: Delta) -> Result<GspMessage<'_>> {
     /// # Errors
     ///
     /// - `ETIMEDOUT` if `timeout` has elapsed before any message becomes available.
-    /// - `EIO` if there was some inconsistency (e.g. message shorter than advertised) on the
-    ///   message queue.
+    /// - `EIO` if the queue is poisoned or the message fails framing or checksum validation (see
+    ///   [`Self::wait_for_msg`]), or if the matched message is too short for `M::Message`.
     /// - `ERANGE` if the message was not the awaited reply.
     ///
     /// Error codes returned by [`MessageFromGsp::read`] are propagated as-is.
@@ -838,23 +871,26 @@ fn receive_msg<M: MessageFromGsp>(&mut self, timeout: Delta) -> Result<M>
         let function = message.header.function();
         let seq = message.header.sequence();
 
-        // Bind the result rather than returning early. The read pointer must advance past this
-        // message on every path.
+        // Every path must advance the read pointer past this message, including a failed decode.
         let result = if matches!(function, Ok(f) if f == M::FUNCTION) {
-            let (cmd, contents_1) = M::Message::from_bytes_prefix(message.contents.0).ok_or(EIO)?;
-            let mut sbuffer = SBufferIter::new_reader([contents_1, message.contents.1]);
-
-            M::read(cmd, &mut sbuffer)
-                .map_err(|e| e.into())
-                .inspect(|_| {
-                    if !sbuffer.is_empty() {
-                        dev_warn!(
-                            &self.dev,
-                            "GSP message {:?} has unprocessed data\n",
-                            M::FUNCTION
-                        );
-                    }
-                })
+            match M::Message::from_bytes_prefix(message.contents.0) {
+                Some((cmd, contents_1)) => {
+                    let mut sbuffer = SBufferIter::new_reader([contents_1, message.contents.1]);
+
+                    M::read(cmd, &mut sbuffer)
+                        .map_err(|e| e.into())
+                        .inspect(|_| {
+                            if !sbuffer.is_empty() {
+                                dev_warn!(
+                                    &self.dev,
+                                    "GSP message {:?} has unprocessed data\n",
+                                    M::FUNCTION
+                                );
+                            }
+                        })
+                }
+                None => Err(EIO),
+            }
         } else {
             self.classify_event(function, seq);
 
-- 
2.55.0


  parent reply	other threads:[~2026-09-03  3:15 UTC|newest]

Thread overview: 27+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-03  3:14 [PATCH v3 00/14] nova-core: GPU interrupt support and GSP event delivery John Hubbard
2026-09-03  3:15 ` [PATCH v3 01/14] rust: pci: declare IrqType and IrqTypes with impl_flags John Hubbard
2026-09-03  3:15 ` [PATCH v3 02/14] rust: sync: completion: add wait_for_completion_timeout() John Hubbard
2026-09-03  3:15 ` [PATCH v3 03/14] gpu: nova-core: add the GIN vector and subtree newtypes John Hubbard
2026-09-03  3:15 ` [PATCH v3 04/14] gpu: nova-core: add the GIN CPU interrupt tree and MSI EOI registers John Hubbard
2026-09-03  3:15 ` [PATCH v3 05/14] gpu: nova-core: add the per-architecture GIN CPU interrupt HAL John Hubbard
2026-09-03  3:15 ` [PATCH v3 06/14] gpu: nova-core: add the GIN interrupt tree and allocate its vectors John Hubbard
2026-09-03  3:15 ` [PATCH v3 07/14] gpu: nova-core: add an interrupt delivery self-test John Hubbard
2026-09-03  3:29   ` sashiko-bot
2026-09-03  3:57     ` John Hubbard
2026-09-03  3:15 ` [PATCH v3 08/14] gpu: nova-core: log GSP events instead of discarding them John Hubbard
2026-09-03  3:15 ` John Hubbard [this message]
2026-09-04 10:53   ` [PATCH v3 09/14] gpu: nova-core: recover the GSP receive path from corrupt framing Alexandre Courbot
2026-09-04 11:17     ` Gary Guo
2026-09-04 13:45       ` Alexandre Courbot
2026-09-03  3:15 ` [PATCH v3 10/14] gpu: nova-core: bound a GSP wait by a single deadline John Hubbard
2026-09-04 11:13   ` Alexandre Courbot
2026-09-04 11:26     ` Gary Guo
2026-09-04 13:32       ` Alexandre Courbot
2026-09-04 13:41         ` Gary Guo
2026-09-03  3:15 ` [PATCH v3 11/14] gpu: nova-core: add the falcon interrupt status and routing registers John Hubbard
2026-09-03  3:15 ` [PATCH v3 12/14] gpu: nova-core: drive GSP events with the SWGEN0 interrupt John Hubbard
2026-09-03  3:28   ` sashiko-bot
2026-09-03  3:55     ` John Hubbard
2026-09-04  1:53       ` John Hubbard
2026-09-03  3:15 ` [PATCH v3 13/14] gpu: nova-core: add KUnit tests for the interrupt tree and HALs John Hubbard
2026-09-03  3:15 ` [PATCH v3 14/14] gpu: nova-core: document the GIN interrupt controller and GSP events John Hubbard

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260903031514.1515905-10-jhubbard@nvidia.com \
    --to=jhubbard@nvidia.com \
    --cc=a.hindborg@kernel.org \
    --cc=acourbot@nvidia.com \
    --cc=airlied@gmail.com \
    --cc=alex.gaynor@gmail.com \
    --cc=aliceryhl@google.com \
    --cc=apopple@nvidia.com \
    --cc=bhelgaas@google.com \
    --cc=bjorn3_gh@protonmail.com \
    --cc=boqun.feng@gmail.com \
    --cc=dakr@kernel.org \
    --cc=ecourtney@nvidia.com \
    --cc=gary@garyguo.net \
    --cc=linux-kernel@vger.kernel.org \
    --cc=lossin@kernel.org \
    --cc=nova-gpu@lists.linux.dev \
    --cc=ojeda@kernel.org \
    --cc=simona@ffwll.ch \
    --cc=tmgross@umich.edu \
    --cc=ttabi@nvidia.com \
    --cc=zhiw@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox