* [PATCH v12 1/8] cxl/mbox: Flag support for Dynamic Capacity Devices (DCD)
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 8:48 ` [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device Anisa Su
` (6 subsequent siblings)
7 siblings, 0 replies; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Per the CXL 4.0 specification software must check the Command Effects
Log (CEL) for dynamic capacity command support.
Detect support for the DCD commands while reading the CEL, including:
Get DC Config
Get DC Extent List
Add DC Response
Release DC
Based on an original patch by Navneet Singh.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
---
Changes:
1. mbox.c: leave mds->dcd_supported false so the hardware enablement
patches can be upstreamed ahead of the extent and DAX work; the
event handling patch re-enables it.
---
drivers/cxl/core/mbox.c | 44 +++++++++++++++++++++++++++++++++++++++++
drivers/cxl/cxlmem.h | 20 +++++++++++++++++++
2 files changed, 64 insertions(+)
diff --git a/drivers/cxl/core/mbox.c b/drivers/cxl/core/mbox.c
index 7c6c5b7450a5..4790524c32a7 100644
--- a/drivers/cxl/core/mbox.c
+++ b/drivers/cxl/core/mbox.c
@@ -165,6 +165,38 @@ static void cxl_set_security_cmd_enabled(struct cxl_security_state *security,
}
}
+static bool cxl_is_dcd_command(u16 opcode)
+{
+#define CXL_MBOX_OP_DCD_CMDS 0x48
+
+ return (opcode >> 8) == CXL_MBOX_OP_DCD_CMDS;
+}
+
+static void cxl_set_dcd_cmd_enabled(u16 opcode, unsigned long *cmd_mask)
+{
+ switch (opcode) {
+ case CXL_MBOX_OP_GET_DC_CONFIG:
+ set_bit(CXL_DCD_ENABLED_GET_CONFIG, cmd_mask);
+ break;
+ case CXL_MBOX_OP_GET_DC_EXTENT_LIST:
+ set_bit(CXL_DCD_ENABLED_GET_EXTENT_LIST, cmd_mask);
+ break;
+ case CXL_MBOX_OP_ADD_DC_RESPONSE:
+ set_bit(CXL_DCD_ENABLED_ADD_RESPONSE, cmd_mask);
+ break;
+ case CXL_MBOX_OP_RELEASE_DC:
+ set_bit(CXL_DCD_ENABLED_RELEASE, cmd_mask);
+ break;
+ default:
+ break;
+ }
+}
+
+static bool cxl_verify_dcd_cmds(unsigned long *cmds_seen)
+{
+ return bitmap_full(cmds_seen, CXL_DCD_ENABLED_MAX);
+}
+
static bool cxl_is_poison_command(u16 opcode)
{
#define CXL_MBOX_OP_POISON_CMDS 0x43
@@ -757,6 +789,7 @@ static void cxl_walk_cel(struct cxl_memdev_state *mds, size_t size, u8 *cel)
struct cxl_mailbox *cxl_mbox = &mds->cxlds.cxl_mbox;
struct cxl_cel_entry *cel_entry;
const int cel_entries = size / sizeof(*cel_entry);
+ DECLARE_BITMAP(dcd_cmds, CXL_DCD_ENABLED_MAX) = {};
struct device *dev = mds->cxlds.dev;
int i, ro_cmds = 0, wr_cmds = 0;
@@ -785,11 +818,22 @@ static void cxl_walk_cel(struct cxl_memdev_state *mds, size_t size, u8 *cel)
enabled++;
}
+ if (cxl_is_dcd_command(opcode)) {
+ cxl_set_dcd_cmd_enabled(opcode, dcd_cmds);
+ enabled++;
+ }
+
dev_dbg(dev, "Opcode 0x%04x %s\n", opcode,
enabled ? "enabled" : "unsupported by driver");
}
set_features_cap(cxl_mbox, ro_cmds, wr_cmds);
+ /*
+ * Disabled until event handling implemented.
+ */
+ if (cxl_verify_dcd_cmds(dcd_cmds))
+ dev_dbg(dev, "Device supports DCD; capability disabled\n");
+ mds->dcd_supported = false;
}
static struct cxl_mbox_get_supported_logs *cxl_get_gsl(struct cxl_memdev_state *mds)
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index ed419d0c59f2..616eeaeee5b9 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -252,6 +252,20 @@ struct cxl_event_state {
struct mutex log_lock;
};
+/**
+ * CXL r4.0 Section 8.2.10.9 - Memory Device Command Sets. See Table 8-308.
+ *
+ * The 48h Command Set (Opcodes 4800h - 4803h) defines the device-enabled DCD
+ * commands.
+ */
+enum dcd_cmd_enabled_bits {
+ CXL_DCD_ENABLED_GET_CONFIG,
+ CXL_DCD_ENABLED_GET_EXTENT_LIST,
+ CXL_DCD_ENABLED_ADD_RESPONSE,
+ CXL_DCD_ENABLED_RELEASE,
+ CXL_DCD_ENABLED_MAX
+};
+
/* Device enabled poison commands */
enum poison_cmd_enabled_bits {
CXL_POISON_ENABLED_LIST,
@@ -427,6 +441,7 @@ static inline struct cxl_dev_state *mbox_to_cxlds(struct cxl_mailbox *cxl_mbox)
* @partition_align_bytes: alignment size for partition-able capacity
* @active_volatile_bytes: sum of hard + soft volatile
* @active_persistent_bytes: sum of hard + soft persistent
+ * @dcd_supported: all DCD commands are supported
* @event: event log driver state
* @poison: poison driver state info
* @security: security driver state info
@@ -446,6 +461,7 @@ struct cxl_memdev_state {
u64 partition_align_bytes;
u64 active_volatile_bytes;
u64 active_persistent_bytes;
+ bool dcd_supported;
struct cxl_event_state event;
struct cxl_poison_state poison;
@@ -507,6 +523,10 @@ enum cxl_opcode {
CXL_MBOX_OP_UNLOCK = 0x4503,
CXL_MBOX_OP_FREEZE_SECURITY = 0x4504,
CXL_MBOX_OP_PASSPHRASE_SECURE_ERASE = 0x4505,
+ CXL_MBOX_OP_GET_DC_CONFIG = 0x4800,
+ CXL_MBOX_OP_GET_DC_EXTENT_LIST = 0x4801,
+ CXL_MBOX_OP_ADD_DC_RESPONSE = 0x4802,
+ CXL_MBOX_OP_RELEASE_DC = 0x4803,
CXL_MBOX_OP_MAX = 0x10000
};
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
2026-07-31 8:48 ` [PATCH v12 1/8] cxl/mbox: Flag support for Dynamic Capacity Devices (DCD) Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 9:01 ` sashiko-bot
2026-07-31 8:48 ` [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions Anisa Su
` (5 subsequent siblings)
7 siblings, 1 reply; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Devices which optionally support Dynamic Capacity (DC) are configured
via mailbox commands. CXL r4.0 section 9.13.3 requires the host to issue
the Get DC Configuration command in order to properly configure DCDs.
Without the Get DC Configuration command DCD can't be supported.
Implement the DC mailbox commands as specified in CXL 4.0 section
8.2.10.9.9 (opcodes 48XXh) to read and store the DCD configuration
information. Disable DCD if an invalid configuration is found.
Linux has no support for more than one dynamic capacity partition. Read
and validate all the partitions but configure only the first partition
as 'dynamic ram 1'. Additional partitions can be added in the future if
such a device ever materializes. Additionally it is anticipated that no
skips will be present from the end of the pmem partition. Check for and
disallow this configuration as well.
Based on an original patch by Navneet Singh.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
---
Changes:
1. mbox.c: bound the Get DC Config request by the negotiated mailbox
payload. Requesting all CXL_MAX_DC_PARTITIONS (8) needs a 328-byte
response, which exceeds a spec-minimum 256-byte mailbox and fails
-E2BIG, disabling DCD. Cap partition_count per call to what the
payload holds and read the rest via the existing start_partition
loop.
---
drivers/cxl/core/hdm.c | 2 +
drivers/cxl/core/mbox.c | 220 ++++++++++++++++++++++++++++++++++++++++
drivers/cxl/cxlmem.h | 55 ++++++++++
drivers/cxl/pci.c | 3 +
include/cxl/cxl.h | 3 +-
5 files changed, 282 insertions(+), 1 deletion(-)
diff --git a/drivers/cxl/core/hdm.c b/drivers/cxl/core/hdm.c
index 0c80b76a5f9b..0ef076c08ed2 100644
--- a/drivers/cxl/core/hdm.c
+++ b/drivers/cxl/core/hdm.c
@@ -446,6 +446,8 @@ static const char *cxl_mode_name(enum cxl_partition_mode mode)
return "ram";
case CXL_PARTMODE_PMEM:
return "pmem";
+ case CXL_PARTMODE_DYNAMIC_RAM_1:
+ return "dynamic_ram_1";
default:
return "";
};
diff --git a/drivers/cxl/core/mbox.c b/drivers/cxl/core/mbox.c
index 4790524c32a7..d79019fbd790 100644
--- a/drivers/cxl/core/mbox.c
+++ b/drivers/cxl/core/mbox.c
@@ -1352,6 +1352,197 @@ int cxl_mem_sanitize(struct cxl_memdev *cxlmd, u16 cmd)
return -EBUSY;
}
+static int cxl_dc_check(struct device *dev, struct cxl_dc_partition_info *part_array,
+ u8 index, struct cxl_dc_partition *dev_part)
+{
+ u64 blk_size = le64_to_cpu(dev_part->block_size);
+ u64 len = le64_to_cpu(dev_part->length);
+
+ part_array[index].start = le64_to_cpu(dev_part->base);
+ part_array[index].size = le64_to_cpu(dev_part->decode_length);
+ part_array[index].size *= CXL_CAPACITY_MULTIPLIER;
+
+ /* Check partitions are in increasing DPA order */
+ if (index > 0) {
+ struct cxl_dc_partition_info *prev_part = &part_array[index - 1];
+
+ if ((prev_part->start + prev_part->size) >
+ part_array[index].start) {
+ dev_err(dev,
+ "DPA ordering violation for DC partition %d and %d\n",
+ index - 1, index);
+ return -EINVAL;
+ }
+ }
+
+ if (part_array[index].size == 0 || len == 0 ||
+ part_array[index].size < len || !IS_ALIGNED(len, blk_size)) {
+ dev_err(dev, "DC partition %d invalid length; size %llu len %llu blk size %llu\n",
+ index, part_array[index].size, len, blk_size);
+ return -EINVAL;
+ }
+
+ if (blk_size == 0 || blk_size % CXL_DCD_BLOCK_LINE_SIZE ||
+ !is_power_of_2(blk_size)) {
+ dev_err(dev, "DC partition %d invalid block size %llu\n",
+ index, blk_size);
+ return -EINVAL;
+ }
+
+ if (!IS_ALIGNED(part_array[index].start, SZ_256M) ||
+ !IS_ALIGNED(part_array[index].start, blk_size)) {
+ dev_err(dev, "DC partition %d invalid start %llu blk size %llu\n",
+ index, part_array[index].start, blk_size);
+ return -EINVAL;
+ }
+
+ dev_dbg(dev, "DC partition %d start %llu size %llu blk_size: %llu\n",
+ index, part_array[index].start, part_array[index].size,
+ blk_size);
+
+ return 0;
+}
+
+/* Returns the number of partitions in dc_resp or -ERRNO */
+static int cxl_get_dc_config(struct cxl_mailbox *mbox, u8 start_partition,
+ u8 partition_count,
+ struct cxl_mbox_get_dc_config_out *dc_resp,
+ size_t dc_resp_size)
+{
+ struct cxl_mbox_get_dc_config_in get_dc = (struct cxl_mbox_get_dc_config_in) {
+ .partition_count = partition_count,
+ .start_partition_index = start_partition,
+ };
+ struct cxl_mbox_cmd mbox_cmd = (struct cxl_mbox_cmd) {
+ .opcode = CXL_MBOX_OP_GET_DC_CONFIG,
+ .payload_in = &get_dc,
+ .size_in = sizeof(get_dc),
+ .size_out = dc_resp_size,
+ .payload_out = dc_resp,
+ .min_out = 8,
+ };
+ size_t expected_sz;
+ int rc;
+
+ rc = cxl_internal_send_cmd(mbox, &mbox_cmd);
+ if (rc < 0)
+ return rc;
+
+ if (dc_resp->partitions_returned > partition_count) {
+ dev_err(mbox->host, "Device returned %u partitions, requested %u\n",
+ dc_resp->partitions_returned, partition_count);
+ return -EIO;
+ }
+
+ /*
+ * The payload carries trailing extent/tag count fields after the
+ * partition array (CXL r4.0 Table 8-179) which the driver ignores, so
+ * the response is at least, not exactly, expected_sz.
+ */
+ expected_sz = struct_size(dc_resp, partition,
+ dc_resp->partitions_returned);
+
+ if (mbox_cmd.size_out < expected_sz) {
+ dev_err(mbox->host,
+ "Payload size %zu less than expected %zu for %u partitions\n",
+ mbox_cmd.size_out,
+ expected_sz,
+ dc_resp->partitions_returned);
+ return -EIO;
+ }
+
+ dev_dbg(mbox->host, "Read %d/%d DC partitions\n",
+ dc_resp->partitions_returned, dc_resp->avail_partition_count);
+ return dc_resp->partitions_returned;
+}
+
+/**
+ * cxl_dev_dc_identify() - Reads the dynamic capacity information from the
+ * device.
+ * @mbox: Mailbox to query
+ * @dc_info: The dynamic partition information to return
+ *
+ * Read Dynamic Capacity information from the device and return the partition
+ * information.
+ *
+ * Return: 0 if identify was executed successfully, -ERRNO on error.
+ * on error only dc_info is left unchanged.
+ */
+int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
+ struct cxl_dc_partition_info *dc_info)
+{
+ struct cxl_dc_partition_info partitions[CXL_MAX_DC_PARTITIONS];
+ struct cxl_mbox_get_dc_config_out *dc_resp __free(kfree) = NULL;
+ struct device *dev = mbox->host;
+ u8 start_partition;
+ u8 num_partitions;
+ u8 partition_count;
+ size_t dc_resp_size;
+
+ /* Bound requested number of partitions by mailbox payload size */
+ partition_count = min_t(size_t, CXL_MAX_DC_PARTITIONS,
+ (mbox->payload_size - sizeof(*dc_resp) -
+ sizeof(struct cxl_mbox_get_dc_config_tail)) /
+ sizeof(struct cxl_dc_partition));
+ dc_resp_size = struct_size(dc_resp, partition, partition_count) +
+ sizeof(struct cxl_mbox_get_dc_config_tail);
+
+ dc_resp = kmalloc(dc_resp_size, GFP_KERNEL);
+ if (!dc_resp)
+ return -ENOMEM;
+
+ /*
+ * Read and check all partition information for validity and potential
+ * debugging; see debug output in cxl_dc_check()
+ */
+ start_partition = 0;
+ num_partitions = 0;
+ do {
+ int rc, i, j;
+
+ rc = cxl_get_dc_config(mbox, start_partition, partition_count,
+ dc_resp, dc_resp_size);
+ if (rc < 0) {
+ dev_err(dev, "Failed to get DC config: %d\n", rc);
+ return rc;
+ }
+
+ if (rc == 0) {
+ dev_err(dev,
+ "Device reported %u partitions available but returned none at index %u\n",
+ dc_resp->avail_partition_count, start_partition);
+ return -EIO;
+ }
+
+ num_partitions += rc;
+
+ if (num_partitions < 1 || num_partitions > CXL_MAX_DC_PARTITIONS) {
+ dev_err(dev, "Invalid num of dynamic capacity partitions %d\n",
+ num_partitions);
+ return -EINVAL;
+ }
+
+ for (i = start_partition, j = 0; i < num_partitions; i++, j++) {
+ rc = cxl_dc_check(dev, partitions, i,
+ &dc_resp->partition[j]);
+ if (rc)
+ return rc;
+ }
+
+ start_partition = num_partitions;
+
+ } while (num_partitions < dc_resp->avail_partition_count);
+
+ /* Return 1st partition */
+ dc_info->start = partitions[0].start;
+ dc_info->size = partitions[0].size;
+ dev_dbg(dev, "Returning partition 0 %llu size %llu\n",
+ dc_info->start, dc_info->size);
+
+ return 0;
+}
+EXPORT_SYMBOL_NS_GPL(cxl_dev_dc_identify, "CXL");
+
static void add_part(struct cxl_dpa_info *info, u64 start, u64 size, enum cxl_partition_mode mode)
{
int i = info->nr_partitions;
@@ -1422,6 +1613,35 @@ int cxl_get_dirty_count(struct cxl_memdev_state *mds, u32 *count)
}
EXPORT_SYMBOL_NS_GPL(cxl_get_dirty_count, "CXL");
+void cxl_configure_dcd(struct cxl_memdev_state *mds, struct cxl_dpa_info *info)
+{
+ struct cxl_dc_partition_info dc_info = { 0 };
+ struct device *dev = mds->cxlds.dev;
+ int rc;
+
+ rc = cxl_dev_dc_identify(&mds->cxlds.cxl_mbox, &dc_info);
+ if (rc) {
+ dev_warn(dev,
+ "Failed to read Dynamic Capacity config: %d\n", rc);
+ cxl_disable_dcd(mds);
+ return;
+ }
+
+ /* Skips between pmem and the dynamic partition are not supported */
+ if (dc_info.start != info->size) {
+ dev_warn(dev,
+ "Dynamic Capacity skip from pmem not supported\n");
+ cxl_disable_dcd(mds);
+ return;
+ }
+
+ info->size += dc_info.size;
+ dev_dbg(dev, "Adding dynamic ram partition 1; %llu size %llu\n",
+ dc_info.start, dc_info.size);
+ add_part(info, dc_info.start, dc_info.size, CXL_PARTMODE_DYNAMIC_RAM_1);
+}
+EXPORT_SYMBOL_NS_GPL(cxl_configure_dcd, "CXL");
+
int cxl_arm_dirty_shutdown(struct cxl_memdev_state *mds)
{
struct cxl_mailbox *cxl_mbox = &mds->cxlds.cxl_mbox;
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index 616eeaeee5b9..a9782939d82b 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -407,6 +407,8 @@ struct cxl_security_state {
struct kernfs_node *sanitize_node;
};
+#define CXL_MAX_DC_PARTITIONS 8
+
static inline resource_size_t cxl_pmem_size(struct cxl_dev_state *cxlds)
{
/*
@@ -691,6 +693,39 @@ struct cxl_mbox_set_shutdown_state_in {
u8 state;
} __packed;
+/* See CXL r4.0 Table 8-178 get dynamic capacity config Input Payload */
+struct cxl_mbox_get_dc_config_in {
+ u8 partition_count;
+ u8 start_partition_index;
+} __packed;
+
+/* See CXL r4.0 Table 8-179 get dynamic capacity config Output Payload */
+struct cxl_mbox_get_dc_config_out {
+ u8 avail_partition_count;
+ u8 partitions_returned;
+ u8 rsvd[6];
+ /* See CXL r4.0 Table 8-180 */
+ struct cxl_dc_partition {
+ __le64 base;
+ __le64 decode_length;
+ __le64 length;
+ __le64 block_size;
+ __le32 dsmad_handle;
+ u8 flags;
+ u8 rsvd[3];
+ } __packed partition[] __counted_by(partitions_returned);
+ /* Trailing extent/tag count fields unused */
+} __packed;
+
+/* Trailing counts; cannot be a member after the flex array above */
+struct cxl_mbox_get_dc_config_tail {
+ __le32 num_extents_supported;
+ __le32 num_extents_available;
+ __le32 num_tags_supported;
+ __le32 num_tags_available;
+} __packed;
+#define CXL_DCD_BLOCK_LINE_SIZE 0x40
+
/* Set Timestamp CXL 3.0 Spec 8.2.9.4.2 */
struct cxl_mbox_set_timestamp_in {
__le64 timestamp;
@@ -814,9 +849,18 @@ enum {
int cxl_internal_send_cmd(struct cxl_mailbox *cxl_mbox,
struct cxl_mbox_cmd *cmd);
int cxl_dev_state_identify(struct cxl_memdev_state *mds);
+
+struct cxl_dc_partition_info {
+ u64 start;
+ u64 size;
+};
+
+int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
+ struct cxl_dc_partition_info *dc_info);
int cxl_await_media_ready(struct cxl_dev_state *cxlds);
int cxl_enumerate_cmds(struct cxl_memdev_state *mds);
int cxl_mem_dpa_fetch(struct cxl_memdev_state *mds, struct cxl_dpa_info *info);
+void cxl_configure_dcd(struct cxl_memdev_state *mds, struct cxl_dpa_info *info);
struct cxl_memdev_state *cxl_memdev_state_create(struct device *dev, u64 serial,
u16 dvsec);
void set_exclusive_cxl_commands(struct cxl_memdev_state *mds,
@@ -830,6 +874,17 @@ void cxl_event_trace_record(struct cxl_memdev *cxlmd,
const uuid_t *uuid, union cxl_event *evt);
int cxl_get_dirty_count(struct cxl_memdev_state *mds, u32 *count);
int cxl_arm_dirty_shutdown(struct cxl_memdev_state *mds);
+
+static inline bool cxl_dcd_supported(struct cxl_memdev_state *mds)
+{
+ return mds->dcd_supported;
+}
+
+static inline void cxl_disable_dcd(struct cxl_memdev_state *mds)
+{
+ mds->dcd_supported = false;
+}
+
int cxl_set_timestamp(struct cxl_memdev_state *mds);
int cxl_poison_state_init(struct cxl_memdev_state *mds);
int cxl_mem_get_poison(struct cxl_memdev *cxlmd, u64 offset, u64 len,
diff --git a/drivers/cxl/pci.c b/drivers/cxl/pci.c
index 267c679b0b3c..9b320a2b6fe0 100644
--- a/drivers/cxl/pci.c
+++ b/drivers/cxl/pci.c
@@ -870,6 +870,9 @@ static int cxl_pci_probe(struct pci_dev *pdev, const struct pci_device_id *id)
if (rc)
return rc;
+ if (cxl_dcd_supported(mds))
+ cxl_configure_dcd(mds, &range_info);
+
rc = cxl_dpa_setup(cxlds, &range_info);
if (rc)
return rc;
diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
index 016c74fb747c..87c2bd73df21 100644
--- a/include/cxl/cxl.h
+++ b/include/cxl/cxl.h
@@ -133,6 +133,7 @@ struct cxl_dpa_perf {
enum cxl_partition_mode {
CXL_PARTMODE_RAM,
CXL_PARTMODE_PMEM,
+ CXL_PARTMODE_DYNAMIC_RAM_1,
};
/**
@@ -147,7 +148,7 @@ struct cxl_dpa_partition {
enum cxl_partition_mode mode;
};
-#define CXL_NR_PARTITIONS_MAX 2
+#define CXL_NR_PARTITIONS_MAX 3
/**
* struct cxl_dev_state - The driver device state
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* Re: [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device
2026-07-31 8:48 ` [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device Anisa Su
@ 2026-07-31 9:01 ` sashiko-bot
0 siblings, 0 replies; 13+ messages in thread
From: sashiko-bot @ 2026-07-31 9:01 UTC (permalink / raw)
To: Anisa Su; +Cc: nvdimm, linux-cxl
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [Medium] False-positive error logging and initialization failure when a device has 0 dynamic capacity partitions configured.
- [Medium] The validation check `is_power_of_2(blk_size)` truncates the 64-bit block size on 32-bit architectures, allowing invalid block sizes to pass validation.
--
--- Patch [2]: [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device ---
commit 7a0233410b5ae133da2a756fa11cf61b2145cace
Author: Ira Weiny <iweiny@kernel.org>
cxl/mem: Read dynamic capacity configuration from the device
This commit implements the Dynamic Capacity (DC) mailbox commands to read
and store the DCD configuration information from a CXL device. It reads
and validates all partitions but configures only the first one.
> diff --git a/drivers/cxl/core/mbox.c b/drivers/cxl/core/mbox.c
> --- a/drivers/cxl/core/mbox.c
> +++ b/drivers/cxl/core/mbox.c
[ ... ]
> +static int cxl_dc_check(struct device *dev, struct cxl_dc_partition_info *part_array,
> + u8 index, struct cxl_dc_partition *dev_part)
> +{
> + u64 blk_size = le64_to_cpu(dev_part->block_size);
[ ... ]
> + if (blk_size == 0 || blk_size % CXL_DCD_BLOCK_LINE_SIZE ||
> + !is_power_of_2(blk_size)) {
[Severity: Medium]
Could passing the 64-bit blk_size to is_power_of_2() cause silent truncation
on 32-bit architectures?
is_power_of_2() takes an unsigned long, which might allow an invalid 64-bit
block size that is not a power of 2 (but whose lower 32 bits form a power of 2)
to bypass the validation check.
> + dev_err(dev, "DC partition %d invalid block size %llu\n",
> + index, blk_size);
> + return -EINVAL;
> + }
[ ... ]
> +int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
> + struct cxl_dc_partition_info *dc_info)
> +{
[ ... ]
> + do {
> + int rc, i, j;
> +
> + rc = cxl_get_dc_config(mbox, start_partition, partition_count,
> + dc_resp, dc_resp_size);
> + if (rc < 0) {
> + dev_err(dev, "Failed to get DC config: %d\n", rc);
> + return rc;
> + }
> +
> + if (rc == 0) {
> + dev_err(dev,
> + "Device reported %u partitions available but returned none at index %u\n",
> + dc_resp->avail_partition_count, start_partition);
> + return -EIO;
> + }
[Severity: Medium]
Does this code incorrectly fail initialization when a valid CXL device has 0
dynamic capacity partitions configured?
Since this loop is implemented as a do-while loop, it will always execute at
least once. If a device has 0 available partitions, it will return 0
partitions (rc == 0), which causes this path to log an error and abort
initialization with -EIO.
> +
> + num_partitions += rc;
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260731084901.1512819-1-anisa.su@samsung.com?part=2
^ permalink raw reply [flat|nested] 13+ messages in thread
* [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
2026-07-31 8:48 ` [PATCH v12 1/8] cxl/mbox: Flag support for Dynamic Capacity Devices (DCD) Anisa Su
2026-07-31 8:48 ` [PATCH v12 2/8] cxl/mem: Read dynamic capacity configuration from the device Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 9:02 ` sashiko-bot
2026-07-31 8:48 ` [PATCH v12 4/8] cxl/events: Split event msgnum configuration from irq setup Anisa Su
` (4 subsequent siblings)
7 siblings, 1 reply; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Additional DCD partition (AKA region) information is contained in the
DSMAS CDAT tables, including performance, read only, and shareable
attributes.
Match DCD partitions with DSMAS tables and store the meta data.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Co-developed-by: Anisa Su <anisa.su@samsung.com>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
---
drivers/cxl/core/cdat.c | 13 +++++++++++++
drivers/cxl/core/hdm.c | 1 +
drivers/cxl/core/mbox.c | 22 ++++++++++++++++------
drivers/cxl/cxlmem.h | 2 ++
include/cxl/cxl.h | 4 ++++
5 files changed, 36 insertions(+), 6 deletions(-)
diff --git a/drivers/cxl/core/cdat.c b/drivers/cxl/core/cdat.c
index 5c9f07262513..37136b2cf7e4 100644
--- a/drivers/cxl/core/cdat.c
+++ b/drivers/cxl/core/cdat.c
@@ -17,6 +17,7 @@ struct dsmas_entry {
struct access_coordinate cdat_coord[ACCESS_COORDINATE_MAX];
int entries;
int qos_class;
+ bool shareable;
};
static u32 cdat_normalize(u16 entry, u64 base, u8 type)
@@ -74,6 +75,7 @@ static int cdat_dsmas_handler(union acpi_subtable_headers *header, void *arg,
return -ENOMEM;
dent->handle = dsmas->dsmad_handle;
+ dent->shareable = dsmas->flags & ACPI_CDAT_DSMAS_SHAREABLE;
dent->dpa_range.start = le64_to_cpu((__force __le64)dsmas->dpa_base_address);
dent->dpa_range.end = le64_to_cpu((__force __le64)dsmas->dpa_base_address) +
le64_to_cpu((__force __le64)dsmas->dpa_length) - 1;
@@ -266,15 +268,26 @@ static void cxl_memdev_set_qos_class(struct cxl_dev_state *cxlds,
bool found = false;
for (int i = 0; i < cxlds->nr_partitions; i++) {
+ enum cxl_partition_mode mode = cxlds->part[i].mode;
struct resource *res = &cxlds->part[i].res;
+ u8 handle = cxlds->part[i].handle;
struct range range = {
.start = res->start,
.end = res->end,
};
if (range_contains(&range, &dent->dpa_range)) {
+ if (mode == CXL_PARTMODE_DYNAMIC_RAM_1 &&
+ dent->handle != handle) {
+ dev_warn(dev,
+ "Dynamic RAM perf mismatch; %pra (%u) vs %pra (%u)\n",
+ &range, handle, &dent->dpa_range,
+ dent->handle);
+ continue;
+ }
update_perf_entry(dev, dent,
&cxlds->part[i].perf);
+ cxlds->part[i].shareable = dent->shareable;
found = true;
break;
}
diff --git a/drivers/cxl/core/hdm.c b/drivers/cxl/core/hdm.c
index 0ef076c08ed2..7f63b86887f4 100644
--- a/drivers/cxl/core/hdm.c
+++ b/drivers/cxl/core/hdm.c
@@ -477,6 +477,7 @@ int cxl_dpa_setup(struct cxl_dev_state *cxlds, const struct cxl_dpa_info *info)
cxlds->part[i].perf.qos_class = CXL_QOS_CLASS_INVALID;
cxlds->part[i].mode = part->mode;
+ cxlds->part[i].handle = part->handle;
/* Require ordered + contiguous partitions */
if (i) {
diff --git a/drivers/cxl/core/mbox.c b/drivers/cxl/core/mbox.c
index d79019fbd790..a6cdea9f4080 100644
--- a/drivers/cxl/core/mbox.c
+++ b/drivers/cxl/core/mbox.c
@@ -1357,10 +1357,16 @@ static int cxl_dc_check(struct device *dev, struct cxl_dc_partition_info *part_a
{
u64 blk_size = le64_to_cpu(dev_part->block_size);
u64 len = le64_to_cpu(dev_part->length);
+ u32 handle = le32_to_cpu(dev_part->dsmad_handle);
part_array[index].start = le64_to_cpu(dev_part->base);
part_array[index].size = le64_to_cpu(dev_part->decode_length);
part_array[index].size *= CXL_CAPACITY_MULTIPLIER;
+ if (handle & ~0xFF) {
+ dev_warn(dev, "DSMAD handle 0x%x has non-zero reserved bits\n", handle);
+ return -EINVAL;
+ }
+ part_array[index].handle = handle;
/* Check partitions are in increasing DPA order */
if (index > 0) {
@@ -1536,6 +1542,7 @@ int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
/* Return 1st partition */
dc_info->start = partitions[0].start;
dc_info->size = partitions[0].size;
+ dc_info->handle = partitions[0].handle;
dev_dbg(dev, "Returning partition 0 %llu size %llu\n",
dc_info->start, dc_info->size);
@@ -1543,7 +1550,8 @@ int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
}
EXPORT_SYMBOL_NS_GPL(cxl_dev_dc_identify, "CXL");
-static void add_part(struct cxl_dpa_info *info, u64 start, u64 size, enum cxl_partition_mode mode)
+static void add_part(struct cxl_dpa_info *info, u64 start, u64 size,
+ enum cxl_partition_mode mode, u8 handle)
{
int i = info->nr_partitions;
@@ -1555,6 +1563,7 @@ static void add_part(struct cxl_dpa_info *info, u64 start, u64 size, enum cxl_pa
.end = start + size - 1,
};
info->part[i].mode = mode;
+ info->part[i].handle = handle;
info->nr_partitions++;
}
@@ -1572,9 +1581,9 @@ int cxl_mem_dpa_fetch(struct cxl_memdev_state *mds, struct cxl_dpa_info *info)
info->size = mds->total_bytes;
if (mds->partition_align_bytes == 0) {
- add_part(info, 0, mds->volatile_only_bytes, CXL_PARTMODE_RAM);
+ add_part(info, 0, mds->volatile_only_bytes, CXL_PARTMODE_RAM, 0);
add_part(info, mds->volatile_only_bytes,
- mds->persistent_only_bytes, CXL_PARTMODE_PMEM);
+ mds->persistent_only_bytes, CXL_PARTMODE_PMEM, 0);
return 0;
}
@@ -1584,9 +1593,9 @@ int cxl_mem_dpa_fetch(struct cxl_memdev_state *mds, struct cxl_dpa_info *info)
return rc;
}
- add_part(info, 0, mds->active_volatile_bytes, CXL_PARTMODE_RAM);
+ add_part(info, 0, mds->active_volatile_bytes, CXL_PARTMODE_RAM, 0);
add_part(info, mds->active_volatile_bytes, mds->active_persistent_bytes,
- CXL_PARTMODE_PMEM);
+ CXL_PARTMODE_PMEM, 0);
return 0;
}
@@ -1638,7 +1647,8 @@ void cxl_configure_dcd(struct cxl_memdev_state *mds, struct cxl_dpa_info *info)
info->size += dc_info.size;
dev_dbg(dev, "Adding dynamic ram partition 1; %llu size %llu\n",
dc_info.start, dc_info.size);
- add_part(info, dc_info.start, dc_info.size, CXL_PARTMODE_DYNAMIC_RAM_1);
+ add_part(info, dc_info.start, dc_info.size, CXL_PARTMODE_DYNAMIC_RAM_1,
+ dc_info.handle);
}
EXPORT_SYMBOL_NS_GPL(cxl_configure_dcd, "CXL");
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index a9782939d82b..eb1e6f7a2038 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -140,6 +140,7 @@ struct cxl_dpa_info {
struct cxl_dpa_part_info {
struct range range;
enum cxl_partition_mode mode;
+ u8 handle;
} part[CXL_NR_PARTITIONS_MAX];
int nr_partitions;
};
@@ -853,6 +854,7 @@ int cxl_dev_state_identify(struct cxl_memdev_state *mds);
struct cxl_dc_partition_info {
u64 start;
u64 size;
+ u8 handle;
};
int cxl_dev_dc_identify(struct cxl_mailbox *mbox,
diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
index 87c2bd73df21..baf0132c70a3 100644
--- a/include/cxl/cxl.h
+++ b/include/cxl/cxl.h
@@ -141,11 +141,15 @@ enum cxl_partition_mode {
* @res: shortcut to the partition in the DPA resource tree (cxlds->dpa_res)
* @perf: performance attributes of the partition from CDAT
* @mode: operation mode for the DPA capacity, e.g. ram, pmem, dynamic...
+ * @handle: DSMAS handle intended to represent this partition
+ * @shareable: Is the partition sharable (from its CDAT DSMAS entry)
*/
struct cxl_dpa_partition {
struct resource res;
struct cxl_dpa_perf perf;
enum cxl_partition_mode mode;
+ u8 handle;
+ bool shareable;
};
#define CXL_NR_PARTITIONS_MAX 3
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* Re: [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions
2026-07-31 8:48 ` [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions Anisa Su
@ 2026-07-31 9:02 ` sashiko-bot
0 siblings, 0 replies; 13+ messages in thread
From: sashiko-bot @ 2026-07-31 9:02 UTC (permalink / raw)
To: Anisa Su; +Cc: nvdimm, linux-cxl
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [Medium] The commit message claims to extract and store the 'read only' attribute from DSMAS tables, but this logic is completely missing from the code.
- [Medium] Spurious `dev_warn` on valid overlapping CDAT DSMAS entries for DCD partitions.
--
--- Patch [3]: [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions ---
commit 2a6d0666da69329a29cf2ac94ee05e10012974cc
Author: Ira Weiny <iweiny@kernel.org>
cxl/cdat: Gather DSMAS data for DCD partitions
This commit extracts additional DCD partition information from DSMAS CDAT
tables. It matches DCD partitions with DSMAS tables to store performance
and shareable attributes.
> cxl/cdat: Gather DSMAS data for DCD partitions
>
> Additional DCD partition (AKA region) information is contained in the
> DSMAS CDAT tables, including performance, read only, and shareable
> attributes.
[Severity: Medium]
The commit message states that the read only attribute is being extracted,
but I am not seeing the logic for it in the patch below.
> diff --git a/drivers/cxl/core/cdat.c b/drivers/cxl/core/cdat.c
> index 5c9f072625135..37136b2cf7e48 100644
> --- a/drivers/cxl/core/cdat.c
> +++ b/drivers/cxl/core/cdat.c
[ ... ]
> @@ -74,6 +75,7 @@ static int cdat_dsmas_handler(union acpi_subtable_headers *header, void *arg,
> return -ENOMEM;
>
> dent->handle = dsmas->dsmad_handle;
> + dent->shareable = dsmas->flags & ACPI_CDAT_DSMAS_SHAREABLE;
[Severity: Medium]
Is the parsing for ACPI_CDAT_DSMAS_READ_ONLY missing from cdat_dsmas_handler()?
The code only appears to extract the shareable flag.
[ ... ]
> @@ -266,15 +268,26 @@ static void cxl_memdev_set_qos_class(struct cxl_dev_state *cxlds,
> bool found = false;
>
> for (int i = 0; i < cxlds->nr_partitions; i++) {
> + enum cxl_partition_mode mode = cxlds->part[i].mode;
> struct resource *res = &cxlds->part[i].res;
> + u8 handle = cxlds->part[i].handle;
> struct range range = {
> .start = res->start,
> .end = res->end,
> };
>
> if (range_contains(&range, &dent->dpa_range)) {
> + if (mode == CXL_PARTMODE_DYNAMIC_RAM_1 &&
> + dent->handle != handle) {
> + dev_warn(dev,
> + "Dynamic RAM perf mismatch; %pra (%u) vs %pra (%u)\n",
> + &range, handle, &dent->dpa_range,
> + dent->handle);
> + continue;
> + }
[Severity: Medium]
Could this trigger spurious warnings on valid hardware configurations?
The CDAT specification allows overlapping DSMAS ranges with different handles
to support multiple memory types or sharing characteristics for the same DPA
space.
If a compliant device provides multiple valid DSMAS entries that overlap
the partition's range but have different handles, won't iterating over them
incorrectly flag the unselected ones as a perf mismatch?
[ ... ]
> diff --git a/include/cxl/cxl.h b/include/cxl/cxl.h
> index 87c2bd73df21d..baf0132c70a34 100644
> --- a/include/cxl/cxl.h
> +++ b/include/cxl/cxl.h
> @@ -141,11 +141,15 @@ enum cxl_partition_mode {
> * @res: shortcut to the partition in the DPA resource tree (cxlds->dpa_res)
> * @perf: performance attributes of the partition from CDAT
> * @mode: operation mode for the DPA capacity, e.g. ram, pmem, dynamic...
> + * @handle: DSMAS handle intended to represent this partition
> + * @shareable: Is the partition sharable (from its CDAT DSMAS entry)
> */
> struct cxl_dpa_partition {
> struct resource res;
> struct cxl_dpa_perf perf;
> enum cxl_partition_mode mode;
> + u8 handle;
> + bool shareable;
> };
[Severity: Medium]
Does struct cxl_dpa_partition also need a read_only boolean field added here
to match the commit message description?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260731084901.1512819-1-anisa.su@samsung.com?part=3
^ permalink raw reply [flat|nested] 13+ messages in thread
* [PATCH v12 4/8] cxl/events: Split event msgnum configuration from irq setup
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
` (2 preceding siblings ...)
2026-07-31 8:48 ` [PATCH v12 3/8] cxl/cdat: Gather DSMAS data for DCD partitions Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 8:48 ` [PATCH v12 5/8] cxl/pci: Factor out interrupt policy check Anisa Su
` (3 subsequent siblings)
7 siblings, 0 replies; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Dynamic Capacity Devices (DCD) require event interrupts to process
memory addition or removal. BIOS may have control over non-DCD event
processing. DCD interrupt configuration needs to be separate from
memory event interrupt configuration.
Split cxl_event_config_msgnums() from irq setup in preparation for
separate DCD interrupts configuration.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
Reviewed-by: Jonathan Cameron <jic23@kernel.org>
Reviewed-by: Fan Ni <nifan.cxl@gmail.com>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
Reviewed-by: Li Ming <ming.li@zohomail.com>
---
drivers/cxl/pci.c | 22 +++++++++++-----------
1 file changed, 11 insertions(+), 11 deletions(-)
diff --git a/drivers/cxl/pci.c b/drivers/cxl/pci.c
index 9b320a2b6fe0..8e122de1876b 100644
--- a/drivers/cxl/pci.c
+++ b/drivers/cxl/pci.c
@@ -599,35 +599,31 @@ static int cxl_event_config_msgnums(struct cxl_memdev_state *mds,
return cxl_event_get_int_policy(mds, policy);
}
-static int cxl_event_irqsetup(struct cxl_memdev_state *mds)
+static int cxl_event_irqsetup(struct cxl_memdev_state *mds,
+ struct cxl_event_interrupt_policy *policy)
{
struct cxl_dev_state *cxlds = &mds->cxlds;
- struct cxl_event_interrupt_policy policy;
int rc;
- rc = cxl_event_config_msgnums(mds, &policy);
- if (rc)
- return rc;
-
- rc = cxl_event_req_irq(cxlds, policy.info_settings);
+ rc = cxl_event_req_irq(cxlds, policy->info_settings);
if (rc) {
dev_err(cxlds->dev, "Failed to get interrupt for event Info log\n");
return rc;
}
- rc = cxl_event_req_irq(cxlds, policy.warn_settings);
+ rc = cxl_event_req_irq(cxlds, policy->warn_settings);
if (rc) {
dev_err(cxlds->dev, "Failed to get interrupt for event Warn log\n");
return rc;
}
- rc = cxl_event_req_irq(cxlds, policy.failure_settings);
+ rc = cxl_event_req_irq(cxlds, policy->failure_settings);
if (rc) {
dev_err(cxlds->dev, "Failed to get interrupt for event Failure log\n");
return rc;
}
- rc = cxl_event_req_irq(cxlds, policy.fatal_settings);
+ rc = cxl_event_req_irq(cxlds, policy->fatal_settings);
if (rc) {
dev_err(cxlds->dev, "Failed to get interrupt for event Fatal log\n");
return rc;
@@ -674,11 +670,15 @@ static int cxl_event_config(struct pci_host_bridge *host_bridge,
return -EBUSY;
}
+ rc = cxl_event_config_msgnums(mds, &policy);
+ if (rc)
+ return rc;
+
rc = cxl_mem_alloc_event_buf(mds);
if (rc)
return rc;
- rc = cxl_event_irqsetup(mds);
+ rc = cxl_event_irqsetup(mds, &policy);
if (rc)
return rc;
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* [PATCH v12 5/8] cxl/pci: Factor out interrupt policy check
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
` (3 preceding siblings ...)
2026-07-31 8:48 ` [PATCH v12 4/8] cxl/events: Split event msgnum configuration from irq setup Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 8:48 ` [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts Anisa Su
` (2 subsequent siblings)
7 siblings, 0 replies; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Dan Williams, Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Dynamic Capacity Devices (DCD) require event interrupts to process
memory addition or removal. BIOS may have control over non-DCD event
processing. DCD interrupt configuration needs to be separate from
memory event interrupt configuration.
Factor out event interrupt setting validation.
Link: https://lore.kernel.org/all/663922b475e50_d54d72945b@dwillia2-xfh.jf.intel.com.notmuch/ [1]
Suggested-by: Dan Williams <djbw@kernel.org>
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
Reviewed-by: Jonathan Cameron <jic23@kernel.org>
Reviewed-by: Fan Ni <nifan.cxl@gmail.com>
Reviewed-by: Li Ming <ming.li@zohomail.com>
---
drivers/cxl/pci.c | 23 ++++++++++++++++-------
1 file changed, 16 insertions(+), 7 deletions(-)
diff --git a/drivers/cxl/pci.c b/drivers/cxl/pci.c
index 8e122de1876b..df13fb8802c3 100644
--- a/drivers/cxl/pci.c
+++ b/drivers/cxl/pci.c
@@ -639,6 +639,21 @@ static bool cxl_event_int_is_fw(u8 setting)
return mode == CXL_INT_FW;
}
+static bool cxl_event_validate_mem_policy(struct cxl_memdev_state *mds,
+ struct cxl_event_interrupt_policy *policy)
+{
+ if (cxl_event_int_is_fw(policy->info_settings) ||
+ cxl_event_int_is_fw(policy->warn_settings) ||
+ cxl_event_int_is_fw(policy->failure_settings) ||
+ cxl_event_int_is_fw(policy->fatal_settings)) {
+ dev_err(mds->cxlds.dev,
+ "FW still in control of Event Logs despite _OSC settings\n");
+ return false;
+ }
+
+ return true;
+}
+
static int cxl_event_config(struct pci_host_bridge *host_bridge,
struct cxl_memdev_state *mds, bool irq_avail)
{
@@ -661,14 +676,8 @@ static int cxl_event_config(struct pci_host_bridge *host_bridge,
if (rc)
return rc;
- if (cxl_event_int_is_fw(policy.info_settings) ||
- cxl_event_int_is_fw(policy.warn_settings) ||
- cxl_event_int_is_fw(policy.failure_settings) ||
- cxl_event_int_is_fw(policy.fatal_settings)) {
- dev_err(mds->cxlds.dev,
- "FW still in control of Event Logs despite _OSC settings\n");
+ if (!cxl_event_validate_mem_policy(mds, &policy))
return -EBUSY;
- }
rc = cxl_event_config_msgnums(mds, &policy);
if (rc)
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
` (4 preceding siblings ...)
2026-07-31 8:48 ` [PATCH v12 5/8] cxl/pci: Factor out interrupt policy check Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 9:04 ` sashiko-bot
2026-07-31 8:48 ` [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search Anisa Su
2026-07-31 8:48 ` [PATCH v12 8/8] cxl/core: Enforce partition order/simplify partition calls Anisa Su
7 siblings, 1 reply; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Dynamic Capacity Devices (DCD) support extent change notifications
through the event log mechanism. The interrupt mailbox commands were
extended in CXL 3.1 to support these notifications. Firmware can't
configure DCD events to be FW controlled but can retain control of
memory events.
Configure DCD event log interrupts on devices supporting dynamic
capacity. Disable DCD if interrupts are not supported.
Care is taken to preserve the interrupt policy set by the FW if FW first
has been selected by the BIOS.
Based on an original patch by Navneet Singh.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
---
Changes:
1. pci.c: size the Set Event Interrupt Policy payload by the device's
policy layout (from the Get reply length) rather than by DCD command
support. A CXL 3.0+ device carries the dcd_settings field for spec
compliance even without DCD commands, so keying the size on
cxl_dcd_supported() sent a short payload and failed the set. Reported
by Benjamin Cheatham.
2. pci.c: add cxl_event_drain_mask() (standard logs when native_cxl, DCD
when supported) and use it for both cxl_event_thread()'s drain mask and
the initial drain in cxl_event_config(). Gating DCD on support stops a
disabled-but-armed DCD from spinning the thread; draining the DCD log
initially even when !native_cxl stops pre-existing DCD events (and
future edge-triggered interrupts) from being stranded.
---
drivers/cxl/cxl.h | 4 +-
drivers/cxl/cxlmem.h | 2 +
drivers/cxl/pci.c | 119 +++++++++++++++++++++++++++++++++++--------
3 files changed, 104 insertions(+), 21 deletions(-)
diff --git a/drivers/cxl/cxl.h b/drivers/cxl/cxl.h
index c0e5308e4d1b..51396eb993e7 100644
--- a/drivers/cxl/cxl.h
+++ b/drivers/cxl/cxl.h
@@ -192,11 +192,13 @@ static inline int ways_to_eiw(unsigned int ways, u8 *eiw)
#define CXLDEV_EVENT_STATUS_WARN BIT(1)
#define CXLDEV_EVENT_STATUS_FAIL BIT(2)
#define CXLDEV_EVENT_STATUS_FATAL BIT(3)
+#define CXLDEV_EVENT_STATUS_DCD BIT(4)
#define CXLDEV_EVENT_STATUS_ALL (CXLDEV_EVENT_STATUS_INFO | \
CXLDEV_EVENT_STATUS_WARN | \
CXLDEV_EVENT_STATUS_FAIL | \
- CXLDEV_EVENT_STATUS_FATAL)
+ CXLDEV_EVENT_STATUS_FATAL | \
+ CXLDEV_EVENT_STATUS_DCD)
/* CXL rev 3.0 section 8.2.9.2.4; Table 8-52 */
#define CXLDEV_EVENT_INT_MODE_MASK GENMASK(1, 0)
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index eb1e6f7a2038..666e01326dae 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -240,7 +240,9 @@ struct cxl_event_interrupt_policy {
u8 warn_settings;
u8 failure_settings;
u8 fatal_settings;
+ u8 dcd_settings;
} __packed;
+#define CXL_EVENT_INT_POLICY_BASE_SIZE 4 /* info, warn, failure, fatal */
/**
* struct cxl_event_state - Event log driver state
diff --git a/drivers/cxl/pci.c b/drivers/cxl/pci.c
index df13fb8802c3..6cb344ec4f3a 100644
--- a/drivers/cxl/pci.c
+++ b/drivers/cxl/pci.c
@@ -509,11 +509,30 @@ static bool cxl_alloc_irq_vectors(struct pci_dev *pdev)
return true;
}
+/*
+ * Event logs the driver drains: standard logs when native_cxl, DCD when
+ * supported.
+ */
+static u32 cxl_event_drain_mask(struct pci_host_bridge *host_bridge,
+ struct cxl_memdev_state *mds)
+{
+ u32 mask = 0;
+
+ if (host_bridge->native_cxl_error)
+ mask |= CXLDEV_EVENT_STATUS_ALL & ~CXLDEV_EVENT_STATUS_DCD;
+ if (cxl_dcd_supported(mds))
+ mask |= CXLDEV_EVENT_STATUS_DCD;
+ return mask;
+}
+
static irqreturn_t cxl_event_thread(int irq, void *id)
{
struct cxl_dev_id *dev_id = id;
struct cxl_dev_state *cxlds = dev_id->cxlds;
struct cxl_memdev_state *mds = to_cxl_memdev_state(cxlds);
+ struct pci_host_bridge *host_bridge =
+ pci_find_host_bridge(to_pci_dev(cxlds->dev)->bus);
+ u32 mask = cxl_event_drain_mask(host_bridge, mds);
u32 status;
do {
@@ -522,8 +541,8 @@ static irqreturn_t cxl_event_thread(int irq, void *id)
* ignore the reserved upper 32 bits
*/
status = readl(cxlds->regs.status + CXLDEV_DEV_EVENT_STATUS_OFFSET);
- /* Ignore logs unknown to the driver */
- status &= CXLDEV_EVENT_STATUS_ALL;
+ /* Ignore logs unknown to the driver or owned by BIOS */
+ status &= mask;
if (!status)
break;
cxl_mem_get_event_records(mds, status);
@@ -550,42 +569,62 @@ static int cxl_event_req_irq(struct cxl_dev_state *cxlds, u8 setting)
}
static int cxl_event_get_int_policy(struct cxl_memdev_state *mds,
- struct cxl_event_interrupt_policy *policy)
+ struct cxl_event_interrupt_policy *policy,
+ size_t *policy_size)
{
struct cxl_mailbox *cxl_mbox = &mds->cxlds.cxl_mbox;
struct cxl_mbox_cmd mbox_cmd = {
.opcode = CXL_MBOX_OP_GET_EVT_INT_POLICY,
.payload_out = policy,
.size_out = sizeof(*policy),
+ /* CXL 2.0 firmware omits dcd_settings; accept the shorter reply */
+ .min_out = CXL_EVENT_INT_POLICY_BASE_SIZE,
};
int rc;
rc = cxl_internal_send_cmd(cxl_mbox, &mbox_cmd);
- if (rc < 0)
+ if (rc < 0) {
dev_err(mds->cxlds.dev,
"Failed to get event interrupt policy : %d", rc);
+ return rc;
+ }
+ if (policy_size)
+ *policy_size = mbox_cmd.size_out;
return rc;
}
static int cxl_event_config_msgnums(struct cxl_memdev_state *mds,
- struct cxl_event_interrupt_policy *policy)
+ struct cxl_event_interrupt_policy *policy,
+ bool native_cxl, size_t policy_size)
{
struct cxl_mailbox *cxl_mbox = &mds->cxlds.cxl_mbox;
struct cxl_mbox_cmd mbox_cmd;
int rc;
- *policy = (struct cxl_event_interrupt_policy) {
- .info_settings = CXL_INT_MSI_MSIX,
- .warn_settings = CXL_INT_MSI_MSIX,
- .failure_settings = CXL_INT_MSI_MSIX,
- .fatal_settings = CXL_INT_MSI_MSIX,
- };
+ /* memory event policy is left if FW has control */
+ if (native_cxl) {
+ *policy = (struct cxl_event_interrupt_policy) {
+ .info_settings = CXL_INT_MSI_MSIX,
+ .warn_settings = CXL_INT_MSI_MSIX,
+ .failure_settings = CXL_INT_MSI_MSIX,
+ .fatal_settings = CXL_INT_MSI_MSIX,
+ .dcd_settings = 0,
+ };
+ }
+
+ /*
+ * A CXL 3.0+ device can carry dcd_settings field without DCD command
+ * support, so size the request by the device's policy_size and only
+ * enable the DCD interrupt when DCD commands are supported.
+ */
+ if (cxl_dcd_supported(mds))
+ policy->dcd_settings = CXL_INT_MSI_MSIX;
mbox_cmd = (struct cxl_mbox_cmd) {
.opcode = CXL_MBOX_OP_SET_EVT_INT_POLICY,
.payload_in = policy,
- .size_in = sizeof(*policy),
+ .size_in = policy_size,
};
rc = cxl_internal_send_cmd(cxl_mbox, &mbox_cmd);
@@ -596,7 +635,7 @@ static int cxl_event_config_msgnums(struct cxl_memdev_state *mds,
}
/* Retrieve final interrupt settings */
- return cxl_event_get_int_policy(mds, policy);
+ return cxl_event_get_int_policy(mds, policy, NULL);
}
static int cxl_event_irqsetup(struct cxl_memdev_state *mds,
@@ -632,6 +671,30 @@ static int cxl_event_irqsetup(struct cxl_memdev_state *mds,
return 0;
}
+static int cxl_irqsetup(struct cxl_memdev_state *mds,
+ struct cxl_event_interrupt_policy *policy,
+ bool native_cxl)
+{
+ struct cxl_dev_state *cxlds = &mds->cxlds;
+ int rc;
+
+ if (native_cxl) {
+ rc = cxl_event_irqsetup(mds, policy);
+ if (rc)
+ return rc;
+ }
+
+ if (cxl_dcd_supported(mds)) {
+ rc = cxl_event_req_irq(cxlds, policy->dcd_settings);
+ if (rc) {
+ dev_err(cxlds->dev, "Failed to get interrupt for DCD event log\n");
+ cxl_disable_dcd(mds);
+ }
+ }
+
+ return 0;
+}
+
static bool cxl_event_int_is_fw(u8 setting)
{
u8 mode = FIELD_GET(CXLDEV_EVENT_INT_MODE_MASK, setting);
@@ -657,29 +720,39 @@ static bool cxl_event_validate_mem_policy(struct cxl_memdev_state *mds,
static int cxl_event_config(struct pci_host_bridge *host_bridge,
struct cxl_memdev_state *mds, bool irq_avail)
{
- struct cxl_event_interrupt_policy policy;
+ struct cxl_event_interrupt_policy policy = { 0 };
+ bool native_cxl = host_bridge->native_cxl_error;
+ size_t policy_size;
+ u32 status;
int rc;
/*
* When BIOS maintains CXL error reporting control, it will process
* event records. Only one agent can do so.
+ *
+ * If BIOS has control of events and DCD is not supported skip event
+ * configuration.
*/
- if (!host_bridge->native_cxl_error)
+ if (!native_cxl && !cxl_dcd_supported(mds))
return 0;
if (!irq_avail) {
dev_info(mds->cxlds.dev, "No interrupt support, disable event processing.\n");
+ if (cxl_dcd_supported(mds)) {
+ dev_info(mds->cxlds.dev, "DCD requires interrupts, disable DCD\n");
+ cxl_disable_dcd(mds);
+ }
return 0;
}
- rc = cxl_event_get_int_policy(mds, &policy);
+ rc = cxl_event_get_int_policy(mds, &policy, &policy_size);
if (rc)
return rc;
- if (!cxl_event_validate_mem_policy(mds, &policy))
+ if (native_cxl && !cxl_event_validate_mem_policy(mds, &policy))
return -EBUSY;
- rc = cxl_event_config_msgnums(mds, &policy);
+ rc = cxl_event_config_msgnums(mds, &policy, native_cxl, policy_size);
if (rc)
return rc;
@@ -687,11 +760,17 @@ static int cxl_event_config(struct pci_host_bridge *host_bridge,
if (rc)
return rc;
- rc = cxl_event_irqsetup(mds, &policy);
+ rc = cxl_irqsetup(mds, &policy, native_cxl);
if (rc)
return rc;
- cxl_mem_get_event_records(mds, CXLDEV_EVENT_STATUS_ALL);
+ status = cxl_event_drain_mask(host_bridge, mds);
+ if (status)
+ cxl_mem_get_event_records(mds, status);
+
+ dev_dbg(mds->cxlds.dev, "Event config : %s DCD %s\n",
+ native_cxl ? "OS" : "BIOS",
+ cxl_dcd_supported(mds) ? "supported" : "not supported");
return 0;
}
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* Re: [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts
2026-07-31 8:48 ` [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts Anisa Su
@ 2026-07-31 9:04 ` sashiko-bot
0 siblings, 0 replies; 13+ messages in thread
From: sashiko-bot @ 2026-07-31 9:04 UTC (permalink / raw)
To: Anisa Su; +Cc: nvdimm, linux-cxl
Thank you for your contribution! Sashiko AI review found 3 potential issue(s) to consider:
New issues:
- [High] Infinite IRQ Thread Loop / CPU Spin when handling DCD events
Pre-existing issues:
- [High] Out-of-Bounds Read and Information Leak in Event logs
- [High] Out-of-Bounds Read and Information Leak in Poison logs
--
--- Patch [6]: [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts ---
commit 5b173c122c3602992c4ceeafd1c2d87a80accacb
Author: Ira Weiny <iweiny@kernel.org>
cxl/mem: Configure dynamic capacity interrupts
Dynamic Capacity Devices (DCD) support extent change notifications
through the event log mechanism. Configure DCD event log interrupts
on devices supporting dynamic capacity. Disable DCD if interrupts
are not supported.
> diff --git a/drivers/cxl/pci.c b/drivers/cxl/pci.c
> index df13fb8802c38..6cb344ec4f3a7 100644
> --- a/drivers/cxl/pci.c
> +++ b/drivers/cxl/pci.c
> @@ -509,11 +509,30 @@ static bool cxl_alloc_irq_vectors(struct pci_dev *pdev)
> return true;
> }
>
> +/*
> + * Event logs the driver drains: standard logs when native_cxl, DCD when
> + * supported.
> + */
> +static u32 cxl_event_drain_mask(struct pci_host_bridge *host_bridge,
> + struct cxl_memdev_state *mds)
> +{
> + u32 mask = 0;
> +
> + if (host_bridge->native_cxl_error)
> + mask |= CXLDEV_EVENT_STATUS_ALL & ~CXLDEV_EVENT_STATUS_DCD;
> + if (cxl_dcd_supported(mds))
> + mask |= CXLDEV_EVENT_STATUS_DCD;
> + return mask;
> +}
> +
> static irqreturn_t cxl_event_thread(int irq, void *id)
> {
> struct cxl_dev_id *dev_id = id;
> struct cxl_dev_state *cxlds = dev_id->cxlds;
> struct cxl_memdev_state *mds = to_cxl_memdev_state(cxlds);
> + struct pci_host_bridge *host_bridge =
> + pci_find_host_bridge(to_pci_dev(cxlds->dev)->bus);
> + u32 mask = cxl_event_drain_mask(host_bridge, mds);
> u32 status;
>
> do {
> @@ -522,8 +541,8 @@ static irqreturn_t cxl_event_thread(int irq, void *id)
> * ignore the reserved upper 32 bits
> */
> status = readl(cxlds->regs.status + CXLDEV_DEV_EVENT_STATUS_OFFSET);
> - /* Ignore logs unknown to the driver */
> - status &= CXLDEV_EVENT_STATUS_ALL;
> + /* Ignore logs unknown to the driver or owned by BIOS */
> + status &= mask;
> if (!status)
> break;
> cxl_mem_get_event_records(mds, status);
[Severity: High]
Since mask now includes CXLDEV_EVENT_STATUS_DCD, status can have this bit set.
However, cxl_mem_get_event_records() does not appear to handle the DCD event:
drivers/cxl/core/mbox.c:cxl_mem_get_event_records() {
...
if (status & CXLDEV_EVENT_STATUS_FATAL)
cxl_mem_get_records_log(mds, CXL_EVENT_TYPE_FATAL);
if (status & CXLDEV_EVENT_STATUS_FAIL)
cxl_mem_get_records_log(mds, CXL_EVENT_TYPE_FAIL);
if (status & CXLDEV_EVENT_STATUS_WARN)
cxl_mem_get_records_log(mds, CXL_EVENT_TYPE_WARN);
if (status & CXLDEV_EVENT_STATUS_INFO)
cxl_mem_get_records_log(mds, CXL_EVENT_TYPE_INFO);
...
}
Will this fail to clear the DCD event from the device? If the event is never
cleared, will the hardware register retain the bit and cause the while (status)
loop to spin infinitely on the exact same status bit, potentially causing a CPU
soft lockup?
> cond_resched();
> } while (status);
[ ... ]
[Severity: High]
This is a pre-existing issue, but I noticed a potential out-of-bounds read when
fetching event logs:
drivers/cxl/core/mbox.c:cxl_mem_get_records_log() {
...
nr_rec = le16_to_cpu(payload->record_count);
if (!nr_rec)
break;
for (i = 0; i < nr_rec; i++)
__cxl_event_trace_record(cxlmd, type,
&payload->records[i]);
...
}
Does this code validate the hardware-provided nr_rec against the allocated
payload size? If an untrusted or compromised device returns an artificially
large count, could this read out of bounds?
Similarly, when clearing those records:
drivers/cxl/core/mbox.c:cxl_clear_event_record() {
...
for (cnt = 0; cnt < total; cnt++) {
struct cxl_event_record_raw *raw = &get_pl->records[cnt];
struct cxl_event_generic *gen = &raw->event.generic;
payload->handles[i++] = gen->hdr.handle;
...
}
Could this write the out-of-bounds memory values into the clear event payload
sent back to the device, leading to an information leak?
[Severity: High]
This is also a pre-existing issue, but a similar pattern exists in the poison
log handling:
drivers/cxl/core/mbox.c:cxl_mem_get_poison() {
...
for (int i = 0; i < le16_to_cpu(po->count); i++)
trace_cxl_poison(cxlmd, cxlr, &po->record[i],
po->flags, po->overflow_ts,
CXL_POISON_TRACE_LIST);
...
}
Does this validate po->count against the buffer size, or could it result
in an out-of-bounds read if the device returns a large count?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260731084901.1512819-1-anisa.su@samsung.com?part=6
^ permalink raw reply [flat|nested] 13+ messages in thread
* [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
` (5 preceding siblings ...)
2026-07-31 8:48 ` [PATCH v12 6/8] cxl/mem: Configure dynamic capacity interrupts Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
2026-07-31 9:01 ` sashiko-bot
2026-07-31 8:48 ` [PATCH v12 8/8] cxl/core: Enforce partition order/simplify partition calls Anisa Su
7 siblings, 1 reply; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
cxl_dpa_to_region() finds the region from a <DPA, device> tuple.
The search involves finding the device endpoint decoder as well.
Dynamic capacity extent processing uses the endpoint decoder HPA
information to calculate the HPA offset. In addition, well behaved
extents should be contained within an endpoint decoder.
Return the endpoint decoder found to be used in subsequent DCD code.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
Reviewed-by: Jonathan Cameron <jic23@kernel.org>
Reviewed-by: Fan Ni <nifan.cxl@gmail.com>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
Reviewed-by: Li Ming <ming.li@zohomail.com>
Reviewed-by: Alison Schofield <alison.schofield@intel.com>
---
drivers/cxl/core/core.h | 6 ++++--
drivers/cxl/core/mbox.c | 2 +-
drivers/cxl/core/memdev.c | 4 ++--
drivers/cxl/core/region.c | 8 +++++++-
4 files changed, 14 insertions(+), 6 deletions(-)
diff --git a/drivers/cxl/core/core.h b/drivers/cxl/core/core.h
index 07555ae63859..e4bd220faa92 100644
--- a/drivers/cxl/core/core.h
+++ b/drivers/cxl/core/core.h
@@ -47,7 +47,8 @@ int cxl_decoder_detach(struct cxl_region *cxlr,
int cxl_region_init(void);
void cxl_region_exit(void);
int cxl_get_poison_by_endpoint(struct cxl_port *port);
-struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa);
+struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa,
+ struct cxl_endpoint_decoder **cxled);
u64 cxl_dpa_to_hpa(struct cxl_region *cxlr, const struct cxl_memdev *cxlmd,
u64 dpa);
int devm_cxl_add_dax_region(struct cxl_region *cxlr);
@@ -61,7 +62,8 @@ static inline u64 cxl_dpa_to_hpa(struct cxl_region *cxlr,
return ULLONG_MAX;
}
static inline
-struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
+struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa,
+ struct cxl_endpoint_decoder **cxled)
{
return NULL;
}
diff --git a/drivers/cxl/core/mbox.c b/drivers/cxl/core/mbox.c
index a6cdea9f4080..b18ea02ed2e6 100644
--- a/drivers/cxl/core/mbox.c
+++ b/drivers/cxl/core/mbox.c
@@ -969,7 +969,7 @@ void cxl_event_trace_record(struct cxl_memdev *cxlmd,
guard(rwsem_read)(&cxl_rwsem.dpa);
dpa = le64_to_cpu(evt->media_hdr.phys_addr) & CXL_DPA_MASK;
- cxlr = cxl_dpa_to_region(cxlmd, dpa);
+ cxlr = cxl_dpa_to_region(cxlmd, dpa, NULL);
if (cxlr) {
u64 cache_size = cxlr->params.cache_size;
diff --git a/drivers/cxl/core/memdev.c b/drivers/cxl/core/memdev.c
index 33a3d2e7b13a..1565a5cf0f32 100644
--- a/drivers/cxl/core/memdev.c
+++ b/drivers/cxl/core/memdev.c
@@ -317,7 +317,7 @@ int cxl_inject_poison_locked(struct cxl_memdev *cxlmd, u64 dpa)
if (rc)
return rc;
- cxlr = cxl_dpa_to_region(cxlmd, dpa);
+ cxlr = cxl_dpa_to_region(cxlmd, dpa, NULL);
if (cxlr)
dev_warn_once(cxl_mbox->host,
"poison inject dpa:%#llx region: %s\n", dpa,
@@ -386,7 +386,7 @@ int cxl_clear_poison_locked(struct cxl_memdev *cxlmd, u64 dpa)
if (rc)
return rc;
- cxlr = cxl_dpa_to_region(cxlmd, dpa);
+ cxlr = cxl_dpa_to_region(cxlmd, dpa, NULL);
if (cxlr)
dev_warn_once(cxl_mbox->host,
"poison clear dpa:%#llx region: %s\n", dpa,
diff --git a/drivers/cxl/core/region.c b/drivers/cxl/core/region.c
index 1e211542b6b6..ec5e5b7090cf 100644
--- a/drivers/cxl/core/region.c
+++ b/drivers/cxl/core/region.c
@@ -3012,6 +3012,7 @@ int cxl_get_poison_by_endpoint(struct cxl_port *port)
struct cxl_dpa_to_region_context {
struct cxl_region *cxlr;
u64 dpa;
+ struct cxl_endpoint_decoder *cxled;
};
static int __cxl_dpa_to_region(struct device *dev, void *arg)
@@ -3045,11 +3046,13 @@ static int __cxl_dpa_to_region(struct device *dev, void *arg)
dev_name(dev));
ctx->cxlr = cxlr;
+ ctx->cxled = cxled;
return 1;
}
-struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
+struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa,
+ struct cxl_endpoint_decoder **cxled)
{
struct cxl_dpa_to_region_context ctx;
struct cxl_port *port = cxlmd->endpoint;
@@ -3063,6 +3066,9 @@ struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
if (cxl_num_decoders_committed(port))
device_for_each_child(&port->dev, &ctx, __cxl_dpa_to_region);
+ if (cxled)
+ *cxled = ctx.cxled;
+
return ctx.cxlr;
}
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread* Re: [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search
2026-07-31 8:48 ` [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search Anisa Su
@ 2026-07-31 9:01 ` sashiko-bot
0 siblings, 0 replies; 13+ messages in thread
From: sashiko-bot @ 2026-07-31 9:01 UTC (permalink / raw)
To: Anisa Su; +Cc: nvdimm, linux-cxl
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Medium] The `cxled` out-parameter is left uninitialized on early return paths in `cxl_dpa_to_region` and its inline stub, creating a vulnerability for uninitialized variable usage.
--
--- Patch [7]: [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search ---
commit c2b37a56d99a44aee4d4340c7154d9180c620839
Author: Ira Weiny <iweiny@kernel.org>
cxl/core: Return endpoint decoder information from region search
This commit updates cxl_dpa_to_region() to return the endpoint decoder
found during the region search via a new out-parameter. This information
is needed for dynamic capacity extent processing to calculate HPA offsets.
> diff --git a/drivers/cxl/core/core.h b/drivers/cxl/core/core.h
> index 07555ae638594..e4bd220faa92c 100644
> --- a/drivers/cxl/core/core.h
> +++ b/drivers/cxl/core/core.h
[ ... ]
> @@ -61,7 +62,8 @@ static inline u64 cxl_dpa_to_hpa(struct cxl_region *cxlr,
> return ULLONG_MAX;
> }
> static inline
> -struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
> +struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa,
> + struct cxl_endpoint_decoder **cxled)
> {
> return NULL;
> }
[Severity: Medium]
If a future caller passes an uninitialized pointer variable for cxled, will
this static inline stub leave it uninitialized?
Could we set "*cxled = NULL" here when cxled is provided, to prevent callers
from evaluating stack garbage on return?
> diff --git a/drivers/cxl/core/region.c b/drivers/cxl/core/region.c
> index 1e211542b6b64..ec5e5b7090cf2 100644
> --- a/drivers/cxl/core/region.c
> +++ b/drivers/cxl/core/region.c
[ ... ]
> @@ -3045,11 +3046,13 @@ static int __cxl_dpa_to_region(struct device *dev, void *arg)
> dev_name(dev));
>
> ctx->cxlr = cxlr;
> + ctx->cxled = cxled;
>
> return 1;
> }
>
> -struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
> +struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa,
> + struct cxl_endpoint_decoder **cxled)
> {
> struct cxl_dpa_to_region_context ctx;
> struct cxl_port *port = cxlmd->endpoint;
[Severity: Medium]
Similar to the inline stub above, cxl_dpa_to_region() has an early return
path right after these variable declarations:
if (!cxlmd->dev.driver)
return NULL;
If a caller doesn't pre-initialize their cxled variable and hits this early
return, won't cxled remain uninitialized?
While current callers in this patch pass NULL, a future caller expecting
cxled to be populated might evaluate uninitialized stack memory. Should this
early return also initialize *cxled to NULL if cxled is provided?
> @@ -3063,6 +3066,9 @@ struct cxl_region *cxl_dpa_to_region(const struct cxl_memdev *cxlmd, u64 dpa)
> if (cxl_num_decoders_committed(port))
> device_for_each_child(&port->dev, &ctx, __cxl_dpa_to_region);
>
> + if (cxled)
> + *cxled = ctx.cxled;
> +
> return ctx.cxlr;
> }
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260731084901.1512819-1-anisa.su@samsung.com?part=7
^ permalink raw reply [flat|nested] 13+ messages in thread
* [PATCH v12 8/8] cxl/core: Enforce partition order/simplify partition calls
2026-07-31 8:48 [PATCH v12 0/8] DCD Prep Series Anisa Su
` (6 preceding siblings ...)
2026-07-31 8:48 ` [PATCH v12 7/8] cxl/core: Return endpoint decoder information from region search Anisa Su
@ 2026-07-31 8:48 ` Anisa Su
7 siblings, 0 replies; 13+ messages in thread
From: Anisa Su @ 2026-07-31 8:48 UTC (permalink / raw)
To: linux-cxl
Cc: nvdimm, linux-kernel, Dave Jiang, Alison Schofield,
Jonathan Cameron, Fan Ni, Li Ming, Vishal Verma, Davidlohr Bueso,
Ira Weiny, Benjamin Cheatham, Wonjae Lee, Junhee Park, Heesoo Kim,
Anisa Su
From: Ira Weiny <iweiny@kernel.org>
Device partitions have an implied order which is made more complex by
the addition of a dynamic partition
Remove the ram special case information calls in favor of generic calls
with a check ahead of time to ensure the preservation of the implied
partition order.
Signed-off-by: Ira Weiny <iweiny@kernel.org>
Signed-off-by: Anisa Su <anisa.su@samsung.com>
Tested-by: Wonjae Lee <wj28.lee@samsung.com>
Tested-by: Junhee Park <jh9934.park@samsung.com>
Tested-by: Heesoo Kim <habil.kim@samsung.com>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
---
drivers/cxl/core/hdm.c | 11 ++++++++++-
drivers/cxl/core/memdev.c | 32 +++++++++-----------------------
drivers/cxl/cxlmem.h | 9 +++------
drivers/cxl/mem.c | 2 +-
4 files changed, 23 insertions(+), 31 deletions(-)
diff --git a/drivers/cxl/core/hdm.c b/drivers/cxl/core/hdm.c
index 7f63b86887f4..d87dfbcac6a5 100644
--- a/drivers/cxl/core/hdm.c
+++ b/drivers/cxl/core/hdm.c
@@ -457,6 +457,7 @@ static const char *cxl_mode_name(enum cxl_partition_mode mode)
int cxl_dpa_setup(struct cxl_dev_state *cxlds, const struct cxl_dpa_info *info)
{
struct device *dev = cxlds->dev;
+ int i;
guard(rwsem_write)(&cxl_rwsem.dpa);
@@ -469,9 +470,17 @@ int cxl_dpa_setup(struct cxl_dev_state *cxlds, const struct cxl_dpa_info *info)
return 0;
}
+ /* Verify partitions are in expected order. */
+ for (i = 1; i < info->nr_partitions; i++) {
+ if (info->part[i].mode < info->part[i - 1].mode) {
+ dev_err(dev, "Partition order mismatch\n");
+ return -EINVAL;
+ }
+ }
+
cxlds->dpa_res = DEFINE_RES_MEM(0, info->size);
- for (int i = 0; i < info->nr_partitions; i++) {
+ for (i = 0; i < info->nr_partitions; i++) {
const struct cxl_dpa_part_info *part = &info->part[i];
int rc;
diff --git a/drivers/cxl/core/memdev.c b/drivers/cxl/core/memdev.c
index 1565a5cf0f32..136443bffcb8 100644
--- a/drivers/cxl/core/memdev.c
+++ b/drivers/cxl/core/memdev.c
@@ -77,20 +77,12 @@ static ssize_t label_storage_size_show(struct device *dev,
}
static DEVICE_ATTR_RO(label_storage_size);
-static resource_size_t cxl_ram_size(struct cxl_dev_state *cxlds)
-{
- /* Static RAM is only expected at partition 0. */
- if (cxlds->part[0].mode != CXL_PARTMODE_RAM)
- return 0;
- return resource_size(&cxlds->part[0].res);
-}
-
static ssize_t ram_size_show(struct device *dev, struct device_attribute *attr,
char *buf)
{
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
struct cxl_dev_state *cxlds = cxlmd->cxlds;
- unsigned long long len = cxl_ram_size(cxlds);
+ unsigned long long len = cxl_part_size(cxlds, CXL_PARTMODE_RAM);
return sysfs_emit(buf, "%#llx\n", len);
}
@@ -103,7 +95,7 @@ static ssize_t pmem_size_show(struct device *dev, struct device_attribute *attr,
{
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
struct cxl_dev_state *cxlds = cxlmd->cxlds;
- unsigned long long len = cxl_pmem_size(cxlds);
+ unsigned long long len = cxl_part_size(cxlds, CXL_PARTMODE_PMEM);
return sysfs_emit(buf, "%#llx\n", len);
}
@@ -426,10 +418,11 @@ static struct attribute *cxl_memdev_attributes[] = {
NULL,
};
-static struct cxl_dpa_perf *to_pmem_perf(struct cxl_dev_state *cxlds)
+static struct cxl_dpa_perf *part_perf(struct cxl_dev_state *cxlds,
+ enum cxl_partition_mode mode)
{
for (int i = 0; i < cxlds->nr_partitions; i++)
- if (cxlds->part[i].mode == CXL_PARTMODE_PMEM)
+ if (cxlds->part[i].mode == mode)
return &cxlds->part[i].perf;
return NULL;
}
@@ -440,7 +433,7 @@ static ssize_t pmem_qos_class_show(struct device *dev,
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
struct cxl_dev_state *cxlds = cxlmd->cxlds;
- return sysfs_emit(buf, "%d\n", to_pmem_perf(cxlds)->qos_class);
+ return sysfs_emit(buf, "%d\n", part_perf(cxlds, CXL_PARTMODE_PMEM)->qos_class);
}
static struct device_attribute dev_attr_pmem_qos_class =
@@ -452,20 +445,13 @@ static struct attribute *cxl_memdev_pmem_attributes[] = {
NULL,
};
-static struct cxl_dpa_perf *to_ram_perf(struct cxl_dev_state *cxlds)
-{
- if (cxlds->part[0].mode != CXL_PARTMODE_RAM)
- return NULL;
- return &cxlds->part[0].perf;
-}
-
static ssize_t ram_qos_class_show(struct device *dev,
struct device_attribute *attr, char *buf)
{
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
struct cxl_dev_state *cxlds = cxlmd->cxlds;
- return sysfs_emit(buf, "%d\n", to_ram_perf(cxlds)->qos_class);
+ return sysfs_emit(buf, "%d\n", part_perf(cxlds, CXL_PARTMODE_RAM)->qos_class);
}
static struct device_attribute dev_attr_ram_qos_class =
@@ -501,7 +487,7 @@ static umode_t cxl_ram_visible(struct kobject *kobj, struct attribute *a, int n)
{
struct device *dev = kobj_to_dev(kobj);
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
- struct cxl_dpa_perf *perf = to_ram_perf(cxlmd->cxlds);
+ struct cxl_dpa_perf *perf = part_perf(cxlmd->cxlds, CXL_PARTMODE_RAM);
if (a == &dev_attr_ram_qos_class.attr &&
(!perf || perf->qos_class == CXL_QOS_CLASS_INVALID))
@@ -520,7 +506,7 @@ static umode_t cxl_pmem_visible(struct kobject *kobj, struct attribute *a, int n
{
struct device *dev = kobj_to_dev(kobj);
struct cxl_memdev *cxlmd = to_cxl_memdev(dev);
- struct cxl_dpa_perf *perf = to_pmem_perf(cxlmd->cxlds);
+ struct cxl_dpa_perf *perf = part_perf(cxlmd->cxlds, CXL_PARTMODE_PMEM);
if (a == &dev_attr_pmem_qos_class.attr &&
(!perf || perf->qos_class == CXL_QOS_CLASS_INVALID))
diff --git a/drivers/cxl/cxlmem.h b/drivers/cxl/cxlmem.h
index 666e01326dae..77f6417a1de7 100644
--- a/drivers/cxl/cxlmem.h
+++ b/drivers/cxl/cxlmem.h
@@ -412,14 +412,11 @@ struct cxl_security_state {
#define CXL_MAX_DC_PARTITIONS 8
-static inline resource_size_t cxl_pmem_size(struct cxl_dev_state *cxlds)
+static inline resource_size_t cxl_part_size(struct cxl_dev_state *cxlds,
+ enum cxl_partition_mode mode)
{
- /*
- * Static PMEM may be at partition index 0 when there is no static RAM
- * capacity.
- */
for (int i = 0; i < cxlds->nr_partitions; i++)
- if (cxlds->part[i].mode == CXL_PARTMODE_PMEM)
+ if (cxlds->part[i].mode == mode)
return resource_size(&cxlds->part[i].res);
return 0;
}
diff --git a/drivers/cxl/mem.c b/drivers/cxl/mem.c
index 798e5c369cfc..9c4ef401cbad 100644
--- a/drivers/cxl/mem.c
+++ b/drivers/cxl/mem.c
@@ -139,7 +139,7 @@ static int cxl_mem_probe(struct device *dev)
return -ENXIO;
}
- if (cxl_pmem_size(cxlds) && IS_ENABLED(CONFIG_CXL_PMEM)) {
+ if (cxl_part_size(cxlds, CXL_PARTMODE_PMEM) && IS_ENABLED(CONFIG_CXL_PMEM)) {
rc = devm_cxl_add_nvdimm(dev, parent_port, cxlmd);
if (rc) {
if (rc == -ENODEV)
--
2.43.0
^ permalink raw reply related [flat|nested] 13+ messages in thread