From: Zhen Ni <zhen.ni@easystack.cn>
To: ast@kernel.org, daniel@iogearbox.net, andrii@kernel.org,
eddyz87@gmail.com, memxor@gmail.com, martin.lau@linux.dev,
song@kernel.org, yonghong.song@linux.dev, jolsa@kernel.org,
emil@etsalapatis.com, ihor.solodrai@linux.dev,
akpm@linux-foundation.org, vbabka@kernel.org, surenb@google.com,
mhocko@suse.com, brendan.jackman@linux.dev, hannes@cmpxchg.org,
ziy@nvidia.com, shuah@kernel.org
Cc: bpf@vger.kernel.org, linux-mm@kvack.org,
linux-kselftest@vger.kernel.org, linux-kernel@vger.kernel.org,
Zhen Ni <zhen.ni@easystack.cn>
Subject: [PATCH 1/6] mm/page_owner: extract page_owner_next_eligible() from read loop
Date: Fri, 9 Oct 2026 19:23:03 +0800 [thread overview]
Message-ID: <20261009112308.240769-2-zhen.ni@easystack.cn> (raw)
In-Reply-To: <20261009112308.240769-1-zhen.ni@easystack.cn>
Move the page scanning logic out of read_page_owner() into a dedicated
helper, in preparation for using it in the page_owner bpf_iter target.
The helper advances a struct page_owner_scan cursor to the next page
eligible for page_owner output and returns true on a hit. On a hit it
takes a snapshot of the page_owner record inside the page_ext RCU
window.
No change to the output is intended. handle is read from the snapshot,
taken before the non-zero check, so the validated and reported handle
are the same value.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
mm/page_owner.c | 105 ++++++++++++++++++++++++++++++------------------
1 file changed, 66 insertions(+), 39 deletions(-)
diff --git a/mm/page_owner.c b/mm/page_owner.c
index fbbda7ba914b..bafe474f3360 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -36,6 +36,16 @@ struct page_owner {
pid_t free_tgid;
};
+/*
+ * Cursor and per-page result of page_owner_next_eligible().
+ */
+struct page_owner_scan {
+ /* resume point */
+ unsigned long pfn;
+ struct page *page;
+ struct page_owner po_snap;
+};
+
struct stack {
struct stack_record *stack_record;
struct stack *next;
@@ -725,37 +735,24 @@ void __dump_page_owner(const struct page *page)
page_ext_put(page_ext);
}
-static ssize_t
-read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
+/*
+ * Advance @scan to the next page eligible for page_owner output.
+ * On a hit, fill scan->pfn/page/po_snap and return true; the caller
+ * must advance scan->pfn past the hit before calling again to resume
+ * the scan.
+ */
+static bool page_owner_next_eligible(struct page_owner_scan *scan)
{
- unsigned long pfn;
- struct page *page;
- struct page_ext *page_ext;
- struct page_owner *page_owner;
- depot_stack_handle_t handle;
- struct page_owner_filter_state *state = file->private_data;
+ unsigned long pfn = scan->pfn;
- if (!static_branch_unlikely(&page_owner_inited))
- return -EINVAL;
-
- page = NULL;
- if (*ppos == 0)
- pfn = min_low_pfn;
- else
- pfn = *ppos;
/* Find a valid PFN or the start of a MAX_ORDER_NR_PAGES area */
while (!pfn_valid(pfn) && (pfn & (MAX_ORDER_NR_PAGES - 1)) != 0)
pfn++;
- /* Find an allocated page */
for (; pfn < max_pfn; pfn++) {
- /*
- * This temporary page_owner is required so
- * that we can avoid the context switches while holding
- * the rcu lock and copying the page owner information to
- * user through copy_to_user() or GFP_KERNEL allocations.
- */
- struct page_owner page_owner_tmp;
+ struct page_owner *page_owner;
+ struct page_ext *page_ext;
+ struct page *page;
/*
* If the new page is in a new MAX_ORDER_NR_PAGES area,
@@ -798,37 +795,67 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
goto ext_put_continue;
/*
- * Access to page_ext->handle isn't synchronous so we should
- * be careful to access it.
+ * Take a snapshot of the page_owner record inside the RCU
+ * window, then validate it, so that print_page_owner()
+ * reads the checked handle from the same snapshot.
*/
- handle = READ_ONCE(page_owner->handle);
- if (!handle)
- goto ext_put_continue;
+ scan->pfn = pfn;
+ scan->page = page;
+ scan->po_snap = *page_owner;
+ page_ext_put(page_ext);
+ if (!scan->po_snap.handle)
+ continue;
+
+ return true;
+
+ext_put_continue:
+ page_ext_put(page_ext);
+ cond_resched();
+ }
+
+ return false;
+}
+
+static ssize_t
+read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
+{
+ struct page_owner_scan scan;
+ struct page_owner_filter_state *state = file->private_data;
+
+ if (!static_branch_unlikely(&page_owner_inited))
+ return -EINVAL;
+
+ if (*ppos == 0)
+ scan.pfn = min_low_pfn;
+ else
+ scan.pfn = *ppos;
+
+ /* Find an allocated page */
+ while (page_owner_next_eligible(&scan)) {
if (state->nid_filter_enabled) {
int nid;
- memdesc_flags_t page_flags = READ_ONCE(page->flags);
+ memdesc_flags_t page_flags =
+ READ_ONCE(scan.page->flags);
/*
* Bypass PF_POISONED_CHECK() in page_to_nid() to avoid
* VM_BUG_ON when accessing poisoned pages.
*/
if (page_flags.f == PAGE_POISON_PATTERN)
- goto ext_put_continue;
+ goto skip_continue;
nid = memdesc_nid(&page_flags);
if (!node_isset(nid, state->nid_filter))
- goto ext_put_continue;
+ goto skip_continue;
}
/* Record the next PFN to read in the file offset */
- *ppos = pfn + 1;
+ *ppos = scan.pfn + 1;
- page_owner_tmp = *page_owner;
- page_ext_put(page_ext);
- return print_page_owner(buf, count, pfn, page,
- &page_owner_tmp, handle, state);
-ext_put_continue:
- page_ext_put(page_ext);
+ return print_page_owner(buf, count, scan.pfn, scan.page,
+ &scan.po_snap, scan.po_snap.handle, state);
+skip_continue:
+ scan.pfn++;
cond_resched();
}
--
2.20.1
next prev parent reply other threads:[~2026-10-09 11:23 UTC|newest]
Thread overview: 8+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
2026-10-09 11:23 ` Zhen Ni [this message]
2026-10-09 11:23 ` [PATCH 2/6] mm/page_owner: add bpf_iter target "page_owner" Zhen Ni
2026-10-09 11:23 ` [PATCH 3/6] mm/page_owner: add open-coded page_owner iterator Zhen Ni
2026-10-09 11:23 ` [PATCH 4/6] mm/page_owner: add bpf_page_owner_get_nid() kfunc Zhen Ni
2026-10-09 11:23 ` [PATCH 5/6] mm/page_owner: add bpf_page_owner_get_memcg_info() kfunc Zhen Ni
2026-10-09 11:23 ` [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target Zhen Ni
2026-10-09 13:48 ` bot+bpf-ci
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261009112308.240769-2-zhen.ni@easystack.cn \
--to=zhen.ni@easystack.cn \
--cc=akpm@linux-foundation.org \
--cc=andrii@kernel.org \
--cc=ast@kernel.org \
--cc=bpf@vger.kernel.org \
--cc=brendan.jackman@linux.dev \
--cc=daniel@iogearbox.net \
--cc=eddyz87@gmail.com \
--cc=emil@etsalapatis.com \
--cc=hannes@cmpxchg.org \
--cc=ihor.solodrai@linux.dev \
--cc=jolsa@kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=martin.lau@linux.dev \
--cc=memxor@gmail.com \
--cc=mhocko@suse.com \
--cc=shuah@kernel.org \
--cc=song@kernel.org \
--cc=surenb@google.com \
--cc=vbabka@kernel.org \
--cc=yonghong.song@linux.dev \
--cc=ziy@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox