Linux IOMMU Development
 help / color / mirror / Atom feed
From: Binbin Wu <binbin.wu@linux.intel.com>
To: Jason Gunthorpe <jgg@nvidia.com>
Cc: Anthony Krowiak <akrowiak@linux.ibm.com>,
	Alex Williamson <alex.williamson@redhat.com>,
	Bagas Sanjaya <bagasdotme@gmail.com>,
	Lu Baolu <baolu.lu@linux.intel.com>,
	Chaitanya Kulkarni <chaitanyak@nvidia.com>,
	Cornelia Huck <cohuck@redhat.com>,
	Jonathan Corbet <corbet@lwn.net>,
	Daniel Jordan <daniel.m.jordan@oracle.com>,
	David Gibson <david@gibson.dropbear.id.au>,
	Eric Auger <eric.auger@redhat.com>,
	Eric Farman <farman@linux.ibm.com>,
	iommu@lists.linux.dev, Jason Wang <jasowang@redhat.com>,
	Jean-Philippe Brucker <jean-philippe@linaro.org>,
	Jason Herne <jjherne@linux.ibm.com>,
	Joao Martins <joao.m.martins@oracle.com>,
	Kevin Tian <kevin.tian@intel.com>,
	kvm@vger.kernel.org, Lixiao Yang <lixiao.yang@intel.com>,
	Matthew Rosato <mjrosato@linux.ibm.com>,
	"Michael S. Tsirkin" <mst@redhat.com>,
	Nicolin Chen <nicolinc@nvidia.com>,
	Halil Pasic <pasic@linux.ibm.com>,
	Niklas Schnelle <schnelle@linux.ibm.com>,
	Shameerali Kolothum Thodi <shameerali.kolothum.thodi@huawei.com>,
	Yi Liu <yi.l.liu@intel.com>, Yu He <yu.he@intel.com>,
	Keqian Zhu <zhukeqian1@huawei.com>
Subject: Re: [PATCH v6 08/19] iommufd: PFN handling for iopt_pages
Date: Mon, 5 Dec 2022 23:58:41 +0800	[thread overview]
Message-ID: <7403f46e-90ff-9761-0b92-8dc8c163ebf8@linux.intel.com> (raw)
In-Reply-To: <8-v6-a196d26f289e+11787-iommufd_jgg@nvidia.com>


On 11/30/2022 4:29 AM, Jason Gunthorpe wrote:
> +#ifndef __IO_PAGETABLE_H
> +#define __IO_PAGETABLE_H
> +
> +#include <linux/interval_tree.h>
> +#include <linux/mutex.h>
> +#include <linux/kref.h>
> +#include <linux/xarray.h>
> +
> +#include "iommufd_private.h"
> +
> +struct iommu_domain;
> +
> +/*
> + * Each io_pagetable is composed of intervals of areas which cover regions of
> + * the iova that are backed by something. iova not covered by areas is not
> + * populated in the page table. Each area is fully populated with pages.
> + *
> + * iovas are in byte units, but must be iopt->iova_alignment aligned.
> + *
> + * pages can be NULL, this means some other thread is still working on setting
> + * up or tearing down the area. When observed under the write side of the
> + * domain_rwsem a NULL pages must mean the area is still being setup and no
> + * domains are filled.
> + *
> + * storage_domain points at an arbitrary iommu_domain that is holding the PFNs
> + * for this area. It is locked by the pages->mutex. This simplifies the locking
> + * as the pages code can rely on the storage_domain without having to get the
> + * iopt->domains_rwsem.
> + *
> + * The io_pagetable::iova_rwsem protects node
> + * The iopt_pages::mutex protects pages_node
> + * iopt and immu_prot

typo, immu_prot -> iommu_prot


>   are immutable

> +
> +/*
> + * Carry means we carry a portion of the final hugepage over to the front of the
> + * batch
> + */
> +static void batch_clear_carry(struct pfn_batch *batch, unsigned int keep_pfns)
> +{
> +	if (!keep_pfns)
> +		return batch_clear(batch);
> +
> +	batch->total_pfns = keep_pfns;
> +	batch->npfns[0] = keep_pfns;
> +	batch->pfns[0] = batch->pfns[batch->end - 1] +
> +			 (batch->npfns[batch->end - 1] - keep_pfns);

The range of the skip_pfns is checked in batch_skip_carry, should 
keep_pfns also be checked in this function?


> +	batch->end = 0;
> +}
> +
> +static void batch_skip_carry(struct pfn_batch *batch, unsigned int skip_pfns)
> +{
> +	if (!batch->total_pfns)
> +		return;
> +	skip_pfns = min(batch->total_pfns, skip_pfns);

Should use batch->npfns[0] instead of batch->total_pfns?



> +	batch->pfns[0] += skip_pfns;
> +	batch->npfns[0] -= skip_pfns;
> +	batch->total_pfns -= skip_pfns;
> +}
> +
> +static int __batch_init(struct pfn_batch *batch, size_t max_pages, void *backup,
> +			size_t backup_len)
> +{
> +	const size_t elmsz = sizeof(*batch->pfns) + sizeof(*batch->npfns);
> +	size_t size = max_pages * elmsz;
> +
> +	batch->pfns = temp_kmalloc(&size, backup, backup_len);
> +	if (!batch->pfns)
> +		return -ENOMEM;
> +	batch->array_size = size / elmsz;
> +	batch->npfns = (u32 *)(batch->pfns + batch->array_size);
> +	batch_clear(batch);
> +	return 0;
> +}
> +
> +static int batch_init(struct pfn_batch *batch, size_t max_pages)
> +{
> +	return __batch_init(batch, max_pages, NULL, 0);
> +}
> +
> +static void batch_init_backup(struct pfn_batch *batch, size_t max_pages,
> +			      void *backup, size_t backup_len)
> +{
> +	__batch_init(batch, max_pages, backup, backup_len);
> +}
> +
> +static void batch_destroy(struct pfn_batch *batch, void *backup)
> +{
> +	if (batch->pfns != backup)
> +		kfree(batch->pfns);
> +}
> +
> +/* true if the pfn could be added, false otherwise */

It is not accurate to use "could be" here because returning ture means 
the pfn has been added.


> +static bool batch_add_pfn(struct pfn_batch *batch, unsigned long pfn)
> +{
> +	const unsigned int MAX_NPFNS = type_max(typeof(*batch->npfns));
> +
> +	if (batch->end &&
> +	    pfn == batch->pfns[batch->end - 1] + batch->npfns[batch->end - 1] &&
> +	    batch->npfns[batch->end - 1] != MAX_NPFNS) {
> +		batch->npfns[batch->end - 1]++;
> +		batch->total_pfns++;
> +		return true;
> +	}
> +	if (batch->end == batch->array_size)
> +		return false;
> +	batch->total_pfns++;
> +	batch->pfns[batch->end] = pfn;
> +	batch->npfns[batch->end] = 1;
> +	batch->end++;
> +	return true;
> +}
> +
> +/*
> + * Fill the batch with pfns from the domain. When the batch is full, or it
> + * reaches last_index, the function will return. The caller should use
> + * batch->total_pfns to determine the starting point for the next iteration.
> + */
> +static void batch_from_domain(struct pfn_batch *batch,
> +			      struct iommu_domain *domain,
> +			      struct iopt_area *area, unsigned long start_index,
> +			      unsigned long last_index)
> +{
> +	unsigned int page_offset = 0;
> +	unsigned long iova;
> +	phys_addr_t phys;
> +
> +	iova = iopt_area_index_to_iova(area, start_index);
> +	if (start_index == iopt_area_index(area))
> +		page_offset = area->page_offset;
> +	while (start_index <= last_index) {
> +		/*
> +		 * This is pretty slow, it would be nice to get the page size
> +		 * back from the driver, or have the driver directly fill the
> +		 * batch.
> +		 */
> +		phys = iommu_iova_to_phys(domain, iova) - page_offset;

seems no need to handle the page_offset, since PHYS_PFN(phys) is used in 
batch_add_pfn below?



> +		if (!batch_add_pfn(batch, PHYS_PFN(phys)))
> +			return;
> +		iova += PAGE_SIZE - page_offset;
> +		page_offset = 0;
> +		start_index++;
> +	}
> +}
> +
> +static struct page **raw_pages_from_domain(struct iommu_domain *domain,
> +					   struct iopt_area *area,
> +					   unsigned long start_index,
> +					   unsigned long last_index,
> +					   struct page **out_pages)
> +{
> +	unsigned int page_offset = 0;
> +	unsigned long iova;
> +	phys_addr_t phys;
> +
> +	iova = iopt_area_index_to_iova(area, start_index);
> +	if (start_index == iopt_area_index(area))
> +		page_offset = area->page_offset;
> +	while (start_index <= last_index) {
> +		phys = iommu_iova_to_phys(domain, iova) - page_offset;

ditto, since only PHYS_PFN(phys) is actually used, no need to handle 
page_offset



> +		*(out_pages++) = pfn_to_page(PHYS_PFN(phys));
> +		iova += PAGE_SIZE - page_offset;
> +		page_offset = 0;
> +		start_index++;
> +	}
> +	return out_pages;
> +}
> +
> +/* Continues reading a domain until we reach a discontiguity

typo, discontiguity -> discontinuity


>   in the pfns. */
> +static void batch_from_domain_continue(struct pfn_batch *batch,
> +				       struct iommu_domain *domain,
> +				       struct iopt_area *area,
> +				       unsigned long start_index,
> +				       unsigned long last_index)
> +{
> +	unsigned int array_size = batch->array_size;
> +
> +	batch->array_size = batch->end;
> +	batch_from_domain(batch, domain, area, start_index, last_index);
> +	batch->array_size = array_size;
> +}
> +

BTW, this is a quite big patch, maybe break into smaller ones?



  reply	other threads:[~2022-12-05 15:58 UTC|newest]

Thread overview: 39+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2022-11-29 20:29 [PATCH v6 00/19] IOMMUFD Generic interface Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 01/19] iommu: Add IOMMU_CAP_ENFORCE_CACHE_COHERENCY Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 02/19] iommu: Add device-centric DMA ownership interfaces Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 03/19] interval-tree: Add a utility to iterate over spans in an interval tree Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 04/19] scripts/kernel-doc: support EXPORT_SYMBOL_NS_GPL() with -export Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 05/19] iommufd: Document overview of iommufd Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 06/19] iommufd: File descriptor, context, kconfig and makefiles Jason Gunthorpe
2022-11-30 14:02   ` Eric Auger
2022-12-04 10:58   ` Binbin Wu
2022-11-29 20:29 ` [PATCH v6 07/19] kernel/user: Allow user::locked_vm to be usable for iommufd Jason Gunthorpe
2022-11-29 20:42   ` Michael S. Tsirkin
2022-11-29 20:48     ` Jason Gunthorpe
2022-11-29 21:10       ` Michael S. Tsirkin
2022-11-29 20:29 ` [PATCH v6 08/19] iommufd: PFN handling for iopt_pages Jason Gunthorpe
2022-12-05 15:58   ` Binbin Wu [this message]
2022-12-06 20:53     ` Jason Gunthorpe
2022-12-06 12:36   ` Binbin Wu
2022-12-06 20:57     ` Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 09/19] iommufd: Algorithms for PFN storage Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 10/19] iommufd: Data structure to provide IOVA to PFN mapping Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 11/19] iommufd: IOCTLs for the io_pagetable Jason Gunthorpe
2022-11-30 14:04   ` Eric Auger
2022-11-29 20:29 ` [PATCH v6 12/19] iommufd: Add a HW pagetable object Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 13/19] iommufd: Add kAPI toward external drivers for physical devices Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 14/19] iommufd: Add kAPI toward external drivers for kernel access Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 15/19] iommufd: vfio container FD ioctl compatibility Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 16/19] iommufd: Add kernel support for testing iommufd Jason Gunthorpe
2024-04-22  7:27   ` Geert Uytterhoeven
2024-04-22 11:54     ` Jason Gunthorpe
2024-04-22 12:48       ` Geert Uytterhoeven
2024-04-22 12:50         ` Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 17/19] iommufd: Add some fault injection points Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 18/19] iommufd: Add additional invariant assertions Jason Gunthorpe
2022-11-29 20:29 ` [PATCH v6 19/19] iommufd: Add a selftest Jason Gunthorpe
2022-11-30  7:14   ` Yi Liu
2022-11-30 13:51     ` Jason Gunthorpe
2022-11-30 17:18       ` Eric Auger
2022-12-01  0:13         ` Jason Gunthorpe
2022-12-01  4:59         ` Yi Liu

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=7403f46e-90ff-9761-0b92-8dc8c163ebf8@linux.intel.com \
    --to=binbin.wu@linux.intel.com \
    --cc=akrowiak@linux.ibm.com \
    --cc=alex.williamson@redhat.com \
    --cc=bagasdotme@gmail.com \
    --cc=baolu.lu@linux.intel.com \
    --cc=chaitanyak@nvidia.com \
    --cc=cohuck@redhat.com \
    --cc=corbet@lwn.net \
    --cc=daniel.m.jordan@oracle.com \
    --cc=david@gibson.dropbear.id.au \
    --cc=eric.auger@redhat.com \
    --cc=farman@linux.ibm.com \
    --cc=iommu@lists.linux.dev \
    --cc=jasowang@redhat.com \
    --cc=jean-philippe@linaro.org \
    --cc=jgg@nvidia.com \
    --cc=jjherne@linux.ibm.com \
    --cc=joao.m.martins@oracle.com \
    --cc=kevin.tian@intel.com \
    --cc=kvm@vger.kernel.org \
    --cc=lixiao.yang@intel.com \
    --cc=mjrosato@linux.ibm.com \
    --cc=mst@redhat.com \
    --cc=nicolinc@nvidia.com \
    --cc=pasic@linux.ibm.com \
    --cc=schnelle@linux.ibm.com \
    --cc=shameerali.kolothum.thodi@huawei.com \
    --cc=yi.l.liu@intel.com \
    --cc=yu.he@intel.com \
    --cc=zhukeqian1@huawei.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox