Groups | Search | Server Info | Keyboard shortcuts | Login | Register [http] [https] [nntp] [nntps]


Groups > linux.kernel > #1673117 > unrolled thread

Re: [RFC 8/9] iommu/intel-svm: notify page request to guest

Started byAlex Williamson <alex.williamson@redhat.com>
First post2017-06-23 01:00 +0200
Last post2017-06-23 23:40 +0200
Articles 4 — 2 participants

Back to article view | Back to linux.kernel

This discussion starts older than the indexed window; earlier articles aren't shown. The article labeled Started by below is the oldest one visible, not the original post.


Contents

  Re: [RFC 8/9] iommu/intel-svm: notify page request to guest Alex Williamson <alex.williamson@redhat.com> - 2017-06-23 01:00 +0200
    Re: [RFC 8/9] iommu/intel-svm: notify page request to guest Jacob Pan <jacob.jun.pan@linux.intel.com> - 2017-06-23 22:20 +0200
      Re: [RFC 8/9] iommu/intel-svm: notify page request to guest Alex Williamson <alex.williamson@redhat.com> - 2017-06-23 22:40 +0200
        Re: [RFC 8/9] iommu/intel-svm: notify page request to guest Jacob Pan <jacob.jun.pan@linux.intel.com> - 2017-06-23 23:40 +0200

#1673117 — Re: [RFC 8/9] iommu/intel-svm: notify page request to guest

FromAlex Williamson <alex.williamson@redhat.com>
Date2017-06-23 01:00 +0200
SubjectRe: [RFC 8/9] iommu/intel-svm: notify page request to guest
Message-ID<tVjeV-2og-3@gated-at.bofh.it>
On Wed, 14 Jun 2017 15:23:02 -0700
Jacob Pan <jacob.jun.pan@linux.intel.com> wrote:

> If the source device of a page request has its PASID table pointer
> bond to a guest, the first level page tables are owned by the guest.
> In this case, we shall let guest OS to manage page fault.
> 
> This patch uses the IOMMU fault notification API to send notifications,
> possibly via VFIO, to the guest OS. Once guest pages are fault in, guest
> will issue page response which will be passed down via the invalidation
> passdown APIs.
> 
> Signed-off-by: Jacob Pan <jacob.jun.pan@linux.intel.com>
> Signed-off-by: Ashok Raj <ashok.raj@intel.com>
> ---
>  drivers/iommu/intel-svm.c | 81 ++++++++++++++++++++++++++++++++++++++++++++++-
>  1 file changed, 80 insertions(+), 1 deletion(-)
> 
> diff --git a/drivers/iommu/intel-svm.c b/drivers/iommu/intel-svm.c
> index 23c4276..d1d2d23 100644
> --- a/drivers/iommu/intel-svm.c
> +++ b/drivers/iommu/intel-svm.c
> @@ -525,6 +525,80 @@ static bool access_error(struct vm_area_struct *vma, struct page_req_dsc *req)
>  	return (requested & ~vma->vm_flags) != 0;
>  }
>  
> +static int prq_to_iommu_prot(struct page_req_dsc *req)
> +{
> +	int prot = 0;
> +
> +	if (req->rd_req)
> +		prot |= IOMMU_READ;
> +	if (req->wr_req)
> +		prot |= IOMMU_WRITE;
> +	if (req->exe_req)
> +		prot |= IOMMU_EXEC;
> +	if (req->priv_req)
> +		prot |= IOMMU_PRIV;
> +
> +	return prot;
> +}
> +
> +static int intel_svm_prq_notify(struct device *dev, struct page_req_dsc *desc)
> +{
> +	int ret = 0;
> +	struct iommu_fault_event *event;
> +	struct pci_dev *pdev;
> +	struct device_domain_info *info;
> +	unsigned long buf_offset;
> +
> +	/**
> +	 * If caller does not provide struct device, this is the case where
> +	 * guest PASID table is bond to the device. So we need to retrieve
> +	 * struct device from the page request deescriptor then proceed.
> +	 */
> +	if (!dev) {
> +		pdev = pci_get_bus_and_slot(desc->bus, desc->devfn);
> +		if (!pdev) {
> +			pr_err("No PCI device found for PRQ [%02x:%02x.%d]\n",
> +				desc->bus, PCI_SLOT(desc->devfn),
> +				PCI_FUNC(desc->devfn));
> +			return -ENODEV;
> +		}
> +		/**
> +		 * Make sure PASID table pointer is bond to guest, if yes notify
> +		 * handler in the guest, e.g. via VFIO.
> +		 */
> +		info = pdev->dev.archdata.iommu;
> +		if (!info || !info->pasid_tbl_bond) {
> +			pr_debug("PRQ device pasid table not bond.\n");

I can "bond" two things together, they are then "bound".

> +			return -EINVAL;
> +		}
> +		dev = &pdev->dev;

Leaks pdev reference.  Both normal and error path.

> +	}
> +
> +	pr_debug("Notify PRQ device [%02x:%02x.%d]\n",
> +		desc->bus, PCI_SLOT(desc->devfn),
> +		PCI_FUNC(desc->devfn));
> +	event = kzalloc(sizeof(*event) + sizeof(*desc), GFP_KERNEL);
> +	if (!event)
> +		return -ENOMEM;
> +
> +	get_device(dev);
> +	/* Fill in event data for device specific processing */
> +	event->dev = dev;
> +	buf_offset = offsetofend(struct iommu_fault_event, length);
> +	memcpy(buf_offset + event, desc, sizeof(*desc));
> +	event->addr = desc->addr;
> +	event->pasid = desc->pasid;
> +	event->prot = prq_to_iommu_prot(desc);
> +	event->length = sizeof(*desc);
> +	event->flags = IOMMU_FAULT_PAGE_REQ;
> +
> +	ret = iommu_fault_notifier_call_chain(event);
> +	put_device(dev);
> +	kfree(event);
> +
> +	return ret;
> +}
> +
>  static irqreturn_t prq_event_thread(int irq, void *d)
>  {
>  	struct intel_iommu *iommu = d;
> @@ -548,7 +622,12 @@ static irqreturn_t prq_event_thread(int irq, void *d)
>  		handled = 1;
>  
>  		req = &iommu->prq[head / sizeof(*req)];
> -
> +		/**
> +		 * If prq is to be handled outside iommu driver via receiver of
> +		 * the fault notifiers, we skip the page response here.
> +		 */
> +		if (!intel_svm_prq_notify(NULL, req))
> +			continue;
>  		result = QI_RESP_FAILURE;
>  		address = (u64)req->addr << VTD_PAGE_SHIFT;
>  		if (!req->pasid_present) {

[toc] | [next] | [standalone]


#1673815

FromJacob Pan <jacob.jun.pan@linux.intel.com>
Date2017-06-23 22:20 +0200
Message-ID<tVDdD-6Lb-3@gated-at.bofh.it>
In reply to#1673117
On Thu, 22 Jun 2017 16:53:58 -0600
Alex Williamson <alex.williamson@redhat.com> wrote:

> On Wed, 14 Jun 2017 15:23:02 -0700
> Jacob Pan <jacob.jun.pan@linux.intel.com> wrote:
> 
> > If the source device of a page request has its PASID table pointer
> > bond to a guest, the first level page tables are owned by the guest.
> > In this case, we shall let guest OS to manage page fault.
> > 
> > This patch uses the IOMMU fault notification API to send
> > notifications, possibly via VFIO, to the guest OS. Once guest pages
> > are fault in, guest will issue page response which will be passed
> > down via the invalidation passdown APIs.
> > 
> > Signed-off-by: Jacob Pan <jacob.jun.pan@linux.intel.com>
> > Signed-off-by: Ashok Raj <ashok.raj@intel.com>
> > ---
> >  drivers/iommu/intel-svm.c | 81
> > ++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 80
> > insertions(+), 1 deletion(-)
> > 
> > diff --git a/drivers/iommu/intel-svm.c b/drivers/iommu/intel-svm.c
> > index 23c4276..d1d2d23 100644
> > --- a/drivers/iommu/intel-svm.c
> > +++ b/drivers/iommu/intel-svm.c
> > @@ -525,6 +525,80 @@ static bool access_error(struct vm_area_struct
> > *vma, struct page_req_dsc *req) return (requested &
> > ~vma->vm_flags) != 0; }
> >  
> > +static int prq_to_iommu_prot(struct page_req_dsc *req)
> > +{
> > +	int prot = 0;
> > +
> > +	if (req->rd_req)
> > +		prot |= IOMMU_READ;
> > +	if (req->wr_req)
> > +		prot |= IOMMU_WRITE;
> > +	if (req->exe_req)
> > +		prot |= IOMMU_EXEC;
> > +	if (req->priv_req)
> > +		prot |= IOMMU_PRIV;
> > +
> > +	return prot;
> > +}
> > +
> > +static int intel_svm_prq_notify(struct device *dev, struct
> > page_req_dsc *desc) +{
> > +	int ret = 0;
> > +	struct iommu_fault_event *event;
> > +	struct pci_dev *pdev;
> > +	struct device_domain_info *info;
> > +	unsigned long buf_offset;
> > +
> > +	/**
> > +	 * If caller does not provide struct device, this is the
> > case where
> > +	 * guest PASID table is bond to the device. So we need to
> > retrieve
> > +	 * struct device from the page request deescriptor then
> > proceed.
> > +	 */
> > +	if (!dev) {
> > +		pdev = pci_get_bus_and_slot(desc->bus,
> > desc->devfn);
> > +		if (!pdev) {
> > +			pr_err("No PCI device found for PRQ
> > [%02x:%02x.%d]\n",
> > +				desc->bus, PCI_SLOT(desc->devfn),
> > +				PCI_FUNC(desc->devfn));
> > +			return -ENODEV;
> > +		}
> > +		/**
> > +		 * Make sure PASID table pointer is bond to guest,
> > if yes notify
> > +		 * handler in the guest, e.g. via VFIO.
> > +		 */
> > +		info = pdev->dev.archdata.iommu;
> > +		if (!info || !info->pasid_tbl_bond) {
> > +			pr_debug("PRQ device pasid table not
> > bond.\n");  
> 
> I can "bond" two things together, they are then "bound".
> 
will fix that :)
> > +			return -EINVAL;
> > +		}
> > +		dev = &pdev->dev;  
> 
> Leaks pdev reference.  Both normal and error path.
> 
I guess you are referring to ref count in pci_get_bus_and_slot()? I did
look at the code, it does not seem to do the count increment. Perhaps
the comment is stale?

 * pointer to its data structure.  The caller must decrement the
 * reference count by calling pci_dev_put().  If no device is found,
 * %NULL is returned.
 */
struct pci_dev *pci_get_domain_bus_and_slot(int domain, unsigned int bus,
					    unsigned int devfn)
{
	struct pci_dev *dev = NULL;

	for_each_pci_dev(dev) {
		if (pci_domain_nr(dev->bus) == domain &&
		    (dev->bus->number == bus && dev->devfn == devfn))
			return dev;
	}
	return NULL;
}
EXPORT_SYMBOL(pci_get_domain_bus_and_slot);



> > +	}
> > +
> > +	pr_debug("Notify PRQ device [%02x:%02x.%d]\n",
> > +		desc->bus, PCI_SLOT(desc->devfn),
> > +		PCI_FUNC(desc->devfn));
> > +	event = kzalloc(sizeof(*event) + sizeof(*desc),
> > GFP_KERNEL);
> > +	if (!event)
> > +		return -ENOMEM;
> > +
> > +	get_device(dev);
> > +	/* Fill in event data for device specific processing */
> > +	event->dev = dev;
> > +	buf_offset = offsetofend(struct iommu_fault_event, length);
> > +	memcpy(buf_offset + event, desc, sizeof(*desc));
> > +	event->addr = desc->addr;
> > +	event->pasid = desc->pasid;
> > +	event->prot = prq_to_iommu_prot(desc);
> > +	event->length = sizeof(*desc);
> > +	event->flags = IOMMU_FAULT_PAGE_REQ;
> > +
> > +	ret = iommu_fault_notifier_call_chain(event);
> > +	put_device(dev);
> > +	kfree(event);
> > +
> > +	return ret;
> > +}
> > +
> >  static irqreturn_t prq_event_thread(int irq, void *d)
> >  {
> >  	struct intel_iommu *iommu = d;
> > @@ -548,7 +622,12 @@ static irqreturn_t prq_event_thread(int irq,
> > void *d) handled = 1;
> >  
> >  		req = &iommu->prq[head / sizeof(*req)];
> > -
> > +		/**
> > +		 * If prq is to be handled outside iommu driver
> > via receiver of
> > +		 * the fault notifiers, we skip the page response
> > here.
> > +		 */
> > +		if (!intel_svm_prq_notify(NULL, req))
> > +			continue;
> >  		result = QI_RESP_FAILURE;
> >  		address = (u64)req->addr << VTD_PAGE_SHIFT;
> >  		if (!req->pasid_present) {  
> 

[Jacob Pan]

[toc] | [prev] | [next] | [standalone]


#1673855

FromAlex Williamson <alex.williamson@redhat.com>
Date2017-06-23 22:40 +0200
Message-ID<tVDwZ-6Sb-15@gated-at.bofh.it>
In reply to#1673815
On Fri, 23 Jun 2017 13:16:29 -0700
Jacob Pan <jacob.jun.pan@linux.intel.com> wrote:

> On Thu, 22 Jun 2017 16:53:58 -0600
> Alex Williamson <alex.williamson@redhat.com> wrote:
> 
> > On Wed, 14 Jun 2017 15:23:02 -0700
> > Jacob Pan <jacob.jun.pan@linux.intel.com> wrote:
> >   
> > > If the source device of a page request has its PASID table pointer
> > > bond to a guest, the first level page tables are owned by the guest.
> > > In this case, we shall let guest OS to manage page fault.
> > > 
> > > This patch uses the IOMMU fault notification API to send
> > > notifications, possibly via VFIO, to the guest OS. Once guest pages
> > > are fault in, guest will issue page response which will be passed
> > > down via the invalidation passdown APIs.
> > > 
> > > Signed-off-by: Jacob Pan <jacob.jun.pan@linux.intel.com>
> > > Signed-off-by: Ashok Raj <ashok.raj@intel.com>
> > > ---
> > >  drivers/iommu/intel-svm.c | 81
> > > ++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 80
> > > insertions(+), 1 deletion(-)
> > > 
> > > diff --git a/drivers/iommu/intel-svm.c b/drivers/iommu/intel-svm.c
> > > index 23c4276..d1d2d23 100644
> > > --- a/drivers/iommu/intel-svm.c
> > > +++ b/drivers/iommu/intel-svm.c
> > > @@ -525,6 +525,80 @@ static bool access_error(struct vm_area_struct
> > > *vma, struct page_req_dsc *req) return (requested &
> > > ~vma->vm_flags) != 0; }
> > >  
> > > +static int prq_to_iommu_prot(struct page_req_dsc *req)
> > > +{
> > > +	int prot = 0;
> > > +
> > > +	if (req->rd_req)
> > > +		prot |= IOMMU_READ;
> > > +	if (req->wr_req)
> > > +		prot |= IOMMU_WRITE;
> > > +	if (req->exe_req)
> > > +		prot |= IOMMU_EXEC;
> > > +	if (req->priv_req)
> > > +		prot |= IOMMU_PRIV;
> > > +
> > > +	return prot;
> > > +}
> > > +
> > > +static int intel_svm_prq_notify(struct device *dev, struct
> > > page_req_dsc *desc) +{
> > > +	int ret = 0;
> > > +	struct iommu_fault_event *event;
> > > +	struct pci_dev *pdev;
> > > +	struct device_domain_info *info;
> > > +	unsigned long buf_offset;
> > > +
> > > +	/**
> > > +	 * If caller does not provide struct device, this is the
> > > case where
> > > +	 * guest PASID table is bond to the device. So we need to
> > > retrieve
> > > +	 * struct device from the page request deescriptor then
> > > proceed.
> > > +	 */
> > > +	if (!dev) {
> > > +		pdev = pci_get_bus_and_slot(desc->bus,
> > > desc->devfn);
> > > +		if (!pdev) {
> > > +			pr_err("No PCI device found for PRQ
> > > [%02x:%02x.%d]\n",
> > > +				desc->bus, PCI_SLOT(desc->devfn),
> > > +				PCI_FUNC(desc->devfn));
> > > +			return -ENODEV;
> > > +		}
> > > +		/**
> > > +		 * Make sure PASID table pointer is bond to guest,
> > > if yes notify
> > > +		 * handler in the guest, e.g. via VFIO.
> > > +		 */
> > > +		info = pdev->dev.archdata.iommu;
> > > +		if (!info || !info->pasid_tbl_bond) {
> > > +			pr_debug("PRQ device pasid table not
> > > bond.\n");    
> > 
> > I can "bond" two things together, they are then "bound".
> >   
> will fix that :)
> > > +			return -EINVAL;
> > > +		}
> > > +		dev = &pdev->dev;    
> > 
> > Leaks pdev reference.  Both normal and error path.
> >   
> I guess you are referring to ref count in pci_get_bus_and_slot()? I did
> look at the code, it does not seem to do the count increment. Perhaps
> the comment is stale?
> 
>  * pointer to its data structure.  The caller must decrement the
>  * reference count by calling pci_dev_put().  If no device is found,
>  * %NULL is returned.
>  */
> struct pci_dev *pci_get_domain_bus_and_slot(int domain, unsigned int bus,
> 					    unsigned int devfn)
> {
> 	struct pci_dev *dev = NULL;
> 
> 	for_each_pci_dev(dev) {
        ^^^^^^^^^^^^^^^^ <-- look in here, it's trickier than it appears


> 		if (pci_domain_nr(dev->bus) == domain &&
> 		    (dev->bus->number == bus && dev->devfn == devfn))
> 			return dev;
> 	}
> 	return NULL;
> }
> EXPORT_SYMBOL(pci_get_domain_bus_and_slot);
> 
> 
> 
> > > +	}
> > > +
> > > +	pr_debug("Notify PRQ device [%02x:%02x.%d]\n",
> > > +		desc->bus, PCI_SLOT(desc->devfn),
> > > +		PCI_FUNC(desc->devfn));
> > > +	event = kzalloc(sizeof(*event) + sizeof(*desc),
> > > GFP_KERNEL);
> > > +	if (!event)
> > > +		return -ENOMEM;
> > > +
> > > +	get_device(dev);
> > > +	/* Fill in event data for device specific processing */
> > > +	event->dev = dev;
> > > +	buf_offset = offsetofend(struct iommu_fault_event, length);
> > > +	memcpy(buf_offset + event, desc, sizeof(*desc));
> > > +	event->addr = desc->addr;
> > > +	event->pasid = desc->pasid;
> > > +	event->prot = prq_to_iommu_prot(desc);
> > > +	event->length = sizeof(*desc);
> > > +	event->flags = IOMMU_FAULT_PAGE_REQ;
> > > +
> > > +	ret = iommu_fault_notifier_call_chain(event);
> > > +	put_device(dev);
> > > +	kfree(event);
> > > +
> > > +	return ret;
> > > +}
> > > +
> > >  static irqreturn_t prq_event_thread(int irq, void *d)
> > >  {
> > >  	struct intel_iommu *iommu = d;
> > > @@ -548,7 +622,12 @@ static irqreturn_t prq_event_thread(int irq,
> > > void *d) handled = 1;
> > >  
> > >  		req = &iommu->prq[head / sizeof(*req)];
> > > -
> > > +		/**
> > > +		 * If prq is to be handled outside iommu driver
> > > via receiver of
> > > +		 * the fault notifiers, we skip the page response
> > > here.
> > > +		 */
> > > +		if (!intel_svm_prq_notify(NULL, req))
> > > +			continue;
> > >  		result = QI_RESP_FAILURE;
> > >  		address = (u64)req->addr << VTD_PAGE_SHIFT;
> > >  		if (!req->pasid_present) {    
> >   
> 
> [Jacob Pan]

[toc] | [prev] | [next] | [standalone]


#1673886

FromJacob Pan <jacob.jun.pan@linux.intel.com>
Date2017-06-23 23:40 +0200
Message-ID<tVEt3-7qG-1@gated-at.bofh.it>
In reply to#1673855
On Fri, 23 Jun 2017 14:34:34 -0600
Alex Williamson <alex.williamson@redhat.com> wrote:

> > 	for_each_pci_dev(dev) {  
>         ^^^^^^^^^^^^^^^^ <-- look in here, it's trickier than it
> appears
you are right, thanks.

[toc] | [prev] | [standalone]


Back to top | Article view | linux.kernel


csiph-web