414 lines
12 KiB
Diff
414 lines
12 KiB
Diff
diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
|
|
--- a/drivers/iommu/amd/amd_iommu.h
|
|
+++ b/drivers/iommu/amd/amd_iommu.h
|
|
@@ -48,6 +48,9 @@ extern u8 amd_iommu_hpt_vasize;
|
|
extern unsigned long amd_iommu_pgsize_bitmap;
|
|
extern bool amd_iommu_hatdis;
|
|
|
|
+int amd_iommu_perfopt_clear(struct amd_iommu *iommu);
|
|
+int amd_iommu_perfopt_restore(struct amd_iommu *iommu);
|
|
+
|
|
/* Protection domain ops */
|
|
void amd_iommu_init_identity_domain(void);
|
|
struct protection_domain *protection_domain_alloc(void);
|
|
diff --git a/drivers/iommu/amd/amd_iommu_types.h b/drivers/iommu/amd/amd_iommu_types.h
|
|
--- a/drivers/iommu/amd/amd_iommu_types.h
|
|
+++ b/drivers/iommu/amd/amd_iommu_types.h
|
|
@@ -65,6 +65,7 @@
|
|
#define MMIO_MSI_ADDR_LO_OFFSET 0x015C
|
|
#define MMIO_MSI_ADDR_HI_OFFSET 0x0160
|
|
#define MMIO_MSI_DATA_OFFSET 0x0164
|
|
+#define MMIO_PERF_OPT_OFFSET 0x016C
|
|
#define MMIO_INTCAPXT_EVT_OFFSET 0x0170
|
|
#define MMIO_INTCAPXT_PPR_OFFSET 0x0178
|
|
#define MMIO_INTCAPXT_GALOG_OFFSET 0x0180
|
|
@@ -99,6 +100,8 @@
|
|
#define FEATURE_GLX GENMASK_ULL(15, 14)
|
|
#define FEATURE_GAM_VAPIC BIT_ULL(21)
|
|
#define FEATURE_PASMAX GENMASK_ULL(36, 32)
|
|
+#define FEATURE_PERF_OPT BIT_ULL(45)
|
|
+#define PERF_OPT_EN BIT(13)
|
|
#define FEATURE_GIOSUP BIT_ULL(48)
|
|
#define FEATURE_HASUP BIT_ULL(49)
|
|
#define FEATURE_EPHSUP BIT_ULL(50)
|
|
@@ -663,6 +666,9 @@ struct amd_iommu {
|
|
/* Extended features 2 */
|
|
u64 features2;
|
|
|
|
+ /* Devices requesting PerfOpt; the shared PERF_OPT_EN bit is on while >0. Protected by @lock. */
|
|
+ int perfopt_refcount;
|
|
+
|
|
/* PCI device id of the IOMMU device */
|
|
u16 devid;
|
|
|
|
@@ -824,6 +830,7 @@ struct iommu_dev_data {
|
|
u8 ppr :1; /* Enable device PPR support */
|
|
bool use_vapic; /* Enable device to use vapic mode */
|
|
bool defer_attach;
|
|
+ bool perfopt;
|
|
|
|
struct ratelimit_state rs; /* Ratelimit IOPF messages */
|
|
};
|
|
diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c
|
|
--- a/drivers/iommu/amd/init.c
|
|
+++ b/drivers/iommu/amd/init.c
|
|
@@ -1945,6 +1945,9 @@ static int __init init_iommu_one(struct amd_iommu *iommu, struct ivhd_header *h,
|
|
if (!iommu->mmio_base)
|
|
return -ENOMEM;
|
|
|
|
+ if (amd_iommu_perfopt_clear(iommu))
|
|
+ pr_err("IOMMU%d: failed to clear PerfOpt\n", iommu->index);
|
|
+
|
|
return init_iommu_from_acpi(iommu, h);
|
|
}
|
|
|
|
@@ -3035,10 +3038,46 @@ static void enable_iommus_vapic(void)
|
|
#endif
|
|
}
|
|
|
|
+static int clear_perfopt_all(void)
|
|
+{
|
|
+ struct amd_iommu *iommu;
|
|
+ int err, ret = 0;
|
|
+
|
|
+ for_each_iommu(iommu) {
|
|
+ err = amd_iommu_perfopt_clear(iommu);
|
|
+ if (err)
|
|
+ ret = err;
|
|
+ }
|
|
+
|
|
+ return ret;
|
|
+}
|
|
+
|
|
+static int restore_perfopt_all(void)
|
|
+{
|
|
+ struct amd_iommu *iommu;
|
|
+ int err, ret = 0;
|
|
+
|
|
+ for_each_iommu(iommu) {
|
|
+ err = amd_iommu_perfopt_restore(iommu);
|
|
+ if (err)
|
|
+ ret = err;
|
|
+ }
|
|
+
|
|
+ return ret;
|
|
+}
|
|
+
|
|
static void disable_iommus(void)
|
|
{
|
|
struct amd_iommu *iommu;
|
|
|
|
+ /*
|
|
+ * PerfOpt is an optional performance bit, so a failure to clear it must
|
|
+ * not skip the mandatory disable below. This also runs from the void
|
|
+ * amd_iommu_disable() shutdown/kexec path, which cannot report an error.
|
|
+ */
|
|
+ if (clear_perfopt_all())
|
|
+ pr_err("Failed to clear PerfOpt while disabling IOMMUs\n");
|
|
+
|
|
for_each_iommu(iommu)
|
|
iommu_disable(iommu);
|
|
|
|
@@ -3064,6 +3103,10 @@ static void amd_iommu_resume(void *data)
|
|
for_each_iommu(iommu)
|
|
early_enable_iommu(iommu);
|
|
|
|
+ /* early_enable_iommu() cleared PERF_OPT_EN; re-assert it from the refcount. */
|
|
+ if (restore_perfopt_all())
|
|
+ pr_err("Failed to restore PerfOpt after IOMMU resume\n");
|
|
+
|
|
iommu_enable_event_buffer();
|
|
amd_iommu_enable_interrupts();
|
|
}
|
|
diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c
|
|
--- a/drivers/iommu/amd/iommu.c
|
|
+++ b/drivers/iommu/amd/iommu.c
|
|
@@ -2352,6 +2352,9 @@ static int attach_device(struct device *dev,
|
|
if (ret)
|
|
goto out;
|
|
|
|
+ if (dev_data->perfopt)
|
|
+ goto skip_caps;
|
|
+
|
|
/* Setup GCR3 table */
|
|
if (pdom_is_sva_capable(domain)) {
|
|
ret = init_gcr3_table(dev_data, domain);
|
|
@@ -2376,6 +2379,7 @@ static int attach_device(struct device *dev,
|
|
pdev_enable_cap_ats(pdev);
|
|
}
|
|
|
|
+skip_caps:
|
|
/* Update data structures */
|
|
dev_data->domain = domain;
|
|
spin_lock_irqsave(&domain->lock, flags);
|
|
@@ -2444,6 +2448,192 @@ static void detach_device(struct device *dev)
|
|
mutex_unlock(&dev_data->mutex);
|
|
}
|
|
|
|
+/* Program the per-IOMMU PerfOpt enable bit. Caller must hold iommu->lock. */
|
|
+static int __perfopt_write(struct amd_iommu *iommu, bool enable)
|
|
+{
|
|
+ u32 old, val, readback;
|
|
+
|
|
+ if (!(readq(iommu->mmio_base + MMIO_EXT_FEATURES) & FEATURE_PERF_OPT))
|
|
+ return enable ? -ENODEV : 0;
|
|
+
|
|
+ old = readl(iommu->mmio_base + MMIO_PERF_OPT_OFFSET);
|
|
+ if (old == U32_MAX)
|
|
+ return -EIO;
|
|
+
|
|
+ val = enable ? old | PERF_OPT_EN : old & ~PERF_OPT_EN;
|
|
+ if (val != old)
|
|
+ writel(val, iommu->mmio_base + MMIO_PERF_OPT_OFFSET);
|
|
+ readback = readl(iommu->mmio_base + MMIO_PERF_OPT_OFFSET);
|
|
+ if (readback == U32_MAX ||
|
|
+ (readback & PERF_OPT_EN) != (val & PERF_OPT_EN))
|
|
+ return -EIO;
|
|
+ return 0;
|
|
+}
|
|
+
|
|
+/*
|
|
+ * PERF_OPT_EN is a single bit shared by every device behind @iommu, so it is
|
|
+ * reference counted: armed on the first requesting device, cleared on the last.
|
|
+ */
|
|
+static int perfopt_get(struct amd_iommu *iommu)
|
|
+{
|
|
+ unsigned long flags;
|
|
+ int ret = 0;
|
|
+
|
|
+ if (!iommu->mmio_base)
|
|
+ return 0;
|
|
+
|
|
+ raw_spin_lock_irqsave(&iommu->lock, flags);
|
|
+ if (iommu->perfopt_refcount == 0) {
|
|
+ ret = __perfopt_write(iommu, true);
|
|
+ if (ret)
|
|
+ goto out;
|
|
+ }
|
|
+ iommu->perfopt_refcount++;
|
|
+out:
|
|
+ raw_spin_unlock_irqrestore(&iommu->lock, flags);
|
|
+ return ret;
|
|
+}
|
|
+
|
|
+static int perfopt_put(struct amd_iommu *iommu)
|
|
+{
|
|
+ unsigned long flags;
|
|
+ int ret = 0;
|
|
+
|
|
+ if (!iommu->mmio_base)
|
|
+ return 0;
|
|
+
|
|
+ raw_spin_lock_irqsave(&iommu->lock, flags);
|
|
+ if (iommu->perfopt_refcount > 0 && --iommu->perfopt_refcount == 0)
|
|
+ ret = __perfopt_write(iommu, false);
|
|
+ raw_spin_unlock_irqrestore(&iommu->lock, flags);
|
|
+ return ret;
|
|
+}
|
|
+
|
|
+/*
|
|
+ * Force PERF_OPT_EN off without touching the refcount (used on init, shutdown,
|
|
+ * and suspend). The count is preserved so amd_iommu_perfopt_restore() can
|
|
+ * re-arm on resume.
|
|
+ */
|
|
+int amd_iommu_perfopt_clear(struct amd_iommu *iommu)
|
|
+{
|
|
+ unsigned long flags;
|
|
+ int ret;
|
|
+
|
|
+ if (!iommu->mmio_base)
|
|
+ return 0;
|
|
+
|
|
+ raw_spin_lock_irqsave(&iommu->lock, flags);
|
|
+ ret = __perfopt_write(iommu, false);
|
|
+ raw_spin_unlock_irqrestore(&iommu->lock, flags);
|
|
+ return ret;
|
|
+}
|
|
+
|
|
+/*
|
|
+ * Re-assert PERF_OPT_EN from the refcount after the hardware was reprogrammed on
|
|
+ * resume, so devices armed before suspend keep the optimization without each
|
|
+ * consumer driver re-arming.
|
|
+ */
|
|
+int amd_iommu_perfopt_restore(struct amd_iommu *iommu)
|
|
+{
|
|
+ unsigned long flags;
|
|
+ int ret;
|
|
+
|
|
+ if (!iommu->mmio_base)
|
|
+ return 0;
|
|
+
|
|
+ raw_spin_lock_irqsave(&iommu->lock, flags);
|
|
+ ret = __perfopt_write(iommu, iommu->perfopt_refcount > 0);
|
|
+ raw_spin_unlock_irqrestore(&iommu->lock, flags);
|
|
+ return ret;
|
|
+}
|
|
+
|
|
+int amd_iommu_enable_perfopt(struct pci_dev *pdev)
|
|
+{
|
|
+ struct iommu_dev_data *dev_data = dev_iommu_priv_get(&pdev->dev);
|
|
+ struct amd_iommu *iommu = rlookup_amd_iommu(&pdev->dev);
|
|
+ struct protection_domain *domain;
|
|
+ int ret;
|
|
+
|
|
+ if (!iommu || !dev_data)
|
|
+ return -ENODEV;
|
|
+
|
|
+ if (!(iommu->features & FEATURE_PERF_OPT))
|
|
+ return -ENODEV;
|
|
+
|
|
+ domain = dev_data->domain;
|
|
+ if (!domain)
|
|
+ return -ENODEV;
|
|
+
|
|
+ /* Already armed for this device (e.g. re-entry on resume). */
|
|
+ if (dev_data->perfopt)
|
|
+ return 0;
|
|
+
|
|
+ /*
|
|
+ * The bit is only architecturally valid while the device is untranslated:
|
|
+ * identity domain with ATS/PRI/PASID off. The identity domain is
|
|
+ * SVA-capable so attach_device() enabled ATS/PRI/PASID and built a GCR3
|
|
+ * table. Re-home the device onto the same identity domain with
|
|
+ * perfopt set, so the attach_device() skip_caps path leaves
|
|
+ * ATS/PRI/PASID off and no GCR3 table. This follows the detach/attach
|
|
+ * pattern used by amd_iommu_attach_device().
|
|
+ *
|
|
+ * Locking: this and amd_iommu_disable_perfopt() run only from the
|
|
+ * consumer driver's bind/unbind path. group->mutex is not exposed to
|
|
+ * drivers, but a device bound to its native driver cannot have its domain
|
|
+ * changed concurrently by the core (VFIO ownership is mutually exclusive;
|
|
+ * sysfs domain changes require an unused group), so the detach/attach pair
|
|
+ * is serialized without it.
|
|
+ */
|
|
+ dev_data->perfopt = true;
|
|
+ detach_device(&pdev->dev);
|
|
+ ret = attach_device(&pdev->dev, domain);
|
|
+ if (ret)
|
|
+ goto err_restore;
|
|
+
|
|
+ ret = perfopt_get(iommu);
|
|
+ if (ret)
|
|
+ goto err_rearm;
|
|
+
|
|
+ dev_info_once(&pdev->dev, "PerfOpt armed on IOMMU%d\n", iommu->index);
|
|
+ return 0;
|
|
+
|
|
+err_rearm:
|
|
+ detach_device(&pdev->dev);
|
|
+err_restore:
|
|
+ dev_data->perfopt = false;
|
|
+ if (attach_device(&pdev->dev, domain))
|
|
+ pci_err(pdev, "failed to restore state after PerfOpt setup; device left detached\n");
|
|
+ dev_err_once(&pdev->dev, "PerfOpt failed to arm on IOMMU%d (%d)\n",
|
|
+ iommu->index, ret);
|
|
+ return ret;
|
|
+}
|
|
+EXPORT_SYMBOL_GPL(amd_iommu_enable_perfopt);
|
|
+
|
|
+void amd_iommu_disable_perfopt(struct pci_dev *pdev)
|
|
+{
|
|
+ struct iommu_dev_data *dev_data = dev_iommu_priv_get(&pdev->dev);
|
|
+ struct amd_iommu *iommu = rlookup_amd_iommu(&pdev->dev);
|
|
+ struct protection_domain *domain;
|
|
+
|
|
+ if (!iommu || !dev_data || !dev_data->perfopt || !dev_data->domain)
|
|
+ return;
|
|
+
|
|
+ if (WARN_ON(perfopt_put(iommu)))
|
|
+ pci_err(pdev, "failed to clear PerfOpt\n");
|
|
+
|
|
+ /*
|
|
+ * Restore ATS/PRI/PASID (and thus SVA) by re-homing the device onto its
|
|
+ * identity domain with the flag cleared, so a later bind without PerfOpt
|
|
+ * sees a normally-capable device. See the locking note in
|
|
+ * amd_iommu_enable_perfopt().
|
|
+ */
|
|
+ domain = dev_data->domain;
|
|
+ dev_data->perfopt = false;
|
|
+ detach_device(&pdev->dev);
|
|
+ if (attach_device(&pdev->dev, domain))
|
|
+ pci_err(pdev, "failed to restore caps after PerfOpt disable\n");
|
|
+}
|
|
+EXPORT_SYMBOL_GPL(amd_iommu_disable_perfopt);
|
|
static struct iommu_device *amd_iommu_probe_device(struct device *dev)
|
|
{
|
|
struct iommu_device *iommu_dev;
|
|
@@ -2511,6 +2701,14 @@ static struct iommu_device *amd_iommu_probe_device(struct device *dev)
|
|
static void amd_iommu_release_device(struct device *dev)
|
|
{
|
|
struct iommu_dev_data *dev_data = dev_iommu_priv_get(dev);
|
|
+ struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data);
|
|
+
|
|
+ if (dev_data->perfopt) {
|
|
+ if (WARN_ON(perfopt_put(iommu)))
|
|
+ dev_err(dev, "IOMMU%d: failed to clear PerfOpt on release\n",
|
|
+ iommu->index);
|
|
+ dev_data->perfopt = false;
|
|
+ }
|
|
|
|
WARN_ON(dev_data->domain);
|
|
|
|
@@ -2885,6 +3083,19 @@ static int blocked_domain_attach_device(struct iommu_domain *domain,
|
|
struct iommu_domain *old)
|
|
{
|
|
struct iommu_dev_data *dev_data = dev_iommu_priv_get(dev);
|
|
+ struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data);
|
|
+
|
|
+ /*
|
|
+ * blocked_domain is also the .release_domain, so this is the normal
|
|
+ * teardown path: drop the reference and clear the flag here too, and
|
|
+ * don't fail teardown if the WARN-guarded write doesn't stick.
|
|
+ */
|
|
+ if (dev_data->perfopt) {
|
|
+ if (WARN_ON(perfopt_put(iommu)))
|
|
+ dev_err(dev, "IOMMU%d: failed to clear PerfOpt for blocked domain\n",
|
|
+ iommu->index);
|
|
+ dev_data->perfopt = false;
|
|
+ }
|
|
|
|
if (dev_data->domain)
|
|
detach_device(dev);
|
|
@@ -2953,6 +3164,9 @@ static int amd_iommu_attach_device(struct iommu_domain *dom, struct device *dev,
|
|
struct amd_iommu *iommu = get_amd_iommu_from_dev(dev);
|
|
int ret;
|
|
|
|
+ if (dev_data->perfopt && !pdom_is_in_pt_mode(domain))
|
|
+ return -EBUSY;
|
|
+
|
|
/*
|
|
* Skip attach device to domain if new domain is same as
|
|
* devices current domain
|
|
diff --git a/include/linux/amd-iommu.h b/include/linux/amd-iommu.h
|
|
--- a/include/linux/amd-iommu.h
|
|
+++ b/include/linux/amd-iommu.h
|
|
@@ -11,11 +11,11 @@
|
|
#include <linux/types.h>
|
|
|
|
struct amd_iommu;
|
|
+struct pci_dev;
|
|
|
|
#ifdef CONFIG_AMD_IOMMU
|
|
|
|
struct task_struct;
|
|
-struct pci_dev;
|
|
|
|
extern void amd_iommu_detect(void);
|
|
|
|
@@ -76,4 +76,15 @@ static inline int amd_iommu_snp_disable(void) { return 0; }
|
|
static inline bool amd_iommu_sev_tio_supported(void) { return false; }
|
|
#endif
|
|
|
|
+#ifdef CONFIG_AMD_IOMMU
|
|
+int amd_iommu_enable_perfopt(struct pci_dev *pdev);
|
|
+void amd_iommu_disable_perfopt(struct pci_dev *pdev);
|
|
+#else
|
|
+static inline int amd_iommu_enable_perfopt(struct pci_dev *pdev)
|
|
+{
|
|
+ return 0;
|
|
+}
|
|
+static inline void amd_iommu_disable_perfopt(struct pci_dev *pdev) { }
|
|
+#endif
|
|
+
|
|
#endif /* _ASM_X86_AMD_IOMMU_H */
|