2#include "pedigree/kernel/LockGuard.h"
3#include "pedigree/kernel/Log.h"
4#include "pedigree/kernel/machine/Device.h"
5#include "pedigree/kernel/machine/Pci.h"
6#include "pedigree/kernel/panic.h"
7#include "pedigree/kernel/processor/PhysicalMemoryManager.h"
8#include "pedigree/kernel/processor/Processor.h"
9#include "pedigree/kernel/processor/VirtualAddressSpace.h"
10#include "pedigree/kernel/utilities/utility.h"
14#include "../../core/processor/x64/utils.h"
16#include "IntelIommu.h"
19constexpr uint32_t TranslationEnable = 1U << 31;
20constexpr uint32_t SetRootPointer = 1U << 30;
21constexpr uint32_t QueuedInvalidationEnabled = 1U << 26;
22constexpr uint64_t PageBytes = 4096;
23constexpr uint64_t LargePageBytes = 2 * 1024 * 1024;
24constexpr size_t PollLimit = 1000000;
27 if (info.hardwareUnitCount != 1 || info.segmentZeroUnitCount != 1) {
30 const uint8_t bus = device->getPciBusPosition();
31 const uint16_t requester =
32 (uint16_t(bus) << 8) | (device->getPciDevicePosition() << 3) | device->getPciFunctionNumber();
33 if (info.segmentZeroIncludeAllCount ||
34 info.includesDirectEndpoint(bus, device->getPciDevicePosition(),
35 device->getPciFunctionNumber())) {
39 auto& pci = PciBus::instance();
40 for (
size_t index = 0; index < info.directBridgeCount; ++index) {
41 const uint16_t source = info.directBridges[index];
43 bridge.
setPciPosition(source >> 8, (source >> 3) & 31, source & 7);
46 uint32_t classCode = 0, buses = 0;
47 if (!pci.readConfig16(&bridge, 0, vendor) || !vendor || vendor == 0xffff ||
48 !pci.readConfig8(&bridge, 0x0e, header) || (header & 0x7fU) != 1 ||
49 !pci.readConfig32(&bridge, 8, classCode) || (classCode >> 16) != 0x0604 ||
50 !pci.readConfig32(&bridge, 0x18, buses)) {
53 const uint8_t primary = buses;
54 const uint8_t secondary = buses >> 8;
55 const uint8_t subordinate = buses >> 16;
56 if (primary != source >> 8 || secondary <= primary || subordinate < secondary) {
60 if (requester == source || (bus >= secondary && bus <= subordinate)) {
73IntelIommu::IntelIommu()
74 : m_Registers(
"Intel VT-d registers"), m_ReservedTokens(
"Intel VT-d IOVA reserve") {}
76void IntelIommu::flushLines(
const void* address,
size_t bytes)
const {
78 __atomic_thread_fence(__ATOMIC_SEQ_CST);
81 const uintptr_t start =
reinterpret_cast<uintptr_t
>(address) & ~(m_CacheLine - 1);
83 (
reinterpret_cast<uintptr_t
>(address) + bytes + m_CacheLine - 1) & ~(m_CacheLine - 1);
84 for (uintptr_t line = start; line < end; line += m_CacheLine) {
85 asm volatile(
"clflush (%0)" : :
"r"(line) :
"memory");
87 asm volatile(
"mfence" : : :
"memory");
90bool IntelIommu::allocateTable(physical_uintptr_t& page) {
95 if ((page & (PageBytes - 1)) || page > m_MaxPhysical - (PageBytes - 1)) {
104bool IntelIommu::waitForStatus(uint32_t bit,
bool set) {
105 for (
size_t attempt = 0; attempt < PollLimit; ++attempt) {
106 if (
bool(m_Registers.
read32(0x1c) & bit) == set) {
109 asm volatile(
"pause");
114bool IntelIommu::command(uint32_t bit,
bool set) {
115 const uint32_t status = m_Registers.
read32(0x1c);
116 constexpr uint32_t persistent =
117 TranslationEnable | QueuedInvalidationEnabled | (1U << 25) | (1U << 23);
118 const uint32_t next = (status & persistent & ~bit) | (set ? bit : 0);
119 m_Registers.
write32(next, 0x18);
120 return waitForStatus(bit, set);
123bool IntelIommu::invalidateContext() {
124 m_Registers.
write64((1ULL << 63) | (1ULL << 61), 0x28);
125 for (
size_t attempt = 0; attempt < PollLimit; ++attempt) {
126 const uint64_t result = m_Registers.
read64(0x28);
127 if (!(result & (1ULL << 63))) {
128 return ((result >> 59) & 3) != 0;
130 asm volatile(
"pause");
135bool IntelIommu::invalidateIotlb() {
136 m_Registers.
write64((1ULL << 63) | (1ULL << 60), m_IotlbOffset + 8);
137 for (
size_t attempt = 0; attempt < PollLimit; ++attempt) {
138 const uint64_t result = m_Registers.
read64(m_IotlbOffset + 8);
139 if (!(result & (1ULL << 63))) {
140 return ((result >> 57) & 3) != 0;
142 asm volatile(
"pause");
147bool IntelIommu::initialise() {
149 if (!info || info->hardwareUnitCount != 1 || info->segmentZeroUnitCount != 1 ||
150 (!info->segmentZeroIncludeAllCount && !info->directEndpointCount &&
151 !info->directBridgeCount) ||
152 info->reservedMemoryRegions || info->firstSegmentZeroUnitRegisterPagesLog2 > 4) {
154 NOTICE(
"Intel VT-d: unsupported DMAR layout, units="
155 <<
Dec << info->hardwareUnitCount <<
" include-all="
156 << info->segmentZeroIncludeAllCount <<
" RMRR=" << info->reservedMemoryRegions);
158 NOTICE(
"Intel VT-d: ACPI DMAR information unavailable");
163 const size_t registerPages = size_t(1) << info->firstSegmentZeroUnitRegisterPagesLog2;
165 m_Registers, registerPages,
170 info->firstSegmentZeroUnitAddress)) {
174 if (m_DefaultContext) {
176 m_DefaultContext = 0;
182 if (m_ReservedTokens) {
183 m_ReservedTokens.free();
189 const uint32_t version = m_Registers.
read32(0);
190 const uint64_t cap = m_Registers.
read64(8);
191 const uint64_t ecap = m_Registers.
read64(0x10);
192 const uint32_t status = m_Registers.
read32(0x1c);
193 const uint8_t mgaw =
static_cast<uint8_t
>(((cap >> 16) & 0x3f) + 1);
194 const uint8_t domainCountEncoding =
static_cast<uint8_t
>(cap & 7);
195 const bool aw39 = cap & (1ULL << 9);
196 const bool aw48 = cap & (1ULL << 10);
197 m_IotlbOffset = size_t((ecap >> 8) & 0x3ff) * 16;
198 if (((version >> 4) & 0xf) == 0 || ((version >> 4) & 0xf) >= 6 || domainCountEncoding == 7 ||
199 !(ecap & (1ULL << 6)) || !(cap & (1ULL << 34)) || mgaw < 32 || (!aw39 && !aw48) ||
200 m_IotlbOffset + 16 > m_Registers.
size() ||
201 (status & (TranslationEnable | QueuedInvalidationEnabled))) {
202 NOTICE(
"Intel VT-d: unsupported register capabilities, version="
203 <<
Hex << version <<
" CAP=" << cap <<
" ECAP=" << ecap <<
" GSTS=" << status);
209 for (
int aw = 4; aw >= 0; --aw) {
210 if (cap & (1ULL << (8 + aw))) {
211 m_PassThroughAw = aw;
216 info->hostAddressWidth == 64 ? ~uint64_t(0) : (uint64_t(1) << info->hostAddressWidth) - 1;
217 m_Coherent = ecap & 1;
219 uint32_t eax, ebx, ecx, edx;
221 m_CacheLine = ((ebx >> 8) & 0xff) * 8;
222 if (!(edx & (1U << 19)) || m_CacheLine < 32 || m_CacheLine > 256 ||
223 (m_CacheLine & (m_CacheLine - 1))) {
229 m_ReservedTokens, TokenPages,
235 if (!tokenBase || tokenBase + TokenPages * PageBytes > (1ULL << 32)) {
239 if (!allocateTable(m_Root) || !allocateTable(m_DefaultContext)) {
242 auto* contexts =
reinterpret_cast<uint64_t*
>(
physicalAddress(m_DefaultContext));
243 for (
size_t index = 0; index < 256; ++index) {
244 contexts[index * 2] = (2ULL << 2) | 1;
245 contexts[index * 2 + 1] = (1ULL << 8) | m_PassThroughAw;
247 flushLines(contexts, PageBytes);
250 for (
size_t bus = 0; bus < 256; ++bus) {
251 roots[bus * 2] = m_DefaultContext | 1;
253 flushLines(roots, PageBytes);
255 m_Registers.
write64(m_Root, 0x20);
256 if (!command(SetRootPointer,
true)) {
259 if (!invalidateContext() || !invalidateIotlb()) {
262 if (!command(TranslationEnable,
true)) {
263 panic(
"Intel VT-d: translation enable did not complete");
266 const uint8_t tableWidth = m_Aw == 2 ? 48 : 39;
267 const uint8_t iovaWidth = mgaw < tableWidth ? mgaw : tableWidth;
268 NOTICE(
"Intel VT-d: DMA remapping enabled, " <<
Dec <<
static_cast<uint32_t
>(iovaWidth)
269 <<
"-bit IOVA, " <<
Hex << tokenBase
274void IntelIommu::freeDomain(Domain& domain) {
275 for (
size_t i = 0; i < domain.pageCount; ++i) {
281bool IntelIommu::makeDomain(Domain& domain) {
282 auto addPage = [&](physical_uintptr_t& page) {
283 if (domain.pageCount == 8 || !allocateTable(page)) {
286 domain.pages[domain.pageCount++] = page;
290 if (!addPage(domain.root)) {
293 physical_uintptr_t pdptPage = domain.root;
295 if (!addPage(pdptPage)) {
299 auto* root =
reinterpret_cast<uint64_t*
>(
physicalAddress(domain.root));
300 root[0] = pdptPage | 3;
304 physical_uintptr_t directoryPages[4] = {};
305 domain.tokenBase = domain.isolated ? IsolatedTokenBase : m_ReservedTokens.
physicalAddress();
306 domain.tokenCount = domain.isolated ? IsolatedTokenPages : TokenPages;
307 const size_t firstGroup = domain.isolated ? domain.tokenBase / (512 * LargePageBytes) : 0;
308 const size_t lastGroup = domain.isolated ? firstGroup + 1 : 4;
309 for (
size_t group = firstGroup; group < lastGroup; ++group) {
310 if (!addPage(directoryPages[group])) {
314 pdpt[group] = directoryPages[group] | 3;
315 if (!domain.isolated) {
316 auto* directory =
reinterpret_cast<uint64_t*
>(
physicalAddress(directoryPages[group]));
317 for (
size_t entry = 0; entry < 512; ++entry) {
318 directory[entry] = (uint64_t(group * 512 + entry) * LargePageBytes) | (1U << 7) | 3;
323 const uint64_t firstBlock = domain.tokenBase / LargePageBytes;
324 const uint64_t lastBlock =
325 (domain.tokenBase + domain.tokenCount * PageBytes - 1) / LargePageBytes;
326 for (uint64_t block = firstBlock; block <= lastBlock; ++block) {
327 physical_uintptr_t tablePage = 0;
328 if (!addPage(tablePage)) {
333 domain.tokenTables[block - firstBlock] = table;
334 if (!domain.isolated) {
335 for (
size_t index = 0; index < 512; ++index) {
336 table[index] = (block * LargePageBytes + index * PageBytes) | 3;
339 for (
size_t slot = 0; slot < domain.tokenCount; ++slot) {
340 const uint64_t iova = domain.tokenBase + slot * PageBytes;
341 if (iova / LargePageBytes == block) {
342 table[(iova / PageBytes) & 511] = 0;
345 auto* directory =
reinterpret_cast<uint64_t*
>(
physicalAddress(directoryPages[block / 512]));
346 directory[block & 511] = tablePage | 3;
349 for (
size_t i = 0; i < domain.pageCount; ++i) {
350 flushLines(
reinterpret_cast<void*
>(
physicalAddress(domain.pages[i])), PageBytes);
355uint64_t* IntelIommu::tokenEntry(
const Domain& domain,
size_t slot) {
356 if (slot >= domain.tokenCount) {
359 const uint64_t iova = domain.tokenBase + slot * PageBytes;
360 const size_t table = iova / LargePageBytes - domain.tokenBase / LargePageBytes;
361 return table < 2 && domain.tokenTables[table]
362 ? &domain.tokenTables[table][(iova / PageBytes) & 511]
367 for (
size_t i = 0; i < MaxDomains; ++i) {
368 if (m_Domains[i].device && m_Domains[i].device == device) {
369 return &m_Domains[i];
376 for (
size_t i = 0; i < MaxDomains; ++i) {
377 if (m_Domains[i].device && m_Domains[i].device == device) {
378 return &m_Domains[i];
384bool IntelIommu::attach(
Device* device,
bool isolated) {
385 if (!device || device->getPciBusPosition() > 255 || device->getPciDevicePosition() > 31 ||
386 device->getPciFunctionNumber() > 7) {
390 if (
const Domain* domain = findDomain(device)) {
391 return domain->isolated == isolated;
393 size_t slot = MaxDomains;
394 for (
size_t i = 0; i < MaxDomains; ++i) {
395 Device* other = m_Domains[i].device;
396 if (other && other->getPciBusPosition() == device->getPciBusPosition() &&
397 other->getPciDevicePosition() == device->getPciDevicePosition() &&
398 other->getPciFunctionNumber() == device->getPciFunctionNumber()) {
401 if (!other && slot == MaxDomains) {
406 if (!info || !includesRequester(*info, device)) {
407 NOTICE(
"Intel VT-d: requester outside supported DMAR scope, PCI "
408 <<
Dec << device->getPciBusPosition() <<
":" << device->getPciDevicePosition() <<
"."
409 << device->getPciFunctionNumber());
412 uint16_t pciCommand = 0;
413 const bool commandRead = PciBus::instance().readConfig16(device, 4, pciCommand);
414 if (!commandRead || (pciCommand & 4)) {
415 NOTICE(
"Intel VT-d: attach requires bus mastering disabled, PCI command=" <<
Hex << pciCommand);
418 if (m_Failed || slot == MaxDomains) {
421 if (!m_Enabled && !initialise()) {
427 domain.device = device;
428 domain.id =
static_cast<uint16_t
>(slot + 2);
429 domain.isolated = isolated;
430 if (!makeDomain(domain)) {
434 const size_t bus = device->getPciBusPosition();
435 const size_t function = (device->getPciDevicePosition() << 3) | device->getPciFunctionNumber();
436 if (!m_BusContexts[bus]) {
437 physical_uintptr_t busContext = 0;
438 if (!allocateTable(busContext)) {
443 reinterpret_cast<const void*
>(
physicalAddress(m_DefaultContext)), PageBytes);
444 flushLines(
reinterpret_cast<void*
>(
physicalAddress(busContext)), PageBytes);
446 roots[bus * 2] = busContext | 1;
447 flushLines(&roots[bus * 2],
sizeof(uint64_t));
448 if (!invalidateContext() || !invalidateIotlb()) {
449 panic(
"Intel VT-d: failed to publish bus context");
451 m_BusContexts[bus] = busContext;
454 auto* contexts =
reinterpret_cast<uint64_t*
>(
physicalAddress(m_BusContexts[bus]));
455 contexts[function * 2 + 1] = (uint64_t(domain.id) << 8) | m_Aw;
456 contexts[function * 2] = domain.root | 1;
457 flushLines(&contexts[function * 2], 2 *
sizeof(uint64_t));
458 if (!invalidateContext() || !invalidateIotlb()) {
459 panic(
"Intel VT-d: failed to publish device context");
462 m_Domains[slot] = domain;
463 NOTICE(
"Intel VT-d: attached PCI " <<
Dec <<
static_cast<uint32_t
>(bus) <<
":"
464 <<
static_cast<uint32_t
>(function >> 3) <<
"."
465 <<
static_cast<uint32_t
>(function & 7)
466 << (isolated ?
" to isolated domain " :
" to domain ")
467 << static_cast<uint32_t>(domain.id));
471bool IntelIommu::isolatedIdle(
Device* device)
const {
473 const Domain* domain = findDomain(device);
474 if (!m_Enabled || !domain || !domain->isolated) {
477 uint16_t command = 0;
478 if (!PciBus::instance().readConfig16(device, 4, command) || (command & 4)) {
481 for (uint32_t used : domain->tokenUsed) {
489bool IntelIommu::detachIsolated(
Device* device,
bool requesterDisabled) {
494 Domain* domain = findDomain(device);
495 if (!m_Enabled || !domain || !domain->isolated) {
498 uint16_t pciCommand = 0;
499 if (!requesterDisabled &&
500 (!PciBus::instance().readConfig16(device, 4, pciCommand) || (pciCommand & 4))) {
503 for (uint32_t used : domain->tokenUsed) {
509 const size_t bus = device->getPciBusPosition();
510 const size_t function = (device->getPciDevicePosition() << 3) | device->getPciFunctionNumber();
511 if (!m_BusContexts[bus]) {
512 panic(
"Intel VT-d: isolated domain context missing");
514 auto* contexts =
reinterpret_cast<uint64_t*
>(
physicalAddress(m_BusContexts[bus]));
515 if (contexts[function * 2] != (domain->root | 1) ||
516 contexts[function * 2 + 1] != ((uint64_t(domain->id) << 8) | m_Aw)) {
517 panic(
"Intel VT-d: isolated domain context mismatch");
519 contexts[function * 2] = 0;
520 flushLines(&contexts[function * 2],
sizeof(uint64_t));
521 if (!invalidateContext() || !invalidateIotlb()) {
522 panic(
"Intel VT-d: failed to revoke isolated domain context");
524 contexts[function * 2 + 1] = 0;
525 flushLines(&contexts[function * 2 + 1],
sizeof(uint64_t));
530bool IntelIommu::attached(
const Device* device)
const {
532 return m_Enabled && findDomain(device);
535bool IntelIommu::isolated(
const Device* device)
const {
537 const Domain* domain = findDomain(device);
538 return m_Enabled && domain && domain->isolated;
541bool IntelIommu::mapPage(
Device* device, physical_uintptr_t physical, uint32_t& dmaAddress,
543 if (!device || !physical || (physical & (PageBytes - 1))) {
547 Domain* domain = findDomain(device);
548 if (!m_Enabled || !domain || physical > m_MaxPhysical - (PageBytes - 1)) {
552 if (!domain->isolated && physical < (1ULL << 32)) {
553 if (physical >= domain->tokenBase &&
554 physical < domain->tokenBase + domain->tokenCount * PageBytes) {
557 dmaAddress =
static_cast<uint32_t
>(physical);
562 for (
size_t slot = 0; slot < domain->tokenCount; ++slot) {
563 const uint32_t mask = 1U << (slot & 31);
564 if (domain->tokenUsed[slot / 32] & mask) {
567 uint64_t* entry = tokenEntry(*domain, slot);
569 panic(
"Intel VT-d: IOVA page table missing");
571 *entry = physical | 3;
572 flushLines(entry,
sizeof(*entry));
573 if (!invalidateIotlb()) {
574 panic(
"Intel VT-d: failed to publish IOVA mapping");
576 domain->tokenUsed[slot / 32] |= mask;
577 dmaAddress =
static_cast<uint32_t
>(domain->tokenBase + slot * PageBytes);
578 token =
static_cast<uint16_t
>(slot + 1);
579 if (physical >= (1ULL << 32) && !domain->loggedHighMapping) {
580 NOTICE(
"Intel VT-d: high physical page " <<
Hex << physical <<
" mapped to IOVA "
581 << dmaAddress <<
" in domain " <<
Dec << domain->id);
582 domain->loggedHighMapping =
true;
589void IntelIommu::unmapPage(
Device* device, uint16_t token) {
594 Domain* domain = findDomain(device);
595 const size_t slot = token - 1;
596 const uint32_t mask = 1U << (slot & 31);
597 if (!m_Enabled || !domain || slot >= domain->tokenCount ||
598 !(domain->tokenUsed[slot / 32] & mask)) {
599 panic(
"Intel VT-d: invalid IOVA unmap");
601 uint64_t* entry = tokenEntry(*domain, slot);
603 panic(
"Intel VT-d: IOVA page table missing");
606 flushLines(entry,
sizeof(*entry));
607 if (!invalidateIotlb()) {
608 panic(
"Intel VT-d: failed to revoke IOVA mapping");
610 domain->tokenUsed[slot / 32] &= ~mask;
614void IntelIommu::logFault() {
615 const uint32_t status = m_Registers.
read32(0x34);
619 const uint64_t cap = m_Registers.
read64(8);
620 const size_t offset = size_t((cap >> 24) & 0x3ff) * 16;
621 const size_t index = (status >> 8) & 0xff;
622 if ((status & 2) && offset + (index + 1) * 16 <= m_Registers.
size()) {
623 const uint64_t low = m_Registers.
read64(offset + index * 16);
624 const uint64_t high = m_Registers.
read64(offset + index * 16 + 8);
625 WARNING(
"Intel VT-d fault: SID=" <<
Hex << (high & 0xffff) <<
" IOVA=" << (low & ~0xfffULL)
626 <<
" reason=" << ((high >> 32) & 0xff));
627 m_Registers.
write64(1ULL << 63, offset + index * 16 + 8);
630 WARNING(
"Intel VT-d fault log overflow");
void setPciPosition(uint32_t bus, uint32_t device, uint32_t func)
virtual uint32_t read32(size_t offset=0)
virtual void write64(uint64_t value, size_t offset=0)
virtual uint64_t read64(size_t offset=0)
virtual size_t size() const
virtual void write32(uint32_t value, size_t offset=0)
physical_uintptr_t physicalAddress() const
static const size_t continuous
virtual physical_uintptr_t allocatePage(size_t pageConstraints=0)=0
static PhysicalMemoryManager & instance()
static const size_t force
static const size_t nonRamMemory
virtual void freePage(physical_uintptr_t page)=0
static const size_t below4GB
static const size_t CacheDisable
static const size_t KernelMode
static const size_t Write
static void cpuid(uint32_t inEax, uint32_t inEcx, uint32_t &eax, uint32_t &ebx, uint32_t &ecx, uint32_t &edx)
void EXPORTED_PUBLIC panic(const char *msg) NORETURN
uintptr_t physicalAddress(physical_uintptr_t address) PURE