The Pedigree Project 0.1
x64/VirtualAddressSpace.cc
1/*
2 * Copyright (c) 2008-2014, Pedigree Developers
3 *
4 * Please see the CONTRIB file in the root of the source tree for a full
5 * list of contributors.
6 *
7 * Permission to use, copy, modify, and distribute this software for any
8 * purpose with or without fee is hereby granted, provided that the above
9 * copyright notice and this permission notice appear in all copies.
10 *
11 * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
12 * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
13 * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
14 * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
15 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
16 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
17 * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
18 */
19
20#include "VirtualAddressSpace.h"
21#include "pedigree/kernel/LockGuard.h"
22#include "pedigree/kernel/Log.h"
23#include "pedigree/kernel/panic.h"
24#include "pedigree/kernel/process/Process.h"
25#include "pedigree/kernel/process/Scheduler.h"
26#include "pedigree/kernel/process/Thread.h"
27#include "pedigree/kernel/processor/PhysicalMemoryManager.h"
28#include "pedigree/kernel/processor/Processor.h"
29#include "pedigree/kernel/processor/ProcessorInformation.h"
30#include "pedigree/kernel/utilities/utility.h"
31
32#include "VirtualAddressSpace-internal.h"
33#include "utils.h"
34
35// Defined in boot-standalone.s
36extern void* pml4;
37
39 KERNEL_VIRTUAL_HEAP,
40 reinterpret_cast<uintptr_t>(&pml4) - reinterpret_cast<uintptr_t>(KERNEL_VIRTUAL_ADDRESS),
41 KERNEL_VIRTUAL_STACK);
42
43static void trackPages(VirtualAddressSpace& space, ssize_t v, ssize_t p, ssize_t s) {
44 // Track, if we can.
45 Thread* pThread = Processor::information().getCurrentThread();
46 if (pThread) {
47 Process* pProcess = pThread->getParent();
48 if (pProcess) {
49 if (pProcess->getAddressSpace() == &space)
50 pProcess = pProcess->addressSpaceOwner();
51 pProcess->trackPages(v, p, s);
52 }
53 }
54}
55
58}
59
61 return new X64VirtualAddressSpace();
62}
63
65 if (pMem < KERNEL_VIRTUAL_HEAP) {
66 return false;
67 } else if (pMem >= adjust_pointer(KERNEL_VIRTUAL_HEAP, KERNEL_VIRTUAL_HEAP_SIZE)) {
68 return false;
69 }
70
71 return true;
72}
73
75 if (pMem < m_Heap) {
76 WARNING("memIsInHeap: " << pMem << " is below the kernel heap.");
77 return false;
78 } else if (pMem >= getEndOfHeap()) {
79 WARNING("memIsInHeap: " << pMem << " is beyond the end of the heap (" << getEndOfHeap()
80 << ").");
81 return false;
82 } else
83 return true;
84}
86 if (m_Heap == KERNEL_VIRTUAL_HEAP) {
87 return reinterpret_cast<void*>(reinterpret_cast<uintptr_t>(KERNEL_VIRTUAL_HEAP) +
88 KERNEL_VIRTUAL_HEAP_SIZE);
89 } else {
90 return m_HeapEnd;
91 }
92}
93
94bool X64VirtualAddressSpace::isAddressValid(void* virtualAddress) {
95 if (reinterpret_cast<uint64_t>(virtualAddress) < 0x0000800000000000ULL ||
96 reinterpret_cast<uint64_t>(virtualAddress) >= 0xFFFF800000000000ULL) {
97 return true;
98 }
99 return false;
100}
101bool X64VirtualAddressSpace::isMapped(void* virtualAddress) {
103
104 size_t pml4Index = PML4_INDEX(virtualAddress);
105 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
106
107 // Is a page directory pointer table present?
108 if ((*pml4Entry & PAGE_PRESENT) != PAGE_PRESENT)
109 return false;
110
111 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
112 uint64_t* pageDirectoryPointerEntry =
113 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
114
115 // Is a page directory present?
116 if ((*pageDirectoryPointerEntry & PAGE_PRESENT) != PAGE_PRESENT)
117 return false;
118
119 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
120 uint64_t* pageDirectoryEntry =
121 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
122
123 // Is a page table or 2MB page present?
124 if ((*pageDirectoryEntry & PAGE_PRESENT) != PAGE_PRESENT)
125 return false;
126
127 // Is it a 2MB page?
128 if ((*pageDirectoryEntry & PAGE_2MB) == PAGE_2MB)
129 return true;
130
131 size_t pageTableIndex = PAGE_TABLE_INDEX(virtualAddress);
132 uint64_t* pageTableEntry =
133 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), pageTableIndex);
134
135 // Is a page present?
136 return (*pageTableEntry & (PAGE_PRESENT | PAGE_NO_ACCESS)) != 0;
137}
138
139bool X64VirtualAddressSpace::map(physical_uintptr_t physAddress, void* virtualAddress,
140 size_t flags) {
142 mutation.lock(m_Lock);
143
144 const bool mapped = mapUnlocked(physAddress, virtualAddress, flags, mutation, true);
145 if (mutation.failed()) {
146 mutation.panicInvalidationFailure();
147 }
148 return mapped;
149}
150
151bool X64VirtualAddressSpace::mapHuge(physical_uintptr_t physAddress, void* virtualAddress,
152 size_t count, size_t flags) {
153 const size_t smallPageSize = PhysicalMemoryManager::getPageSize();
154 const size_t twoMiB = 1UL << 21UL;
155 const size_t pagesPerTwoMiB = twoMiB / smallPageSize;
156 const uintptr_t virtualValue = reinterpret_cast<uintptr_t>(virtualAddress);
157
158 if (count < pagesPerTwoMiB || (physAddress % twoMiB) || (virtualValue % twoMiB)) {
159 return VirtualAddressSpace::mapHuge(physAddress, virtualAddress, count, flags);
160 }
161
162 // The page-table walkers only recognize the page-size bit at the page
163 // directory level. Keep this path at 2 MiB until every walker can safely
164 // stop at a 1 GiB page-directory-pointer entry.
165 const size_t numHugePages = count / pagesPerTwoMiB;
166 const size_t mappedPages = numHugePages * pagesPerTwoMiB;
167 {
169 mutation.lock(m_Lock);
170
171 // Clean up existing mappings before installing the huge-page entries.
172 for (size_t i = 0; i < mappedPages; ++i) {
173 unmapUnlocked(adjust_pointer(virtualAddress, i * smallPageSize), mutation, false);
174 if (mutation.failed()) {
175 mutation.panicInvalidationFailure();
176 }
177 }
178
179 size_t Flags = toFlags(flags, virtualAddress, true);
180 for (size_t i = 0; i < numHugePages; ++i) {
181 size_t pml4Index = PML4_INDEX(virtualAddress);
182 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
183
184 // Is a page directory pointer table present?
185 if (conditionalTableEntryAllocation(pml4Entry, flags) == false) {
186 return false;
187 }
188
189 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
190 uint64_t* pageDirectoryPointerEntry =
191 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
192
193 // Is a page directory present?
194 if (conditionalTableEntryAllocation(pageDirectoryPointerEntry, flags) == false) {
195 return false;
196 }
197
198 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
199 uint64_t* pageDirectoryEntry =
200 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
201
202 // The full 2 MiB range was unmapped above. Replacing its retained table
203 // must release that storage only after invalidating the paging structure.
204 const physical_uintptr_t oldPageTable =
205 (*pageDirectoryEntry & PAGE_PRESENT) && !(*pageDirectoryEntry & PAGE_2MB)
206 ? PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry)
207 : 0;
208 *pageDirectoryEntry = physAddress | PAGE_2MB | Flags;
209 if (!invalidateMapping(virtualAddress, mutation)) {
210 mutation.panicInvalidationFailure();
211 }
212 if (oldPageTable) {
214 }
215
216 virtualAddress = adjust_pointer(virtualAddress, twoMiB);
217 physAddress += twoMiB;
218 }
219 }
220
221 if (mappedPages < count) {
222 return mapHuge(physAddress, virtualAddress, count - mappedPages, flags);
223 }
224
225 return true;
226}
227
228bool X64VirtualAddressSpace::mapUnlocked(physical_uintptr_t physAddress, void* virtualAddress,
229 size_t flags, X64MappingMutationScope& mutation,
230 bool locked) {
231 size_t Flags = toFlags(flags, virtualAddress, true);
232 size_t pml4Index = PML4_INDEX(virtualAddress);
233 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
234
235 // Check if a page directory pointer table was present *before* the
236 // conditional allocation.
237 const bool pml4WasPresent = (*pml4Entry & PAGE_PRESENT) == PAGE_PRESENT;
238
239 // Is a page directory pointer table present?
240 if (conditionalTableEntryAllocation(pml4Entry, flags) == false) {
241 return false;
242 }
243
244 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
245 uint64_t* pageDirectoryPointerEntry =
246 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
247
248 // Is a page directory present?
249 if (conditionalTableEntryAllocation(pageDirectoryPointerEntry, flags) == false) {
250 return false;
251 }
252
253 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
254 uint64_t* pageDirectoryEntry =
255 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
256
257 // Is a page table present?
258 if (conditionalTableEntryAllocation(pageDirectoryEntry, flags) == false) {
259 return false;
260 }
261
262 size_t pageTableIndex = PAGE_TABLE_INDEX(virtualAddress);
263 uint64_t* pageTableEntry =
264 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), pageTableIndex);
265
266 // Is a page already present?
267 if (*pageTableEntry & (PAGE_PRESENT | PAGE_NO_ACCESS)) {
268 return false;
269 }
270
271 // Map the page
272 *pageTableEntry = physAddress | Flags;
273
274 trackPages(*this, 1, 0, 0);
275
276 // We don't need the lock to propagate the PDPT.
277 if (locked) {
278 mutation.unlock(m_Lock);
279 }
280
281 // If there wasn't a PDPT already present, and the address is in the kernel
282 // area of memory, we need to propagate this change across all address
283 // spaces.
284 if (!pml4WasPresent && Processor::m_Initialised == 2 && virtualAddress >= KERNEL_SPACE_START) {
285 uint64_t thisPml4Entry = *pml4Entry;
286 for (size_t i = 0; i < Scheduler::instance().getNumProcesses(); i++) {
288 if (!Scheduler::instance().acquireProcess(process, i)) {
289 continue;
290 }
291
292 X64VirtualAddressSpace* x64VAS =
293 reinterpret_cast<X64VirtualAddressSpace*>(process->getAddressSpace());
294 uint64_t* otherPml4Entry = TABLE_ENTRY(x64VAS->m_PhysicalPML4, pml4Index);
295 *otherPml4Entry = thisPml4Entry;
296 }
297 }
298
299 // If we were locked before, take the lock to enforce that.
300 if (locked) {
301 mutation.relock(m_Lock);
302 }
303
304 // A previously non-present entry can be cached, and upper-half entries are
305 // shared by every active address space. Invalidate only after any new PML4
306 // entry has been propagated.
307 if (!invalidateMapping(virtualAddress, mutation)) {
308 return false;
309 }
310
311 return true;
312}
313
314bool X64VirtualAddressSpace::getMapping(void* virtualAddress, physical_uintptr_t& physAddress,
315 size_t& flags) {
316 // Get a pointer to the page-table entry (Also checks whether the page is
317 // actually present or marked swapped out)
318 uint64_t* pageTableEntry = 0;
319 if (getPageTableEntry(virtualAddress, pageTableEntry) == false) {
320 return false;
321 }
322
323 // Extract the physical address and the flags
324 physAddress = PAGE_GET_PHYSICAL_ADDRESS(pageTableEntry);
325 flags = fromFlags(PAGE_GET_FLAGS(pageTableEntry), true);
326
327 return true;
328}
329
330bool X64VirtualAddressSpace::handleCopyOnWriteFault(void* virtualAddress, bool userMode) {
331 virtualAddress = page_align(virtualAddress);
332
333 {
335 uint64_t* pageTableEntry = nullptr;
336 if (!getPageTableEntry(virtualAddress, pageTableEntry)) {
337 return false;
338 }
339
340 const uint64_t pageFlags = *pageTableEntry;
341 if ((userMode && !(pageFlags & PAGE_USER)) ||
342 (pageFlags & (PAGE_NO_ACCESS | PAGE_WRITE_PROTECTED))) {
343 return false;
344 }
345 if ((pageFlags & PAGE_PRESENT) && (pageFlags & PAGE_WRITE) &&
346 !(pageFlags & PAGE_COPY_ON_WRITE)) {
347 return true;
348 }
349 if (!(pageFlags & PAGE_PRESENT) || !(pageFlags & PAGE_COPY_ON_WRITE) ||
350 (pageFlags & PAGE_SWAPPED)) {
351 return false;
352 }
353 }
354
356 const physical_uintptr_t replacement = physicalMemory.allocatePage();
357 if (!replacement) {
358 return false;
359 }
360
361 bool retireReplacement = true;
362 bool resolved = false;
363 physical_uintptr_t oldPhysical = 0;
364 {
366 mutation.lock(m_Lock);
367
368 uint64_t* pageTableEntry = nullptr;
369 if (getPageTableEntry(virtualAddress, pageTableEntry)) {
370 const uint64_t pageFlags = *pageTableEntry;
371 if ((userMode && !(pageFlags & PAGE_USER)) ||
372 (pageFlags & (PAGE_NO_ACCESS | PAGE_WRITE_PROTECTED))) {
373 resolved = false;
374 } else if ((pageFlags & PAGE_PRESENT) && (pageFlags & PAGE_WRITE) &&
375 !(pageFlags & PAGE_COPY_ON_WRITE)) {
376 resolved = true;
377 } else if ((pageFlags & PAGE_PRESENT) && (pageFlags & PAGE_COPY_ON_WRITE) &&
378 !(pageFlags & PAGE_SWAPPED)) {
379 oldPhysical = PAGE_GET_PHYSICAL_ADDRESS(pageTableEntry);
380 MemoryCopy(reinterpret_cast<void*>(physicalAddress(replacement)),
381 reinterpret_cast<void*>(physicalAddress(oldPhysical)),
383
384 uint64_t replacementFlags = PAGE_GET_FLAGS(pageTableEntry);
385 replacementFlags |= PAGE_WRITE;
386 replacementFlags &= ~(PAGE_COPY_ON_WRITE | PAGE_BORROWED | PAGE_SHARED);
387 __atomic_store_n(pageTableEntry, replacement | replacementFlags, __ATOMIC_RELEASE);
388 if (!invalidateMapping(virtualAddress, mutation)) {
389 mutation.panicInvalidationFailure();
390 }
391
392 retireReplacement = false;
393 resolved = true;
394 }
395 }
396 }
397
398 if (retireReplacement) {
399 physicalMemory.freePage(replacement);
400 }
401 if (oldPhysical) {
402 physicalMemory.freePage(oldPhysical);
403 }
404 return resolved;
405}
406
407bool X64VirtualAddressSpace::tryReadUser32(uintptr_t address, uint32_t& value) {
408 uintptr_t word = 0;
409 if (!tryAccessUserWord(address, sizeof(value), word, nullptr)) {
410 return false;
411 }
412 value = static_cast<uint32_t>(word);
413 return true;
414}
415
416bool X64VirtualAddressSpace::tryReadUserPointer(uintptr_t address, uintptr_t& value) {
417 return tryAccessUserWord(address, sizeof(value), value, nullptr);
418}
419
420bool X64VirtualAddressSpace::tryCompareExchangeUser32(uintptr_t address, uint32_t& expected,
421 uint32_t desired, bool& exchanged) {
422 const uint32_t original = expected;
423 uintptr_t observed = expected;
424 const uintptr_t replacement = desired;
425 exchanged = false;
426 if (!tryAccessUserWord(address, sizeof(expected), observed, &replacement)) {
427 return false;
428 }
429 expected = static_cast<uint32_t>(observed);
430 exchanged = expected == original;
431 return true;
432}
433
434VirtualAddressSpace::ResidentCopyStatus X64VirtualAddressSpace::copyResidentUserPage(
435 uintptr_t address, void* kernelBuffer, size_t bytes, bool write) {
436 const uintptr_t userEnd = 0x0000800000000000ULL;
437 const size_t pageSize = PhysicalMemoryManager::getPageSize();
438 if (!kernelBuffer || !bytes || bytes > pageSize || address < getUserStart() ||
439 address >= userEnd || address >= getKernelStart() || bytes > userEnd - address ||
440 bytes > getKernelStart() - address || (address & (pageSize - 1)) > pageSize - bytes)
441 return ResidentCopyStatus::Inaccessible;
442
444 uint64_t table = m_PhysicalPML4;
445 const size_t indices[] = {(address >> 39) & 0x1ff, (address >> 30) & 0x1ff,
446 (address >> 21) & 0x1ff};
447 for (const size_t index : indices) {
448 uint64_t* entry = TABLE_ENTRY(table, index);
449 const uint64_t flags = __atomic_load_n(entry, __ATOMIC_ACQUIRE);
450 if (!(flags & PAGE_PRESENT) || !(flags & PAGE_USER) || (flags & PAGE_2MB) ||
451 (write && !(flags & PAGE_WRITE)))
452 return ResidentCopyStatus::Inaccessible;
453 table = PAGE_GET_PHYSICAL_ADDRESS(entry);
454 }
455 uint64_t* entry = TABLE_ENTRY(table, (address >> 12) & 0x1ff);
456 const uint64_t flags = __atomic_load_n(entry, __ATOMIC_ACQUIRE);
457 if (!(flags & PAGE_PRESENT) || !(flags & PAGE_USER) ||
458 (flags &
459 (PAGE_SWAPPED | PAGE_NO_ACCESS | PAGE_CACHE_DISABLE | PAGE_WRITE_COMBINE | PAGE_PAT)) ||
460 (write && (!(flags & PAGE_WRITE) || (flags & (PAGE_COPY_ON_WRITE | PAGE_WRITE_PROTECTED)))))
461 return ResidentCopyStatus::Inaccessible;
462
463 // The owner cannot detach this latest leaf while its physical alias is in use.
464 void* userBytes = reinterpret_cast<void*>(
465 physicalAddress(PAGE_GET_PHYSICAL_ADDRESS(entry) + (address & (pageSize - 1))));
466 if (write)
467 MemoryCopy(userBytes, kernelBuffer, bytes);
468 else
469 MemoryCopy(kernelBuffer, userBytes, bytes);
470 __atomic_fetch_or(entry, PAGE_ACCESSED | (write ? PAGE_DIRTY : 0), __ATOMIC_RELEASE);
471 return ResidentCopyStatus::Success;
472}
473
474bool X64VirtualAddressSpace::tryAccessUserWord(uintptr_t address, size_t width, uintptr_t& value,
475 const uintptr_t* replacement) {
476 if (!address || (address % width) || address < getUserStart() ||
477 address >= 0x0000800000000000ULL || address >= getKernelStart() ||
478 address > getKernelStart() - width) {
479 return false;
480 }
481
482 const size_t pageSize = PhysicalMemoryManager::getPageSize();
483 const size_t pageOffset = address & (pageSize - 1);
484 if (pageOffset > pageSize - width) {
485 return false;
486 }
487
489 uint64_t* pageTableEntry = nullptr;
490 if (!getPageTableEntry(reinterpret_cast<void*>(address), pageTableEntry)) {
491 return false;
492 }
493
494 const uint64_t pageFlags = *pageTableEntry;
495 if (!(pageFlags & PAGE_PRESENT) || !(pageFlags & PAGE_USER) ||
496 (pageFlags & (PAGE_SWAPPED | PAGE_NO_ACCESS)) ||
497 (replacement &&
498 (!(pageFlags & PAGE_WRITE) || (pageFlags & (PAGE_COPY_ON_WRITE | PAGE_WRITE_PROTECTED))))) {
499 return false;
500 }
501
502 void* target = reinterpret_cast<void*>(
503 physicalAddress(PAGE_GET_PHYSICAL_ADDRESS(pageTableEntry) + pageOffset));
504 if (replacement) {
505 uint32_t expected = static_cast<uint32_t>(value);
506 __atomic_compare_exchange_n(reinterpret_cast<uint32_t*>(target), &expected,
507 static_cast<uint32_t>(*replacement), false, __ATOMIC_ACQ_REL,
508 __ATOMIC_ACQUIRE);
509 value = expected;
510 } else if (width == sizeof(uint32_t)) {
511 value = __atomic_load_n(reinterpret_cast<uint32_t*>(target), __ATOMIC_ACQUIRE);
512 } else {
513 value = __atomic_load_n(reinterpret_cast<uintptr_t*>(target), __ATOMIC_ACQUIRE);
514 }
515 return true;
516}
517
518bool X64VirtualAddressSpace::tryWriteUser32(uintptr_t address, uint32_t value) {
519 if (!address || (address % alignof(uint32_t)) || address < getUserStart() ||
520 address >= 0x0000800000000000ULL || address >= getKernelStart() ||
521 address > getKernelStart() - sizeof(value)) {
522 return false;
523 }
524
525 const size_t pageSize = PhysicalMemoryManager::getPageSize();
526 const size_t pageOffset = address & (pageSize - 1);
527 if (pageOffset > pageSize - sizeof(value)) {
528 return false;
529 }
530
532 uint64_t* pageTableEntry = nullptr;
533 if (!getPageTableEntry(reinterpret_cast<void*>(address), pageTableEntry)) {
534 return false;
535 }
536
537 const uint64_t pageFlags = *pageTableEntry;
538 if (!(pageFlags & PAGE_PRESENT) || !(pageFlags & PAGE_USER) || !(pageFlags & PAGE_WRITE) ||
539 (pageFlags & (PAGE_COPY_ON_WRITE | PAGE_SWAPPED | PAGE_NO_ACCESS | PAGE_WRITE_PROTECTED))) {
540 return false;
541 }
542
543 const physical_uintptr_t target = PAGE_GET_PHYSICAL_ADDRESS(pageTableEntry) + pageOffset;
544 __atomic_store_n(reinterpret_cast<uint32_t*>(physicalAddress(target)), value, __ATOMIC_RELEASE);
545 return true;
546}
547
548void X64VirtualAddressSpace::setFlags(void* virtualAddress, size_t newFlags) {
550 mutation.lock(m_Lock);
551
552 // Get a pointer to the page-table entry (Also checks whether the page is
553 // actually present or marked swapped out)
554 uint64_t* pageTableEntry = 0;
555 if (getPageTableEntry(virtualAddress, pageTableEntry) == false) {
556 mutation.panicWithoutRestoringInterrupts("VirtualAddressSpace::setFlags(): function misused");
557 }
558
559 // Set the flags
560 PAGE_SET_FLAGS(pageTableEntry, toFlags(newFlags, virtualAddress, true));
561
562 // Flush TLB - modified the mapping for this address.
563 if (!invalidateMapping(virtualAddress, mutation)) {
564 mutation.panicInvalidationFailure();
565 }
566}
567
568bool X64VirtualAddressSpace::trySetFlags(void* virtualAddress, size_t newFlags) {
570 mutation.lock(m_Lock);
571
572 // Get a pointer to the page-table entry (Also checks whether the page is
573 // actually present or marked swapped out)
574 uint64_t* pageTableEntry = 0;
575 if (!getPageTableEntry(virtualAddress, pageTableEntry) ||
576 !(*pageTableEntry & (PAGE_PRESENT | PAGE_NO_ACCESS)) || (*pageTableEntry & PAGE_SWAPPED)) {
577 return false;
578 }
579
580 // Set the flags
581 PAGE_SET_FLAGS(pageTableEntry, toFlags(newFlags, virtualAddress, true));
582
583 // Flush TLB - modified the mapping for this address.
584 if (!invalidateMapping(virtualAddress, mutation)) {
585 mutation.panicInvalidationFailure();
586 }
587 return true;
588}
589
590void X64VirtualAddressSpace::unmap(void* virtualAddress) {
592 mutation.lock(m_Lock);
593
594 if (!unmapUnlocked(virtualAddress, mutation)) {
595 if (mutation.failed()) {
596 mutation.panicInvalidationFailure();
597 }
598 mutation.panicWithoutRestoringInterrupts("VirtualAddressSpace::unmap(): function misused");
599 }
600}
601
602bool X64VirtualAddressSpace::detachMapping(void* virtualAddress, physical_uintptr_t& physical,
603 size_t& flags, size_t requiredFlags) {
605 mutation.lock(m_Lock);
606 physical = 0;
607 flags = 0;
608 uint64_t* entry = nullptr;
609 if (!getPageTableEntry(virtualAddress, entry)) {
610 return false;
611 }
612 physical = PAGE_GET_PHYSICAL_ADDRESS(entry);
613 flags = fromFlags(PAGE_GET_FLAGS(entry), true);
614 if ((flags & requiredFlags) != requiredFlags) {
615 return false;
616 }
617 if (!unmapUnlocked(virtualAddress, mutation)) {
618 mutation.panicInvalidationFailure();
619 }
620 return true;
621}
622
624 bool requireMapped) {
625 // Get a pointer to the page-table entry (Also checks whether the page is
626 // actually present or marked swapped out)
627 uint64_t* pageTableEntry = 0;
628 if (getPageTableEntry(virtualAddress, pageTableEntry) == false) {
629 // Not mapped! This is a panic for most cases, but private usage of
630 // unmap within X64VirtualAddressSpace is allowed to do this.
631 if (requireMapped) {
632 return false;
633 } else {
634 return true;
635 }
636 }
637
638 // Unmap the page
639 *pageTableEntry = 0;
640
641 trackPages(*this, -1, 0, 0);
642
643 // Keep empty paging structures for reuse; teardown releases private tables.
644 if (!invalidateMapping(virtualAddress, mutation)) {
645 return false;
646 }
647 return true;
648}
649
651 UserMemoryOperation operation(*this);
653
654 // Create a new virtual address space
655 X64VirtualAddressSpace* pClone =
657 if (pClone == 0) {
658 WARNING("X64VirtualAddressSpace: Clone() failed!");
659 return 0;
660 }
661
662 if (rawUserMemory().cloneInto(pClone->rawUserMemory()) != MemoryLockStatus::Success) {
663 delete pClone;
664 return nullptr;
665 }
666 pClone->m_HeapRegionId = m_HeapRegionId;
667
668 {
669 // Lock both address spaces so we can clone their mappings safely.
671 mutation.lock(pClone->m_Lock);
672 mutation.lock(m_Lock);
673
674 // The userspace area is only the bottom half of the address space - the top
675 // 256 PML4 entries are for the kernel, and these should be mapped anyway.
676 for (uint64_t i = 0; i < 256; i++) {
677 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, i);
678 if ((*pml4Entry & PAGE_PRESENT) != PAGE_PRESENT)
679 continue;
680
681 for (uint64_t j = 0; j < 512; j++) {
682 uint64_t* pdptEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), j);
683 if ((*pdptEntry & PAGE_PRESENT) != PAGE_PRESENT)
684 continue;
685
686 for (uint64_t k = 0; k < 512; k++) {
687 uint64_t* pdEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pdptEntry), k);
688 if ((*pdEntry & PAGE_PRESENT) != PAGE_PRESENT)
689 continue;
690
692 if ((*pdEntry & PAGE_2MB) == PAGE_2MB)
693 continue;
694
695 for (uint64_t l = 0; l < 512; l++) {
696 uint64_t* ptEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pdEntry), l);
697 if (!(*ptEntry & (PAGE_PRESENT | PAGE_NO_ACCESS)))
698 continue;
699
700 const uint64_t originalFlags = PAGE_GET_FLAGS(ptEntry);
701 uint64_t flags = originalFlags;
702 physical_uintptr_t physicalAddress = PAGE_GET_PHYSICAL_ADDRESS(ptEntry);
703
704 void* virtualAddress =
705 reinterpret_cast<void*>(((i & 0x100) ? (~0ULL << 48) : 0ULL) | /* Sign-extension. */
706 (i << 39) | (j << 30) | (k << 21) | (l << 12));
707
708 if (flags & PAGE_SHARED) {
709 // The physical address is now referenced (shared) in
710 // two address spaces, so make sure we hold another
711 // reference on it. Otherwise, if one of the two
712 // address spaces frees the page, the other may still
713 // refer to the bad page (and eventually double-free).
714 if (!(flags & PAGE_BORROWED)) {
716 }
717
718 // Handle shared mappings - don't copy the original
719 // page.
720 pClone->mapUnlocked(physicalAddress, virtualAddress, fromFlags(flags, true),
721 mutation);
722 if (mutation.failed()) {
723 mutation.panicInvalidationFailure();
724 }
725 continue;
726 }
727
728 // Map the new page in to the new address space for
729 // copy-on-write. This implies read-only (so we #PF for copy
730 // on write).
731 bool bWasCopyOnWrite = (flags & PAGE_COPY_ON_WRITE);
732 if (copyOnWrite) {
733 if (!(flags & (PAGE_WRITE | PAGE_COPY_ON_WRITE))) {
734 flags |= PAGE_WRITE_PROTECTED;
735 }
736 flags |= PAGE_COPY_ON_WRITE;
737 flags &= ~PAGE_WRITE;
738 }
739 pClone->mapUnlocked(physicalAddress, virtualAddress, fromFlags(flags, true), mutation);
740 if (mutation.failed()) {
741 mutation.panicInvalidationFailure();
742 }
743
744 // We need to modify the entry in *this* address space as
745 // well to also have the read-only and copy-on-write flag
746 // set, as otherwise writes in the parent process will cause
747 // the child process to see those changes immediately. An
748 // already protected CoW entry needs no second update.
749 if (copyOnWrite && flags != originalFlags) {
750 PAGE_SET_FLAGS(ptEntry, flags);
751 if (!invalidateMapping(virtualAddress, mutation)) {
752 mutation.panicInvalidationFailure();
753 }
754 }
755
756 // Pin the page twice - once for each side of the clone.
757 // But only pin for the parent if the parent page is not
758 // already copy on write. If we pin the CoW page, it'll be
759 // leaked when both parent and child terminate if the parent
760 // clone()s again.
761 if (!bWasCopyOnWrite)
764 }
765 }
766 }
767 }
768
769 // Before returning the address space, bring across metadata.
770 // Note though that if the parent of the clone (ie, this address space)
771 // is the kernel address space, we mustn't copy metadata or else the
772 // userspace defaults in the constructor get wiped out.
773
774 if (m_Heap < KERNEL_SPACE_START) {
775 pClone->m_Heap = m_Heap;
776 pClone->m_HeapEnd = m_HeapEnd;
777 }
778 }
779
780 // Now we pick up the stacks lock, so we can copy safely. However, we don't
781 // have the VirtualAddressSpace lock, so we can still safely use the heap
782 // without worrying about re-entering.
783 LockGuard<Spinlock> cloneStacksGuard(pClone->m_StacksLock);
785
786 if (m_pStackTop < KERNEL_SPACE_START) {
787 pClone->m_pStackTop = m_pStackTop;
788 for (Vector<Stack*>::Iterator it = m_freeStacks.begin(); it != m_freeStacks.end(); ++it) {
789 Stack* pNewStack = new Stack(**it);
790 pClone->m_freeStacks.pushBack(pNewStack);
791 }
792 }
793
795
796 return pClone;
797}
798
801 mutation.lock(m_Lock);
802
803 // The userspace area is only the bottom half of the address space - the top
804 // 256 PML4 entries are for the kernel, and these should be mapped anyway.
805 for (uint64_t i = 0; i < 256; i++) {
806 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, i);
807 if ((*pml4Entry & PAGE_PRESENT) != PAGE_PRESENT)
808 continue;
809
810 for (uint64_t j = 0; j < 512; j++) {
811 uint64_t* pdptEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), j);
812 if ((*pdptEntry & PAGE_PRESENT) != PAGE_PRESENT)
813 continue;
814
815 void* pdptVirtualAddress =
816 reinterpret_cast<void*>(((i & 0x100) ? (~0ULL << 48) : 0ULL) | (i << 39) | (j << 30));
817
818 for (uint64_t k = 0; k < 512; k++) {
819 uint64_t* pdEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pdptEntry), k);
820 if ((*pdEntry & PAGE_PRESENT) != PAGE_PRESENT)
821 continue;
822
823 // Address this region begins at.
824 void* regionVirtualAddress =
825 reinterpret_cast<void*>(((i & 0x100) ? (~0ULL << 48) : 0ULL) | /* Sign-extension. */
826 (i << 39) | (j << 30) | (k << 21));
827
828 if (regionVirtualAddress < USERSPACE_VIRTUAL_START)
829 continue;
830 if (regionVirtualAddress > KERNEL_SPACE_START)
831 break;
832
834 if ((*pdEntry & PAGE_2MB) == PAGE_2MB)
835 continue;
836
837 for (uint64_t l = 0; l < 512; l++) {
838 uint64_t* ptEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pdEntry), l);
839 if (!(*ptEntry & (PAGE_PRESENT | PAGE_NO_ACCESS)))
840 continue;
841
842 void* virtualAddress = reinterpret_cast<void*>(
843 reinterpret_cast<uintptr_t>(regionVirtualAddress) | (l << 12));
844
845 size_t flags = PAGE_GET_FLAGS(ptEntry);
846 physical_uintptr_t physicalAddress = PAGE_GET_PHYSICAL_ADDRESS(ptEntry);
847
848 // Release the physical memory if it is not shared with
849 // another process (eg, memory mapped file) Also avoid
850 // stumbling over a swapped out page.
854 const bool releasePhysicalPage =
855 (flags & (PAGE_SHARED | PAGE_SWAPPED | PAGE_BORROWED)) == 0;
856
857 // Free the page.
858 trackPages(*this, -1, 0, 0);
859 *ptEntry = 0;
860 if (!invalidateMapping(virtualAddress, mutation)) {
861 mutation.panicInvalidationFailure();
862 }
863 if (releasePhysicalPage) {
865 }
866 }
867
868 // Remove the table.
869 const physical_uintptr_t pageTable = PAGE_GET_PHYSICAL_ADDRESS(pdEntry);
870 *pdEntry = 0;
871 if (!invalidateMapping(regionVirtualAddress, mutation)) {
872 mutation.panicInvalidationFailure();
873 }
875 }
876
877 const physical_uintptr_t pageDirectory = PAGE_GET_PHYSICAL_ADDRESS(pdptEntry);
878 *pdptEntry = 0;
879 if (!invalidateMapping(pdptVirtualAddress, mutation)) {
880 mutation.panicInvalidationFailure();
881 }
883 }
884
885 const physical_uintptr_t pageDirectoryPointerTable = PAGE_GET_PHYSICAL_ADDRESS(pml4Entry);
886 *pml4Entry = 0;
887 void* pml4VirtualAddress =
888 reinterpret_cast<void*>(((i & 0x100) ? (~0ULL << 48) : 0ULL) | (i << 39));
889 if (!invalidateMapping(pml4VirtualAddress, mutation)) {
890 mutation.panicInvalidationFailure();
891 }
892 PhysicalMemoryManager::instance().freePage(pageDirectoryPointerTable);
893 }
894
895 // Reset heap; it's been wiped out by this reversion.
897}
898
899bool X64VirtualAddressSpace::mapPageStructures(physical_uintptr_t physAddress, void* virtualAddress,
900 size_t flags) {
901 // PageStack capacity is fully populated before AP startup. Keeping these
902 // special no-shootdown mappings bootstrap-only makes that lifetime
903 // invariant executable if a future caller tries to expand it at runtime.
904 if (Processor::m_Initialised == 2) {
905 panic("PageStack paging structures cannot expand after processor startup");
906 }
908
909 size_t Flags = toFlags(flags, virtualAddress);
910 size_t pml4Index = PML4_INDEX(virtualAddress);
911 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
912
913 // Is a page directory pointer table present?
914 if (conditionalTableEntryMapping(pml4Entry, physAddress, Flags) == true)
915 return true;
916
917 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
918 uint64_t* pageDirectoryPointerEntry =
919 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
920
921 // Is a page directory present?
922 if (conditionalTableEntryMapping(pageDirectoryPointerEntry, physAddress, Flags) == true)
923 return true;
924
925 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
926 uint64_t* pageDirectoryEntry =
927 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
928
929 // Is a page table present?
930 if (conditionalTableEntryMapping(pageDirectoryEntry, physAddress, Flags) == true)
931 return true;
932
933 size_t pageTableIndex = PAGE_TABLE_INDEX(virtualAddress);
934 uint64_t* pageTableEntry =
935 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), pageTableIndex);
936
937 // Is a page already present?
938 if ((*pageTableEntry & PAGE_PRESENT) != PAGE_PRESENT) {
939 *pageTableEntry = physAddress | flags;
940 return true;
941 }
942 return false;
943}
944
945bool X64VirtualAddressSpace::mapPageStructuresAbove4GB(physical_uintptr_t physAddress,
946 void* virtualAddress, size_t flags) {
947 if (Processor::m_Initialised == 2) {
948 panic("PageStack paging structures cannot expand after processor startup");
949 }
951
952 size_t Flags = toFlags(flags, virtualAddress);
953 size_t pml4Index = PML4_INDEX(virtualAddress);
954 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
955
956 // Is a page directory pointer table present?
957 if (conditionalTableEntryAllocation(pml4Entry, Flags) == false)
958 return true;
959
960 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
961 uint64_t* pageDirectoryPointerEntry =
962 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
963
964 // Is a page directory present?
965 if (conditionalTableEntryAllocation(pageDirectoryPointerEntry, Flags) == false)
966 return true;
967
968 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
969 uint64_t* pageDirectoryEntry =
970 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
971
972 // Is a page table present?
973 if (conditionalTableEntryAllocation(pageDirectoryEntry, Flags) == false)
974 return true;
975
976 size_t pageTableIndex = PAGE_TABLE_INDEX(virtualAddress);
977 uint64_t* pageTableEntry =
978 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), pageTableIndex);
979
980 // Is a page already present?
981 if ((*pageTableEntry & PAGE_PRESENT) != PAGE_PRESENT) {
982 *pageTableEntry = physAddress | flags;
983 return true;
984 }
985 return false;
986}
987
988size_t X64VirtualAddressSpace::runtimeMappingPages(uintptr_t base, size_t length) {
989 if (length > ~uintptr_t(0) - base)
990 return 0;
992 const uintptr_t end = base + length;
993 size_t count = 0;
994 for (uintptr_t address = base; address < end;) {
995 uint64_t table = m_PhysicalPML4;
996 size_t missingShift = 0;
997 for (size_t shift = 39; shift > 12; shift -= 9) {
998 const uint64_t entry = *TABLE_ENTRY(table, (address >> shift) & 511);
999 if (!(entry & PAGE_PRESENT) || (entry & PAGE_2MB)) {
1000 missingShift = shift;
1001 break;
1002 }
1003 table = entry & ~0x8780000000000FFFULL;
1004 }
1005 if (missingShift) {
1006 const uintptr_t next = ((address >> missingShift) + 1) << missingShift;
1007 if (next <= address)
1008 break;
1009 address = next;
1010 continue;
1011 }
1012 const uint64_t leaf = *TABLE_ENTRY(table, (address >> 12) & 511);
1013 if ((leaf & PAGE_RUNTIME) && (leaf & (PAGE_PRESENT | PAGE_NO_ACCESS)))
1014 ++count;
1016 }
1017 return count;
1018}
1019
1021 size_t sz = USERSPACE_VIRTUAL_STACK_SIZE;
1022 if (this == &m_KernelSpace)
1023 sz = KERNEL_STACK_SIZE;
1024 return doAllocateStack(sz);
1025}
1026
1028 if (stackSz == 0)
1029 return allocateStack();
1030 return doAllocateStack(stackSz);
1031}
1032
1034 if (this != &m_KernelSpace && Processor::getInterrupts())
1035 return allocateTrackedUserStack(sSize);
1036 // Native generic events retain their existing IRQ-phase fallback. Such an
1037 // image cannot advertise complete CURRENT/FUTURE memory-lock coverage.
1038 if (this != &m_KernelSpace && rawUserMemory().completeInventory())
1039 FATAL("Unprepared user stack in a complete memory-lock inventory");
1040 size_t flags = 0;
1041 bool bMapAll = false;
1042 if (this == &m_KernelSpace) {
1043 // Don't demand map kernel mode stacks.
1045 bMapAll = true;
1046 }
1047
1048 const size_t pageSz = PhysicalMemoryManager::getPageSize();
1049
1050 // Grab a new stack pointer. Use the list of freed stacks if we can,
1051 // otherwise adjust the internal stack pointer. Using the list of freed
1052 // stacks helps avoid having the virtual address creep downwards.
1053 void* pStack = 0;
1055 if (m_freeStacks.count() != 0) {
1056 Stack* poppedStack = m_freeStacks.popBack();
1057 if (poppedStack->getSize() >= sSize) {
1058 pStack = poppedStack->getTop();
1059 }
1060 delete poppedStack;
1061 }
1063
1064 if (!pStack) {
1065 // Need the main address space lock now so we can adjust the next stack
1066 // pointer without interference.
1067 m_Lock.acquire();
1068 pStack = m_pStackTop;
1069
1070 // Always leave one page unmapped between each stack to catch overflow.
1071 m_pStackTop = adjust_pointer(m_pStackTop, -(sSize + pageSz));
1072 m_Lock.release();
1073 }
1074
1075 // Map the top of the stack in proper.
1076 uintptr_t firstPage = reinterpret_cast<uintptr_t>(pStack) - pageSz;
1077 physical_uintptr_t phys = PhysicalMemoryManager::instance().allocatePage();
1078 if (!bMapAll)
1080 if (!map(phys, reinterpret_cast<void*>(firstPage), flags | VirtualAddressSpace::Write))
1081 WARNING("map() failed in doAllocateStack");
1082
1083 // Bring in the rest of the stack as CoW.
1084 uintptr_t stackBottom = reinterpret_cast<uintptr_t>(pStack) - sSize;
1085 for (uintptr_t addr = stackBottom; addr < firstPage; addr += pageSz) {
1086 size_t map_flags = 0;
1087
1088 if (!bMapAll) {
1089 // Copy first stack page on write.
1092 } else {
1094 map_flags = VirtualAddressSpace::Write;
1095 }
1096
1097 if (!map(phys, reinterpret_cast<void*>(addr), flags | map_flags))
1098 WARNING("CoW map() failed in doAllocateStack");
1099 }
1100
1101 Stack* stackInfo = new Stack(pStack, sSize);
1102 return stackInfo;
1103}
1104
1105VirtualAddressSpace::Stack* X64VirtualAddressSpace::allocateTrackedUserStack(size_t size) {
1106 const size_t page = PhysicalMemoryManager::getPageSize();
1107 if (!size || size > ~size_t(0) - (page - 1))
1108 return nullptr;
1109 size = (size + page - 1) & ~(page - 1);
1110 UserMemoryOperation operation(*this);
1111 const uint64_t id = rawUserMemory().nextRegionId();
1112 if (!id)
1113 return nullptr;
1114 // The operation gate serializes user stack allocation; kernel stacks use
1115 // their separate address space and retain the scheduler-safe path above.
1116 Stack* reusable = nullptr;
1117 {
1119 if (m_freeStacks.count() && m_freeStacks[m_freeStacks.count() - 1]->getSize() == size)
1120 reusable = m_freeStacks.popBack();
1121 }
1122 if (reusable && userMemoryPolicy() &&
1123 userMemoryPolicy()->overlapsManagedMemory(
1124 *this, reinterpret_cast<uintptr_t>(reusable->getBase()), size)) {
1125 delete reusable;
1126 reusable = nullptr;
1127 }
1128 void* top = reusable ? reusable->getTop() : m_pStackTop;
1129 const uintptr_t topValue = reinterpret_cast<uintptr_t>(top);
1130 if (topValue < size + page || topValue - size < getUserStart()) {
1131 if (reusable) {
1133 m_freeStacks.pushBack(reusable);
1134 }
1135 return nullptr;
1136 }
1137 Stack* stack = new Stack(top, size, id);
1138 if (!stack) {
1139 if (reusable) {
1141 m_freeStacks.pushBack(reusable);
1142 }
1143 return nullptr;
1144 }
1145 const uintptr_t base = topValue - size;
1146 UserRegion region{id, base, size, UserRegion::Kind::Stack, true};
1148 MemoryLockCharge charge;
1149 bool ready = rawUserMemory().prepareChange(nullptr, &region, plan) == MemoryLockStatus::Success;
1150 ready = ready && admitRawMemoryChange(*plan.get(), operation.privileged(), charge) &&
1151 prepareZeroPage();
1152 size_t mapped = 0;
1153 if (ready) {
1154 for (; mapped < size; mapped += page) {
1156 if (!map(m_ZeroPage, reinterpret_cast<void*>(base + mapped), CopyOnWrite)) {
1158 ready = false;
1159 break;
1160 }
1161 }
1162 // Eager admission breaks private CoW before publishing the new region.
1163 if (ready)
1164 ready = plan.get()->populate() == PopulationStatus::Success;
1165 }
1166 if (!ready) {
1167 for (size_t offset = 0; offset < mapped; offset += page) {
1168 physical_uintptr_t physical = 0;
1169 size_t flags = 0;
1170 if (detachMapping(reinterpret_cast<void*>(base + offset), physical, flags))
1172 }
1173 delete stack;
1174 if (reusable) {
1176 m_freeStacks.pushBack(reusable);
1177 }
1178 return nullptr;
1179 }
1180 commitRawMemoryChange(*plan.get(), charge);
1181 if (!reusable)
1182 m_pStackTop = reinterpret_cast<void*>(base - page);
1183 delete reusable;
1184 return stack;
1185}
1186
1188 if (!pStack)
1189 return;
1190 if (pStack->regionId()) {
1191 UserMemoryOperation operation(*this);
1192 const size_t removed = rawUserMemory().retireRegion(pStack->regionId());
1193 if (MemoryLockAccount* account = memoryLockAccount()) {
1194 MemoryLockCharge charge = account->charge();
1195 assert(removed <= charge.rawPages);
1196 charge.rawPages -= removed;
1197 account->publish(charge, account->futureMode());
1198 }
1200 m_freeStacks.pushBack(pStack);
1201 return;
1202 }
1203 const size_t pageSz = PhysicalMemoryManager::getPageSize();
1204
1205 // Clean up the stack
1206 uintptr_t stackTop = reinterpret_cast<uintptr_t>(pStack->getTop());
1207 for (size_t i = 0; i < pStack->getSize(); i += pageSz) {
1208 stackTop -= pageSz;
1209 void* v = reinterpret_cast<void*>(stackTop);
1210 if (!isMapped(v)) {
1211 continue;
1212 }
1213
1214 size_t flags = 0;
1215 physical_uintptr_t phys = 0;
1216 getMapping(v, phys, flags);
1217
1218 unmap(v);
1220 }
1221
1222 // Add the stack to the list; using the stacks lock, not the main address
1223 // space lock, as pushing could require mapping pages via the heap.
1225 m_freeStacks.pushBack(pStack);
1227}
1228
1230 assert(m_bKernelSpace || !m_ResidentProcessors.value());
1231 PhysicalMemoryManager& physicalMemoryManager = PhysicalMemoryManager::instance();
1232
1235
1236 // Drop back to the kernel address space. This will blow away the child's
1237 // mappings, but maintains shared pages as needed.
1239
1240 // Free the PageMapLevel4
1241 physicalMemoryManager.freePage(m_PhysicalPML4);
1242}
1243
1245 : VirtualAddressSpace(USERSPACE_VIRTUAL_HEAP),
1246 m_PhysicalPML4(0),
1247 m_ResidentProcessors(0),
1248 m_pStackTop(USERSPACE_VIRTUAL_STACK),
1249 m_freeStacks(),
1250 m_bKernelSpace(false),
1251 m_Lock(false, false),
1252 m_StacksLock(false) {
1253 // Allocate a new PageMapLevel4
1254 PhysicalMemoryManager& physicalMemoryManager = PhysicalMemoryManager::instance();
1255 m_PhysicalPML4 = physicalMemoryManager.allocatePage();
1256
1257 // Initialise the page directory
1258 ByteSet(reinterpret_cast<void*>(physicalAddress(m_PhysicalPML4)), 0, 0x800);
1259
1260 // Copy the kernel PageMapLevel4
1261 MemoryCopy(reinterpret_cast<void*>(physicalAddress(m_PhysicalPML4) + 0x800),
1262 reinterpret_cast<void*>(physicalAddress(m_KernelSpace.m_PhysicalPML4) + 0x800), 0x800);
1263}
1264
1265X64VirtualAddressSpace::X64VirtualAddressSpace(void* Heap, physical_uintptr_t PhysicalPML4,
1266 void* VirtualStack)
1267 : VirtualAddressSpace(Heap),
1268 m_PhysicalPML4(PhysicalPML4),
1269 m_ResidentProcessors(0),
1270 m_pStackTop(VirtualStack),
1271 m_freeStacks(),
1272 m_bKernelSpace(true),
1273 m_Lock(false, false),
1274 m_StacksLock(false) {}
1275
1277 uint64_t*& pageTableEntry) const {
1278 size_t pml4Index = PML4_INDEX(virtualAddress);
1279 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
1280
1281 // Is a page directory pointer table present?
1282 if ((*pml4Entry & PAGE_PRESENT) != PAGE_PRESENT)
1283 return false;
1284
1285 size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
1286 uint64_t* pageDirectoryPointerEntry =
1287 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
1288
1289 // Is a page directory present?
1290 if ((*pageDirectoryPointerEntry & PAGE_PRESENT) != PAGE_PRESENT)
1291 return false;
1292
1293 size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
1294 uint64_t* pageDirectoryEntry =
1295 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
1296
1297 // Is a page table or 2MB page present?
1298 if ((*pageDirectoryEntry & PAGE_PRESENT) != PAGE_PRESENT)
1299 return false;
1300 if ((*pageDirectoryEntry & PAGE_2MB) == PAGE_2MB)
1301 return false;
1302
1303 size_t pageTableIndex = PAGE_TABLE_INDEX(virtualAddress);
1304 pageTableEntry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), pageTableIndex);
1305
1306 // Is a page present?
1307 if ((*pageTableEntry & PAGE_PRESENT) != PAGE_PRESENT &&
1308 (*pageTableEntry & PAGE_SWAPPED) != PAGE_SWAPPED && !(*pageTableEntry & PAGE_NO_ACCESS))
1309 return false;
1310
1311 return true;
1312}
1313
1315 X64MappingMutationScope& mutation) {
1316#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1317 Thread* diagnosticThread = Processor::information().getCurrentThread();
1318 Process* diagnosticProcess = diagnosticThread ? diagnosticThread->getParent() : nullptr;
1319 if (diagnosticProcess && virtualAddress < KERNEL_SPACE_START) {
1320 diagnosticProcess->recordBenchmarkVmCounter(diagnosticProcess->getAddressSpace() == this
1321 ? Process::VmInvalidationActive
1322 : Process::VmInvalidationInactive);
1323 }
1324#endif
1325 if (!m_bKernelSpace && virtualAddress < KERNEL_SPACE_START) {
1326 // Pair PTE publication with switchAddressSpace's incoming residency bit.
1327 // A CPU missed here must load CR3 after the new PTE is visible; a CPU
1328 // leaving the mask has already flushed its private translations.
1329 __atomic_thread_fence(__ATOMIC_SEQ_CST);
1330 return mutation.invalidate(virtualAddress, m_ResidentProcessors.value());
1331 }
1332 // The upper half is shared; bootstrap also installs the kernel CR3 before
1333 // residency tracking is available.
1334 return mutation.invalidate(virtualAddress);
1335}
1336
1338 physical_uintptr_t* detachedTables) {
1339#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1340 Thread* diagnosticThread = Processor::information().getCurrentThread();
1341 Process* diagnosticProcess = diagnosticThread ? diagnosticThread->getParent() : nullptr;
1342 if (diagnosticProcess && diagnosticProcess->getAddressSpace() != this) {
1343 diagnosticProcess = nullptr;
1344 }
1345 size_t pteEntries = 0, pdeEntries = 0, pdptEntries = 0;
1346 if (diagnosticProcess) {
1347 diagnosticProcess->recordBenchmarkVmCounter(Process::VmTableRetirementScans);
1348 }
1349 auto publishDiagnostics = [&](size_t detachedCount) {
1350 if (!diagnosticProcess)
1351 return;
1352 diagnosticProcess->recordBenchmarkVmCounter(Process::VmDetachPteEntries, pteEntries);
1353 diagnosticProcess->recordBenchmarkVmCounter(Process::VmDetachPdeEntries, pdeEntries);
1354 diagnosticProcess->recordBenchmarkVmCounter(Process::VmDetachPdptEntries, pdptEntries);
1355 diagnosticProcess->recordBenchmarkVmCounter(Process::VmDetachTables, detachedCount);
1356 };
1357#endif
1358 const size_t pml4Index = PML4_INDEX(virtualAddress);
1359 uint64_t* pml4Entry = TABLE_ENTRY(m_PhysicalPML4, pml4Index);
1360 if ((*pml4Entry & PAGE_PRESENT) != PAGE_PRESENT) {
1361#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1362 publishDiagnostics(0);
1363#endif
1364 return 0;
1365 }
1366
1367 const size_t pageDirectoryPointerIndex = PAGE_DIRECTORY_POINTER_INDEX(virtualAddress);
1368 uint64_t* pageDirectoryPointerEntry =
1369 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), pageDirectoryPointerIndex);
1370 if ((*pageDirectoryPointerEntry & PAGE_PRESENT) != PAGE_PRESENT) {
1371#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1372 publishDiagnostics(0);
1373#endif
1374 return 0;
1375 }
1376
1377 const size_t pageDirectoryIndex = PAGE_DIRECTORY_INDEX(virtualAddress);
1378 uint64_t* pageDirectoryEntry =
1379 TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), pageDirectoryIndex);
1380 if ((*pageDirectoryEntry & PAGE_PRESENT) != PAGE_PRESENT ||
1381 (*pageDirectoryEntry & PAGE_2MB) == PAGE_2MB) {
1382#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1383 publishDiagnostics(0);
1384#endif
1385 return 0;
1386 }
1387
1388 for (size_t i = 0; i < 0x200; ++i) {
1389#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1390 ++pteEntries;
1391#endif
1392 uint64_t* entry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry), i);
1393 if (*entry & (PAGE_PRESENT | PAGE_SWAPPED | PAGE_NO_ACCESS)) {
1394#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1395 publishDiagnostics(0);
1396#endif
1397 return 0;
1398 }
1399 }
1400
1401 size_t detachedCount = 0;
1402 detachedTables[detachedCount++] = PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryEntry);
1403 *pageDirectoryEntry = 0;
1404
1405 for (size_t i = 0; i < 0x200; ++i) {
1406#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1407 ++pdeEntries;
1408#endif
1409 uint64_t* entry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry), i);
1410 if ((*entry & PAGE_PRESENT) == PAGE_PRESENT) {
1411#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1412 publishDiagnostics(detachedCount);
1413#endif
1414 return detachedCount;
1415 }
1416 }
1417
1418 detachedTables[detachedCount++] = PAGE_GET_PHYSICAL_ADDRESS(pageDirectoryPointerEntry);
1419 *pageDirectoryPointerEntry = 0;
1420
1421 // Every process PML4 contains its own copy of the upper-half entry. Retain
1422 // the shared PDPT root so clearing one PML4 cannot leave the others pointing
1423 // at freed storage.
1424 if (reinterpret_cast<uintptr_t>(virtualAddress) >=
1425 reinterpret_cast<uintptr_t>(KERNEL_SPACE_START)) {
1426#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1427 publishDiagnostics(detachedCount);
1428#endif
1429 return detachedCount;
1430 }
1431
1432 for (size_t i = 0; i < 0x200; ++i) {
1433#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1434 ++pdptEntries;
1435#endif
1436 uint64_t* entry = TABLE_ENTRY(PAGE_GET_PHYSICAL_ADDRESS(pml4Entry), i);
1437 if ((*entry & PAGE_PRESENT) == PAGE_PRESENT) {
1438#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1439 publishDiagnostics(detachedCount);
1440#endif
1441 return detachedCount;
1442 }
1443 }
1444
1445 detachedTables[detachedCount++] = PAGE_GET_PHYSICAL_ADDRESS(pml4Entry);
1446 *pml4Entry = 0;
1447#if PEDIGREE_BENCHMARK_VM_DIAGNOSTICS
1448 publishDiagnostics(detachedCount);
1449#endif
1450 return detachedCount;
1451}
1452
1453uint64_t X64VirtualAddressSpace::toFlags(size_t flags, void* virtualAddress, bool bFinal) const {
1454 uint64_t Flags = 0;
1455 if ((flags & KernelMode) == KernelMode) {
1456 // Private supervisor mappings must also be flushed by a CR3 switch.
1457 if (virtualAddress >= KERNEL_SPACE_START) {
1458 Flags |= PAGE_GLOBAL;
1459 }
1460 } else {
1461 Flags |= PAGE_USER;
1462 }
1463 if ((flags & Write) == Write)
1464 Flags |= PAGE_WRITE;
1465 if ((flags & WriteCombine) == WriteCombine)
1466 Flags |= PAGE_WRITE_COMBINE;
1467 if ((flags & CacheDisable) == CacheDisable)
1468 Flags |= PAGE_CACHE_DISABLE;
1469 if ((flags & Execute) != Execute)
1470 Flags |= PAGE_NX;
1471 if ((flags & Swapped) == Swapped)
1472 Flags |= PAGE_SWAPPED;
1473 else
1474 Flags |= PAGE_PRESENT;
1475 if (flags & Borrowed) {
1476 Flags |= PAGE_BORROWED;
1477 }
1478 if (flags & RuntimeMapping)
1479 Flags |= PAGE_RUNTIME;
1480 if (flags & NoAccess) {
1481 Flags &= ~PAGE_PRESENT;
1482 Flags |= PAGE_NO_ACCESS;
1483 }
1484 if (flags & WriteProtected) {
1485 Flags &= ~PAGE_WRITE;
1486 Flags |= PAGE_WRITE_PROTECTED;
1487 }
1488 if ((flags & CopyOnWrite) == CopyOnWrite)
1489 Flags |= PAGE_COPY_ON_WRITE;
1490 if ((flags & Shared) == Shared)
1491 Flags |= PAGE_SHARED;
1492 if (bFinal) {
1493 if ((flags & WriteThrough) == WriteThrough)
1494 Flags |= PAGE_WRITE_THROUGH;
1495 if ((flags & Accessed) == Accessed)
1496 Flags |= PAGE_ACCESSED;
1497 if ((flags & Dirty) == Dirty)
1498 Flags |= PAGE_DIRTY;
1499 if ((flags & ClearDirty) == ClearDirty)
1500 Flags &= ~PAGE_DIRTY;
1501 }
1502 return Flags;
1503}
1504
1505size_t X64VirtualAddressSpace::fromFlags(uint64_t Flags, bool bFinal) const {
1506 size_t flags = 0;
1507 if ((Flags & PAGE_USER) != PAGE_USER)
1508 flags |= KernelMode;
1509 if ((Flags & PAGE_WRITE) == PAGE_WRITE)
1510 flags |= Write;
1511 if ((Flags & PAGE_WRITE_COMBINE) == PAGE_WRITE_COMBINE)
1512 flags |= WriteCombine;
1513 if ((Flags & PAGE_CACHE_DISABLE) == PAGE_CACHE_DISABLE)
1514 flags |= CacheDisable;
1515 if ((Flags & PAGE_NX) != PAGE_NX)
1516 flags |= Execute;
1517 if ((Flags & PAGE_SWAPPED) == PAGE_SWAPPED)
1518 flags |= Swapped;
1519 if (Flags & PAGE_BORROWED)
1520 flags |= Borrowed;
1521 if (Flags & PAGE_RUNTIME)
1522 flags |= RuntimeMapping;
1523 if (Flags & PAGE_NO_ACCESS)
1524 flags |= NoAccess;
1525 if (Flags & PAGE_WRITE_PROTECTED)
1526 flags |= WriteProtected;
1527 if ((Flags & PAGE_COPY_ON_WRITE) == PAGE_COPY_ON_WRITE)
1528 flags |= CopyOnWrite;
1529 if ((Flags & PAGE_SHARED) == PAGE_SHARED)
1530 flags |= Shared;
1531 if (bFinal) {
1532 if ((Flags & PAGE_WRITE_THROUGH) == PAGE_WRITE_THROUGH)
1533 flags |= WriteThrough;
1534 if ((Flags & PAGE_ACCESSED) == PAGE_ACCESSED)
1535 flags |= Accessed;
1536 if ((Flags & PAGE_DIRTY) == PAGE_DIRTY)
1537 flags |= Dirty;
1538 }
1539 return flags;
1540}
1541
1542bool X64VirtualAddressSpace::conditionalTableEntryAllocation(uint64_t* tableEntry, uint64_t flags) {
1543 // Convert VirtualAddressSpace::* flags to X64 flags.
1544 flags = toFlags(flags, nullptr);
1545
1546 if ((*tableEntry & PAGE_PRESENT) != PAGE_PRESENT) {
1547 // Allocate a page
1549 uint64_t page = PMemoryManager.allocatePage();
1550 if (page == 0) {
1551 ERROR(
1552 "OOM in "
1553 "X64VirtualAddressSpace::conditionalTableEntryAllocation!");
1554 return false;
1555 }
1556
1557 // Add the WRITE and USER flags so that these can be controlled
1558 // on a page-granularity level.
1559 flags &= ~(PAGE_GLOBAL | PAGE_NX | PAGE_SWAPPED | PAGE_COPY_ON_WRITE | PAGE_NO_ACCESS |
1560 PAGE_WRITE_PROTECTED | PAGE_BORROWED);
1561 flags |= PAGE_WRITE | PAGE_USER | PAGE_PRESENT;
1562
1563 // Map the page.
1564 *tableEntry = page | flags;
1565
1566 // Zero the page directory pointer table.
1567 ByteSet(physicalAddress(reinterpret_cast<void*>(page)), 0,
1569 } else if (((*tableEntry & PAGE_USER) != PAGE_USER) && (flags & PAGE_USER)) {
1570 // Flags request user mapping, entry doesn't have that.
1571 *tableEntry |= PAGE_USER;
1572 }
1573
1574 return true;
1575}
1576
1578 uint64_t physAddress, uint64_t flags) {
1579 // Convert VirtualAddressSpace::* flags to X64 flags.
1580 flags = toFlags(flags, nullptr, true);
1581
1582 if ((*tableEntry & PAGE_PRESENT) != PAGE_PRESENT) {
1583 // Map the page. Add the WRITE and USER flags so that these can be
1584 // controlled on a page-granularity level.
1585 *tableEntry =
1586 physAddress | ((flags & ~(PAGE_GLOBAL | PAGE_NX | PAGE_SWAPPED | PAGE_COPY_ON_WRITE |
1587 PAGE_NO_ACCESS | PAGE_WRITE_PROTECTED | PAGE_BORROWED)) |
1588 PAGE_WRITE | PAGE_USER | PAGE_PRESENT);
1589
1590 // Zero the page directory pointer table
1591 ByteSet(physicalAddress(reinterpret_cast<void*>(physAddress)), 0,
1593
1594 return true;
1595 } else if (((*tableEntry & PAGE_USER) != PAGE_USER) && (flags & PAGE_USER)) {
1596 // Flags request user mapping, entry doesn't have that.
1597 *tableEntry |= PAGE_USER;
1598 }
1599
1600 return false;
1601}
virtual physical_uintptr_t allocatePage(size_t pageConstraints=0)=0
static PhysicalMemoryManager & instance()
virtual void freePage(physical_uintptr_t page)=0
virtual void pin(physical_uintptr_t page)=0
VirtualAddressSpace * getAddressSpace()
Definition Process.h:472
Process * addressSpaceOwner()
Definition Process.h:487
static bool getInterrupts()
static ProcessorInformation & information()
static size_t m_Initialised
Definition Processor.h:483
static Scheduler & instance()
Definition Scheduler.h:96
size_t getNumProcesses()
Definition Scheduler.cc:268
MUST_USE_RESULT bool acquireProcess(ProcessLease &lease, size_t n)
Definition Scheduler.cc:275
void release()
Definition Spinlock.cc:168
bool acquire(bool recurse=false, bool safe=true)
Definition Spinlock.cc:36
Process * getParent() const
Definition Thread.h:325
Iterator end()
Definition Vector.h:172
Iterator begin()
Definition Vector.h:162
static VirtualAddressSpace * create()
virtual bool mapHuge(physical_uintptr_t physAddress, void *virtualAddress, size_t count, size_t flags)
static EXPORTED_PUBLIC VirtualAddressSpace & getKernelAddressSpace()
MUST_USE_RESULT bool trySetFlags(void *virtualAddress, size_t newFlags) override
bool unmapUnlocked(void *virtualAddress, X64MappingMutationScope &mutation, bool requireMapped=true)
size_t fromFlags(uint64_t Flags, bool bFinal=false) const PURE
virtual bool tryReadUser32(uintptr_t address, uint32_t &value)
virtual bool handleCopyOnWriteFault(void *virtualAddress, bool userMode)
virtual void freeStack(Stack *pStack)
uint64_t toFlags(size_t flags, void *virtualAddress, bool bFinal=false) const PURE
virtual bool mapHuge(physical_uintptr_t physAddress, void *virtualAddress, size_t count, size_t flags)
virtual void unmap(void *virtualAddress)
virtual bool isAddressValid(void *virtualAddress)
virtual bool detachMapping(void *virtualAddress, physical_uintptr_t &physical, size_t &flags, size_t requiredFlags=0)
virtual bool memIsInKernelHeap(void *pMem)
virtual bool map(physical_uintptr_t physAddress, void *virtualAddress, size_t flags)
bool invalidateMapping(void *virtualAddress, X64MappingMutationScope &mutation)
bool conditionalTableEntryAllocation(uint64_t *tableEntry, uint64_t flags)
ResidentCopyStatus copyResidentUserPage(uintptr_t userAddress, void *kernelBuffer, size_t bytes, bool write) override
virtual VirtualAddressSpace * clone(bool copyOnWrite=true)
bool conditionalTableEntryMapping(uint64_t *tableEntry, uint64_t physAddress, uint64_t flags)
virtual bool tryCompareExchangeUser32(uintptr_t address, uint32_t &expected, uint32_t desired, bool &exchanged)
size_t detachEmptyTables(void *virtualAddress, physical_uintptr_t *detachedTables)
virtual bool memIsInHeap(void *pMem)
virtual bool tryWriteUser32(uintptr_t address, uint32_t value)
Stack * doAllocateStack(size_t sSize)
virtual bool getMapping(void *virtualAddress, physical_uintptr_t &physAddress, size_t &flags)
bool mapPageStructures(physical_uintptr_t physAddress, void *virtualAddress, size_t flags)
virtual void setFlags(void *virtualAddress, size_t newFlags)
virtual bool isMapped(void *virtualAddress)
bool getPageTableEntry(void *virtualAddress, uint64_t *&pageTableEntry) const
bool mapUnlocked(physical_uintptr_t physAddress, void *virtualAddress, size_t flags, X64MappingMutationScope &mutation, bool locked=false)
void EXPORTED_PUBLIC panic(const char *msg) NORETURN
Definition panic.cc:118
uintptr_t physicalAddress(physical_uintptr_t address) PURE
Definition utils.h:39
T popBack()
Definition Vector.h:303
void pushBack(const T &value)
Definition Vector.h:275
EXPORTED_PUBLIC void * page_align(void *p) PURE
Definition utility.cc:29
size_t count() const
Definition Vector.h:270