The Pedigree Project 0.1
Ext2Filesystem.cc
1/*
2 * Copyright (c) 2008-2014, Pedigree Developers
3 *
4 * Please see the CONTRIB file in the root of the source tree for a full
5 * list of contributors.
6 *
7 * Permission to use, copy, modify, and distribute this software for any
8 * purpose with or without fee is hereby granted, provided that the above
9 * copyright notice and this permission notice appear in all copies.
10 *
11 * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
12 * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
13 * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
14 * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
15 * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
16 * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
17 * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
18 */
19
20#include "Ext2Filesystem.h"
21#include "pedigree/kernel/LockGuard.h"
22#include "pedigree/kernel/Log.h"
23#include "pedigree/kernel/TargetInfo.h"
24#include "pedigree/kernel/compiler.h"
25#include "pedigree/kernel/machine/Disk.h"
26#include "pedigree/kernel/machine/Machine.h"
27#include "pedigree/kernel/machine/Timer.h"
28#include "pedigree/kernel/process/Process.h"
29#include "pedigree/kernel/process/Thread.h"
30#include "pedigree/kernel/processor/Processor.h"
31#include "pedigree/kernel/processor/ProcessorInformation.h"
32#include "pedigree/kernel/syscallError.h"
33#include "pedigree/kernel/utilities/StaticString.h"
34#include "pedigree/kernel/utilities/Vector.h"
35#include "pedigree/kernel/utilities/assert.h"
36#include "pedigree/kernel/utilities/utility.h"
37
38#include "Ext2Directory.h"
39#include "Ext2File.h"
40#include "Ext2Node.h"
41#include "Ext2Symlink.h"
42#include "ext2.h"
43#include "modules/system/users/Group.h"
44#include "modules/system/users/User.h"
45#include "modules/system/vfs/File.h"
46#include "modules/system/vfs/VFS.h"
47
48#ifndef EXT2_STANDALONE
49#include "modules/Module.h"
50#endif
51
52// The sparse block page. This is zeroed and made read-only. A handler is set
53// and if written, it traps.
56static uint8_t g_pSparseBlock[4096] ALIGN(4096) SECTION(".bss");
57
58#ifdef EXT2_STANDALONE
59extern uint32_t getUnixTimestamp();
60#else
61static uint32_t getUnixTimestamp() {
62 Timer* pTimer = Machine::instance().getTimer();
63 return pTimer->getUnixTimestamp();
64}
65#endif
66
67Ext2Filesystem::Ext2Filesystem()
68 : m_pSuperblock(0),
69 m_pGroupDescriptors(0),
70 m_pInodeTables(0),
71 m_pInodeBitmaps(0),
72 m_pBlockBitmaps(0),
73 m_BlockSize(0),
74 m_InodeSize(0),
75 m_nGroupDescriptors(0),
76#if THREADS || defined(STANDALONE_MUTEXES)
77 m_WriteLock(),
78 m_InodeTableLoadLock(),
79#endif
80 m_pRoot(0) {
81}
82
83Ext2Filesystem::~Ext2Filesystem() {
84 closeQuotaFiles();
85 delete m_pRoot;
86 drainAttributeWrites();
87
88 for (auto it = m_InodeStates.begin(); it != m_InodeStates.end(); ++it) {
89 assert(!it.value()->references);
90 delete it.value();
91 }
92 m_InodeStates.clear();
93
94 releaseMetadata();
95}
96
97void Ext2Filesystem::releaseMetadata() {
98 if (m_pDisk && m_pSuperblock) {
100 for (size_t group = 0; group < m_nGroupDescriptors; ++group) {
101 GroupDesc* descriptor = m_pGroupDescriptors[group];
102 if (!descriptor) {
103 continue;
104 }
105
106 if (m_pBlockBitmaps) {
107 const uint32_t start = LITTLE_TO_HOST32(descriptor->bg_block_bitmap);
108 for (size_t i = 0; i < m_pBlockBitmaps[group].count(); ++i) {
109 unpinBlock(start + i);
110 }
111 }
112 if (m_pInodeBitmaps) {
113 const uint32_t start = LITTLE_TO_HOST32(descriptor->bg_inode_bitmap);
114 for (size_t i = 0; i < m_pInodeBitmaps[group].count(); ++i) {
115 unpinBlock(start + i);
116 }
117 }
118 if (m_pInodeTables) {
119 const uint32_t start = LITTLE_TO_HOST32(descriptor->bg_inode_table);
120 for (size_t i = 0; i < m_pInodeTables[group].count(); ++i) {
121 if (m_pInodeTables[group][i])
122 unpinBlock(start + i);
123 }
124 }
125 }
126
127 const uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
128 for (size_t i = 0; i < m_nGroupDescriptors; ++i) {
129 if (m_pGroupDescriptors[i]) {
130 const size_t block = gdBlock + ((i * sizeof(GroupDesc)) / m_BlockSize);
131 unpinBlock(block);
132 }
133 }
134 }
135
136 // The successful read() is the persistent superblock reference.
137 m_pDisk->unpin(1024ULL);
138 }
139
140 delete[] m_pBlockBitmaps;
141 delete[] m_pInodeBitmaps;
142 delete[] m_pInodeTables;
143 delete[] m_pGroupDescriptors;
144 m_pBlockBitmaps = nullptr;
145 m_pInodeBitmaps = nullptr;
146 m_pInodeTables = nullptr;
147 m_pGroupDescriptors = nullptr;
148 m_pSuperblock = nullptr;
149}
150
151bool Ext2Filesystem::deviceRemoved(bool deviceAvailable) {
152 if (!m_pDisk) {
153 return true;
154 }
155 if (!beginDeviceRemoval(deviceAvailable)) {
156 return false;
157 }
158 m_bReadOnly = true;
159 {
160 LockGuard<Mutex> registry(m_InodeStateLock);
161 for (auto it = m_InodeStates.begin(); it != m_InodeStates.end(); ++it) {
162 auto* state = it.value();
163 LockGuard<Mutex> data(state->dataLock);
164 {
165 LockGuard<Mutex> writeback(state->writebackLock);
166 state->removedMetadata = *state->metadata;
167 state->metadata = &state->removedMetadata;
168 state->orphan = false;
169 }
170 if (state->cache && !state->cache->fill.shutdown(Cache::ShutdownMode::DiscardDeferred)) {
171 return false;
172 }
173 }
174 }
175 closeQuotaFiles();
176 for (size_t n = 0; n < m_AttributeWriteCount; ++n) {
177 if (m_AttributeWrites[n].ownsPin) {
178 m_pDisk->unpin(m_AttributeWrites[n].location);
179 }
180 }
181 m_AttributeWriteCount = 0;
182 releaseMetadata();
184}
185
187 String devName;
188 m_pDisk = pDisk;
189 pDisk->getName(devName);
190
191 // Attempt to read the superblock. A successful Disk::read() transfers the
192 // persistent reference that this filesystem holds until destruction.
193 const BufferView block = m_pDisk->read(1024ULL);
194 if (!block || block.size() < sizeof(Superblock)) {
195 if (block) {
196 m_pDisk->unpin(1024ULL);
197 }
198 ERROR("Ext2: Failed to read a superblock on " << devName);
199 return false;
200 }
201 m_pSuperblock = block.as<Superblock>();
202
203 // Read correctly?
204 if (LITTLE_TO_HOST16(m_pSuperblock->s_magic) != 0xEF53) {
205 ERROR("Ext2: Superblock was not found on device " << devName);
206 return false;
207 }
208
209 // Other creator formats assign different meanings to inode owner-high fields.
210 const uint32_t creator = LITTLE_TO_HOST32(m_pSuperblock->s_creator_os);
211 if (creator != 0 && creator != 1) {
212 ERROR("Ext2: unsupported inode creator format on " << devName);
213 return false;
214 }
215
216 // Clean?
217 m_MountState = LITTLE_TO_HOST16(m_pSuperblock->s_state);
218 if (m_MountState != EXT2_STATE_CLEAN) {
219 WARNING("Ext2: filesystem on device " << devName << " is not clean.");
220 }
221
222 // Compressed filesystem?
223 if (checkRequiredFeature(1)) {
224 WARNING("Ext2: filesystem on device " << devName
225 << " requires compression, some files may fail to read.");
226
227 // Compression type.
228 uint32_t algo_bitmap = LITTLE_TO_HOST32(m_pSuperblock->s_algo_bitmap);
229 switch (algo_bitmap) {
230 case EXT2_LZV1_ALG:
231 NOTICE("Ext2: filesystem on device '" << devName << "' uses compression algorithm LZV1.");
232 break;
233 case EXT2_LZRW3A_ALG:
234 NOTICE("Ext2: filesystem on device '" << devName << "' uses compression algorithm LZRW3A.");
235 break;
236 case EXT2_GZIP_ALG:
237 NOTICE("Ext2: filesystem on device '" << devName << "' uses compression algorithm gzip.");
238 break;
239 case EXT2_BZIP2_ALG:
240 NOTICE("Ext2: filesystem on device '" << devName << "' uses compression algorithm bzip2.");
241 break;
242 case EXT2_LZO_ALG:
243 NOTICE("Ext2: filesystem on device '" << devName << "' uses compression algorithm LZO.");
244 break;
245 default:
246 ERROR("Ext2: unknown compression algorithm " << algo_bitmap << " on device '" << devName
247 << "' -- cannot mount!");
248 return false;
249 }
250 }
251
254
255 // If we can, check extended superblock fields.
256 if (LITTLE_TO_HOST32(m_pSuperblock->s_rev_level) >= 1) {
257 // Non-standard inode sizes are permitted, handle that.
258 m_InodeSize = LITTLE_TO_HOST16(m_pSuperblock->s_inode_size);
259 } else {
260 m_InodeSize = sizeof(Inode);
261 }
262
263 // Calculate the block size.
264 m_BlockSize = 1024 << LITTLE_TO_HOST32(m_pSuperblock->s_log_block_size);
265
266 if (m_BlockSize > 4096) {
267 ERROR("Ext2: filesystem's block size is too large (must be 4096 or less, but is " << m_BlockSize
268 << ")");
269 return false;
270 }
272 ERROR("Ext2: filesystem block size " << m_BlockSize << " exceeds the target page size "
274 return false;
275 }
276
277 // Where is the group descriptor table?
278 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
279
280 // How many group descriptors do we have? Round up the result.
281 uint32_t inodeCount = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_count);
282 uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
283 if (!inodeCount || !inodesPerGroup) {
284 ERROR("Ext2: filesystem on device '" << devName << "' has invalid inode geometry.");
285 return false;
286 }
287 m_nGroupDescriptors = (inodeCount / inodesPerGroup) + ((inodeCount % inodesPerGroup) ? 1 : 0);
288
289 // Add an entry to the group descriptor tree for each GD.
291 for (size_t i = 0; i < m_nGroupDescriptors; ++i) {
292 m_pGroupDescriptors[i] = 0;
293 }
294 for (size_t i = 0; i < m_nGroupDescriptors; i++) {
295 uintptr_t idx = (i * sizeof(GroupDesc)) / m_BlockSize;
296 uintptr_t off = (i * sizeof(GroupDesc)) % m_BlockSize;
297
298 uintptr_t groupBlock = readBlock(gdBlock + idx);
299 if (!groupBlock) {
300 ERROR("Ext2: Failed to read block group descriptor " << i);
301 return false;
302 }
303 m_pGroupDescriptors[i] = reinterpret_cast<GroupDesc*>(groupBlock + off);
304 }
305
306 // Create our bitmap arrays and tables.
310
312
313 // load root directory and sanity check it
314 Inode* inode = getInode(EXT2_ROOT_INO);
315 if (!inode) {
316 ERROR("failed to retrieve root directory inode (corrupted inode table?");
317 return false;
318 }
319 if ((LITTLE_TO_HOST16(inode->i_mode) & 0xF000) != EXT2_S_IFDIR) {
320 ERROR("root directory is not a directory");
321 return false;
322 }
323 m_pRoot = new Ext2Directory(String(""), EXT2_ROOT_INO, inode, this, 0);
324
325 // cache volume label
326 bool hasVolumeLabel = LITTLE_TO_HOST32(m_pSuperblock->s_rev_level) >= 1;
327 if ((!hasVolumeLabel) || (m_pSuperblock->s_volume_name[0] == '\0')) {
329 str += "no-volume-label@";
330 str.append(reinterpret_cast<uintptr_t>(this), 16);
331 m_VolumeLabel.assign(str, str.length(), true);
332 } else {
333 char buffer[17];
334 StringCopyN(buffer, m_pSuperblock->s_volume_name, 16);
335 buffer[16] = '\0';
336 m_VolumeLabel.assign(buffer);
337 }
338
339 return m_bReadOnly || beginWritableMount();
340}
341
342Filesystem* Ext2Filesystem::probe(Disk* pDisk) {
343 Ext2Filesystem* pFs = new Ext2Filesystem();
344 if (!pFs->initialise(pDisk)) {
345 // No ext2 filesystem found - don't leak the filesystem object.
346 delete pFs;
347 return 0;
348 } else
349 return pFs;
350}
351
353 return m_pRoot;
354}
355
357 return m_VolumeLabel;
358}
359
361 OperationBarrier::Lease operation;
362 if (!tryAcquireOperation(operation) || !m_pSuperblock) {
363 uuid.clear();
364 return false;
365 }
366 const uint8_t* value = reinterpret_cast<const uint8_t*>(m_pSuperblock->s_uuid);
367 static const char digits[] = "0123456789abcdef";
368 char text[37];
369 size_t pos = 0;
370 for (size_t i = 0; i < 16; ++i) {
371 if (i == 4 || i == 6 || i == 8 || i == 10) {
372 text[pos++] = '-';
373 }
374 text[pos++] = digits[value[i] >> 4];
375 text[pos++] = digits[value[i] & 15];
376 }
377 text[pos] = 0;
378 uuid.assign(text, pos);
379 return true;
380}
381
382bool Ext2Filesystem::createNode(File* parent, const String& filename, uint32_t mask,
383 const String& value, size_t type, uint32_t inodeOverride) {
384 LockGuard<Mutex> quotaNamespace(m_QuotaNamespaceLock);
385 if (inodeOverride && isQuotaFile(inodeOverride)) {
386 SYSCALL_ERROR(NotEnoughPermissions);
387 return false;
388 }
389 // Quick sanity check;
390 if (!parent->isDirectory()) {
391 SYSCALL_ERROR(NotADirectory);
392 return false;
393 }
394
395 // The filename cannot be the special entries "." or "..".
396 if (filename.length() == 0 || !StringCompare(filename.cstr(), ".") ||
397 !StringCompare(filename.cstr(), "..")) {
398 SYSCALL_ERROR(InvalidArgument);
399 return false;
400 }
401
402#ifdef EXT2_STANDALONE
403 uint32_t uid = 0, gid = 0;
404#else
405 FilesystemCredentials credentials;
406 if (!Process::currentFilesystemCredentials(credentials)) {
407 SYSCALL_ERROR(PermissionDenied);
408 return false;
409 }
410 const uint32_t uid = credentials.uid, gid = credentials.gid;
411#endif
412
413 // Find a free inode.
414 uint32_t inode_num = inodeOverride;
415 if (!inode_num) {
416 inode_num = findFreeInode(uid, gid);
417 if (inode_num == 0) {
418 return false;
419 }
420 }
421
422 uint32_t timestamp = getUnixTimestamp();
423
424 // Populate the inode.
426 Inode* newInode = getInode(inode_num);
427 if (!inodeOverride) {
428 // Allocation has already cleared the inode and advanced its generation.
429 newInode->i_mode = HOST_TO_LITTLE16(mask | type);
430 newInode->i_atime = newInode->i_ctime = newInode->i_mtime = HOST_TO_LITTLE32(timestamp);
431 }
432
433 // If we have a value to store, and it's small enough, use the block
434 // indices.
435 if (value.length() && value.length() < 4 * 15) {
436 MemoryCopy(reinterpret_cast<void*>(newInode->i_block), value.cstr(), value.length());
437 newInode->i_size = HOST_TO_LITTLE32(value.length());
438 }
439 // Else case comes later, after pFile is created.
440
441 Ext2Directory* pE2Parent = reinterpret_cast<Ext2Directory*>(parent);
442 Ext2Node* pNewNode = 0;
443 Ext2Directory* pNewDirectory = nullptr;
444 bool dotEntryCreated = false;
445 bool dotDotEntryCreated = false;
446
447 // Create the new File object.
448 File* pFile = 0;
449 switch (type) {
450 case EXT2_S_IFREG: {
451 Ext2File* pNewFile = new Ext2File(filename, inode_num, newInode, this, parent);
452 if (!pNewFile || !pNewFile->valid()) {
453 if (!inodeOverride) {
454 releaseInode(inode_num, pNewFile);
455 }
456 delete pNewFile;
457 SYSCALL_ERROR(OutOfMemory);
458 return false;
459 }
460 pFile = pNewFile;
461 pNewNode = pNewFile;
462 break;
463 }
464 case EXT2_S_IFDIR: {
465 Ext2Directory* pE2Dir = new Ext2Directory(filename, inode_num, newInode, this, parent);
466 pFile = pE2Dir;
467 pNewNode = pE2Dir;
468 pNewDirectory = pE2Dir;
469
470 // If we already have an inode, assume we already have dot/dotdot
471 // entries and so don't need to make them.
472 if (!inodeOverride) {
473 // Dot entries are backing metadata, not separate VFS objects. Avoid
474 // manufacturing a second directory object and mutex for this inode.
475 syscallError(0);
476 dotEntryCreated = pE2Dir->addEntry(String("."), pE2Dir, EXT2_S_IFDIR);
477 if (dotEntryCreated) {
478 dotDotEntryCreated = pE2Dir->addEntry(String(".."), pE2Parent, EXT2_S_IFDIR);
479 }
480 if (!dotEntryCreated || !dotDotEntryCreated) {
481 const int failure = currentIoError();
482 if (dotEntryCreated && !pE2Dir->removeEntry(String("."), pE2Dir)) {
483 ERROR("EXT2: Failed to unwind a new directory's self link");
484 releaseInode(inode_num, pE2Dir);
485 } else if (!dotEntryCreated) {
486 releaseInode(inode_num, pE2Dir);
487 }
488 delete pE2Dir;
489 syscallError(failure);
490 return false;
491 }
492 }
493 break;
494 }
495 case EXT2_S_IFLNK: {
496 Ext2Symlink* pNewSymlink = new Ext2Symlink(filename, inode_num, newInode, this, parent);
497 pFile = pNewSymlink;
498 pNewNode = pNewSymlink;
499 break;
500 }
501 default:
502 FATAL("EXT2: Unrecognised file type: " << Hex << type);
503 break;
504 }
505
506 // Else case from earlier.
507 if (value.length() && value.length() >= 4 * 15) {
508 syscallError(0);
509 if (pFile->write(0ULL, value.length(), reinterpret_cast<uintptr_t>(value.cstr())) !=
510 value.length()) {
511 const int failure = currentIoError();
512 if (!inodeOverride)
513 releaseInode(inode_num, pNewNode);
514 delete pFile;
515 syscallError(failure);
516 return false;
517 }
518 }
519
520 // Add to the parent directory.
521 syscallError(0);
522 if (!pE2Parent->addEntry(filename, pFile, type)) {
523 const int failure = currentIoError();
524 ERROR("EXT2: Internal error adding directory entry.");
525 if (!inodeOverride) {
526 if (pNewDirectory && dotDotEntryCreated &&
527 !pNewDirectory->removeEntry(String(".."), pE2Parent)) {
528 ERROR("EXT2: Failed to unwind a new directory's parent link");
529 releaseInode(pE2Parent->getInodeNumber(), pE2Parent);
530 }
531 if (pNewDirectory && dotEntryCreated) {
532 if (!pNewDirectory->removeEntry(String("."), pNewDirectory)) {
533 ERROR("EXT2: Failed to unwind a new directory's self link");
534 releaseInode(inode_num, pNewNode);
535 }
536 } else {
537 releaseInode(inode_num, pNewNode);
538 }
539 }
540 delete pFile;
541 syscallError(failure);
542 return false;
543 }
544
545 // Edit the atime and mtime of the parent directory.
546 parent->setAccessedTime(timestamp);
547 parent->setModifiedTime(timestamp);
548
549 // Write updated inodes.
550 writeInode(inode_num);
551 writeInode(pE2Parent->getInodeNumber());
552
553 // Update directory count in the group descriptor.
554 if (type == EXT2_S_IFDIR) {
555#if THREADS || defined(STANDALONE_MUTEXES)
557#endif
558 uint32_t group = (inode_num - 1) / LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
559 GroupDesc* pDesc = m_pGroupDescriptors[group];
560
561 pDesc->bg_used_dirs_count++;
562
563 // Update group descriptor on disk.
565 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
566 uint32_t groupBlock = (group * sizeof(GroupDesc)) / m_BlockSize;
567 writeBlock(gdBlock + groupBlock);
568 }
569
570 // OK, now we can preallocate blocks if desired.
571 // Note: don't preallocate symlinks, which can store data in i_blocks.
572 if (m_pSuperblock->s_prealloc_blocks && !(pFile->isDirectory() || pFile->isSymlink())) {
573 pNewNode->ensureLargeEnough(m_pSuperblock->s_prealloc_blocks * m_BlockSize, 0, 0, true);
574 } else if (m_pSuperblock->s_prealloc_dir_blocks && pFile->isDirectory()) {
575 pNewNode->ensureLargeEnough(m_pSuperblock->s_prealloc_dir_blocks * m_BlockSize, 0, 0, true);
576 }
577
578 return true;
579}
580
581bool Ext2Filesystem::createFile(File* parent, const String& filename, uint32_t mask) {
582 return createNode(parent, filename, mask, String(""), EXT2_S_IFREG);
583}
584
585bool Ext2Filesystem::createDirectory(File* parent, const String& filename, uint32_t mask) {
586 if (!createNode(parent, filename, mask, String(""), EXT2_S_IFDIR)) {
587 return false;
588 }
589 return true;
590}
591
592bool Ext2Filesystem::createSymlink(File* parent, const String& filename, const String& value) {
593 return createNode(parent, filename, 0777, value, EXT2_S_IFLNK);
594}
595
596bool Ext2Filesystem::createLink(File* parent, const String& filename, File* target) {
597 Ext2Directory* pE2Parent = reinterpret_cast<Ext2Directory*>(parent);
598
599 Ext2Node* pNode = 0;
600 if (target->isDirectory()) {
601 Ext2Directory* pDirectory = static_cast<Ext2Directory*>(target);
602 pNode = pDirectory;
603 } else if (target->isSymlink()) {
604 Ext2Symlink* pSymlink = static_cast<Ext2Symlink*>(target);
605 pNode = pSymlink;
606 } else {
607 Ext2File* pFile = static_cast<Ext2File*>(target);
608 pNode = pFile;
609 }
610
611 if (!pNode) {
612 return false;
613 }
614
615 // Extract permissions and entry type.
616 Inode* inode = pNode->getInode();
617 uint32_t mask = LITTLE_TO_HOST16(inode->i_mode) & 0x0FFF;
618 size_t type = LITTLE_TO_HOST16(inode->i_mode) & 0xF000;
619
620 return createNode(parent, filename, mask, String(""), type, pNode->getInodeNumber());
621}
622
623bool Ext2Filesystem::removeNode(File* parent, const String& filename, File* file) {
624 LockGuard<Mutex> quotaNamespace(m_QuotaNamespaceLock);
625 if (isQuotaFile(file->getInode())) {
626 SYSCALL_ERROR(NotEnoughPermissions);
627 return false;
628 }
629 // Quick sanity check.
630 if (!parent->isDirectory()) {
631 SYSCALL_ERROR(IoError);
632 return false;
633 }
634
635 Ext2Node* pNode = 0;
636 if (file->isDirectory()) {
637 Ext2Directory* pDirectory = static_cast<Ext2Directory*>(file);
638 pNode = pDirectory;
639 } else if (file->isSymlink()) {
640 Ext2Symlink* pSymlink = static_cast<Ext2Symlink*>(file);
641 pNode = pSymlink;
642 } else {
643 Ext2File* pFile = static_cast<Ext2File*>(file);
644 pNode = pFile;
645 }
646
647 Ext2Directory* pE2Parent = reinterpret_cast<Ext2Directory*>(parent);
648 const bool ordinaryDirectory =
649 file->isDirectory() && !(filename.compare(".") || filename.compare(".."));
650 bool result = ordinaryDirectory
651 ? static_cast<Ext2Directory*>(file)->removeFromParent(pE2Parent, filename)
652 : pE2Parent->removeEntry(filename, pNode);
653
654 // Update the group descriptor directory count to reflect the deletion.
655 if (result && ordinaryDirectory) {
656#if THREADS || defined(STANDALONE_MUTEXES)
658#endif
659 uint32_t inode_num = pNode->getInodeNumber();
660
661 uint32_t group = (inode_num - 1) / LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
662 GroupDesc* pDesc = m_pGroupDescriptors[group];
663
664 pDesc->bg_used_dirs_count--;
665
666 // Update group descriptor on disk.
668 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
669 uint32_t groupBlock = (group * sizeof(GroupDesc)) / m_BlockSize;
670 writeBlock(gdBlock + groupBlock);
671 }
672
673 return result;
674}
675
676uintptr_t Ext2Filesystem::readBlock(uint32_t block) {
677 OperationBarrier::Lease operation;
678 if (!tryAcquireOperation(operation)) {
679 return 0;
680 }
681 if (block == 0)
682 return reinterpret_cast<uintptr_t>(g_pSparseBlock);
683
684 const uint64_t location = static_cast<uint64_t>(m_BlockSize) * static_cast<uint64_t>(block);
685 const BufferView view = m_pDisk->read(location);
686 if (!view || view.size() < m_BlockSize) {
687 if (view) {
688 m_pDisk->unpin(location);
689 }
690 return 0;
691 }
692 return view.address();
693}
694
695DiskReadView Ext2Filesystem::readBlockView(uint32_t block) {
696 OperationBarrier::Lease operation;
697 if (!tryAcquireOperation(operation)) {
698 return DiskReadView();
699 }
700 DiskReadView view = block ? m_pDisk->readView(static_cast<uint64_t>(m_BlockSize) * block)
701 : DiskReadView::borrowed(g_pSparseBlock, sizeof(g_pSparseBlock));
702 if (!view || !view.truncate(m_BlockSize))
703 return DiskReadView();
704 return view;
705}
706
707void Ext2Filesystem::writeBlock(uint32_t block) {
708 OperationBarrier::Lease operation;
709 if (!tryAcquireOperation(operation)) {
710 return;
711 }
712 if (block == 0)
713 return;
714
715 m_pDisk->write(static_cast<uint64_t>(m_BlockSize) * static_cast<uint64_t>(block));
716}
717
718bool Ext2Filesystem::pinBlock(uint64_t location) {
719 if (!location) {
720 return true;
721 }
722 return m_pDisk->pin(static_cast<uint64_t>(m_BlockSize) * location);
723}
724
725void Ext2Filesystem::unpinBlock(uint64_t location) {
726 if (!location) {
727 return;
728 }
729 m_pDisk->unpin(static_cast<uint64_t>(m_BlockSize) * location);
730}
731
732bool Ext2Filesystem::syncBlock(uint32_t block, bool async) {
733 if (!block) {
734 return true;
735 }
736 const uint64_t location = static_cast<uint64_t>(m_BlockSize) * block;
737 if (async) {
738 return m_pDisk->sync(location, true);
739 }
740 // syncPages skips absent pages; this operation requires a resident block.
741 if (!m_pDisk->pin(location)) {
742 return false;
743 }
744 // The batch path checks dirty generations and pinned writable aliases before
745 // submitting a page; an unchanged metadata page needs no new write or barrier.
746 const bool succeeded = m_pDisk->syncPages(&location, 1);
747 m_pDisk->unpin(location);
748 return succeeded;
749}
750
751bool Ext2Filesystem::syncInode(uint32_t inode, Ext2Node& node, bool includeNamespaceMetadata) {
752 OperationBarrier::Lease operation;
753 if (!tryAcquireOperation(operation)) {
754 return false;
755 }
756 if (!inode || !m_pSuperblock) {
757 return false;
758 }
759#if THREADS || defined(STANDALONE_MUTEXES)
760 LockGuard<Mutex> allocationGuard(m_WriteLock);
761#endif
762 // Keep the current payload resident while the dependency set releases its pins.
763 AttributeRetirement attributes;
764 if (readAttributeBlockLocked(node.m_pInode, attributes) != XattrStatus::Success ||
765 !flushAttributeWritesLocked())
766 return false;
767 const uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
768 const uint32_t blocksPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
769 if (!inodesPerGroup || !blocksPerGroup) {
770 return false;
771 }
772 const uint32_t inodeGroup = (inode - 1) / inodesPerGroup;
773 const uint32_t index = (inode - 1) % inodesPerGroup;
774 const size_t inodeTableIndex = (static_cast<uint64_t>(index) * m_InodeSize) / m_BlockSize;
775 if (inodeGroup >= m_nGroupDescriptors || !loadInodeTableBlock(inodeGroup, inodeTableIndex)) {
776 return false;
777 }
778
780 for (size_t i = 0; i < m_nGroupDescriptors; ++i) {
781 groups.pushBack(i == inodeGroup ? 1 : 0);
782 }
783 bool succeeded = true;
784 const uint32_t firstBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block);
785 auto includeBlockGroup = [&](uint32_t block) {
786 if (!block) {
787 return;
788 }
789 if (block < firstBlock || (block - firstBlock) / blocksPerGroup >= groups.count()) {
790 succeeded = false;
791 return;
792 }
793 groups[(block - firstBlock) / blocksPerGroup] = 1;
794 };
795 const uint32_t attributeBlock = LITTLE_TO_HOST32(node.m_pInode->i_file_acl);
796 includeBlockGroup(attributeBlock);
797 if (attributeBlock && !syncBlock(attributeBlock, false))
798 return false;
799 if (!node.isInlineSymlink()) {
800 for (size_t i = 0; i < 12; ++i)
801 includeBlockGroup(LITTLE_TO_HOST32(node.m_pInode->i_block[i]));
802 }
803
804 const size_t entries = m_BlockSize / sizeof(uint32_t);
806 succeeded = node.collectMappingPages(
807 node.isInlineSymlink() ? 0 : LITTLE_TO_HOST32(node.m_pInode->i_block[12]), 1, 12,
808 entries, mappings) &&
809 succeeded;
810 succeeded = node.collectMappingPages(
811 node.isInlineSymlink() ? 0 : LITTLE_TO_HOST32(node.m_pInode->i_block[13]), 2,
812 12 + entries, entries * entries, mappings) &&
813 succeeded;
814 succeeded = node.collectMappingPages(
815 node.isInlineSymlink() ? 0 : LITTLE_TO_HOST32(node.m_pInode->i_block[14]), 3,
816 12 + entries + entries * entries, entries * entries * entries, mappings) &&
817 succeeded;
818 for (const Ext2Node::MappingPage& mapping : mappings) {
819 includeBlockGroup(mapping.block);
820 if (mapping.depth == 1) {
821 const uint32_t* children = reinterpret_cast<const uint32_t*>(mapping.buffer);
822 for (size_t i = 0; i < entries; ++i) {
823 includeBlockGroup(LITTLE_TO_HOST32(children[i]));
824 }
825 }
826 succeeded = syncBlock(mapping.block, false) && succeeded;
827 unpinBlock(mapping.block);
828 }
829
830 // The inode is durable only once the allocation metadata needed to recover
831 // its data and mapping blocks has also reached the backend.
832 uint64_t locations[Disk::MaxSyncPages];
833 size_t locationCount = 0;
834 auto flushMetadata = [&] {
835 if (locationCount) {
836 succeeded = m_pDisk->syncPages(locations, locationCount) && succeeded;
837 locationCount = 0;
838 }
839 };
840 auto submitMetadata = [&](uint64_t location) {
841 for (size_t i = 0; i < locationCount; ++i) {
842 if (locations[i] == location) {
843 return;
844 }
845 }
846 locations[locationCount++] = location;
847 if (locationCount == Disk::MaxSyncPages) {
848 flushMetadata();
849 }
850 };
851 for (size_t group = 0; group < groups.count(); ++group) {
852 if (!groups[group]) {
853 continue;
854 }
855 GroupDesc* descriptor = m_pGroupDescriptors[group];
856 if (ensureFreeBlockBitmapLoaded(group)) {
857 const uint32_t start = LITTLE_TO_HOST32(descriptor->bg_block_bitmap);
858 for (size_t i = 0; i < m_pBlockBitmaps[group].count(); ++i) {
859 submitMetadata(static_cast<uint64_t>(start + i) * m_BlockSize);
860 }
861 } else {
862 succeeded = false;
863 }
864 const uint32_t descriptorBlock = firstBlock + 1 + (group * sizeof(GroupDesc)) / m_BlockSize;
865 submitMetadata(static_cast<uint64_t>(descriptorBlock) * m_BlockSize);
866 }
867 if (ensureFreeInodeBitmapLoaded(inodeGroup)) {
868 const uint32_t start = LITTLE_TO_HOST32(m_pGroupDescriptors[inodeGroup]->bg_inode_bitmap);
869 for (size_t i = 0; i < m_pInodeBitmaps[inodeGroup].count(); ++i) {
870 submitMetadata(static_cast<uint64_t>(start + i) * m_BlockSize);
871 }
872 } else {
873 succeeded = false;
874 }
875 submitMetadata(1024);
876 flushMetadata();
877 if (!succeeded) {
878 return false;
879 }
880 if (includeNamespaceMetadata) {
881 // Removed entries no longer identify the affected inode or freed blocks.
882 // Include loaded metadata from every group so namespace sync also submits
883 // child creation, link changes, and retirement, including failed retries.
884#if THREADS || defined(STANDALONE_MUTEXES)
886#endif
887 // These tables and bitmaps stay pinned for the filesystem lifetime. Let
888 // the disk share one durability barrier across a bounded set of pages.
889 for (size_t group = 0; group < m_nGroupDescriptors; ++group) {
890 GroupDesc* descriptor = m_pGroupDescriptors[group];
891 const uint32_t inodeTable = LITTLE_TO_HOST32(descriptor->bg_inode_table);
892 bool loadedInodeTable = false;
893 for (size_t i = 0; i < m_pInodeTables[group].count(); ++i) {
894 if (m_pInodeTables[group][i]) {
895 submitMetadata(static_cast<uint64_t>(inodeTable + i) * m_BlockSize);
896 loadedInodeTable = true;
897 }
898 }
899 const uint32_t blockBitmap = LITTLE_TO_HOST32(descriptor->bg_block_bitmap);
900 for (size_t i = 0; i < m_pBlockBitmaps[group].count(); ++i) {
901 submitMetadata(static_cast<uint64_t>(blockBitmap + i) * m_BlockSize);
902 }
903 const uint32_t inodeBitmap = LITTLE_TO_HOST32(descriptor->bg_inode_bitmap);
904 for (size_t i = 0; i < m_pInodeBitmaps[group].count(); ++i) {
905 submitMetadata(static_cast<uint64_t>(inodeBitmap + i) * m_BlockSize);
906 }
907 if (loadedInodeTable || m_pBlockBitmaps[group].count() || m_pInodeBitmaps[group].count()) {
908 const uint32_t descriptorBlock = firstBlock + 1 + (group * sizeof(GroupDesc)) / m_BlockSize;
909 submitMetadata(static_cast<uint64_t>(descriptorBlock) * m_BlockSize);
910 }
911 }
912 flushMetadata();
913 }
914 const uint32_t inodeBlock = LITTLE_TO_HOST32(m_pGroupDescriptors[inodeGroup]->bg_inode_table) +
915 ((index * m_InodeSize) / m_BlockSize);
916 return succeeded && syncBlock(inodeBlock, false);
917}
918
919uint32_t Ext2Filesystem::findFreeBlock(uint32_t inode) {
920 Vector<uint32_t> blocks;
921 if (findFreeBlocks(inode, 1, blocks)) {
922 return blocks[0];
923 }
924
925 return 0;
926}
927
928bool Ext2Filesystem::findFreeBlocks(uint32_t inode, size_t count, Vector<uint32_t>& blocks) {
929#if THREADS || defined(STANDALONE_MUTEXES)
931#endif
932
933 if (!count)
934 return true;
935 if (!blocks.tryReserve(count)) {
936 SYSCALL_ERROR(OutOfMemory);
937 return false;
938 }
939 if (!m_BlockSize || count > ~uint64_t(0) / m_BlockSize) {
940 SYSCALL_ERROR(ValueTooLarge);
941 return false;
942 }
943 const uint64_t reserved = static_cast<uint64_t>(count) * m_BlockSize;
944 auto status = prepareQuotaInodeLocked(inode);
945 if (status == QuotaStatus::Success)
946 status = m_Quota.reserve(inode, reserved);
947 if (!quotaSucceeded(status))
948 return false;
949 const uint32_t inodeNumber = inode;
950 // The ledger includes reservations before callers attach their mappings.
951 --inode;
952
953 // Try to allocate near the inode's group (but we can fall back to a
954 // different group if needed).
955 uint32_t group = inode / LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
956 uint32_t startGroup = group;
957 int error = 0;
958
959 for (; count && group < m_nGroupDescriptors; ++group) {
960 // A zero allocation result also means full storage. Check the fallible
961 // bitmap load separately so I/O failures survive reservation rollback.
962 if (m_pGroupDescriptors[group]->bg_free_blocks_count && !ensureFreeBlockBitmapLoaded(group)) {
963 error = currentIoError();
964 break;
965 }
966 count -= findFreeBlocksInGroup(group, count, blocks);
967 }
968
969 // Try again from the start of the disk if we couldn't find a group (if
970 // we started e.g. halfway through the disk due to the inode closeness
971 // thing above, we need to check the rest of the groups).
972 if (count && !error)
973 ERROR("FALLING BACK TO STARTING FROM ZERO");
974 for (group = 0; count && !error && group < startGroup; ++group) {
975 if (m_pGroupDescriptors[group]->bg_free_blocks_count && !ensureFreeBlockBitmapLoaded(group)) {
976 error = currentIoError();
977 break;
978 }
979 count -= findFreeBlocksInGroup(group, count, blocks);
980 }
981
982 if (count) {
983 for (uint32_t block : blocks) {
984 releaseBlockLocked(block);
985 }
986 blocks.clear();
987 m_Quota.refund(inodeNumber, reserved);
988 syscallError(error ? error : Error::NoSpaceLeftOnDevice);
989 }
990 return count == 0;
991}
992
993size_t Ext2Filesystem::findFreeBlocksInGroup(uint32_t group, size_t maxCount,
994 Vector<uint32_t>& blocks) {
995 if (!maxCount) {
996 return 0;
997 }
998
999 const uint32_t blocksPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
1000 const size_t bitmapBlockBytes = m_BlockSize;
1001 size_t currentCount = 0;
1002
1003 // Any free blocks here?
1004 GroupDesc* pDesc = m_pGroupDescriptors[group];
1005 if (!pDesc->bg_free_blocks_count) {
1006 // No blocks free in this group.
1007 return currentCount;
1008 }
1009
1010 if (!ensureFreeBlockBitmapLoaded(group)) {
1011 return 0;
1012 }
1013
1014 // 8 blocks per byte - i == bitmap offset in bytes.
1015 Vector<size_t>& list = m_pBlockBitmaps[group];
1016 const uint32_t bytesToSearch = blocksPerGroup >> 3;
1017 size_t idx = 0;
1018
1019 // Block bitmap pointer.
1020 typedef uint64_t searchType;
1021 size_t base = list[idx];
1022 searchType* ptr = reinterpret_cast<searchType*>(base);
1023 searchType* ptr_end = adjust_pointer(ptr, bitmapBlockBytes);
1024
1025 // Find a free block in this group.
1026 bool changedBitmap = false;
1027 while (true) {
1028 // Grab the specific block for the bitmap.
1031 searchType tmp = *ptr;
1032
1033 // Bitmap full of bits? Skip it.
1034 if (tmp != static_cast<searchType>(-1)) {
1035 // Check each bit in this field.
1036 for (size_t j = 0; j < (sizeof(searchType) * 8); j++, tmp >>= static_cast<searchType>(1)) {
1037 // Free?
1038 if ((tmp & 1) == 0) {
1039 // This block is free! Mark used.
1040 *ptr |= (static_cast<searchType>(1) << j);
1041 pDesc->bg_free_blocks_count--;
1042
1043 // Yes, we changed the bitmap.
1044 changedBitmap = true;
1045
1046 // Update superblock.
1047 m_pSuperblock->s_free_blocks_count--;
1048
1049 // First block of this group...
1050 uint32_t result = group * LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
1051 // Add the data block offset for this filesystem.
1052 result += LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block);
1053 // Blocks skipped so far (i == offset in bytes)...
1054 result += ((idx * bitmapBlockBytes) + (reinterpret_cast<uintptr_t>(ptr) - base)) << 3;
1055 // Blocks skipped so far (j == bits ie blocks)...
1056 result += j;
1057 // Return block.
1058 blocks.pushBack(result);
1059
1060 // Check if we're done - we have nothing left to do if
1061 // there's no more blocks free in this bitmap.
1062 if ((++currentCount >= maxCount) || (!pDesc->bg_free_blocks_count)) {
1063 break;
1064 }
1065 }
1066 }
1067 }
1068
1069 // Did we make changes to the bitmap? Write back now if so - we don't
1070 // want to keep writing over and over if e.g. we're setting more than
1071 // one block above.
1072 if (changedBitmap) {
1073 // Update bitmap on disk.
1074 uint32_t desc_block = LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_block_bitmap) + idx;
1075 writeBlock(desc_block);
1076
1077 changedBitmap = false;
1078 }
1079
1080 // Are we finished with this loop?
1081 if (currentCount >= maxCount) {
1082 break;
1083 }
1084
1085 // Haven't found anything yet - need to take care here.
1086 if (++ptr >= ptr_end) {
1087 if ((++idx * bitmapBlockBytes) >= bytesToSearch)
1088 break;
1089
1090 base = list[idx];
1091 ptr = reinterpret_cast<searchType*>(base);
1092 ptr_end = adjust_pointer(ptr, bitmapBlockBytes);
1093 }
1094 }
1095
1096 if (currentCount >= maxCount) {
1097 // Write back the superblock/group descriptor updates now.
1098 m_pDisk->write(1024ULL);
1099
1100 // Update group descriptor on disk.
1102 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
1103 uint32_t groupBlock = (group * sizeof(GroupDesc)) / m_BlockSize;
1104 writeBlock(gdBlock + groupBlock);
1105 }
1106
1107 return currentCount;
1108}
1109
1110void Ext2Filesystem::releaseBlock(uint32_t block, uint32_t inode) {
1111#if THREADS || defined(STANDALONE_MUTEXES)
1113#endif
1114
1115 releaseBlockLocked(block, inode);
1116}
1117
1118bool Ext2Filesystem::prepareBlockReleaseLocked(uint32_t block) {
1119 const uint32_t first = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block);
1120 const uint32_t perGroup = LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
1121 if (block <= first || !perGroup || (block - first) / perGroup >= m_nGroupDescriptors) {
1122 SYSCALL_ERROR(IoError);
1123 return false;
1124 }
1125 return ensureFreeBlockBitmapLoaded((block - first) / perGroup);
1126}
1127
1128bool Ext2Filesystem::prepareInodeWrite(uint32_t inode) {
1129 const uint32_t perGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1130 if (!inode || !perGroup || (inode - 1) / perGroup >= m_nGroupDescriptors) {
1131 SYSCALL_ERROR(IoError);
1132 return false;
1133 }
1134 if (!m_BlockSize) {
1135 SYSCALL_ERROR(IoError);
1136 return false;
1137 }
1138 const size_t block = (static_cast<uint64_t>((inode - 1) % perGroup) * m_InodeSize) / m_BlockSize;
1139 return loadInodeTableBlock((inode - 1) / perGroup, block) != 0;
1140}
1141
1142void Ext2Filesystem::releaseBlockLocked(uint32_t block, uint32_t inode) {
1143 // In some ext2 filesystems, this is zero so we don't need to do this. But
1144 // for those that do, not doing this messes up the bit offsets below.
1145 block -= LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block);
1146
1147 uint32_t blocksPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
1148 uint32_t group = block / blocksPerGroup;
1149 uint32_t index = block % blocksPerGroup;
1150
1151 if (!block) {
1152 // Error out, zero is used as a sentinel in a few places - and we almost
1153 // certainly never actually mean to free block zero.
1154 FATAL("Releasing block zero!");
1155 }
1156
1157 if (!ensureFreeBlockBitmapLoaded(group)) {
1158 return;
1159 }
1160
1161 // Free block.
1162 GroupDesc* pDesc = m_pGroupDescriptors[group];
1163
1164 // Index = block offset from the start of this block.
1165 size_t bitmapField = (index / 8) / m_BlockSize;
1166 size_t bitmapOffset = (index / 8) % m_BlockSize;
1167
1168 Vector<size_t>& list = m_pBlockBitmaps[group];
1169 uintptr_t diskBlock = list[bitmapField];
1170 uint8_t* ptr = reinterpret_cast<uint8_t*>(diskBlock + bitmapOffset);
1171 uint8_t bit = (index % 8);
1172 if ((*ptr & (1 << bit)) == 0) {
1173 ERROR("bit already freed for block " << Dec << block << Hex);
1174 return;
1175 }
1176 *ptr &= ~(1 << bit);
1177 if (inode)
1178 m_Quota.refund(inode, m_BlockSize);
1179
1180 // Update hints.
1181 pDesc->bg_free_blocks_count++;
1182 m_pSuperblock->s_free_blocks_count++;
1183
1184 // Update superblock.
1185 m_pDisk->write(1024ULL);
1186
1187 // Update bitmap on disk.
1188 uint32_t desc_block = LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_block_bitmap) + bitmapField;
1189 writeBlock(desc_block);
1190
1191 // Update group descriptor on disk.
1193 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
1194 uint32_t groupBlock = (group * sizeof(GroupDesc)) / m_BlockSize;
1195 writeBlock(gdBlock + groupBlock);
1196}
1197
1198Ext2InodeState* Ext2Filesystem::acquireInodeState(uint32_t inode, Inode* metadata) {
1199 LockGuard<Mutex> guard(m_InodeStateLock);
1200 Ext2InodeState* state = m_InodeStates.lookup(inode);
1201 if (state) {
1202 if (!state->references && !state->cache) {
1203 state->reloadMappings(metadata, this);
1204 }
1205 ++state->references;
1206 return state;
1207 }
1208 state = new Ext2InodeState(metadata, this);
1209 m_InodeStates.insert(inode, state);
1210 return state;
1211}
1212
1213Ext2InodeState* Ext2Filesystem::acquireInodeStateLocked(uint32_t inode, Inode* metadata) {
1214 Ext2InodeState* state = m_InodeStates.lookup(inode);
1215 if (state) {
1216 if (!state->references && !state->cache) {
1217 state->reloadMappings(metadata, this);
1218 }
1219 ++state->references;
1220 return state;
1221 }
1222 state = new Ext2InodeState(metadata, this);
1223 if (!state || !m_InodeStates.tryInsert(inode, state)) {
1224 delete state;
1225 return nullptr;
1226 }
1227 return state;
1228}
1229
1230void Ext2Filesystem::releaseInodeState(uint32_t inode, Ext2InodeState* state, Ext2Node* lastNode) {
1231 LockGuard<Mutex> stateGuard(m_InodeStateLock);
1232 assert(state == m_InodeStates.lookup(inode) && state->references);
1233 if (--state->references) {
1234 return;
1235 }
1236 if (!state->orphan) {
1237 assert(!state->pageLoans && !state->files.count());
1238 // Keep the nonreusable futex identity through the linked inode lifetime,
1239 // but reload potentially large block maps when an alias next opens it.
1240 if (!state->cache) {
1241 state->blocks.clear(true);
1242 state->metadataBlocks = 0;
1243 }
1244 state->files.clear(true);
1245 return;
1246 }
1247 m_InodeStates.remove(inode);
1248 if (state->orphan && !isDeviceRemoved()) {
1249#if THREADS || defined(STANDALONE_MUTEXES)
1251#endif
1252 retireInodeLocked(inode, lastNode);
1253 }
1254 delete state;
1255}
1256
1257bool Ext2Filesystem::releaseInode(uint32_t inodeNumber, Ext2Node* retiringNode) {
1258 LockGuard<Mutex> stateGuard(m_InodeStateLock);
1259 Ext2InodeState* state = m_InodeStates.lookup(inodeNumber);
1260 LockGuard<Mutex> metadataGuard(state ? state->writebackLock : m_InodeStateLock, state != nullptr);
1261#if THREADS || defined(STANDALONE_MUTEXES)
1263#endif
1264 const bool remove = decreaseInodeRefcount(inodeNumber);
1265 if (remove) {
1266 if (state) {
1267 state->orphan = true;
1268 state->inodeEvents.beginRetirement();
1269 } else {
1270 retireInodeLocked(inodeNumber, retiringNode);
1271 }
1272 }
1273 return remove;
1274}
1275
1276void Ext2Filesystem::retireInodeLocked(uint32_t inodeNumber, Ext2Node* retiringNode) {
1277 Inode* pInode = getInode(inodeNumber);
1278 if (!pInode) {
1279 m_TeardownFailed = true;
1280 return;
1281 }
1282 const uint32_t inodeIndex = inodeNumber - 1; // Inode zero is undefined, so it's not used.
1283
1284 uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1285 uint32_t group = inodeIndex / inodesPerGroup;
1286 uint32_t index = inodeIndex % inodesPerGroup;
1287
1288 uint32_t allocatedBlocks = 0;
1289 bool inlineSymlink = false;
1290 AttributeRetirement attributes;
1291 if (!Ext2Node::decodeAllocation(*pInode, m_BlockSize, allocatedBlocks, inlineSymlink) ||
1292 prepareAttributeRetirementLocked(pInode, attributes) != XattrStatus::Success ||
1293 !ensureFreeInodeBitmapLoaded(group) ||
1294 (attributes.block && reserveAttributeWritesLocked(12) != XattrStatus::Success)) {
1295 ERROR("Ext2: retaining an orphan inode whose attributes could not be retired");
1296 m_TeardownFailed = true;
1297 return;
1298 }
1299 {
1300 // Keep the allocation bit set until all old data has been retired. This
1301 // prevents a concurrent creator from reusing the inode before wipe()
1302 // finishes zeroing it.
1303 if (retiringNode && !retiringNode->wipe(true)) {
1304 ERROR("Ext2: retaining an orphan inode whose blocks could not be retired");
1305 m_TeardownFailed = true;
1306 return;
1307 }
1308
1309 if (attributes.block) {
1310 const uint32_t sectors = m_BlockSize / 512;
1311 const uint32_t allocated = LITTLE_TO_HOST32(pInode->i_blocks);
1312 assert(allocated >= sectors);
1313 pInode->i_file_acl = 0;
1314 pInode->i_blocks = HOST_TO_LITTLE32(allocated - sectors);
1315 commitAttributeRetirementLocked(attributes);
1316 recordAttributeInodeLocked(inodeNumber);
1317 }
1318 // Set dtime on inode.
1319 pInode->i_dtime = HOST_TO_LITTLE32(getUnixTimestamp());
1320
1321 if (!ensureFreeInodeBitmapLoaded(group)) {
1322 m_TeardownFailed = true;
1323 return;
1324 }
1325
1326 // Free inode.
1327 GroupDesc* pDesc = m_pGroupDescriptors[group];
1328 pDesc->bg_free_inodes_count++;
1329 m_pSuperblock->s_free_inodes_count++;
1330
1331 // Index = inode offset from the start of this block.
1332 size_t bitmapField = (index / 8) / m_BlockSize;
1333 size_t bitmapOffset = (index / 8) % m_BlockSize;
1334
1335 Vector<size_t>& list = m_pInodeBitmaps[group];
1336 uintptr_t block = list[bitmapField];
1337 uint8_t* ptr = reinterpret_cast<uint8_t*>(block + bitmapOffset);
1338 *ptr &= ~(1 << (index % 8));
1339 m_Quota.forget(inodeNumber);
1340 if (attributes.block) {
1341 const uint32_t bitmap = LITTLE_TO_HOST32(pDesc->bg_inode_bitmap) + bitmapField;
1342 recordAttributeWriteLocked(static_cast<uint64_t>(bitmap) * m_BlockSize,
1343 AttributeWriteKind::Allocation);
1344 recordAttributeAllocationLocked(attributes.block);
1345 const uint32_t descriptor = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1 +
1346 (group * sizeof(GroupDesc)) / m_BlockSize;
1347 recordAttributeWriteLocked(static_cast<uint64_t>(descriptor) * m_BlockSize,
1348 AttributeWriteKind::Allocation);
1349 }
1350
1351 // Update superblock.
1352 m_pDisk->write(1024ULL);
1353
1354 // Update on disk.
1355 uint32_t desc_block =
1356 LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_inode_bitmap) + bitmapField;
1357 writeBlock(desc_block);
1358
1359 // Update group descriptor on disk.
1361 uint32_t gdBlock = LITTLE_TO_HOST32(m_pSuperblock->s_first_data_block) + 1;
1362 uint32_t groupBlock = (group * sizeof(GroupDesc)) / m_BlockSize;
1363 writeBlock(gdBlock + groupBlock);
1364 }
1365
1366 writeInode(inodeNumber);
1367}
1368
1369Inode* Ext2Filesystem::getInode(uint32_t inode) {
1370 OperationBarrier::Lease operation;
1371 if (!tryAcquireOperation(operation)) {
1372 return nullptr;
1373 }
1374 const uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1375 if (!inode || !inodesPerGroup || !m_BlockSize) {
1376 SYSCALL_ERROR(IoError);
1377 return nullptr;
1378 }
1379 --inode;
1380 const uint32_t group = inode / inodesPerGroup;
1381 const uint64_t byteOffset = static_cast<uint64_t>(inode % inodesPerGroup) * m_InodeSize;
1382 const size_t blockNum = byteOffset / m_BlockSize;
1383 const size_t blockOff = byteOffset % m_BlockSize;
1384 if (sizeof(Inode) > m_BlockSize - blockOff) {
1385 SYSCALL_ERROR(IoError);
1386 return nullptr;
1387 }
1388 const uintptr_t block = loadInodeTableBlock(group, blockNum);
1389 if (!block)
1390 return nullptr;
1391
1392 Inode* pInode = reinterpret_cast<Inode*>(block + blockOff);
1393 if (pInode->i_flags & EXT2_COMPRBLK_FL) {
1394 WARNING("Ext2: inode " << inode << " has compressed blocks - not yet supported!");
1395 }
1396 return pInode;
1397}
1398
1399void Ext2Filesystem::writeInode(uint32_t inode) {
1400 if (!prepareInodeWrite(inode))
1401 return;
1402 --inode;
1403 const uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1404 const uint32_t group = inode / inodesPerGroup;
1405 const size_t blockNum =
1406 (static_cast<uint64_t>(inode % inodesPerGroup) * m_InodeSize) / m_BlockSize;
1407 const uint32_t diskBlock =
1408 LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_inode_table) + blockNum;
1409 writeBlock(diskBlock);
1410}
1411
1412bool Ext2Filesystem::checkOptionalFeature(size_t feature) {
1413 if (LITTLE_TO_HOST32(m_pSuperblock->s_rev_level) < 1)
1414 return false;
1415 return m_pSuperblock->s_feature_compat & feature;
1416}
1417
1418bool Ext2Filesystem::checkRequiredFeature(size_t feature) {
1419 if (LITTLE_TO_HOST32(m_pSuperblock->s_rev_level) < 1)
1420 return false;
1421 return m_pSuperblock->s_feature_incompat & feature;
1422}
1423
1424bool Ext2Filesystem::checkReadOnlyFeature(size_t feature) {
1425 if (LITTLE_TO_HOST32(m_pSuperblock->s_rev_level) < 1)
1426 return false;
1427 return m_pSuperblock->s_feature_ro_compat & feature;
1428}
1429
1430bool Ext2Filesystem::ensureFreeBlockBitmapLoaded(size_t group) {
1431 assert(group < m_nGroupDescriptors);
1432 Vector<size_t>& list = m_pBlockBitmaps[group];
1433
1434 if (list.count() > 0)
1435 // Descriptors already loaded.
1436 return true;
1437
1438 // Determine how many blocks to load to bring in the full block bitmap.
1439 // The bitmap works so that 8 blocks fit into one byte.
1440 uint32_t blocksPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_blocks_per_group);
1441 size_t nBlocks = blocksPerGroup / (m_BlockSize * 8);
1442 if (blocksPerGroup % (m_BlockSize * 8))
1443 nBlocks++;
1444
1445 if (!list.tryReserve(nBlocks)) {
1446 SYSCALL_ERROR(OutOfMemory);
1447 return false;
1448 }
1449
1450 const uint32_t start = LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_block_bitmap);
1451 for (size_t i = 0; i < nBlocks; i++) {
1452 uint32_t blockNumber = start + i;
1453 if (!blockNumber) {
1454 while (list.count()) {
1455 unpinBlock(start + list.count() - 1);
1456 list.popBack();
1457 }
1458 SYSCALL_ERROR(IoError);
1459 return false;
1460 }
1461 uintptr_t buffer = readBlock(blockNumber);
1462 if (!buffer) {
1463 while (list.count()) {
1464 unpinBlock(start + list.count() - 1);
1465 list.popBack();
1466 }
1467 SYSCALL_ERROR(IoError);
1468 return false;
1469 }
1470 list.pushBack(buffer);
1471 }
1472
1473 return true;
1474}
1475
1476bool Ext2Filesystem::ensureFreeInodeBitmapLoaded(size_t group) {
1477 assert(group < m_nGroupDescriptors);
1478 Vector<size_t>& list = m_pInodeBitmaps[group];
1479
1480 if (list.count() > 0)
1481 // Descriptors already loaded.
1482 return true;
1483
1484 // Determine how many blocks to load to bring in the full inode bitmap.
1485 // The bitmap works so that 8 inodes fit into one byte.
1486 uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1487 size_t nBlocks = inodesPerGroup / (m_BlockSize * 8);
1488 if (inodesPerGroup % (m_BlockSize * 8))
1489 nBlocks++;
1490
1491 if (!list.tryReserve(nBlocks)) {
1492 SYSCALL_ERROR(OutOfMemory);
1493 return false;
1494 }
1495
1496 const uint32_t start = LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_inode_bitmap);
1497 for (size_t i = 0; i < nBlocks; i++) {
1498 uint32_t blockNumber = start + i;
1499 if (!blockNumber) {
1500 while (list.count()) {
1501 unpinBlock(start + list.count() - 1);
1502 list.popBack();
1503 }
1504 SYSCALL_ERROR(IoError);
1505 return false;
1506 }
1507 uintptr_t buffer = readBlock(blockNumber);
1508 if (!buffer) {
1509 while (list.count()) {
1510 unpinBlock(start + list.count() - 1);
1511 list.popBack();
1512 }
1513 SYSCALL_ERROR(IoError);
1514 return false;
1515 }
1516 list.pushBack(buffer);
1517 }
1518
1519 return true;
1520}
1521
1522uintptr_t Ext2Filesystem::loadInodeTableBlock(size_t group, size_t block) {
1523#if THREADS || defined(STANDALONE_MUTEXES)
1525#endif
1526 if (group >= m_nGroupDescriptors || !m_BlockSize || !m_InodeSize) {
1527 SYSCALL_ERROR(IoError);
1528 return 0;
1529 }
1530 const uint32_t inodesPerGroup = LITTLE_TO_HOST32(m_pSuperblock->s_inodes_per_group);
1531 const uint64_t inodeTableBytes = static_cast<uint64_t>(inodesPerGroup) * m_InodeSize;
1532 const uint64_t blocks = (inodeTableBytes + m_BlockSize - 1) / m_BlockSize;
1533 const uint32_t start = LITTLE_TO_HOST32(m_pGroupDescriptors[group]->bg_inode_table);
1534 if (block >= blocks || !start || block > ~uint32_t{0} - start) {
1535 SYSCALL_ERROR(IoError);
1536 return 0;
1537 }
1538
1539 Vector<size_t>& list = m_pInodeTables[group];
1540 if (block < list.count() && list[block])
1541 return list[block];
1542 if (!list.tryReserve(block + 1)) {
1543 SYSCALL_ERROR(OutOfMemory);
1544 return 0;
1545 }
1546 while (list.count() <= block)
1547 list.pushBack(0);
1548
1549 const uintptr_t buffer = readBlock(start + block);
1550 if (!buffer) {
1551 SYSCALL_ERROR(IoError);
1552 return 0;
1553 }
1554 // Retain the successful read's pin for stable Inode pointers. A missing
1555 // slot stays retryable without disturbing previously loaded table blocks.
1556 list[block] = buffer;
1557 return buffer;
1558}
1559
1560void Ext2Filesystem::increaseInodeRefcount(uint32_t inode) {
1561 LockGuard<Mutex> stateGuard(m_InodeStateLock);
1562 Ext2InodeState* state = m_InodeStates.lookup(inode);
1563 LockGuard<Mutex> metadataGuard(state ? state->writebackLock : m_InodeStateLock, state != nullptr);
1564#if THREADS || defined(STANDALONE_MUTEXES)
1566#endif
1567
1568 Inode* pInode = getInode(inode);
1569 if (!pInode)
1570 return;
1571
1572 uint32_t current_count = LITTLE_TO_HOST16(pInode->i_links_count);
1573 pInode->i_links_count = HOST_TO_LITTLE16(current_count + 1);
1574 if (state) {
1575 state->orphan = false;
1576 }
1577
1578 writeInode(inode);
1579}
1580
1581bool Ext2Filesystem::decreaseInodeRefcount(uint32_t inode) {
1582 Inode* pInode = getInode(inode);
1583 if (!pInode)
1584 return true; // No inode found - but didn't decrement to zero.
1585
1586 uint32_t current_count = LITTLE_TO_HOST16(pInode->i_links_count);
1587 bool bRemove = current_count <= 1;
1588 if (current_count)
1589 pInode->i_links_count = HOST_TO_LITTLE16(current_count - 1);
1590
1591 writeInode(inode);
1592 return bRemove;
1593}
1594
1595#ifndef EXT2_STANDALONE
1596static bool initExt2() {
1597 VFS::instance().addProbeCallback(&Ext2Filesystem::probe);
1598 return true;
1599}
1600
1601static void destroyExt2() {
1602 if (!VFS::instance().removeProbeCallback(&Ext2Filesystem::probe)) {
1603 FATAL("Ext2 probe callback was not registered during unload");
1604 }
1605}
1606
1607MODULE_INFO("ext2", &initExt2, &destroyExt2, "vfs");
1608#endif
static DiskReadView borrowed(const void *data, size_t size)
Definition Disk.cc:60
Definition Disk.h:35
virtual bool sync(uint64_t location, bool async)
Definition Disk.cc:358
virtual BufferView read(uint64_t location)
Definition Disk.cc:163
virtual void getName(String &str)
Definition Disk.cc:155
virtual void unpin(uint64_t location)=0
virtual MUST_USE_RESULT bool syncPages(const uint64_t *locations, size_t count)
Definition Disk.cc:362
virtual void write(uint64_t location)
Definition Disk.cc:340
virtual DiskReadView readView(uint64_t location)
Definition Disk.cc:167
virtual MUST_USE_RESULT bool pin(uint64_t location)=0
Pins a cache page.
virtual bool removeEntry(const String &filename, Ext2Node *pFile)
virtual bool addEntry(const String &filename, File *pFile, size_t type)
virtual bool createNode(File *parent, const String &filename, uint32_t mask, const String &value, size_t type, uint32_t inodeOverride=0)
size_t m_nGroupDescriptors
GroupDesc ** m_pGroupDescriptors
virtual bool removeNode(File *parent, const String &filename, File *file)
void writeBlock(uint32_t block)
void retireInodeLocked(uint32_t inode, Ext2Node *lastNode)
virtual bool initialise(Disk *pDisk)
uintptr_t readBlock(uint32_t block)
Vector< size_t > * m_pInodeTables
virtual bool createDirectory(File *parent, const String &filename, uint32_t mask)
Vector< size_t > * m_pBlockBitmaps
virtual bool createSymlink(File *parent, const String &filename, const String &value)
void releaseBlockLocked(uint32_t block, uint32_t inode=0)
virtual const String & getVolumeLabel() const
size_t findFreeBlocksInGroup(uint32_t group, size_t maxCount, Vector< uint32_t > &blocks)
virtual bool deviceRemoved(bool deviceAvailable=false)
virtual bool getUuid(String &uuid) const
Vector< size_t > * m_pInodeBitmaps
bool releaseInode(uint32_t inode, Ext2Node *retiringNode=nullptr)
virtual bool createFile(File *parent, const String &filename, uint32_t mask)
virtual bool createLink(File *parent, const String &filename, File *target)
Superblock * m_pSuperblock
virtual File * getRoot() const
bool wipe(bool allocationLockHeld=false)
Definition Ext2Node.cc:217
bool ensureLargeEnough(size_t size, uint64_t location, uint64_t opsize, bool onlyBlocks=false, bool nozeroblocks=false)
Definition Ext2Node.cc:258
Definition File.h:75
virtual uint64_t write(uint64_t location, uint64_t size, uintptr_t buffer, bool bCanBlock=true) final
Definition File.cc:342
virtual bool isSymlink()
Definition File.cc:800
virtual bool isDirectory()
Definition File.cc:804
void setModifiedTime(Time::Timestamp t)
Definition File.cc:771
void setAccessedTime(Time::Timestamp t)
Definition File.cc:740
Disk * m_pDisk
Definition Filesystem.h:188
bool m_bReadOnly
Definition Filesystem.h:186
bool remove(const StringView &path, File *pStartNode=0)
virtual bool deviceRemoved(bool deviceAvailable=false)
Definition Filesystem.cc:92
virtual Timer * getTimer()=0
bool compare(const char *s, size_t len) const
Definition String.cc:202
static constexpr size_t getPageSize() noexcept
Definition TargetInfo.h:40
virtual Time::Timestamp getUnixTimestamp()
Definition Timer.cc:29
bool tryInsert(const K &key, const E &value)
Definition Tree.h:169
Iterator begin()
Definition Tree.h:402
void remove(const K &key)
Definition Tree.h:301
E lookup(const K &key) const
Definition Tree.h:193
void clear()
Definition Tree.h:383
void insert(const K &key, const E &value)
Definition Tree.h:149
Iterator end()
Definition Tree.h:427
void addProbeCallback(Filesystem::ProbeCallback callback)
Definition VFS.cc:1050
static VFS & instance()
Definition VFS.cc:311
A vector / dynamic array.
Definition Vector.h:33
@ Dec
Definition Log.h:126
@ Hex
Definition Log.h:124
void pushBack(const T &value)
Definition Vector.h:275
void clear(bool freeMem=false)
Definition Vector.h:378
size_t count() const
Definition Vector.h:270
Definition ext2.h:152