Fix trim-related issues

Fixes:
* Use uint64 instead of off_t when handling offset and size
  of the trimmed range in the fs_trim_data structure
* BlockAllocator::Trim: Correct the size of a buffer
* ram_disk, mmc: Do not trim past device capacity

Improvements:
* BlockAllocator::Trim: Because the received offset and size
  are ignored by BFS (the functionality is not implemented yet),
  return B_UNSUPPORTED if the range does not cover the whole
  partition
* ram_disk, mmc: More accurate calculation of the number
  of trimmed bytes
* devfs: Add a uint64 version of translate_partition_access()

Change-Id: I24f4c08674f123ad33a5fef6e28996a4ada6ff0d
Reviewed-on: https://review.haiku-os.org/c/haiku/+/4155
Tested-by: Commit checker robot <[email protected]>
Reviewed-by: Adrien Destugues <[email protected]>
This commit is contained in:
David Sebek
2021-07-16 18:35:34 +00:00
committed by Adrien Destugues
parent 354b6bcfbd
commit bd02d81c24
7 changed files with 164 additions and 39 deletions
+2 -2
View File
@@ -175,8 +175,8 @@ typedef struct {
uint32 range_count; uint32 range_count;
uint64 trimmed_size; /* filled on return */ uint64 trimmed_size; /* filled on return */
struct range { struct range {
off_t offset; /* offset (in bytes) */ uint64 offset; /* offset (in bytes) */
off_t size; uint64 size;
} ranges[1]; } ranges[1];
} fs_trim_data; } fs_trim_data;
@@ -64,7 +64,7 @@ copy_trim_data_to_user(void* buffer, fs_trim_data* trimData)
return B_BAD_ADDRESS; return B_BAD_ADDRESS;
// Do not copy any ranges // Do not copy any ranges
return user_memcpy(buffer, trimData, sizeof(uint64) * 2); return user_memcpy(buffer, trimData, offsetof(fs_trim_data, ranges));
} }
@@ -491,38 +491,62 @@ mmc_block_trim(mmc_disk_driver_info* info, fs_trim_data* trimData)
}; };
TRACE("trim_device()\n"); TRACE("trim_device()\n");
trimData->trimmed_size = 0;
const off_t deviceSize = info->DeviceSize(); // in bytes
if (deviceSize < 0)
return B_BAD_VALUE;
STATIC_ASSERT(sizeof(deviceSize) <= sizeof(uint64));
ASSERT(deviceSize >= 0);
// Do not trim past device end
for (uint32 i = 0; i < trimData->range_count; i++) {
uint64 offset = trimData->ranges[i].offset;
uint64& size = trimData->ranges[i].size;
if (offset >= (uint64)deviceSize)
return B_BAD_VALUE;
size = min_c(size, (uint64)deviceSize - offset);
}
uint64 trimmedSize = 0; uint64 trimmedSize = 0;
status_t result = B_OK; status_t result = B_OK;
for (uint32 i = 0; i < trimData->range_count; i++) { for (uint32 i = 0; i < trimData->range_count; i++) {
off_t offset = trimData->ranges[i].offset; uint64 offset = trimData->ranges[i].offset;
off_t length = trimData->ranges[i].size; uint64 length = trimData->ranges[i].size;
// Round up offset and length to multiple of the sector size // Round up offset and length to multiple of the sector size
// The offset is rounded up, so some space may be left // The offset is rounded up, so some space may be left
// (not trimmed) at the start of the range. // (not trimmed) at the start of the range.
offset = ROUNDUP(offset, kBlockSize); offset = ROUNDUP(offset, kBlockSize);
// Adjust the length for the possibly skipped range // Adjust the length for the possibly skipped range
length -= trimData->ranges[i].offset - offset; length -= offset - trimData->ranges[i].offset;
// The length is rounded down, so some space at the end may also // The length is rounded down, so some space at the end may also
// be left (not trimmed). // be left (not trimmed).
length &= ~(kBlockSize - 1); length &= ~(kBlockSize - 1);
if (length == 0) { if (length == 0)
trimmedSize += trimData->ranges[i].size;
continue; continue;
}
TRACE("trim %" B_PRIdOFF " bytes from %" B_PRIdOFF "\n", TRACE("trim %" B_PRIu64 " bytes from %" B_PRIu64 "\n",
length, offset); length, offset);
ASSERT(offset % kBlockSize == 0); ASSERT(offset % kBlockSize == 0);
ASSERT(length % kBlockSize == 0); ASSERT(length % kBlockSize == 0);
if (info->flags & kIoCommandOffsetAsSectors) { if ((info->flags & kIoCommandOffsetAsSectors) != 0) {
offset /= kBlockSize; offset /= kBlockSize;
length /= kBlockSize; length /= kBlockSize;
} }
// Parameter of execute_command is uint32_t
if (offset > UINT32_MAX
|| length > UINT32_MAX - offset) {
result = B_BAD_VALUE;
break;
}
uint32_t response; uint32_t response;
result = info->mmc->execute_command(info->parent, info->parentCookie, result = info->mmc->execute_command(info->parent, info->parentCookie,
info->rca, SD_ERASE_WR_BLK_START, offset, &response); info->rca, SD_ERASE_WR_BLK_START, offset, &response);
@@ -537,7 +561,8 @@ mmc_block_trim(mmc_disk_driver_info* info, fs_trim_data* trimData)
if (result != B_OK) if (result != B_OK)
break; break;
trimmedSize += trimData->ranges[i].size; trimmedSize += (info->flags & kIoCommandOffsetAsSectors) != 0
? length * kBlockSize : length;
} }
trimData->trimmed_size = trimmedSize; trimData->trimmed_size = trimmedSize;
@@ -531,54 +531,78 @@ struct RawDevice : Device, DoublyLinkedListLinkImpl<RawDevice> {
{ {
TRACE("trim_device()\n"); TRACE("trim_device()\n");
trimData->trimmed_size = 0;
const off_t deviceSize = fDeviceSize; // in bytes
if (deviceSize < 0)
return B_BAD_VALUE;
STATIC_ASSERT(sizeof(deviceSize) <= sizeof(uint64));
ASSERT(deviceSize >= 0);
// Do not trim past device end
for (uint32 i = 0; i < trimData->range_count; i++) {
uint64 offset = trimData->ranges[i].offset;
uint64& size = trimData->ranges[i].size;
if (offset >= (uint64)deviceSize)
return B_BAD_VALUE;
size = min_c(size, (uint64)deviceSize - offset);
}
status_t result = B_OK;
uint64 trimmedSize = 0; uint64 trimmedSize = 0;
for (uint32 i = 0; i < trimData->range_count; i++) { for (uint32 i = 0; i < trimData->range_count; i++) {
trimmedSize += trimData->ranges[i].size; uint64 offset = trimData->ranges[i].offset;
uint64 length = trimData->ranges[i].size;
off_t offset = trimData->ranges[i].offset;
off_t length = trimData->ranges[i].size;
// Round up offset and length to multiple of the page size // Round up offset and length to multiple of the page size
// The offset is rounded up, so some space may be left // The offset is rounded up, so some space may be left
// (not trimmed) at the start of the range. // (not trimmed) at the start of the range.
offset = (offset + B_PAGE_SIZE - 1) & ~(B_PAGE_SIZE - 1); offset = (offset + B_PAGE_SIZE - 1) & ~(B_PAGE_SIZE - 1);
// Adjust the length for the possibly skipped range // Adjust the length for the possibly skipped range
length -= trimData->ranges[i].offset - offset; length -= offset - trimData->ranges[i].offset;
// The length is rounded down, so some space at the end may also // The length is rounded down, so some space at the end may also
// be left (not trimmed). // be left (not trimmed).
length &= ~(B_PAGE_SIZE - 1); length &= ~(B_PAGE_SIZE - 1);
TRACE("ramdisk: trim %" B_PRIdOFF " bytes from %" B_PRIdOFF "\n", if (length == 0)
continue;
TRACE("ramdisk: trim %" B_PRIu64 " bytes from %" B_PRIu64 "\n",
length, offset); length, offset);
ASSERT(offset % B_PAGE_SIZE == 0); ASSERT(offset % B_PAGE_SIZE == 0);
ASSERT(length % B_PAGE_SIZE == 0); ASSERT(length % B_PAGE_SIZE == 0);
vm_page** pages = new(std::nothrow) vm_page*[length / B_PAGE_SIZE]; vm_page** pages = new(std::nothrow) vm_page*[length / B_PAGE_SIZE];
if (pages == NULL) if (pages == NULL) {
return B_NO_MEMORY; result = B_NO_MEMORY;
break;
}
ArrayDeleter<vm_page*> pagesDeleter(pages); ArrayDeleter<vm_page*> pagesDeleter(pages);
_GetPages(offset, length, false, pages); _GetPages((off_t)offset, (off_t)length, false, pages);
AutoLocker<VMCache> locker(fCache); AutoLocker<VMCache> locker(fCache);
uint32 j; uint64 j;
for (j = 0; j < length / B_PAGE_SIZE; j++) { for (j = 0; j < length / B_PAGE_SIZE; j++) {
// If we run out of pages (some may already be trimmed), stop. // If we run out of pages (some may already be trimmed), stop.
if (pages[j] == NULL) if (pages[j] == NULL)
break; break;
TRACE("free range %" B_PRIu32 ", page %" B_PRIu32 ", offset %" TRACE("free range %" B_PRIu32 ", page %" B_PRIu64 ", offset %"
B_PRIdOFF "\n", i, j, offset); B_PRIu64 "\n", i, j, offset);
if (pages[j]->Cache()) if (pages[j]->Cache())
fCache->RemovePage(pages[j]); fCache->RemovePage(pages[j]);
vm_page_free(NULL, pages[j]); vm_page_free(NULL, pages[j]);
trimmedSize += B_PAGE_SIZE;
} }
} }
trimData->trimmed_size = trimmedSize; trimData->trimmed_size = trimmedSize;
return B_OK; return result;
} }
@@ -1158,9 +1158,18 @@ BlockAllocator::_CheckGroup(int32 groupIndex) const
status_t status_t
BlockAllocator::Trim(uint64 offset, uint64 size, uint64& trimmedSize) BlockAllocator::Trim(uint64 offset, uint64 size, uint64& trimmedSize)
{ {
// TODO: Remove this check when offset and size handling is implemented
if (offset != 0
|| fVolume->NumBlocks() < 0
|| size < (uint64)fVolume->NumBlocks() * fVolume->BlockSize()) {
INFORM(("BFS Trim: Ranges smaller than the file system size"
" are not supported yet.\n"));
return B_UNSUPPORTED;
}
const uint32 kTrimRanges = 128; const uint32 kTrimRanges = 128;
fs_trim_data* trimData = (fs_trim_data*)malloc(sizeof(fs_trim_data) fs_trim_data* trimData = (fs_trim_data*)malloc(sizeof(fs_trim_data)
+ sizeof(uint64) * kTrimRanges); + 2 * sizeof(uint64) * (kTrimRanges - 1));
if (trimData == NULL) if (trimData == NULL)
return B_NO_MEMORY; return B_NO_MEMORY;
@@ -1175,7 +1184,7 @@ BlockAllocator::Trim(uint64 offset, uint64 size, uint64& trimmedSize)
uint32 blockShift = fVolume->BlockShift(); uint32 blockShift = fVolume->BlockShift();
uint64 firstFree = 0; uint64 firstFree = 0;
size_t freeLength = 0; uint64 freeLength = 0;
trimData->range_count = 0; trimData->range_count = 0;
trimmedSize = 0; trimmedSize = 0;
@@ -1191,6 +1200,15 @@ BlockAllocator::Trim(uint64 offset, uint64 size, uint64& trimmedSize)
if (cached.IsUsed(i)) { if (cached.IsUsed(i)) {
// Block is in use // Block is in use
if (freeLength > 0) { if (freeLength > 0) {
// Overflow is unlikely to happen, but check it anyway
if ((firstFree << blockShift) >> blockShift
!= firstFree
|| (freeLength << blockShift) >> blockShift
!= freeLength) {
FATAL(("BlockAllocator::Trim:"
" Overflow detected!\n"));
return B_ERROR;
}
status_t status = _TrimNext(*trimData, kTrimRanges, status_t status = _TrimNext(*trimData, kTrimRanges,
firstFree << blockShift, freeLength << blockShift, firstFree << blockShift, freeLength << blockShift,
false, trimmedSize); false, trimmedSize);
@@ -1306,7 +1324,7 @@ BlockAllocator::CheckBlockRun(block_run run, const char* type, bool allocated)
} }
status_t bool
BlockAllocator::_AddTrim(fs_trim_data& trimData, uint32 maxRanges, BlockAllocator::_AddTrim(fs_trim_data& trimData, uint32 maxRanges,
uint64 offset, uint64 size) uint64 offset, uint64 size)
{ {
@@ -1333,13 +1351,16 @@ BlockAllocator::_TrimNext(fs_trim_data& trimData, uint32 maxRanges,
if (!pushed || force) { if (!pushed || force) {
// Trim now // Trim now
trimData.trimmed_size = 0; trimData.trimmed_size = 0;
dprintf("TRIM FS:\n"); #ifdef DEBUG_TRIM
for (uint32 i = 0; i < trimData.range_count; i++) { dprintf("TRIM: BFS: free ranges (bytes):\n");
dprintf("[%3" B_PRIu32 "] %" B_PRIu64 " : %" B_PRIu64 "\n", i, for (uint32 i = 0; i < trimData.range_count; i++) {
trimData.ranges[i].offset, trimData.ranges[i].size); dprintf("[%3" B_PRIu32 "] %" B_PRIu64 " : %" B_PRIu64 "\n", i,
} trimData.ranges[i].offset, trimData.ranges[i].size);
}
#endif
if (ioctl(fVolume->Device(), B_TRIM_DEVICE, &trimData, if (ioctl(fVolume->Device(), B_TRIM_DEVICE, &trimData,
sizeof(fs_trim_data)) != 0) { sizeof(fs_trim_data)
+ 2 * sizeof(uint64) * (trimData.range_count - 1)) != 0) {
return errno; return errno;
} }
@@ -68,7 +68,7 @@ private:
#ifdef DEBUG_ALLOCATION_GROUPS #ifdef DEBUG_ALLOCATION_GROUPS
void _CheckGroup(int32 group) const; void _CheckGroup(int32 group) const;
#endif #endif
status_t _AddTrim(fs_trim_data& trimData, uint32 maxRanges, bool _AddTrim(fs_trim_data& trimData, uint32 maxRanges,
uint64 offset, uint64 size); uint64 offset, uint64 size);
status_t _TrimNext(fs_trim_data& trimData, uint32 maxRanges, status_t _TrimNext(fs_trim_data& trimData, uint32 maxRanges,
uint64 offset, uint64 size, bool force, uint64 offset, uint64 size, bool force,
+59 -4
View File
@@ -495,18 +495,49 @@ err1:
} }
template<typename size_type> static inline void static inline void
translate_partition_access(devfs_partition* partition, off_t& offset, translate_partition_access(devfs_partition* partition, off_t& offset,
size_type& size) size_t& size)
{ {
ASSERT(offset >= 0); ASSERT(offset >= 0);
ASSERT(offset < partition->info.size); ASSERT(offset < partition->info.size);
size = (size_type)min_c((off_t)size, partition->info.size - offset); size = (size_t)min_c((off_t)size, partition->info.size - offset);
offset += partition->info.offset; offset += partition->info.offset;
} }
static bool
translate_partition_access(devfs_partition* partition, uint64& offset,
uint64& size)
{
const off_t partitionSize = partition->info.size;
const off_t partitionOffset = partition->info.offset;
// Check that off_t values can be cast to uint64,
// partition offset can theoretically be negative
ASSERT(partitionSize >= 0);
STATIC_ASSERT(sizeof(partitionSize) <= sizeof(uint64));
STATIC_ASSERT(sizeof(partitionOffset) <= sizeof(uint64));
// Check that calculations give expected results
if (offset >= (uint64)partitionSize)
return false;
if (partitionOffset >= 0 && offset > UINT64_MAX - (uint64)partitionOffset)
return false;
if (partitionOffset < 0 && offset < (uint64)-partitionOffset)
return false;
size = min_c(size, (uint64)partitionSize - offset);
if (partitionOffset >= 0)
offset += (uint64)partitionOffset;
else
offset -= (uint64)-partitionOffset;
return true;
}
static inline void static inline void
translate_partition_access(devfs_partition* partition, io_request* request) translate_partition_access(devfs_partition* partition, io_request* request)
{ {
@@ -1479,14 +1510,38 @@ devfs_ioctl(fs_volume* _volume, fs_vnode* _vnode, void* _cookie, uint32 op,
if (status != B_OK) if (status != B_OK)
return status; return status;
#ifdef DEBUG_TRIM
dprintf("TRIM: devfs: received TRIM ranges (bytes):\n");
for (uint32 i = 0; i < trimData->range_count; i++) {
dprintf("[%3" B_PRIu32 "] %" B_PRIu64 " : %"
B_PRIu64 "\n", i,
trimData->ranges[i].offset,
trimData->ranges[i].size);
}
#endif
if (partition != NULL) { if (partition != NULL) {
// If there is a partition, offset all ranges according // If there is a partition, offset all ranges according
// to the partition start. // to the partition start.
// Range size may be reduced to fit the partition size.
for (uint32 i = 0; i < trimData->range_count; i++) { for (uint32 i = 0; i < trimData->range_count; i++) {
translate_partition_access(partition, if (!translate_partition_access(partition,
trimData->ranges[i].offset,
trimData->ranges[i].size)) {
return B_BAD_VALUE;
}
}
#ifdef DEBUG_TRIM
dprintf("TRIM: devfs: TRIM ranges after partition"
" translation (bytes):\n");
for (uint32 i = 0; i < trimData->range_count; i++) {
dprintf("[%3" B_PRIu32 "] %" B_PRIu64 " : %"
B_PRIu64 "\n", i,
trimData->ranges[i].offset, trimData->ranges[i].offset,
trimData->ranges[i].size); trimData->ranges[i].size);
} }
#endif
} }
status = vnode->stream.u.dev.device->Control( status = vnode->stream.u.dev.device->Control(