| ... | ... | @@ -34,9 +34,6 @@ fn cShrink(self: *Allocator, old_mem: []u8, old_align: u29, new_size: usize, new |
| 34 | 34 | /// Thread-safe and lock-free. |
| 35 | 35 | pub const DirectAllocator = struct { |
| 36 | 36 | allocator: Allocator, |
| 37 | | heap_handle: ?HeapHandle, |
| 38 | | |
| 39 | | const HeapHandle = if (builtin.os == Os.windows) os.windows.HANDLE else void; |
| 40 | 37 | |
| 41 | 38 | pub fn init() DirectAllocator { |
| 42 | 39 | return DirectAllocator{ |
| ... | ... | @@ -44,18 +41,10 @@ pub const DirectAllocator = struct { |
| 44 | 41 | .reallocFn = realloc, |
| 45 | 42 | .shrinkFn = shrink, |
| 46 | 43 | }, |
| 47 | | .heap_handle = if (builtin.os == Os.windows) null else {}, |
| 48 | 44 | }; |
| 49 | 45 | } |
| 50 | 46 | |
| 51 | | pub fn deinit(self: *DirectAllocator) void { |
| 52 | | switch (builtin.os) { |
| 53 | | Os.windows => if (self.heap_handle) |heap_handle| { |
| 54 | | _ = os.windows.HeapDestroy(heap_handle); |
| 55 | | }, |
| 56 | | else => {}, |
| 57 | | } |
| 58 | | } |
| 47 | pub fn deinit(self: *DirectAllocator) void {} |
| 59 | 48 | |
| 60 | 49 | fn alloc(allocator: *Allocator, n: usize, alignment: u29) error{OutOfMemory}![]u8 { |
| 61 | 50 | const self = @fieldParentPtr(DirectAllocator, "allocator", allocator); |
| ... | ... | @@ -89,21 +78,57 @@ pub const DirectAllocator = struct { |
| 89 | 78 | |
| 90 | 79 | return @intToPtr([*]u8, aligned_addr)[0..n]; |
| 91 | 80 | }, |
| 92 | | Os.windows => { |
| 93 | | const amt = n + alignment + @sizeOf(usize); |
| 94 | | const optional_heap_handle = @atomicLoad(?HeapHandle, &self.heap_handle, builtin.AtomicOrder.SeqCst); |
| 95 | | const heap_handle = optional_heap_handle orelse blk: { |
| 96 | | const hh = os.windows.HeapCreate(0, amt, 0) orelse return error.OutOfMemory; |
| 97 | | const other_hh = @cmpxchgStrong(?HeapHandle, &self.heap_handle, null, hh, builtin.AtomicOrder.SeqCst, builtin.AtomicOrder.SeqCst) orelse break :blk hh; |
| 98 | | _ = os.windows.HeapDestroy(hh); |
| 99 | | break :blk other_hh.?; // can't be null because of the cmpxchg |
| 100 | | }; |
| 101 | | const ptr = os.windows.HeapAlloc(heap_handle, 0, amt) orelse return error.OutOfMemory; |
| 102 | | const root_addr = @ptrToInt(ptr); |
| 103 | | const adjusted_addr = mem.alignForward(root_addr, alignment); |
| 104 | | const record_addr = adjusted_addr + n; |
| 105 | | @intToPtr(*align(1) usize, record_addr).* = root_addr; |
| 106 | | return @intToPtr([*]u8, adjusted_addr)[0..n]; |
| 81 | .windows => { |
| 82 | const w = os.windows; |
| 83 | |
| 84 | // Although officially it's at least aligned to page boundary, |
| 85 | // Windows is known to reserve pages on a 64K boundary. It's |
| 86 | // even more likely that the requested alignment is <= 64K than |
| 87 | // 4K, so we're just allocating blindly and hoping for the best. |
| 88 | // see https://devblogs.microsoft.com/oldnewthing/?p=42223 |
| 89 | const addr = w.VirtualAlloc( |
| 90 | null, |
| 91 | n, |
| 92 | w.MEM_COMMIT | w.MEM_RESERVE, |
| 93 | w.PAGE_READWRITE, |
| 94 | ) orelse return error.OutOfMemory; |
| 95 | |
| 96 | // If the allocation is sufficiently aligned, use it. |
| 97 | if (@ptrToInt(addr) & (alignment - 1) == 0) { |
| 98 | return @ptrCast([*]u8, addr)[0..n]; |
| 99 | } |
| 100 | |
| 101 | // If it wasn't, actually do an explicitely aligned allocation. |
| 102 | if (w.VirtualFree(addr, 0, w.MEM_RELEASE) == 0) unreachable; |
| 103 | const alloc_size = n + alignment; |
| 104 | |
| 105 | const final_addr = while (true) { |
| 106 | // Reserve a range of memory large enough to find a sufficiently |
| 107 | // aligned address. |
| 108 | const reserved_addr = w.VirtualAlloc( |
| 109 | null, |
| 110 | alloc_size, |
| 111 | w.MEM_RESERVE, |
| 112 | w.PAGE_NOACCESS, |
| 113 | ) orelse return error.OutOfMemory; |
| 114 | const aligned_addr = mem.alignForward(@ptrToInt(reserved_addr), alignment); |
| 115 | |
| 116 | // Release the reserved pages (not actually used). |
| 117 | if (w.VirtualFree(reserved_addr, 0, w.MEM_RELEASE) == 0) unreachable; |
| 118 | |
| 119 | // At this point, it is possible that another thread has |
| 120 | // obtained some memory space that will cause the next |
| 121 | // VirtualAlloc call to fail. To handle this, we will retry |
| 122 | // until it succeeds. |
| 123 | if (w.VirtualAlloc( |
| 124 | @intToPtr(*c_void, aligned_addr), |
| 125 | n, |
| 126 | w.MEM_COMMIT | w.MEM_RESERVE, |
| 127 | w.PAGE_READWRITE, |
| 128 | )) |ptr| break ptr; |
| 129 | } else unreachable; // TODO else unreachable should not be necessary |
| 130 | |
| 131 | return @ptrCast([*]u8, final_addr)[0..n]; |
| 107 | 132 | }, |
| 108 | 133 | else => @compileError("Unsupported OS"), |
| 109 | 134 | } |
| ... | ... | @@ -121,13 +146,31 @@ pub const DirectAllocator = struct { |
| 121 | 146 | } |
| 122 | 147 | return old_mem[0..new_size]; |
| 123 | 148 | }, |
| 124 | | Os.windows => return realloc(allocator, old_mem, old_align, new_size, new_align) catch { |
| 125 | | const old_adjusted_addr = @ptrToInt(old_mem.ptr); |
| 126 | | const old_record_addr = old_adjusted_addr + old_mem.len; |
| 127 | | const root_addr = @intToPtr(*align(1) usize, old_record_addr).*; |
| 128 | | const old_ptr = @intToPtr(*c_void, root_addr); |
| 129 | | const new_record_addr = old_record_addr - new_size + old_mem.len; |
| 130 | | @intToPtr(*align(1) usize, new_record_addr).* = root_addr; |
| 149 | .windows => { |
| 150 | const w = os.windows; |
| 151 | if (new_size == 0) { |
| 152 | // From the docs: |
| 153 | // "If the dwFreeType parameter is MEM_RELEASE, this parameter |
| 154 | // must be 0 (zero). The function frees the entire region that |
| 155 | // is reserved in the initial allocation call to VirtualAlloc." |
| 156 | // So we can only use MEM_RELEASE when actually releasing the |
| 157 | // whole allocation. |
| 158 | if (w.VirtualFree(old_mem.ptr, 0, w.MEM_RELEASE) == 0) unreachable; |
| 159 | } else { |
| 160 | const base_addr = @ptrToInt(old_mem.ptr); |
| 161 | const old_addr_end = base_addr + old_mem.len; |
| 162 | const new_addr_end = base_addr + new_size; |
| 163 | const new_addr_end_rounded = mem.alignForward(new_addr_end, os.page_size); |
| 164 | if (old_addr_end > new_addr_end_rounded) { |
| 165 | // For shrinking that is not releasing, we will only |
| 166 | // decommit the pages not needed anymore. |
| 167 | if (w.VirtualFree( |
| 168 | @intToPtr(*c_void, new_addr_end_rounded), |
| 169 | old_addr_end - new_addr_end_rounded, |
| 170 | w.MEM_DECOMMIT, |
| 171 | ) == 0) unreachable; |
| 172 | } |
| 173 | } |
| 131 | 174 | return old_mem[0..new_size]; |
| 132 | 175 | }, |
| 133 | 176 | else => @compileError("Unsupported OS"), |
| ... | ... | @@ -147,43 +190,59 @@ pub const DirectAllocator = struct { |
| 147 | 190 | } |
| 148 | 191 | return result; |
| 149 | 192 | }, |
| 150 | | Os.windows => { |
| 151 | | if (old_mem.len == 0) return alloc(allocator, new_size, new_align); |
| 193 | .windows => { |
| 194 | if (old_mem.len == 0) { |
| 195 | return alloc(allocator, new_size, new_align); |
| 196 | } |
| 152 | 197 | |
| 153 | | const self = @fieldParentPtr(DirectAllocator, "allocator", allocator); |
| 154 | | const old_adjusted_addr = @ptrToInt(old_mem.ptr); |
| 155 | | const old_record_addr = old_adjusted_addr + old_mem.len; |
| 156 | | const root_addr = @intToPtr(*align(1) usize, old_record_addr).*; |
| 157 | | const old_ptr = @intToPtr(*c_void, root_addr); |
| 198 | if (new_size <= old_mem.len and new_align <= old_align) { |
| 199 | return shrink(allocator, old_mem, old_align, new_size, new_align); |
| 200 | } |
| 158 | 201 | |
| 159 | | if (new_size == 0) { |
| 160 | | if (os.windows.HeapFree(self.heap_handle.?, 0, old_ptr) == 0) unreachable; |
| 161 | | return old_mem[0..0]; |
| 202 | const w = os.windows; |
| 203 | const base_addr = @ptrToInt(old_mem.ptr); |
| 204 | |
| 205 | if (new_align > old_align and base_addr & (new_align - 1) != 0) { |
| 206 | // Current allocation doesn't satisfy the new alignment. |
| 207 | // For now we'll do a new one no matter what, but maybe |
| 208 | // there is something smarter to do instead. |
| 209 | const result = try alloc(allocator, new_size, new_align); |
| 210 | assert(old_mem.len != 0); |
| 211 | @memcpy(result.ptr, old_mem.ptr, std.math.min(old_mem.len, result.len)); |
| 212 | if (w.VirtualFree(old_mem.ptr, 0, w.MEM_RELEASE) == 0) unreachable; |
| 213 | |
| 214 | return result; |
| 162 | 215 | } |
| 163 | 216 | |
| 164 | | const amt = new_size + new_align + @sizeOf(usize); |
| 165 | | const new_ptr = os.windows.HeapReAlloc( |
| 166 | | self.heap_handle.?, |
| 167 | | 0, |
| 168 | | old_ptr, |
| 169 | | amt, |
| 170 | | ) orelse return error.OutOfMemory; |
| 171 | | const offset = old_adjusted_addr - root_addr; |
| 172 | | const new_root_addr = @ptrToInt(new_ptr); |
| 173 | | var new_adjusted_addr = new_root_addr + offset; |
| 174 | | const offset_is_valid = new_adjusted_addr + new_size + @sizeOf(usize) <= new_root_addr + amt; |
| 175 | | const offset_is_aligned = new_adjusted_addr % new_align == 0; |
| 176 | | if (!offset_is_valid or !offset_is_aligned) { |
| 177 | | // If HeapReAlloc didn't happen to move the memory to the new alignment, |
| 178 | | // or the memory starting at the old offset would be outside of the new allocation, |
| 179 | | // then we need to copy the memory to a valid aligned address and use that |
| 180 | | const new_aligned_addr = mem.alignForward(new_root_addr, new_align); |
| 181 | | @memcpy(@intToPtr([*]u8, new_aligned_addr), @intToPtr([*]u8, new_adjusted_addr), std.math.min(old_mem.len, new_size)); |
| 182 | | new_adjusted_addr = new_aligned_addr; |
| 217 | const old_addr_end = base_addr + old_mem.len; |
| 218 | const old_addr_end_rounded = mem.alignForward(old_addr_end, os.page_size); |
| 219 | const new_addr_end = base_addr + new_size; |
| 220 | const new_addr_end_rounded = mem.alignForward(new_addr_end, os.page_size); |
| 221 | if (new_addr_end_rounded == old_addr_end_rounded) { |
| 222 | // The reallocation fits in the already allocated pages. |
| 223 | return @ptrCast([*]u8, old_mem.ptr)[0..new_size]; |
| 183 | 224 | } |
| 184 | | const new_record_addr = new_adjusted_addr + new_size; |
| 185 | | @intToPtr(*align(1) usize, new_record_addr).* = new_root_addr; |
| 186 | | return @intToPtr([*]u8, new_adjusted_addr)[0..new_size]; |
| 225 | assert(new_addr_end_rounded > old_addr_end_rounded); |
| 226 | |
| 227 | // We need to commit new pages. |
| 228 | const additional_size = new_addr_end - old_addr_end_rounded; |
| 229 | const realloc_addr = w.VirtualAlloc( |
| 230 | @intToPtr(*c_void, old_addr_end_rounded), |
| 231 | additional_size, |
| 232 | w.MEM_COMMIT | w.MEM_RESERVE, |
| 233 | w.PAGE_READWRITE, |
| 234 | ) orelse { |
| 235 | // Committing new pages at the end of the existing allocation |
| 236 | // failed, we need to try a new one. |
| 237 | const new_alloc_mem = try alloc(allocator, new_size, new_align); |
| 238 | @memcpy(new_alloc_mem.ptr, old_mem.ptr, old_mem.len); |
| 239 | if (w.VirtualFree(old_mem.ptr, 0, w.MEM_RELEASE) == 0) unreachable; |
| 240 | |
| 241 | return new_alloc_mem; |
| 242 | }; |
| 243 | |
| 244 | assert(@ptrToInt(realloc_addr) == old_addr_end_rounded); |
| 245 | return @ptrCast([*]u8, old_mem.ptr)[0..new_size]; |
| 187 | 246 | }, |
| 188 | 247 | else => @compileError("Unsupported OS"), |
| 189 | 248 | } |
| ... | ... | @@ -686,6 +745,17 @@ test "DirectAllocator" { |
| 686 | 745 | try testAllocatorAligned(allocator, 16); |
| 687 | 746 | try testAllocatorLargeAlignment(allocator); |
| 688 | 747 | try testAllocatorAlignedShrink(allocator); |
| 748 | |
| 749 | if (builtin.os == .windows) { |
| 750 | // Trying really large alignment. As mentionned in the implementation, |
| 751 | // VirtualAlloc returns 64K aligned addresses. We want to make sure |
| 752 | // DirectAllocator works beyond that, as it's not tested by |
| 753 | // `testAllocatorLargeAlignment`. |
| 754 | const slice = try allocator.alignedAlloc(u8, 1 << 20, 128); |
| 755 | slice[0] = 0x12; |
| 756 | slice[127] = 0x34; |
| 757 | allocator.free(slice); |
| 758 | } |
| 689 | 759 | } |
| 690 | 760 | |
| 691 | 761 | test "HeapAllocator" { |
| ... | ... | @@ -714,7 +784,7 @@ test "ArenaAllocator" { |
| 714 | 784 | try testAllocatorAlignedShrink(&arena_allocator.allocator); |
| 715 | 785 | } |
| 716 | 786 | |
| 717 | | var test_fixed_buffer_allocator_memory: [40000 * @sizeOf(u64)]u8 = undefined; |
| 787 | var test_fixed_buffer_allocator_memory: [80000 * @sizeOf(u64)]u8 = undefined; |
| 718 | 788 | test "FixedBufferAllocator" { |
| 719 | 789 | var fixed_buffer_allocator = FixedBufferAllocator.init(test_fixed_buffer_allocator_memory[0..]); |
| 720 | 790 | |
| ... | ... | @@ -852,7 +922,11 @@ fn testAllocatorAlignedShrink(allocator: *mem.Allocator) mem.Allocator.Error!voi |
| 852 | 922 | defer allocator.free(slice); |
| 853 | 923 | |
| 854 | 924 | var stuff_to_free = std.ArrayList([]align(16) u8).init(debug_allocator); |
| 855 | | while (@ptrToInt(slice.ptr) == mem.alignForward(@ptrToInt(slice.ptr), os.page_size * 2)) { |
| 925 | // On Windows, VirtualAlloc returns addresses aligned to a 64K boundary, |
| 926 | // which is 16 pages, hence the 32. This test may require to increase |
| 927 | // the size of the allocations feeding the `allocator` parameter if they |
| 928 | // fail, because of this high over-alignment we want to have. |
| 929 | while (@ptrToInt(slice.ptr) == mem.alignForward(@ptrToInt(slice.ptr), os.page_size * 32)) { |
| 856 | 930 | try stuff_to_free.append(slice); |
| 857 | 931 | slice = try allocator.alignedAlloc(u8, 16, alloc_size); |
| 858 | 932 | } |
| ... | ... | @@ -863,7 +937,7 @@ fn testAllocatorAlignedShrink(allocator: *mem.Allocator) mem.Allocator.Error!voi |
| 863 | 937 | slice[60] = 0x34; |
| 864 | 938 | |
| 865 | 939 | // realloc to a smaller size but with a larger alignment |
| 866 | | slice = try allocator.alignedRealloc(slice, os.page_size * 2, alloc_size / 2); |
| 940 | slice = try allocator.alignedRealloc(slice, os.page_size * 32, alloc_size / 2); |
| 867 | 941 | testing.expect(slice[0] == 0x12); |
| 868 | 942 | testing.expect(slice[60] == 0x34); |
| 869 | 943 | } |