authorgravatar for carl@astholm.seCarl Åstholm <carl@astholm.se> 2023-12-19 22:21:03+01:00
committergravatar for carl@astholm.seCarl Åstholm <carl@astholm.se> 2024-01-01 16:18:57+01:00
log59ac0d1eed561483b4d97145eafa7d0763fb8ed8
treebfbb0c4ec4dbbb5168ad8be1c7cc4f21245d929e
parent781c3a985c2c6e31c57165c02582aa79c286e431

Deprecate `suggestVectorSize` in favor of `suggestVectorLength`

The function returns the vector length, not the byte size of the vector or the bit size of individual elements. This distinction is very important and some usages of this function in the stdlib operated under these incorrect assumptions.

5 files changed, 28 insertions(+), 28 deletions(-)

lib/std/crypto/ghash_polyval.zig+1-5
......@@ -158,11 +158,7 @@ fn Hash(comptime endian: std.builtin.Endian, comptime shift_key: bool) type {
158158 /// clmulSoft128_64 is faster on platforms with no native 128-bit registers.
159159 const clmulSoft = switch (builtin.cpu.arch) {
160160 .wasm32, .wasm64 => clmulSoft128_64,
161 else => impl: {
162 const vector_size = std.simd.suggestVectorSize(u128) orelse 0;
163 if (vector_size < 128) break :impl clmulSoft128_64;
164 break :impl clmulSoft128;
165 },
161 else => if (std.simd.suggestVectorLength(u128) != null) clmulSoft128 else clmulSoft128_64,
166162 };
167163
168164 // Software carryless multiplication of two 64-bit integers using native 128-bit registers.
lib/std/http/protocol.zig+1-1
......@@ -84,7 +84,7 @@ pub const HeadersParser = struct {
8484 /// If the amount returned is less than `bytes.len`, you may assume that the parser is in a content state and the
8585 /// first byte of content is located at `bytes[result]`.
8686 pub fn findHeadersEnd(r: *HeadersParser, bytes: []const u8) u32 {
87 const vector_len: comptime_int = @max(std.simd.suggestVectorSize(u8) orelse 1, 8);
87 const vector_len: comptime_int = @max(std.simd.suggestVectorLength(u8) orelse 1, 8);
8888 const len: u32 = @intCast(bytes.len);
8989 var index: u32 = 0;
9090
lib/std/mem.zig+4-4
......@@ -966,7 +966,7 @@ pub fn indexOfSentinel(comptime T: type, comptime sentinel: T, p: [*:sentinel]co
966966 // The below branch assumes that reading past the end of the buffer is valid, as long
967967 // as we don't read into a new page. This should be the case for most architectures
968968 // which use paged memory, however should be confirmed before adding a new arch below.
969 .aarch64, .x86, .x86_64 => if (std.simd.suggestVectorSize(T)) |block_len| {
969 .aarch64, .x86, .x86_64 => if (std.simd.suggestVectorLength(T)) |block_len| {
970970 const Block = @Vector(block_len, T);
971971 const mask: Block = @splat(sentinel);
972972
......@@ -1020,7 +1020,7 @@ test "indexOfSentinel vector paths" {
10201020 const allocator = std.testing.allocator;
10211021
10221022 inline for (Types) |T| {
1023 const block_len = std.simd.suggestVectorSize(T) orelse continue;
1023 const block_len = std.simd.suggestVectorLength(T) orelse continue;
10241024
10251025 // Allocate three pages so we guarantee a page-crossing address with a full page after
10261026 const memory = try allocator.alloc(T, 3 * std.mem.page_size / @sizeOf(T));
......@@ -1111,11 +1111,11 @@ pub fn indexOfScalarPos(comptime T: type, slice: []const T, start_index: usize,
11111111 !@inComptime() and
11121112 (@typeInfo(T) == .Int or @typeInfo(T) == .Float) and std.math.isPowerOfTwo(@bitSizeOf(T)))
11131113 {
1114 if (std.simd.suggestVectorSize(T)) |block_len| {
1114 if (std.simd.suggestVectorLength(T)) |block_len| {
11151115 // For Intel Nehalem (2009) and AMD Bulldozer (2012) or later, unaligned loads on aligned data result
11161116 // in the same execution as aligned loads. We ignore older arch's here and don't bother pre-aligning.
11171117 //
1118 // Use `std.simd.suggestVectorSize(T)` to get the same alignment as used in this function
1118 // Use `std.simd.suggestVectorLength(T)` to get the same alignment as used in this function
11191119 // however this usually isn't necessary unless your arch has a performance penalty due to this.
11201120 //
11211121 // This may differ for other arch's. Arm for example costs a cycle when loading across a cache
lib/std/simd.zig+16-12
......@@ -6,7 +6,9 @@
66const std = @import("std");
77const builtin = @import("builtin");
88
9pub fn suggestVectorSizeForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?comptime_int {
9pub const suggestVectorSizeForCpu = @compileError("deprecated; use 'suggestVectorLengthForCpu'");
10
11pub fn suggestVectorLengthForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?comptime_int {
1012 // This is guesswork, if you have better suggestions can add it or edit the current here
1113 // This can run in comptime only, but stage 1 fails at it, stage 2 can understand it
1214 const element_bit_size = @max(8, std.math.ceilPowerOfTwo(u16, @bitSizeOf(T)) catch unreachable);
......@@ -53,24 +55,26 @@ pub fn suggestVectorSizeForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?
5355 return @divExact(vector_bit_size, element_bit_size);
5456}
5557
56/// Suggests a target-dependant vector size for a given type, or null if scalars are recommended.
58pub const suggestVectorSize = @compileError("deprecated; use 'suggestVectorLength'");
59
60/// Suggests a target-dependant vector length for a given type, or null if scalars are recommended.
5761/// Not yet implemented for every CPU architecture.
58pub fn suggestVectorSize(comptime T: type) ?comptime_int {
59 return suggestVectorSizeForCpu(T, builtin.cpu);
62pub fn suggestVectorLength(comptime T: type) ?comptime_int {
63 return suggestVectorLengthForCpu(T, builtin.cpu);
6064}
6165
62test "suggestVectorSizeForCpu works with signed and unsigned values" {
66test "suggestVectorLengthForCpu works with signed and unsigned values" {
6367 comptime var cpu = std.Target.Cpu.baseline(std.Target.Cpu.Arch.x86_64);
6468 comptime cpu.features.addFeature(@intFromEnum(std.Target.x86.Feature.avx512f));
6569 comptime cpu.features.populateDependencies(&std.Target.x86.all_features);
66 const expected_size: usize = switch (builtin.zig_backend) {
70 const expected_len: usize = switch (builtin.zig_backend) {
6771 .stage2_x86_64 => 8,
6872 else => 16,
6973 };
70 const signed_integer_size = suggestVectorSizeForCpu(i32, cpu).?;
71 const unsigned_integer_size = suggestVectorSizeForCpu(u32, cpu).?;
72 try std.testing.expectEqual(expected_size, unsigned_integer_size);
73 try std.testing.expectEqual(expected_size, signed_integer_size);
74 const signed_integer_len = suggestVectorLengthForCpu(i32, cpu).?;
75 const unsigned_integer_len = suggestVectorLengthForCpu(u32, cpu).?;
76 try std.testing.expectEqual(expected_len, unsigned_integer_len);
77 try std.testing.expectEqual(expected_len, signed_integer_len);
7478}
7579
7680fn vectorLength(comptime VectorType: type) comptime_int {
......@@ -232,7 +236,7 @@ test "vector patterns" {
232236 }
233237}
234238
235/// Joins two vectors, shifts them leftwards (towards lower indices) and extracts the leftmost elements into a vector the size of a and b.
239/// Joins two vectors, shifts them leftwards (towards lower indices) and extracts the leftmost elements into a vector the length of a and b.
236240pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a, b))) @TypeOf(a, b) {
237241 const len = vectorLength(@TypeOf(a, b));
238242
......@@ -240,7 +244,7 @@ pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a,
240244}
241245
242246/// Elements are shifted rightwards (towards higher indices). New elements are added to the left, and the rightmost elements are cut off
243/// so that the size of the vector stays the same.
247/// so that the length of the vector stays the same.
244248pub fn shiftElementsRight(vec: anytype, comptime amount: VectorCount(@TypeOf(vec)), shift_in: std.meta.Child(@TypeOf(vec))) @TypeOf(vec) {
245249 // It may be possible to implement shifts and rotates with a runtime-friendly slice of two joined vectors, as the length of the
246250 // slice would be comptime-known. This would permit vector shifts and rotates by a non-comptime-known amount.
lib/std/unicode.zig+6-6
......@@ -202,7 +202,7 @@ pub fn utf8CountCodepoints(s: []const u8) !usize {
202202pub fn utf8ValidateSlice(input: []const u8) bool {
203203 var remaining = input;
204204
205 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;
205 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
206206 const Chunk = @Vector(chunk_len, u8);
207207
208208 // Fast path. Check for and skip ASCII characters at the start of the input.
......@@ -758,7 +758,7 @@ pub fn utf16leToUtf8Alloc(allocator: mem.Allocator, utf16le: []const u16) ![]u8
758758
759759 var remaining = utf16le;
760760 if (builtin.zig_backend != .stage2_x86_64) {
761 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;
761 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
762762 const Chunk = @Vector(chunk_len, u16);
763763
764764 // Fast path. Check for and encode ASCII characters at the start of the input.
......@@ -801,7 +801,7 @@ pub fn utf16leToUtf8AllocZ(allocator: mem.Allocator, utf16le: []const u16) ![:0]
801801
802802 var remaining = utf16le;
803803 if (builtin.zig_backend != .stage2_x86_64) {
804 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;
804 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
805805 const Chunk = @Vector(chunk_len, u16);
806806
807807 // Fast path. Check for and encode ASCII characters at the start of the input.
......@@ -842,7 +842,7 @@ pub fn utf16leToUtf8(utf8: []u8, utf16le: []const u16) !usize {
842842
843843 var remaining = utf16le;
844844 if (builtin.zig_backend != .stage2_x86_64) {
845 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;
845 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
846846 const Chunk = @Vector(chunk_len, u16);
847847
848848 // Fast path. Check for and encode ASCII characters at the start of the input.
......@@ -941,7 +941,7 @@ pub fn utf8ToUtf16LeWithNull(allocator: mem.Allocator, utf8: []const u8) ![:0]u1
941941 var remaining = utf8;
942942 // Need support for std.simd.interlace
943943 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {
944 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;
944 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
945945 const Chunk = @Vector(chunk_len, u8);
946946
947947 // Fast path. Check for and encode ASCII characters at the start of the input.
......@@ -986,7 +986,7 @@ pub fn utf8ToUtf16Le(utf16le: []u16, utf8: []const u8) !usize {
986986 var remaining = utf8;
987987 // Need support for std.simd.interlace
988988 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {
989 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;
989 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
990990 const Chunk = @Vector(chunk_len, u8);
991991
992992 // Fast path. Check for and encode ASCII characters at the start of the input.