authorgravatar for carl@astholm.seCarl Åstholm <carl@astholm.se> 2023-12-19 22:21:03+01:00
committergravatar for carl@astholm.seCarl Åstholm <carl@astholm.se> 2024-01-01 16:18:57+01:00
log59ac0d1eed561483b4d97145eafa7d0763fb8ed8
treebfbb0c4ec4dbbb5168ad8be1c7cc4f21245d929e
parent781c3a985c2c6e31c57165c02582aa79c286e431

Deprecate `suggestVectorSize` in favor of `suggestVectorLength`

The function returns the vector length, not the byte size of the vector or the bit size of individual elements. This distinction is very important and some usages of this function in the stdlib operated under these incorrect assumptions.

5 files changed, 28 insertions(+), 28 deletions(-)

lib/std/crypto/ghash_polyval.zig+1-5
...@@ -158,11 +158,7 @@ fn Hash(comptime endian: std.builtin.Endian, comptime shift_key: bool) type {...@@ -158,11 +158,7 @@ fn Hash(comptime endian: std.builtin.Endian, comptime shift_key: bool) type {
158 /// clmulSoft128_64 is faster on platforms with no native 128-bit registers.158 /// clmulSoft128_64 is faster on platforms with no native 128-bit registers.
159 const clmulSoft = switch (builtin.cpu.arch) {159 const clmulSoft = switch (builtin.cpu.arch) {
160 .wasm32, .wasm64 => clmulSoft128_64,160 .wasm32, .wasm64 => clmulSoft128_64,
161 else => impl: {161 else => if (std.simd.suggestVectorLength(u128) != null) clmulSoft128 else clmulSoft128_64,
162 const vector_size = std.simd.suggestVectorSize(u128) orelse 0;
163 if (vector_size < 128) break :impl clmulSoft128_64;
164 break :impl clmulSoft128;
165 },
166 };162 };
167163
168 // Software carryless multiplication of two 64-bit integers using native 128-bit registers.164 // Software carryless multiplication of two 64-bit integers using native 128-bit registers.
lib/std/http/protocol.zig+1-1
...@@ -84,7 +84,7 @@ pub const HeadersParser = struct {...@@ -84,7 +84,7 @@ pub const HeadersParser = struct {
84 /// If the amount returned is less than `bytes.len`, you may assume that the parser is in a content state and the84 /// If the amount returned is less than `bytes.len`, you may assume that the parser is in a content state and the
85 /// first byte of content is located at `bytes[result]`.85 /// first byte of content is located at `bytes[result]`.
86 pub fn findHeadersEnd(r: *HeadersParser, bytes: []const u8) u32 {86 pub fn findHeadersEnd(r: *HeadersParser, bytes: []const u8) u32 {
87 const vector_len: comptime_int = @max(std.simd.suggestVectorSize(u8) orelse 1, 8);87 const vector_len: comptime_int = @max(std.simd.suggestVectorLength(u8) orelse 1, 8);
88 const len: u32 = @intCast(bytes.len);88 const len: u32 = @intCast(bytes.len);
89 var index: u32 = 0;89 var index: u32 = 0;
9090
lib/std/mem.zig+4-4
...@@ -966,7 +966,7 @@ pub fn indexOfSentinel(comptime T: type, comptime sentinel: T, p: [*:sentinel]co...@@ -966,7 +966,7 @@ pub fn indexOfSentinel(comptime T: type, comptime sentinel: T, p: [*:sentinel]co
966 // The below branch assumes that reading past the end of the buffer is valid, as long966 // The below branch assumes that reading past the end of the buffer is valid, as long
967 // as we don't read into a new page. This should be the case for most architectures967 // as we don't read into a new page. This should be the case for most architectures
968 // which use paged memory, however should be confirmed before adding a new arch below.968 // which use paged memory, however should be confirmed before adding a new arch below.
969 .aarch64, .x86, .x86_64 => if (std.simd.suggestVectorSize(T)) |block_len| {969 .aarch64, .x86, .x86_64 => if (std.simd.suggestVectorLength(T)) |block_len| {
970 const Block = @Vector(block_len, T);970 const Block = @Vector(block_len, T);
971 const mask: Block = @splat(sentinel);971 const mask: Block = @splat(sentinel);
972972
...@@ -1020,7 +1020,7 @@ test "indexOfSentinel vector paths" {...@@ -1020,7 +1020,7 @@ test "indexOfSentinel vector paths" {
1020 const allocator = std.testing.allocator;1020 const allocator = std.testing.allocator;
10211021
1022 inline for (Types) |T| {1022 inline for (Types) |T| {
1023 const block_len = std.simd.suggestVectorSize(T) orelse continue;1023 const block_len = std.simd.suggestVectorLength(T) orelse continue;
10241024
1025 // Allocate three pages so we guarantee a page-crossing address with a full page after1025 // Allocate three pages so we guarantee a page-crossing address with a full page after
1026 const memory = try allocator.alloc(T, 3 * std.mem.page_size / @sizeOf(T));1026 const memory = try allocator.alloc(T, 3 * std.mem.page_size / @sizeOf(T));
...@@ -1111,11 +1111,11 @@ pub fn indexOfScalarPos(comptime T: type, slice: []const T, start_index: usize,...@@ -1111,11 +1111,11 @@ pub fn indexOfScalarPos(comptime T: type, slice: []const T, start_index: usize,
1111 !@inComptime() and1111 !@inComptime() and
1112 (@typeInfo(T) == .Int or @typeInfo(T) == .Float) and std.math.isPowerOfTwo(@bitSizeOf(T)))1112 (@typeInfo(T) == .Int or @typeInfo(T) == .Float) and std.math.isPowerOfTwo(@bitSizeOf(T)))
1113 {1113 {
1114 if (std.simd.suggestVectorSize(T)) |block_len| {1114 if (std.simd.suggestVectorLength(T)) |block_len| {
1115 // For Intel Nehalem (2009) and AMD Bulldozer (2012) or later, unaligned loads on aligned data result1115 // For Intel Nehalem (2009) and AMD Bulldozer (2012) or later, unaligned loads on aligned data result
1116 // in the same execution as aligned loads. We ignore older arch's here and don't bother pre-aligning.1116 // in the same execution as aligned loads. We ignore older arch's here and don't bother pre-aligning.
1117 //1117 //
1118 // Use `std.simd.suggestVectorSize(T)` to get the same alignment as used in this function1118 // Use `std.simd.suggestVectorLength(T)` to get the same alignment as used in this function
1119 // however this usually isn't necessary unless your arch has a performance penalty due to this.1119 // however this usually isn't necessary unless your arch has a performance penalty due to this.
1120 //1120 //
1121 // This may differ for other arch's. Arm for example costs a cycle when loading across a cache1121 // This may differ for other arch's. Arm for example costs a cycle when loading across a cache
lib/std/simd.zig+16-12
...@@ -6,7 +6,9 @@...@@ -6,7 +6,9 @@
6const std = @import("std");6const std = @import("std");
7const builtin = @import("builtin");7const builtin = @import("builtin");
88
9pub fn suggestVectorSizeForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?comptime_int {9pub const suggestVectorSizeForCpu = @compileError("deprecated; use 'suggestVectorLengthForCpu'");
10
11pub fn suggestVectorLengthForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?comptime_int {
10 // This is guesswork, if you have better suggestions can add it or edit the current here12 // This is guesswork, if you have better suggestions can add it or edit the current here
11 // This can run in comptime only, but stage 1 fails at it, stage 2 can understand it13 // This can run in comptime only, but stage 1 fails at it, stage 2 can understand it
12 const element_bit_size = @max(8, std.math.ceilPowerOfTwo(u16, @bitSizeOf(T)) catch unreachable);14 const element_bit_size = @max(8, std.math.ceilPowerOfTwo(u16, @bitSizeOf(T)) catch unreachable);
...@@ -53,24 +55,26 @@ pub fn suggestVectorSizeForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?...@@ -53,24 +55,26 @@ pub fn suggestVectorSizeForCpu(comptime T: type, comptime cpu: std.Target.Cpu) ?
53 return @divExact(vector_bit_size, element_bit_size);55 return @divExact(vector_bit_size, element_bit_size);
54}56}
5557
56/// Suggests a target-dependant vector size for a given type, or null if scalars are recommended.58pub const suggestVectorSize = @compileError("deprecated; use 'suggestVectorLength'");
59
60/// Suggests a target-dependant vector length for a given type, or null if scalars are recommended.
57/// Not yet implemented for every CPU architecture.61/// Not yet implemented for every CPU architecture.
58pub fn suggestVectorSize(comptime T: type) ?comptime_int {62pub fn suggestVectorLength(comptime T: type) ?comptime_int {
59 return suggestVectorSizeForCpu(T, builtin.cpu);63 return suggestVectorLengthForCpu(T, builtin.cpu);
60}64}
6165
62test "suggestVectorSizeForCpu works with signed and unsigned values" {66test "suggestVectorLengthForCpu works with signed and unsigned values" {
63 comptime var cpu = std.Target.Cpu.baseline(std.Target.Cpu.Arch.x86_64);67 comptime var cpu = std.Target.Cpu.baseline(std.Target.Cpu.Arch.x86_64);
64 comptime cpu.features.addFeature(@intFromEnum(std.Target.x86.Feature.avx512f));68 comptime cpu.features.addFeature(@intFromEnum(std.Target.x86.Feature.avx512f));
65 comptime cpu.features.populateDependencies(&std.Target.x86.all_features);69 comptime cpu.features.populateDependencies(&std.Target.x86.all_features);
66 const expected_size: usize = switch (builtin.zig_backend) {70 const expected_len: usize = switch (builtin.zig_backend) {
67 .stage2_x86_64 => 8,71 .stage2_x86_64 => 8,
68 else => 16,72 else => 16,
69 };73 };
70 const signed_integer_size = suggestVectorSizeForCpu(i32, cpu).?;74 const signed_integer_len = suggestVectorLengthForCpu(i32, cpu).?;
71 const unsigned_integer_size = suggestVectorSizeForCpu(u32, cpu).?;75 const unsigned_integer_len = suggestVectorLengthForCpu(u32, cpu).?;
72 try std.testing.expectEqual(expected_size, unsigned_integer_size);76 try std.testing.expectEqual(expected_len, unsigned_integer_len);
73 try std.testing.expectEqual(expected_size, signed_integer_size);77 try std.testing.expectEqual(expected_len, signed_integer_len);
74}78}
7579
76fn vectorLength(comptime VectorType: type) comptime_int {80fn vectorLength(comptime VectorType: type) comptime_int {
...@@ -232,7 +236,7 @@ test "vector patterns" {...@@ -232,7 +236,7 @@ test "vector patterns" {
232 }236 }
233}237}
234238
235/// Joins two vectors, shifts them leftwards (towards lower indices) and extracts the leftmost elements into a vector the size of a and b.239/// Joins two vectors, shifts them leftwards (towards lower indices) and extracts the leftmost elements into a vector the length of a and b.
236pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a, b))) @TypeOf(a, b) {240pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a, b))) @TypeOf(a, b) {
237 const len = vectorLength(@TypeOf(a, b));241 const len = vectorLength(@TypeOf(a, b));
238242
...@@ -240,7 +244,7 @@ pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a,...@@ -240,7 +244,7 @@ pub fn mergeShift(a: anytype, b: anytype, comptime shift: VectorCount(@TypeOf(a,
240}244}
241245
242/// Elements are shifted rightwards (towards higher indices). New elements are added to the left, and the rightmost elements are cut off246/// Elements are shifted rightwards (towards higher indices). New elements are added to the left, and the rightmost elements are cut off
243/// so that the size of the vector stays the same.247/// so that the length of the vector stays the same.
244pub fn shiftElementsRight(vec: anytype, comptime amount: VectorCount(@TypeOf(vec)), shift_in: std.meta.Child(@TypeOf(vec))) @TypeOf(vec) {248pub fn shiftElementsRight(vec: anytype, comptime amount: VectorCount(@TypeOf(vec)), shift_in: std.meta.Child(@TypeOf(vec))) @TypeOf(vec) {
245 // It may be possible to implement shifts and rotates with a runtime-friendly slice of two joined vectors, as the length of the249 // It may be possible to implement shifts and rotates with a runtime-friendly slice of two joined vectors, as the length of the
246 // slice would be comptime-known. This would permit vector shifts and rotates by a non-comptime-known amount.250 // slice would be comptime-known. This would permit vector shifts and rotates by a non-comptime-known amount.
lib/std/unicode.zig+6-6
...@@ -202,7 +202,7 @@ pub fn utf8CountCodepoints(s: []const u8) !usize {...@@ -202,7 +202,7 @@ pub fn utf8CountCodepoints(s: []const u8) !usize {
202pub fn utf8ValidateSlice(input: []const u8) bool {202pub fn utf8ValidateSlice(input: []const u8) bool {
203 var remaining = input;203 var remaining = input;
204204
205 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;205 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
206 const Chunk = @Vector(chunk_len, u8);206 const Chunk = @Vector(chunk_len, u8);
207207
208 // Fast path. Check for and skip ASCII characters at the start of the input.208 // Fast path. Check for and skip ASCII characters at the start of the input.
...@@ -758,7 +758,7 @@ pub fn utf16leToUtf8Alloc(allocator: mem.Allocator, utf16le: []const u16) ![]u8...@@ -758,7 +758,7 @@ pub fn utf16leToUtf8Alloc(allocator: mem.Allocator, utf16le: []const u16) ![]u8
758758
759 var remaining = utf16le;759 var remaining = utf16le;
760 if (builtin.zig_backend != .stage2_x86_64) {760 if (builtin.zig_backend != .stage2_x86_64) {
761 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;761 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
762 const Chunk = @Vector(chunk_len, u16);762 const Chunk = @Vector(chunk_len, u16);
763763
764 // Fast path. Check for and encode ASCII characters at the start of the input.764 // Fast path. Check for and encode ASCII characters at the start of the input.
...@@ -801,7 +801,7 @@ pub fn utf16leToUtf8AllocZ(allocator: mem.Allocator, utf16le: []const u16) ![:0]...@@ -801,7 +801,7 @@ pub fn utf16leToUtf8AllocZ(allocator: mem.Allocator, utf16le: []const u16) ![:0]
801801
802 var remaining = utf16le;802 var remaining = utf16le;
803 if (builtin.zig_backend != .stage2_x86_64) {803 if (builtin.zig_backend != .stage2_x86_64) {
804 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;804 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
805 const Chunk = @Vector(chunk_len, u16);805 const Chunk = @Vector(chunk_len, u16);
806806
807 // Fast path. Check for and encode ASCII characters at the start of the input.807 // Fast path. Check for and encode ASCII characters at the start of the input.
...@@ -842,7 +842,7 @@ pub fn utf16leToUtf8(utf8: []u8, utf16le: []const u16) !usize {...@@ -842,7 +842,7 @@ pub fn utf16leToUtf8(utf8: []u8, utf16le: []const u16) !usize {
842842
843 var remaining = utf16le;843 var remaining = utf16le;
844 if (builtin.zig_backend != .stage2_x86_64) {844 if (builtin.zig_backend != .stage2_x86_64) {
845 const chunk_len = std.simd.suggestVectorSize(u16) orelse 1;845 const chunk_len = std.simd.suggestVectorLength(u16) orelse 1;
846 const Chunk = @Vector(chunk_len, u16);846 const Chunk = @Vector(chunk_len, u16);
847847
848 // Fast path. Check for and encode ASCII characters at the start of the input.848 // Fast path. Check for and encode ASCII characters at the start of the input.
...@@ -941,7 +941,7 @@ pub fn utf8ToUtf16LeWithNull(allocator: mem.Allocator, utf8: []const u8) ![:0]u1...@@ -941,7 +941,7 @@ pub fn utf8ToUtf16LeWithNull(allocator: mem.Allocator, utf8: []const u8) ![:0]u1
941 var remaining = utf8;941 var remaining = utf8;
942 // Need support for std.simd.interlace942 // Need support for std.simd.interlace
943 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {943 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {
944 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;944 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
945 const Chunk = @Vector(chunk_len, u8);945 const Chunk = @Vector(chunk_len, u8);
946946
947 // Fast path. Check for and encode ASCII characters at the start of the input.947 // Fast path. Check for and encode ASCII characters at the start of the input.
...@@ -986,7 +986,7 @@ pub fn utf8ToUtf16Le(utf16le: []u16, utf8: []const u8) !usize {...@@ -986,7 +986,7 @@ pub fn utf8ToUtf16Le(utf16le: []u16, utf8: []const u8) !usize {
986 var remaining = utf8;986 var remaining = utf8;
987 // Need support for std.simd.interlace987 // Need support for std.simd.interlace
988 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {988 if (builtin.zig_backend != .stage2_x86_64 and comptime !builtin.cpu.arch.isMIPS()) {
989 const chunk_len = std.simd.suggestVectorSize(u8) orelse 1;989 const chunk_len = std.simd.suggestVectorLength(u8) orelse 1;
990 const Chunk = @Vector(chunk_len, u8);990 const Chunk = @Vector(chunk_len, u8);
991991
992 // Fast path. Check for and encode ASCII characters at the start of the input.992 // Fast path. Check for and encode ASCII characters at the start of the input.