authorgravatar for 124872+jedisct1@users.noreply.github.comFrank Denis <124872+jedisct1@users.noreply.github.com> 2025-11-26 12:07:43+01:00
committergravatar for 124872+jedisct1@users.noreply.github.comFrank Denis <124872+jedisct1@users.noreply.github.com> 2025-11-26 10:08:09+01:00
log846082fdf20d82637198d0f04b6921c6eedff943
tree40d615e0ca38f5e874599984b0844d2e24c2eb5c
parent9ede8ee135961bcc268a5077771571437bd32d8c

std.crypto.kt128: add support for threads

Allow KT128 and KT256 to use multiple threads to quickly process very large inputs.

1 files changed, 515 insertions(+), 0 deletions(-)

lib/std/crypto/kangarootwelve.zig+515
...@@ -2,6 +2,8 @@ const std = @import("std");...@@ -2,6 +2,8 @@ const std = @import("std");
2const builtin = @import("builtin");2const builtin = @import("builtin");
3const crypto = std.crypto;3const crypto = std.crypto;
4const Allocator = std.mem.Allocator;4const Allocator = std.mem.Allocator;
5const Io = std.Io;
6const Thread = std.Thread;
57
6const TurboSHAKE128State = crypto.hash.sha3.TurboShake128(0x06);8const TurboSHAKE128State = crypto.hash.sha3.TurboShake128(0x06);
7const TurboSHAKE256State = crypto.hash.sha3.TurboShake256(0x06);9const TurboSHAKE256State = crypto.hash.sha3.TurboShake256(0x06);
...@@ -12,6 +14,13 @@ const cache_line_size = std.atomic.cache_line;...@@ -12,6 +14,13 @@ const cache_line_size = std.atomic.cache_line;
12// Optimal SIMD vector length for u64 on this target platform14// Optimal SIMD vector length for u64 on this target platform
13const optimal_vector_len = std.simd.suggestVectorLength(u64) orelse 1;15const optimal_vector_len = std.simd.suggestVectorLength(u64) orelse 1;
1416
17// Number of bytes processed per SIMD batch in multi-threaded mode
18const bytes_per_batch = 256 * 1024;
19
20// Multi-threading threshold: inputs larger than this will use parallel processing.
21// Benchmarked optimal value for ReleaseFast mode.
22const large_file_threshold: usize = 2 * 1024 * 1024; // 2 MB
23
15// Round constants for Keccak-p[1600,12]24// Round constants for Keccak-p[1600,12]
16const RC = [12]u64{25const RC = [12]u64{
17 0x000000008000808B,26 0x000000008000808B,
...@@ -569,6 +578,16 @@ fn processLeaves(...@@ -569,6 +578,16 @@ fn processLeaves(
569 }578 }
570}579}
571580
581/// Context for processing a batch of leaves in a thread
582const LeafBatchContext = struct {
583 output_cvs: []align(@alignOf(u64)) u8,
584 batch_start: usize,
585 batch_count: usize,
586 view: *const MultiSliceView,
587 scratch_buffer: []u8, // Pre-allocated scratch space (no allocations in worker)
588 total_len: usize, // Total length of input data (for boundary checking)
589};
590
572/// Helper function to process N leaves in parallel, reducing code duplication591/// Helper function to process N leaves in parallel, reducing code duplication
573inline fn processNLeaves(592inline fn processNLeaves(
574 comptime Variant: type,593 comptime Variant: type,
...@@ -593,6 +612,42 @@ inline fn processNLeaves(...@@ -593,6 +612,42 @@ inline fn processNLeaves(
593 }612 }
594}613}
595614
615/// Process a batch of leaves in a single thread using SIMD
616fn processLeafBatch(comptime Variant: type, ctx: LeafBatchContext) void {
617 const cv_size = Variant.cv_size;
618 const leaf_buffer = ctx.scratch_buffer[0 .. 8 * chunk_size];
619
620 var cvs_offset: usize = 0;
621 var j: usize = ctx.batch_start;
622 const batch_end = @min(ctx.batch_start + ctx.batch_count * chunk_size, ctx.total_len);
623
624 // Process leaves using SIMD (8x, 4x, 2x) based on optimal vector length
625 inline for ([_]usize{ 8, 4, 2 }) |batch_size| {
626 while (optimal_vector_len >= batch_size and j + batch_size * chunk_size <= batch_end) {
627 processNLeaves(Variant, batch_size, ctx.view, j, leaf_buffer, @alignCast(ctx.output_cvs[cvs_offset..]));
628 cvs_offset += batch_size * cv_size;
629 j += batch_size * chunk_size;
630 }
631 }
632
633 // Process remaining single leaves
634 while (j < batch_end) {
635 const chunk_len = @min(chunk_size, batch_end - j);
636 if (ctx.view.tryGetSlice(j, j + chunk_len)) |leaf_data| {
637 const cv_slice = MultiSliceView.init(leaf_data, &[_]u8{}, &[_]u8{});
638 Variant.turboShakeToBuffer(&cv_slice, 0x0B, ctx.output_cvs[cvs_offset..][0..cv_size]);
639 } else {
640 ctx.view.copyRange(j, j + chunk_len, leaf_buffer[0..chunk_len]);
641 const cv_slice = MultiSliceView.init(leaf_buffer[0..chunk_len], &[_]u8{}, &[_]u8{});
642 Variant.turboShakeToBuffer(&cv_slice, 0x0B, ctx.output_cvs[cvs_offset..][0..cv_size]);
643 }
644 cvs_offset += cv_size;
645 j += chunk_len;
646 }
647
648 std.debug.assert(cvs_offset == ctx.output_cvs.len);
649}
650
596/// Helper to process N leaves in SIMD and absorb CVs into state651/// Helper to process N leaves in SIMD and absorb CVs into state
597inline fn processAndAbsorbNLeaves(652inline fn processAndAbsorbNLeaves(
598 comptime Variant: type,653 comptime Variant: type,
...@@ -679,6 +734,224 @@ fn ktSingleThreaded(comptime Variant: type, view: *const MultiSliceView, total_l...@@ -679,6 +734,224 @@ fn ktSingleThreaded(comptime Variant: type, view: *const MultiSliceView, total_l
679 final_state.final(output);734 final_state.final(output);
680}735}
681736
737fn BatchResult(comptime Variant: type) type {
738 const cv_size = Variant.cv_size;
739 const leaves_per_batch = bytes_per_batch / chunk_size;
740 const max_cvs_size = leaves_per_batch * cv_size;
741
742 return struct {
743 batch_idx: usize,
744 cv_len: usize,
745 cvs: [max_cvs_size]u8,
746 };
747}
748
749fn SelectLeafContext(comptime Variant: type) type {
750 const cv_size = Variant.cv_size;
751 const Result = BatchResult(Variant);
752
753 return struct {
754 view: *const MultiSliceView,
755 batch_idx: usize,
756 start_offset: usize,
757 num_leaves: usize,
758
759 fn process(ctx: @This()) Result {
760 var result: Result = .{
761 .batch_idx = ctx.batch_idx,
762 .cv_len = ctx.num_leaves * cv_size,
763 .cvs = undefined,
764 };
765
766 var leaf_buffer: [bytes_per_batch]u8 align(cache_line_size) = undefined;
767 var leaves_processed: usize = 0;
768 var byte_offset = ctx.start_offset;
769 var cv_offset: usize = 0;
770 const simd_batch_bytes = optimal_vector_len * chunk_size;
771 while (leaves_processed + optimal_vector_len <= ctx.num_leaves) {
772 if (ctx.view.tryGetSlice(byte_offset, byte_offset + simd_batch_bytes)) |leaf_data| {
773 var leaf_cvs: [optimal_vector_len * Variant.cv_size]u8 = undefined;
774 processLeaves(Variant, optimal_vector_len, leaf_data, &leaf_cvs);
775 @memcpy(result.cvs[cv_offset..][0..leaf_cvs.len], &leaf_cvs);
776 } else {
777 ctx.view.copyRange(byte_offset, byte_offset + simd_batch_bytes, leaf_buffer[0..simd_batch_bytes]);
778 var leaf_cvs: [optimal_vector_len * Variant.cv_size]u8 = undefined;
779 processLeaves(Variant, optimal_vector_len, leaf_buffer[0..simd_batch_bytes], &leaf_cvs);
780 @memcpy(result.cvs[cv_offset..][0..leaf_cvs.len], &leaf_cvs);
781 }
782 leaves_processed += optimal_vector_len;
783 byte_offset += optimal_vector_len * chunk_size;
784 cv_offset += optimal_vector_len * cv_size;
785 }
786
787 while (leaves_processed < ctx.num_leaves) {
788 const leaf_end = byte_offset + chunk_size;
789 var cv_buffer: [64]u8 = undefined;
790
791 if (ctx.view.tryGetSlice(byte_offset, leaf_end)) |leaf_data| {
792 const cv_slice = MultiSliceView.init(leaf_data, &[_]u8{}, &[_]u8{});
793 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
794 } else {
795 ctx.view.copyRange(byte_offset, leaf_end, leaf_buffer[0..chunk_size]);
796 const cv_slice = MultiSliceView.init(leaf_buffer[0..chunk_size], &[_]u8{}, &[_]u8{});
797 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
798 }
799 @memcpy(result.cvs[cv_offset..][0..cv_size], cv_buffer[0..cv_size]);
800
801 leaves_processed += 1;
802 byte_offset += chunk_size;
803 cv_offset += cv_size;
804 }
805
806 return result;
807 }
808 };
809}
810
811fn FinalLeafContext(comptime Variant: type) type {
812 return struct {
813 view: *const MultiSliceView,
814 start_offset: usize,
815 leaf_len: usize,
816 output_cv: []align(@alignOf(u64)) u8,
817
818 fn process(ctx: @This()) void {
819 const cv_size = Variant.cv_size;
820 var leaf_buffer: [chunk_size]u8 = undefined;
821 var cv_buffer: [64]u8 = undefined;
822
823 if (ctx.view.tryGetSlice(ctx.start_offset, ctx.start_offset + ctx.leaf_len)) |leaf_data| {
824 const cv_slice = MultiSliceView.init(leaf_data, &[_]u8{}, &[_]u8{});
825 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
826 } else {
827 ctx.view.copyRange(ctx.start_offset, ctx.start_offset + ctx.leaf_len, leaf_buffer[0..ctx.leaf_len]);
828 const cv_slice = MultiSliceView.init(leaf_buffer[0..ctx.leaf_len], &[_]u8{}, &[_]u8{});
829 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
830 }
831 @memcpy(ctx.output_cv[0..cv_size], cv_buffer[0..cv_size]);
832 }
833 };
834}
835
836fn ktMultiThreaded(
837 comptime Variant: type,
838 allocator: Allocator,
839 io: Io,
840 view: *const MultiSliceView,
841 total_len: usize,
842 output: []u8,
843) !void {
844 comptime std.debug.assert(bytes_per_batch % (optimal_vector_len * chunk_size) == 0);
845
846 const cv_size = Variant.cv_size;
847 const StateType = Variant.StateType;
848 const leaves_per_batch = bytes_per_batch / chunk_size;
849 const remaining_bytes = total_len - chunk_size;
850 const total_leaves = std.math.divCeil(usize, remaining_bytes, chunk_size) catch unreachable;
851
852 var final_state = StateType.init(.{});
853
854 var first_chunk_buffer: [chunk_size]u8 = undefined;
855 if (view.tryGetSlice(0, chunk_size)) |first_chunk| {
856 final_state.update(first_chunk);
857 } else {
858 view.copyRange(0, chunk_size, &first_chunk_buffer);
859 final_state.update(&first_chunk_buffer);
860 }
861
862 const padding = [_]u8{ 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00 };
863 final_state.update(&padding);
864
865 const full_leaves = remaining_bytes / chunk_size;
866 const has_partial_leaf = (remaining_bytes % chunk_size) != 0;
867 const partial_leaf_size = if (has_partial_leaf) remaining_bytes % chunk_size else 0;
868
869 if (full_leaves > 0) {
870 const total_batches = std.math.divCeil(usize, full_leaves, leaves_per_batch) catch unreachable;
871 const max_concurrent: usize = @min(256, total_batches);
872
873 const Result = BatchResult(Variant);
874 const SelectResult = union(enum) { batch: Result };
875 const Select = Io.Select(SelectResult);
876
877 const select_buf = try allocator.alloc(SelectResult, max_concurrent);
878 defer allocator.free(select_buf);
879
880 // Buffer for out-of-order results (select_buf slots get reused)
881 const pending_cv_buf = try allocator.alloc([leaves_per_batch * cv_size]u8, max_concurrent);
882 defer allocator.free(pending_cv_buf);
883 var pending_cv_lens: [256]usize = .{0} ** 256;
884
885 var select: Select = .init(io, select_buf);
886 var batches_spawned: usize = 0;
887 var next_to_process: usize = 0;
888
889 while (next_to_process < total_batches) {
890 while (batches_spawned < total_batches and batches_spawned - next_to_process < max_concurrent) {
891 const batch_start_leaf = batches_spawned * leaves_per_batch;
892 const batch_leaves = @min(leaves_per_batch, full_leaves - batch_start_leaf);
893 const start_offset = chunk_size + batch_start_leaf * chunk_size;
894
895 select.async(.batch, SelectLeafContext(Variant).process, .{SelectLeafContext(Variant){
896 .view = view,
897 .batch_idx = batches_spawned,
898 .start_offset = start_offset,
899 .num_leaves = batch_leaves,
900 }});
901 batches_spawned += 1;
902 }
903
904 const result = select.wait() catch unreachable;
905 const batch = result.batch;
906 const slot = batch.batch_idx % max_concurrent;
907
908 if (batch.batch_idx == next_to_process) {
909 final_state.update(batch.cvs[0..batch.cv_len]);
910 next_to_process += 1;
911
912 // Drain pending batches that are now ready
913 while (next_to_process < total_batches) {
914 const pending_slot = next_to_process % max_concurrent;
915 const pending_len = pending_cv_lens[pending_slot];
916 if (pending_len == 0) break;
917
918 final_state.update(pending_cv_buf[pending_slot][0..pending_len]);
919 pending_cv_lens[pending_slot] = 0;
920 next_to_process += 1;
921 }
922 } else {
923 @memcpy(pending_cv_buf[slot][0..batch.cv_len], batch.cvs[0..batch.cv_len]);
924 pending_cv_lens[slot] = batch.cv_len;
925 }
926 }
927
928 select.group.wait(io);
929 }
930
931 if (has_partial_leaf) {
932 var cv_buffer: [64]u8 = undefined;
933 var leaf_buffer: [chunk_size]u8 = undefined;
934
935 const start_offset = chunk_size + full_leaves * chunk_size;
936 if (view.tryGetSlice(start_offset, start_offset + partial_leaf_size)) |leaf_data| {
937 const cv_slice = MultiSliceView.init(leaf_data, &[_]u8{}, &[_]u8{});
938 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
939 } else {
940 view.copyRange(start_offset, start_offset + partial_leaf_size, leaf_buffer[0..partial_leaf_size]);
941 const cv_slice = MultiSliceView.init(leaf_buffer[0..partial_leaf_size], &[_]u8{}, &[_]u8{});
942 Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
943 }
944 final_state.update(cv_buffer[0..cv_size]);
945 }
946
947 const n_enc = rightEncode(total_leaves);
948 final_state.update(n_enc.slice());
949 const terminator = [_]u8{ 0xFF, 0xFF };
950 final_state.update(&terminator);
951
952 final_state.final(output);
953}
954
682/// Generic KangarooTwelve hash function builder.955/// Generic KangarooTwelve hash function builder.
683/// Creates a public API type with hash and hashParallel methods for a specific variant.956/// Creates a public API type with hash and hashParallel methods for a specific variant.
684fn KTHash(957fn KTHash(
...@@ -974,6 +1247,32 @@ fn KTHash(...@@ -974,6 +1247,32 @@ fn KTHash(
974 // Tree mode - single-threaded SIMD processing1247 // Tree mode - single-threaded SIMD processing
975 ktSingleThreaded(Variant, &view, total_len, out);1248 ktSingleThreaded(Variant, &view, total_len, out);
976 }1249 }
1250
1251 /// Hash with automatic parallelization for large inputs (>2MB).
1252 /// Automatically uses sequential processing for smaller inputs to avoid thread overhead.
1253 /// Allocator required for temporary buffers. IO object required for thread management.
1254 pub fn hashParallel(message: []const u8, out: []u8, options: Options, allocator: Allocator, io: Io) !void {
1255 const custom = options.customization orelse &[_]u8{};
1256
1257 const custom_len_enc = rightEncode(custom.len);
1258 const view = MultiSliceView.init(message, custom, custom_len_enc.slice());
1259 const total_len = view.totalLen();
1260
1261 // Single chunk case
1262 if (total_len <= chunk_size) {
1263 singleChunkFn(&view, 0x07, out);
1264 return;
1265 }
1266
1267 // Use single-threaded processing if below threshold
1268 if (total_len < large_file_threshold) {
1269 ktSingleThreaded(Variant, &view, total_len, out);
1270 return;
1271 }
1272
1273 // Tree mode - multi-threaded processing
1274 try ktMultiThreaded(Variant, allocator, io, &view, total_len, out);
1275 }
977 };1276 };
978}1277}
9791278
...@@ -1006,6 +1305,222 @@ pub const KT128 = KTHash(KT128Variant, turboShake128MultiSliceToBuffer);...@@ -1006,6 +1305,222 @@ pub const KT128 = KTHash(KT128Variant, turboShake128MultiSliceToBuffer);
1006/// For most applications, KT128 offers better performance with adequate security.1305/// For most applications, KT128 offers better performance with adequate security.
1007pub const KT256 = KTHash(KT256Variant, turboShake256MultiSliceToBuffer);1306pub const KT256 = KTHash(KT256Variant, turboShake256MultiSliceToBuffer);
10081307
1308test "KT128 sequential and parallel produce same output for small inputs" {
1309 const allocator = std.testing.allocator;
1310 const io = std.testing.io;
1311
1312 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1313 const random = prng.random();
1314
1315 // Test with different small input sizes
1316 const test_sizes = [_]usize{ 100, 1024, 4096, 8192 }; // 100B, 1KB, 4KB, 8KB
1317
1318 for (test_sizes) |size| {
1319 const input = try allocator.alloc(u8, size);
1320 defer allocator.free(input);
1321
1322 // Fill with random data
1323 random.bytes(input);
1324
1325 var output_seq: [32]u8 = undefined;
1326 var output_par: [32]u8 = undefined;
1327
1328 // Hash with sequential method
1329 try KT128.hash(input, &output_seq, .{});
1330
1331 // Hash with parallel method
1332 try KT128.hashParallel(input, &output_par, .{}, allocator, io);
1333
1334 // Verify outputs match
1335 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1336 }
1337}
1338
1339test "KT128 sequential and parallel produce same output for large inputs" {
1340 const allocator = std.testing.allocator;
1341 const io = std.testing.io;
1342
1343 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1344 const random = prng.random();
1345
1346 // Test with input sizes above the 2MB threshold to trigger parallel processing.
1347 // Include a size with partial final leaf to stress boundary handling.
1348 const test_sizes = [_]usize{
1349 5 * 512 * 1024, // 2.5 MB
1350 5 * 512 * 1024 + 8191, // 2.5 MB + 8191B (partial leaf)
1351 };
1352
1353 for (test_sizes) |size| {
1354 const input = try allocator.alloc(u8, size);
1355 defer allocator.free(input);
1356
1357 // Fill with random data
1358 random.bytes(input);
1359
1360 var output_seq: [64]u8 = undefined;
1361 var output_par: [64]u8 = undefined;
1362
1363 // Hash with sequential method
1364 try KT128.hash(input, &output_seq, .{});
1365
1366 // Hash with parallel method
1367 try KT128.hashParallel(input, &output_par, .{}, allocator, io);
1368
1369 // Verify outputs match
1370 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1371 }
1372}
1373
1374test "KT128 sequential and parallel produce same output for many random lengths" {
1375 const allocator = std.testing.allocator;
1376 const io = std.testing.io;
1377
1378 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1379 const random = prng.random();
1380
1381 const num_tests = if (builtin.mode == .Debug) 10 else 1000;
1382 const max_length = 250000;
1383
1384 for (0..num_tests) |_| {
1385 const length = random.intRangeAtMost(usize, 0, max_length);
1386
1387 const input = try allocator.alloc(u8, length);
1388 defer allocator.free(input);
1389
1390 random.bytes(input);
1391
1392 var output_seq: [32]u8 = undefined;
1393 var output_par: [32]u8 = undefined;
1394
1395 try KT128.hash(input, &output_seq, .{});
1396 try KT128.hashParallel(input, &output_par, .{}, allocator, io);
1397
1398 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1399 }
1400}
1401
1402test "KT128 sequential and parallel produce same output with customization" {
1403 const allocator = std.testing.allocator;
1404 const io = std.testing.io;
1405
1406 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1407 const random = prng.random();
1408
1409 const input_size = 5 * 512 * 1024; // 2.5MB
1410 const input = try allocator.alloc(u8, input_size);
1411 defer allocator.free(input);
1412
1413 // Fill with random data
1414 random.bytes(input);
1415
1416 const customization = "test domain";
1417 var output_seq: [48]u8 = undefined;
1418 var output_par: [48]u8 = undefined;
1419
1420 // Hash with sequential method
1421 try KT128.hash(input, &output_seq, .{ .customization = customization });
1422
1423 // Hash with parallel method
1424 try KT128.hashParallel(input, &output_par, .{ .customization = customization }, allocator, io);
1425
1426 // Verify outputs match
1427 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1428}
1429
1430test "KT256 sequential and parallel produce same output for small inputs" {
1431 const allocator = std.testing.allocator;
1432 const io = std.testing.io;
1433
1434 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1435 const random = prng.random();
1436
1437 // Test with different small input sizes
1438 const test_sizes = [_]usize{ 100, 1024, 4096, 8192 }; // 100B, 1KB, 4KB, 8KB
1439
1440 for (test_sizes) |size| {
1441 const input = try allocator.alloc(u8, size);
1442 defer allocator.free(input);
1443
1444 // Fill with random data
1445 random.bytes(input);
1446
1447 var output_seq: [64]u8 = undefined;
1448 var output_par: [64]u8 = undefined;
1449
1450 // Hash with sequential method
1451 try KT256.hash(input, &output_seq, .{});
1452
1453 // Hash with parallel method
1454 try KT256.hashParallel(input, &output_par, .{}, allocator, io);
1455
1456 // Verify outputs match
1457 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1458 }
1459}
1460
1461test "KT256 sequential and parallel produce same output for large inputs" {
1462 const allocator = std.testing.allocator;
1463 const io = std.testing.io;
1464
1465 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1466 const random = prng.random();
1467
1468 // Test with input sizes above the 2MB threshold to trigger parallel processing.
1469 // Include a size with partial final leaf to stress boundary handling.
1470 const test_sizes = [_]usize{
1471 5 * 512 * 1024, // 2.5 MB
1472 5 * 512 * 1024 + 8191, // 2.5 MB + 8191B (partial leaf)
1473 };
1474
1475 for (test_sizes) |size| {
1476 const input = try allocator.alloc(u8, size);
1477 defer allocator.free(input);
1478
1479 // Fill with random data
1480 random.bytes(input);
1481
1482 var output_seq: [64]u8 = undefined;
1483 var output_par: [64]u8 = undefined;
1484
1485 // Hash with sequential method
1486 try KT256.hash(input, &output_seq, .{});
1487
1488 // Hash with parallel method
1489 try KT256.hashParallel(input, &output_par, .{}, allocator, io);
1490
1491 // Verify outputs match
1492 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1493 }
1494}
1495
1496test "KT256 sequential and parallel produce same output with customization" {
1497 const allocator = std.testing.allocator;
1498 const io = std.testing.io;
1499
1500 var prng = std.Random.DefaultPrng.init(std.testing.random_seed);
1501 const random = prng.random();
1502
1503 const input_size = 5 * 512 * 1024; // 2.5MB
1504 const input = try allocator.alloc(u8, input_size);
1505 defer allocator.free(input);
1506
1507 // Fill with random data
1508 random.bytes(input);
1509
1510 const customization = "test domain";
1511 var output_seq: [80]u8 = undefined;
1512 var output_par: [80]u8 = undefined;
1513
1514 // Hash with sequential method
1515 try KT256.hash(input, &output_seq, .{ .customization = customization });
1516
1517 // Hash with parallel method
1518 try KT256.hashParallel(input, &output_par, .{ .customization = customization }, allocator, io);
1519
1520 // Verify outputs match
1521 try std.testing.expectEqualSlices(u8, &output_seq, &output_par);
1522}
1523
1009/// Helper: Generate pattern data where data[i] = (i % 251)1524/// Helper: Generate pattern data where data[i] = (i % 251)
1010fn generatePattern(allocator: Allocator, len: usize) ![]u8 {1525fn generatePattern(allocator: Allocator, len: usize) ![]u8 {
1011 const data = try allocator.alloc(u8, len);1526 const data = try allocator.alloc(u8, len);