Zig 0.17.0-dev (Split by item)

This is an example of documentation generated by ZigDoc, an alternative to Zig's built-in Auto Doc feature. See also examples in other modes/formats. The project being documented here (as the example) is the Zig library itself.

KTHash

Generic KangarooTwelve hash function builder. Creates a public API type with hash and hashParallel methods for a specific variant.

kangarootwelve.KTHash
fn KTHash(
    comptime Variant: type,
    comptime singleChunkFn: fn (*const MultiSliceView, u8, []u8) void,
) type

File

lib/std/crypto/kangarootwelve.zig:962

Code

fn KTHash(
    comptime Variant: type,
    comptime singleChunkFn: fn (*const MultiSliceView, u8, []u8) void,
) type {
    return struct {
        const Self = @This();
        const StateType = Variant.StateType;

        /// The recommended output length, in bytes.
        pub const digest_length = Variant.security_level / 8 * 2;
        /// The block length, or rate, in bytes.
        pub const block_length = Variant.rate;

        /// Configuration options for KangarooTwelve hashing.
        ///
        /// Options include an optional customization string that provides domain separation,
        /// ensuring that identical inputs with different customization strings
        /// produce completely distinct hash outputs.
        ///
        /// This prevents hash collisions when the same data is hashed in different contexts.
        ///
        /// Customization strings can be of any length.
        ///
        /// Common options for customization::
        ///
        /// - Key derivation or MAC: 16-byte secret for KT128, 32-byte secret for KT256
        /// - Context Separation: domain-specific strings (e.g., "email", "password", "session")
        /// - Composite Keys: concatenation of secret key + context string
        pub const Options = struct {
            customization: ?[]const u8 = null,
        };

        // Message buffer (accumulates message data only, not customization)
        buffer: [chunk_size]u8,
        buffer_len: usize,
        message_len: usize,

        // Customization string (fixed at init)
        customization: []const u8,
        custom_len_enc: RightEncoded,

        // Tree mode state (lazy initialization when buffer overflows first time)
        first_chunk: ?[chunk_size]u8, // Saved first chunk for tree mode
        final_state: ?StateType, // Running TurboSHAKE state for final node
        num_leaves: usize, // Count of leaves processed (after first chunk)

        // SIMD chunk batching
        pending_chunks: [8 * chunk_size]u8 align(cache_line_size), // Buffer for up to 8 chunks
        pending_count: usize, // Number of complete chunks in pending_chunks

        /// Initialize a KangarooTwelve hashing context.
        ///
        /// Options include an optional customization string that provides domain separation,
        /// ensuring that identical inputs with different customization strings
        /// produce completely distinct hash outputs.
        ///
        /// This prevents hash collisions when the same data is hashed in different contexts.
        ///
        /// Customization strings can be of any length.
        ///
        /// Common options for customization::
        ///
        /// - Key derivation or MAC: 16-byte secret for KT128, 32-byte secret for KT256
        /// - Context Separation: domain-specific strings (e.g., "email", "password", "session")
        /// - Composite Keys: concatenation of secret key + context string
        pub fn init(options: Options) Self {
            const custom = options.customization orelse &[_]u8{};
            return .{
                .buffer = undefined,
                .buffer_len = 0,
                .message_len = 0,
                .customization = custom,
                .custom_len_enc = rightEncode(custom.len),
                .first_chunk = null,
                .final_state = null,
                .num_leaves = 0,
                .pending_chunks = undefined,
                .pending_count = 0,
            };
        }

        /// Flush all pending chunks using SIMD when possible
        fn flushPendingChunks(self: *Self) void {
            const cv_size = Variant.cv_size;

            // Process all pending chunks using the largest SIMD batch sizes possible
            while (self.pending_count > 0) {
                // Try SIMD batches in decreasing size order
                inline for ([_]usize{ 8, 4, 2 }) |batch_size| {
                    if (optimal_vector_len >= batch_size and self.pending_count >= batch_size) {
                        var leaf_cvs: [batch_size * cv_size]u8 align(cache_line_size) = undefined;
                        processLeaves(Variant, batch_size, self.pending_chunks[0 .. batch_size * chunk_size], &leaf_cvs);
                        self.final_state.?.update(&leaf_cvs);
                        self.num_leaves += batch_size;
                        self.pending_count -= batch_size;

                        // Shift remaining chunks to the front
                        if (self.pending_count > 0) {
                            const remaining_bytes = self.pending_count * chunk_size;
                            @memcpy(self.pending_chunks[0..remaining_bytes], self.pending_chunks[batch_size * chunk_size ..][0..remaining_bytes]);
                        }
                        break; // Continue outer loop to try next batch
                    }
                }

                // If no SIMD batch was possible, process one chunk with scalar code
                if (self.pending_count > 0 and self.pending_count < 2) {
                    var cv_buffer: [64]u8 = undefined;
                    const cv_slice = MultiSliceView.init(self.pending_chunks[0..chunk_size], &[_]u8{}, &[_]u8{});
                    Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
                    self.final_state.?.update(cv_buffer[0..cv_size]);
                    self.num_leaves += 1;
                    self.pending_count -= 1;
                    break; // No more chunks to process
                }
            }
        }

        /// Absorb data into the hash state.
        /// Can be called multiple times to incrementally add data.
        pub fn update(self: *Self, data: []const u8) void {
            if (data.len == 0) return;

            var remaining = data;

            while (remaining.len > 0) {
                const space_in_buffer = chunk_size - self.buffer_len;
                const to_copy = @min(space_in_buffer, remaining.len);

                // Copy data into buffer
                @memcpy(self.buffer[self.buffer_len..][0..to_copy], remaining[0..to_copy]);
                self.buffer_len += to_copy;
                self.message_len += to_copy;
                remaining = remaining[to_copy..];

                // If buffer is full, process it
                if (self.buffer_len == chunk_size) {
                    if (self.first_chunk == null) {
                        // First time buffer fills - initialize tree mode
                        self.first_chunk = self.buffer;
                        self.final_state = StateType.init(.{});

                        // Absorb first chunk into final state
                        self.final_state.?.update(&self.buffer);

                        // Absorb padding (8 bytes: 0x03 followed by 7 zeros)
                        const padding = [_]u8{ 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00 };
                        self.final_state.?.update(&padding);
                    } else {
                        // Add chunk to pending buffer for SIMD batch processing
                        @memcpy(self.pending_chunks[self.pending_count * chunk_size ..][0..chunk_size], &self.buffer);
                        self.pending_count += 1;

                        // Flush when we have enough chunks for optimal SIMD batch
                        // Determine best batch size for this architecture
                        const optimal_batch_size = comptime blk: {
                            if (optimal_vector_len >= 8) break :blk 8;
                            if (optimal_vector_len >= 4) break :blk 4;
                            if (optimal_vector_len >= 2) break :blk 2;
                            break :blk 1;
                        };
                        if (self.pending_count >= optimal_batch_size) {
                            self.flushPendingChunks();
                        }
                    }
                    self.buffer_len = 0;
                }
            }
        }

        /// Finalize the hash and produce output.
        ///
        /// Unlike traditional hash functions, the output can be of any length.
        ///
        /// When using as a regular hash function, use the recommended `digest_length` value (32 bytes for KT128, 64 bytes for KT256).
        ///
        /// After calling this method, the context should not be reused. However, the structure can be cloned before finalizing
        /// to compute multiple hashes with the same prefix.
        pub fn final(self: *Self, out: []u8) void {
            const cv_size = Variant.cv_size;

            // Calculate total length: message + customization + right_encode(customization.len)
            const total_len = self.message_len + self.customization.len + self.custom_len_enc.len;

            // Single chunk mode: total data fits in one chunk
            if (total_len <= chunk_size) {
                // Build the complete input: buffer + customization + encoded length
                var single_chunk: [chunk_size]u8 = undefined;
                @memcpy(single_chunk[0..self.buffer_len], self.buffer[0..self.buffer_len]);
                @memcpy(single_chunk[self.buffer_len..][0..self.customization.len], self.customization);
                @memcpy(single_chunk[self.buffer_len + self.customization.len ..][0..self.custom_len_enc.len], self.custom_len_enc.slice());

                const view = MultiSliceView.init(single_chunk[0..total_len], &[_]u8{}, &[_]u8{});
                singleChunkFn(&view, 0x07, out);
                return;
            }

            // Flush any pending chunks with SIMD
            self.flushPendingChunks();

            // Build view over remaining data (buffer + customization + encoding)
            const remaining_view = MultiSliceView.init(
                self.buffer[0..self.buffer_len],
                self.customization,
                self.custom_len_enc.slice(),
            );
            const remaining_len = remaining_view.totalLen();

            var final_leaves = self.num_leaves;
            var leaf_start: usize = 0;

            // Tree mode: initialize if not already done (lazy initialization)
            if (self.final_state == null and remaining_len > 0) {
                self.final_state = StateType.init(.{});

                // Absorb first chunk (up to chunk_size bytes from remaining data)
                const first_chunk_len = @min(chunk_size, remaining_len);
                if (remaining_view.tryGetSlice(0, first_chunk_len)) |first_chunk| {
                    // Data is contiguous, use it directly
                    self.final_state.?.update(first_chunk);
                } else {
                    // Data spans boundaries, copy to buffer
                    var first_chunk_buf: [chunk_size]u8 = undefined;
                    remaining_view.copyRange(0, first_chunk_len, first_chunk_buf[0..first_chunk_len]);
                    self.final_state.?.update(first_chunk_buf[0..first_chunk_len]);
                }

                // Absorb padding (8 bytes: 0x03 followed by 7 zeros)
                const padding = [_]u8{ 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00 };
                self.final_state.?.update(&padding);

                // Process remaining data as leaves
                leaf_start = first_chunk_len;
            }

            // Process all remaining data as leaves (starting from leaf_start)
            var offset = leaf_start;
            while (offset < remaining_len) {
                const leaf_end = @min(offset + chunk_size, remaining_len);
                const leaf_size = leaf_end - offset;

                var cv_buffer: [64]u8 = undefined;
                if (remaining_view.tryGetSlice(offset, leaf_end)) |leaf_data| {
                    // Data is contiguous, use it directly
                    const cv_slice = MultiSliceView.init(leaf_data, &[_]u8{}, &[_]u8{});
                    Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
                } else {
                    // Data spans boundaries, copy to buffer
                    var leaf_buf: [chunk_size]u8 = undefined;
                    remaining_view.copyRange(offset, leaf_end, leaf_buf[0..leaf_size]);
                    const cv_slice = MultiSliceView.init(leaf_buf[0..leaf_size], &[_]u8{}, &[_]u8{});
                    Variant.turboShakeToBuffer(&cv_slice, 0x0B, cv_buffer[0..cv_size]);
                }
                self.final_state.?.update(cv_buffer[0..cv_size]);
                final_leaves += 1;
                offset = leaf_end;
            }

            // Absorb right_encode(num_leaves) and terminator
            const n_enc = rightEncode(final_leaves);
            self.final_state.?.update(n_enc.slice());
            const terminator = [_]u8{ 0xFF, 0xFF };
            self.final_state.?.update(&terminator);

            // Squeeze output
            self.final_state.?.final(out);
        }

        /// Hash a message using sequential processing with SIMD acceleration.
        ///
        /// Parameters:
        ///   - message: Input data to hash (any length)
        ///   - out: Output buffer (any length, arbitrary output sizes supported, `digest_length` recommended for standard use)
        ///   - options: Optional settings to include a secret key or a context separation string
        pub fn hash(message: []const u8, out: []u8, options: Options) !void {
            const custom = options.customization orelse &[_]u8{};

            // Right-encode customization length
            const custom_len_enc = rightEncode(custom.len);

            // Create zero-copy multi-slice view (no concatenation)
            const view = MultiSliceView.init(message, custom, custom_len_enc.slice());
            const total_len = view.totalLen();

            // Single chunk case - zero-copy absorption!
            if (total_len <= chunk_size) {
                singleChunkFn(&view, 0x07, out);
                return;
            }

            // Tree mode - single-threaded SIMD processing
            ktSingleThreaded(Variant, &view, total_len, out);
        }

        /// Hash with automatic parallelization for large inputs (>2MB).
        /// Automatically uses sequential processing for smaller inputs to avoid thread overhead.
        /// Allocator required for temporary buffers. IO object required for thread management.
        pub fn hashParallel(message: []const u8, out: []u8, options: Options, allocator: Allocator, io: Io) !void {
            const custom = options.customization orelse &[_]u8{};

            const custom_len_enc = rightEncode(custom.len);
            const view = MultiSliceView.init(message, custom, custom_len_enc.slice());
            const total_len = view.totalLen();

            // Single chunk case
            if (total_len <= chunk_size) {
                singleChunkFn(&view, 0x07, out);
                return;
            }

            // Use single-threaded processing if below threshold
            if (total_len < large_file_threshold) {
                ktSingleThreaded(Variant, &view, total_len, out);
                return;
            }

            // Tree mode - multi-threaded processing
            try ktMultiThreaded(Variant, allocator, io, &view, total_len, out);
        }
    };
}