diff --git a/hash_testing/new_testing/hash_funcs/xxh64/bin_hash b/hash_testing/new_testing/hash_funcs/xxh64/bin_hash new file mode 100755 index 0000000..8262fc3 Binary files /dev/null and b/hash_testing/new_testing/hash_funcs/xxh64/bin_hash differ diff --git a/hash_testing/new_testing/hash_funcs/xxh64/bin_hash.c b/hash_testing/new_testing/hash_funcs/xxh64/bin_hash.c new file mode 100644 index 0000000..2c2ec4c --- /dev/null +++ b/hash_testing/new_testing/hash_funcs/xxh64/bin_hash.c @@ -0,0 +1,224 @@ +/* + * Standalone XXH64 command-line wrapper for avalanche testing. + * + * XXH64 algorithm derived from xxHash by Yann Collet: + * https://github.com/Cyan4973/xxHash + * + * Copyright (C) 2012-2023 Yann Collet + * + * BSD 2-Clause License + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * + * 1. Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * 2. Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES ARE DISCLAIMED. IN NO EVENT SHALL THE + * COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DAMAGES ARISING IN ANY WAY + * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH + * DAMAGE. + */ + +#include +#include +#include +#include +#include + +#define XXH64_SEED UINT64_C(0) +#define XXH_PRIME64_1 UINT64_C(11400714785074694791) +#define XXH_PRIME64_2 UINT64_C(14029467366897019727) +#define XXH_PRIME64_3 UINT64_C(1609587929392839161) +#define XXH_PRIME64_4 UINT64_C(9650029242287828579) +#define XXH_PRIME64_5 UINT64_C(2870177450012600261) + +static uint64_t rotate_left64(uint64_t value, unsigned int count) +{ + return (value << count) | (value >> (64U - count)); +} + +static uint32_t read_little_endian32(const unsigned char *data) +{ + return (uint32_t)data[0] | ((uint32_t)data[1] << 8U) | + ((uint32_t)data[2] << 16U) | ((uint32_t)data[3] << 24U); +} + +static uint64_t read_little_endian64(const unsigned char *data) +{ + return (uint64_t)read_little_endian32(data) | + ((uint64_t)read_little_endian32(data + 4) << 32U); +} + +static uint64_t xxh64_round(uint64_t accumulator, uint64_t input) +{ + accumulator += input * XXH_PRIME64_2; + accumulator = rotate_left64(accumulator, 31U); + accumulator *= XXH_PRIME64_1; + return accumulator; +} + +static uint64_t xxh64_merge_round(uint64_t accumulator, uint64_t value) +{ + value = xxh64_round(UINT64_C(0), value); + accumulator ^= value; + accumulator = accumulator * XXH_PRIME64_1 + XXH_PRIME64_4; + return accumulator; +} + +static uint64_t xxh64(const unsigned char *data, size_t length, uint64_t seed) +{ + const unsigned char *position = data; + const unsigned char *const end = data + length; + uint64_t hash; + + if (length >= 32U) { + const unsigned char *const block_end = end - 32U; + uint64_t accumulator1 = seed + XXH_PRIME64_1 + XXH_PRIME64_2; + uint64_t accumulator2 = seed + XXH_PRIME64_2; + uint64_t accumulator3 = seed; + uint64_t accumulator4 = seed - XXH_PRIME64_1; + + do { + accumulator1 = xxh64_round(accumulator1, read_little_endian64(position)); + position += 8; + accumulator2 = xxh64_round(accumulator2, read_little_endian64(position)); + position += 8; + accumulator3 = xxh64_round(accumulator3, read_little_endian64(position)); + position += 8; + accumulator4 = xxh64_round(accumulator4, read_little_endian64(position)); + position += 8; + } while (position <= block_end); + + hash = rotate_left64(accumulator1, 1U) + + rotate_left64(accumulator2, 7U) + + rotate_left64(accumulator3, 12U) + + rotate_left64(accumulator4, 18U); + hash = xxh64_merge_round(hash, accumulator1); + hash = xxh64_merge_round(hash, accumulator2); + hash = xxh64_merge_round(hash, accumulator3); + hash = xxh64_merge_round(hash, accumulator4); + } else { + hash = seed + XXH_PRIME64_5; + } + + hash += (uint64_t)length; + + while ((size_t)(end - position) >= 8U) { + uint64_t value = xxh64_round(UINT64_C(0), read_little_endian64(position)); + hash ^= value; + hash = rotate_left64(hash, 27U) * XXH_PRIME64_1 + XXH_PRIME64_4; + position += 8; + } + + if ((size_t)(end - position) >= 4U) { + hash ^= (uint64_t)read_little_endian32(position) * XXH_PRIME64_1; + hash = rotate_left64(hash, 23U) * XXH_PRIME64_2 + XXH_PRIME64_3; + position += 4; + } + + while (position < end) { + hash ^= (uint64_t)(*position) * XXH_PRIME64_5; + hash = rotate_left64(hash, 11U) * XXH_PRIME64_1; + ++position; + } + + hash ^= hash >> 33U; + hash *= XXH_PRIME64_2; + hash ^= hash >> 29U; + hash *= XXH_PRIME64_3; + hash ^= hash >> 32U; + return hash; +} + +static int is_ascii_trailing_space(unsigned char character) +{ + return character == ' ' || character == '\t' || character == '\n' || + character == '\r' || character == '\v' || character == '\f'; +} + +static int read_stdin(unsigned char **data, size_t *length) +{ + size_t capacity = 256; + unsigned char *buffer = malloc(capacity); + + if (buffer == NULL) { + return -1; + } + + *length = 0; + for (;;) { + size_t available = capacity - *length; + size_t bytes_read = fread(buffer + *length, 1, available, stdin); + *length += bytes_read; + + if (bytes_read < available) { + if (ferror(stdin)) { + free(buffer); + return -1; + } + break; + } + + if (capacity > SIZE_MAX / 2U) { + free(buffer); + return -1; + } + capacity *= 2U; + + { + unsigned char *larger_buffer = realloc(buffer, capacity); + if (larger_buffer == NULL) { + free(buffer); + return -1; + } + buffer = larger_buffer; + } + } + + *data = buffer; + return 0; +} + +int main(int argc, char **argv) +{ + const unsigned char *word; + unsigned char *stdin_buffer = NULL; + size_t length; + + if (argc > 2) { + fprintf(stderr, "usage: %s [ASCII_WORD]\n", argv[0]); + return EXIT_FAILURE; + } + + if (argc == 2) { + word = (const unsigned char *)argv[1]; + length = strlen(argv[1]); + } else { + if (read_stdin(&stdin_buffer, &length) != 0) { + fprintf(stderr, "failed to read input\n"); + return EXIT_FAILURE; + } + word = stdin_buffer; + } + + while (length > 0U && is_ascii_trailing_space(word[length - 1U])) { + --length; + } + + for (size_t index = 0; index < length; ++index) { + if (word[index] > 0x7fU) { + fprintf(stderr, "input must contain ASCII characters only\n"); + free(stdin_buffer); + return EXIT_FAILURE; + } + } + + printf("%016" PRIx64 "\n", xxh64(word, length, XXH64_SEED)); + free(stdin_buffer); + return EXIT_SUCCESS; +} diff --git a/hash_testing/new_testing/probability_map.py b/hash_testing/new_testing/probability_map.py index d9bfd9c..3feac96 100755 --- a/hash_testing/new_testing/probability_map.py +++ b/hash_testing/new_testing/probability_map.py @@ -33,6 +33,7 @@ DEFAULT_OUTPUT_DIR = ROOT / "probability_maps" # Add directory names from hash_funcs here to include more implementations. HASHES = [ "jenkinsOAAT", + "xxh64", ] HEX_HASH = re.compile(r"(?:0[xX])?([0-9a-fA-F]+)") @@ -84,9 +85,7 @@ def prepare_hash(hash_name: str, hash_funcs_dir: Path = HASH_FUNCS_DIR) -> Path: result = subprocess.run(command, text=True, capture_output=True) if result.returncode != 0: details = result.stderr.strip() or result.stdout.strip() - raise HashToolError( - f"не удалось скомпилировать {source}: {details}" - ) + raise HashToolError(f"не удалось скомпилировать {source}: {details}") return binary python_source = hash_directory / "bin_hash.py" @@ -98,10 +97,7 @@ def prepare_hash(hash_name: str, hash_funcs_dir: Path = HASH_FUNCS_DIR) -> Path: encoding="utf-8", ) python_source.chmod( - python_source.stat().st_mode - | stat.S_IXUSR - | stat.S_IXGRP - | stat.S_IXOTH + python_source.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH ) return python_source @@ -131,9 +127,7 @@ def run_hash(executable: Path, word: str) -> tuple[int, int]: output = result.stdout.strip() match = HEX_HASH.fullmatch(output) if match is None: - raise HashToolError( - f"{executable} вернул не шестнадцатеричный хэш: {output!r}" - ) + raise HashToolError(f"{executable} вернул не шестнадцатеричный хэш: {output!r}") digits = match.group(1) return int(digits, 16), len(digits) * 4 @@ -169,9 +163,7 @@ def write_csv_table(stream: TextIO, rows: Mapping[str, ProbabilityRow]) -> None: bits = widths.pop() writer = csv.writer(stream, lineterminator="\n") - writer.writerow( - ["operation", "pairs", *(f"bit_{bit}" for bit in range(bits))] - ) + writer.writerow(["operation", "pairs", *(f"bit_{bit}" for bit in range(bits))]) for operation, (pair_count, probabilities) in rows.items(): writer.writerow( [ @@ -306,10 +298,7 @@ def build_parser() -> argparse.ArgumentParser: "--output", type=Path, default=DEFAULT_OUTPUT_DIR, - help=( - "каталог для CSV-таблиц " - f"(по умолчанию: {DEFAULT_OUTPUT_DIR})" - ), + help=("каталог для CSV-таблиц " f"(по умолчанию: {DEFAULT_OUTPUT_DIR})"), ) parser.add_argument( "--hash", diff --git a/hash_testing/new_testing/probability_maps/jenkinsOAAT.csv b/hash_testing/new_testing/probability_maps/jenkinsOAAT.csv deleted file mode 100644 index 8d768d0..0000000 --- a/hash_testing/new_testing/probability_maps/jenkinsOAAT.csv +++ /dev/null @@ -1,5 +0,0 @@ -operation,pairs,bit_0,bit_1,bit_2,bit_3,bit_4,bit_5,bit_6,bit_7,bit_8,bit_9,bit_10,bit_11,bit_12,bit_13,bit_14,bit_15,bit_16,bit_17,bit_18,bit_19,bit_20,bit_21,bit_22,bit_23,bit_24,bit_25,bit_26,bit_27,bit_28,bit_29,bit_30,bit_31 -replace,1922,0.486472,0.498959,0.483351,0.501561,0.503122,0.495317,0.523413,0.466701,0.520812,0.496878,0.513007,0.503122,0.490114,0.510926,0.482310,0.494277,0.511446,0.505723,0.498959,0.501561,0.512487,0.509365,0.489594,0.514048,0.495317,0.519771,0.505203,0.515609,0.517690,0.489074,0.498959,0.508845 -add,2111,0.506869,0.505448,0.502132,0.502132,0.496921,0.502132,0.505921,0.505921,0.494552,0.507342,0.493605,0.502605,0.492658,0.501658,0.495026,0.508764,0.496921,0.486973,0.507816,0.518238,0.513501,0.506395,0.513974,0.488394,0.504500,0.482236,0.510185,0.520133,0.500711,0.504027,0.520133,0.497868 -first,186,0.451613,0.435484,0.440860,0.462366,0.521505,0.494624,0.510753,0.435484,0.462366,0.483871,0.505376,0.521505,0.451613,0.532258,0.462366,0.505376,0.473118,0.505376,0.446237,0.473118,0.489247,0.537634,0.440860,0.521505,0.537634,0.500000,0.548387,0.575269,0.575269,0.494624,0.462366,0.505376 -last,186,0.516129,0.500000,0.483871,0.510753,0.510753,0.483871,0.478495,0.521505,0.569892,0.500000,0.478495,0.494624,0.462366,0.521505,0.510753,0.500000,0.569892,0.494624,0.505376,0.510753,0.521505,0.478495,0.478495,0.543011,0.500000,0.548387,0.532258,0.451613,0.500000,0.500000,0.516129,0.521505 diff --git a/hash_testing/new_testing/probability_maps/jenkinsOAAT_bits.png b/hash_testing/new_testing/probability_maps/jenkinsOAAT_bits.png deleted file mode 100644 index e940149..0000000 Binary files a/hash_testing/new_testing/probability_maps/jenkinsOAAT_bits.png and /dev/null differ diff --git a/hash_testing/new_testing/probability_maps/jenkinsOAAT_deviation.png b/hash_testing/new_testing/probability_maps/jenkinsOAAT_deviation.png deleted file mode 100644 index 1da87a7..0000000 Binary files a/hash_testing/new_testing/probability_maps/jenkinsOAAT_deviation.png and /dev/null differ diff --git a/hash_testing/new_testing/test_plot_probability_map.py b/hash_testing/new_testing/tests/test_plot_probability_map.py similarity index 100% rename from hash_testing/new_testing/test_plot_probability_map.py rename to hash_testing/new_testing/tests/test_plot_probability_map.py diff --git a/hash_testing/new_testing/test_probability_map.py b/hash_testing/new_testing/tests/test_probability_map.py similarity index 100% rename from hash_testing/new_testing/test_probability_map.py rename to hash_testing/new_testing/tests/test_probability_map.py diff --git a/hash_testing/new_testing/tests/test_xxh64_bin_hash.py b/hash_testing/new_testing/tests/test_xxh64_bin_hash.py new file mode 100644 index 0000000..8be6a07 --- /dev/null +++ b/hash_testing/new_testing/tests/test_xxh64_bin_hash.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python3 +"""Behavior tests for the standalone XXH64 hash CLI.""" + +from __future__ import annotations + +import re +import subprocess +import tempfile +import unittest +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +SOURCE = ROOT / "hash_funcs" / "xxh64" / "bin_hash.c" + + +class Xxh64BinHashTests(unittest.TestCase): + @classmethod + def setUpClass(cls) -> None: + cls.temporary_directory = tempfile.TemporaryDirectory() + cls.binary = Path(cls.temporary_directory.name) / "bin_hash" + subprocess.run( + [ + "cc", + "-std=c11", + "-O2", + "-Wall", + "-Wextra", + "-Wpedantic", + "-Werror", + str(SOURCE), + "-o", + str(cls.binary), + ], + check=True, + text=True, + capture_output=True, + ) + + @classmethod + def tearDownClass(cls) -> None: + cls.temporary_directory.cleanup() + + def run_hash(self, word: str | None = None, stdin: bytes | None = None) -> subprocess.CompletedProcess[bytes]: + command = [str(self.binary)] + if word is not None: + command.append(word) + return subprocess.run(command, input=stdin, capture_output=True, check=False) + + def test_matches_official_xxh64_seed_zero_vectors(self) -> None: + vectors = { + "": b"ef46db3751d8e999\n", + "hello": b"26c7827d889f6da3\n", + "HashWord": b"3e26fc2935163fbe\n", + } + + for word, expected in vectors.items(): + with self.subTest(word=word): + result = self.run_hash(word) + self.assertEqual(result.returncode, 0) + self.assertEqual(result.stdout, expected) + self.assertRegex(result.stdout.decode(), r"^[0-9a-f]{16}\n$") + + def test_argv_and_stdin_are_equivalent_and_strip_trailing_ascii_space(self) -> None: + argv = self.run_hash("hello") + stdin = self.run_hash(stdin=b"hello \t\r\n") + + self.assertEqual(stdin.returncode, 0) + self.assertEqual(stdin.stdout, argv.stdout) + + def test_handles_long_ascii_input(self) -> None: + payload = b"a" * 100_000 + result = self.run_hash(stdin=payload) + reference = subprocess.run( + ["xxhsum", "-H64"], input=payload, capture_output=True, check=True + ).stdout.split()[0] + + self.assertEqual(result.returncode, 0) + self.assertEqual(result.stdout.strip(), reference) + + def test_rejects_non_ascii_input(self) -> None: + result = self.run_hash(stdin="ёж".encode()) + + self.assertNotEqual(result.returncode, 0) + self.assertIn(b"ASCII", result.stderr) + self.assertEqual(result.stdout, b"") + + +if __name__ == "__main__": + unittest.main()