From f911df50f7443279d1504e8bc079c5a94ae49736 Mon Sep 17 00:00:00 2001 From: jadidbourbaki Date: Fri, 25 Sep 2026 21:02:37 -0400 Subject: [PATCH 1/3] vendor : add unordered_dense for the ngram caches --- scripts/sync_vendor.py | 5 + vendor/CMakeLists.txt | 1 + vendor/ankerl/CMakeLists.txt | 6 + vendor/ankerl/LICENSE | 21 + vendor/ankerl/stl.h | 85 + vendor/ankerl/unordered_dense.h | 4407 +++++++++++++++++++++++++++++++ 6 files changed, 4525 insertions(+) create mode 100644 vendor/ankerl/CMakeLists.txt create mode 100644 vendor/ankerl/LICENSE create mode 100644 vendor/ankerl/stl.h create mode 100644 vendor/ankerl/unordered_dense.h diff --git a/scripts/sync_vendor.py b/scripts/sync_vendor.py index b320884bbeba..23e39ee6595c 100755 --- a/scripts/sync_vendor.py +++ b/scripts/sync_vendor.py @@ -6,6 +6,7 @@ import subprocess HTTPLIB_VERSION = "refs/tags/v0.57.1" +UNORDERED_DENSE_VERSION = "refs/tags/v5.0.1" # used by examples/gguf-hash, these repos have no release tag, so we pin a commit XXHASH_COMMIT = "9f465f1ea932d6ad9a26cd77496311ffa544cd68" @@ -27,6 +28,10 @@ f"https://raw.githubusercontent.com/yhirose/cpp-httplib/{HTTPLIB_VERSION}/split.py": "split.py", f"https://raw.githubusercontent.com/yhirose/cpp-httplib/{HTTPLIB_VERSION}/LICENSE": "vendor/cpp-httplib/LICENSE", + f"https://raw.githubusercontent.com/martinus/unordered_dense/{UNORDERED_DENSE_VERSION}/include/ankerl/unordered_dense.h": "vendor/ankerl/unordered_dense.h", + f"https://raw.githubusercontent.com/martinus/unordered_dense/{UNORDERED_DENSE_VERSION}/include/ankerl/stl.h": "vendor/ankerl/stl.h", + f"https://raw.githubusercontent.com/martinus/unordered_dense/{UNORDERED_DENSE_VERSION}/LICENSE": "vendor/ankerl/LICENSE", + "https://raw.githubusercontent.com/sheredom/subprocess.h/0dccaa9aa176dd6d7ef8afeca3c18d6e80a32795/subprocess.h": "vendor/sheredom/subprocess.h", f"https://raw.githubusercontent.com/Cyan4973/xxHash/{XXHASH_COMMIT}/xxhash.c": "vendor/hash/xxhash/xxhash.c", diff --git a/vendor/CMakeLists.txt b/vendor/CMakeLists.txt index 4479dafcad0c..91ef898509d4 100644 --- a/vendor/CMakeLists.txt +++ b/vendor/CMakeLists.txt @@ -7,5 +7,6 @@ add_subdirectory(stb) # only used by common if (LLAMA_BUILD_COMMON) + add_subdirectory(ankerl) add_subdirectory(cpp-httplib) endif() diff --git a/vendor/ankerl/CMakeLists.txt b/vendor/ankerl/CMakeLists.txt new file mode 100644 index 000000000000..57bac6a70f86 --- /dev/null +++ b/vendor/ankerl/CMakeLists.txt @@ -0,0 +1,6 @@ +# header-only: interface target exposing the vendor/ root so consumers +# can include via +add_library(unordered_dense INTERFACE) +add_library(vendor::unordered_dense ALIAS unordered_dense) + +target_include_directories(unordered_dense INTERFACE ..) diff --git a/vendor/ankerl/LICENSE b/vendor/ankerl/LICENSE new file mode 100644 index 000000000000..c4d1a0e48eba --- /dev/null +++ b/vendor/ankerl/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2022 Martin Leitner-Ankerl + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/vendor/ankerl/stl.h b/vendor/ankerl/stl.h new file mode 100644 index 000000000000..96d14537c2d0 --- /dev/null +++ b/vendor/ankerl/stl.h @@ -0,0 +1,85 @@ +///////////////////////// ankerl::unordered_dense::{map, set} ///////////////////////// + +// A fast & densely stored hashmap and hashset based on robin-hood backward shift deletion. +// Version 5.0.1 +// https://github.com/martinus/unordered_dense +// +// Licensed under the MIT License . +// SPDX-License-Identifier: MIT +// Copyright (c) 2022 Martin Leitner-Ankerl +// +// Permission is hereby granted, free of charge, to any person obtaining a copy +// of this software and associated documentation files (the "Software"), to deal +// in the Software without restriction, including without limitation the rights +// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the Software is +// furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in all +// copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +// SOFTWARE. + +#ifndef ANKERL_STL_H +#define ANKERL_STL_H + +#include // for array +#include // for uint64_t, uint32_t, std::uint8_t, UINT64_C +#include // for size_t, memcpy, memset +#include // for equal_to, hash +#include // for initializer_list +#include // for pair, distance +#include // for numeric_limits +#include // for allocator, allocator_traits, shared_ptr +#include // for optional +#include // for out_of_range +#include // for basic_string +#include // for basic_string_view, hash +#include // for forward_as_tuple +#include // for enable_if_t, declval, conditional_t, ena... +#include // for forward, exchange, pair, as_const, piece... +#include // for vector + +// includes , which fails to compile if +// targeting GCC >= 13 with the (rewritten) win32 thread model, and +// targeting Windows earlier than Vista (0x600). GCC predefines +// _REENTRANT when using the 'posix' model, and doesn't when using the +// 'win32' model. +#if defined __MINGW64__ && defined __GNUC__ && __GNUC__ >= 13 && !defined _REENTRANT +// _WIN32_WINNT is guaranteed to be defined here because of the +// inclusion above. +# ifndef _WIN32_WINNT +# error "_WIN32_WINNT not defined" +# endif +# if _WIN32_WINNT < 0x600 +# define ANKERL_MEMORY_RESOURCE_IS_BAD() 1 // NOLINT(cppcoreguidelines-macro-usage) +# endif +#endif +#ifndef ANKERL_MEMORY_RESOURCE_IS_BAD +# define ANKERL_MEMORY_RESOURCE_IS_BAD() 0 // NOLINT(cppcoreguidelines-macro-usage) +#endif + +#if defined(__has_include) && !defined(ANKERL_UNORDERED_DENSE_DISABLE_PMR) +# if __has_include() && !ANKERL_MEMORY_RESOURCE_IS_BAD() +# define ANKERL_UNORDERED_DENSE_PMR std::pmr // NOLINT(cppcoreguidelines-macro-usage) +# include // for polymorphic_allocator +# elif __has_include() +# define ANKERL_UNORDERED_DENSE_PMR std::experimental::pmr // NOLINT(cppcoreguidelines-macro-usage) +# include // for polymorphic_allocator +# endif +#endif + +#if defined(_MSC_VER) && defined(_M_X64) +# include +# if !defined(_M_ARM64EC) +# pragma intrinsic(_umul128) +# endif +#endif + +#endif diff --git a/vendor/ankerl/unordered_dense.h b/vendor/ankerl/unordered_dense.h new file mode 100644 index 000000000000..0ac177b9a965 --- /dev/null +++ b/vendor/ankerl/unordered_dense.h @@ -0,0 +1,4407 @@ +///////////////////////// ankerl::unordered_dense::{map, set} ///////////////////////// + +// A fast & densely stored hashmap and hashset. +// Version 5.0.1 +// https://github.com/martinus/unordered_dense +// +// Licensed under the MIT License . +// SPDX-License-Identifier: MIT +// Copyright (c) 2022 Martin Leitner-Ankerl +// +// Permission is hereby granted, free of charge, to any person obtaining a copy +// of this software and associated documentation files (the "Software"), to deal +// in the Software without restriction, including without limitation the rights +// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the Software is +// furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in all +// copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +// SOFTWARE. + +#ifndef ANKERL_UNORDERED_DENSE_H +#define ANKERL_UNORDERED_DENSE_H + +// see https://semver.org/spec/v2.0.0.html +#define ANKERL_UNORDERED_DENSE_VERSION_MAJOR 5 // NOLINT(cppcoreguidelines-macro-usage) incompatible API changes +#define ANKERL_UNORDERED_DENSE_VERSION_MINOR 0 // NOLINT(cppcoreguidelines-macro-usage) backwards compatible functionality +#define ANKERL_UNORDERED_DENSE_VERSION_PATCH 1 // NOLINT(cppcoreguidelines-macro-usage) backwards compatible bug fixes + +// API versioning with inline namespace, see https://www.foonathan.net/2018/11/inline-namespaces/ + +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +#define ANKERL_UNORDERED_DENSE_VERSION_CONCAT1(major, minor, patch) v##major##_##minor##_##patch +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +#define ANKERL_UNORDERED_DENSE_VERSION_CONCAT(major, minor, patch) ANKERL_UNORDERED_DENSE_VERSION_CONCAT1(major, minor, patch) +#define ANKERL_UNORDERED_DENSE_NAMESPACE \ + ANKERL_UNORDERED_DENSE_VERSION_CONCAT( \ + ANKERL_UNORDERED_DENSE_VERSION_MAJOR, ANKERL_UNORDERED_DENSE_VERSION_MINOR, ANKERL_UNORDERED_DENSE_VERSION_PATCH) + +#if defined(_MSVC_LANG) +# define ANKERL_UNORDERED_DENSE_CPP_VERSION _MSVC_LANG +#else +# define ANKERL_UNORDERED_DENSE_CPP_VERSION __cplusplus +#endif + +#if defined(__GNUC__) +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_PACK(decl) decl __attribute__((__packed__)) +#elif defined(_MSC_VER) +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_PACK(decl) __pragma(pack(push, 1)) decl __pragma(pack(pop)) +#endif + +// exceptions +#if defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND) +# define ANKERL_UNORDERED_DENSE_HAS_EXCEPTIONS() 1 // NOLINT(cppcoreguidelines-macro-usage) +#else +# define ANKERL_UNORDERED_DENSE_HAS_EXCEPTIONS() 0 // NOLINT(cppcoreguidelines-macro-usage) +#endif +#ifdef _MSC_VER +# define ANKERL_UNORDERED_DENSE_NOINLINE __declspec(noinline) +# define ANKERL_UNORDERED_DENSE_FORCEINLINE __forceinline +#else +# define ANKERL_UNORDERED_DENSE_NOINLINE __attribute__((noinline)) +# define ANKERL_UNORDERED_DENSE_FORCEINLINE inline __attribute__((always_inline)) +#endif + +// Data prefetch hint, a no-op where there is nothing to spell it with. MSVC has no +// __builtin_prefetch and used to get the no-op, which quietly cost it the one the probe issues for +// a group's value indices -- measured at 3 cycles off every hit, so a whole compiler was paying for +// a missing spelling. Both MSVC intrinsics come from , which is included further down; +// that is in time, because a macro needs its declarations where it is expanded and every expansion +// is inside the table. Taken from boost, which covers the same three cases. +#if defined(__GNUC__) || defined(__clang__) +# define ANKERL_UNORDERED_DENSE_PREFETCH(addr) __builtin_prefetch(addr) // NOLINT(cppcoreguidelines-macro-usage) +#elif defined(_MSC_VER) && (defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_PREFETCH(addr) _mm_prefetch(reinterpret_cast(addr), _MM_HINT_T0) +#elif defined(_MSC_VER) && defined(_M_ARM64) +# define ANKERL_UNORDERED_DENSE_PREFETCH(addr) __prefetch(addr) // NOLINT(cppcoreguidelines-macro-usage) +#else +# define ANKERL_UNORDERED_DENSE_PREFETCH(addr) static_cast(addr) // NOLINT(cppcoreguidelines-macro-usage) +#endif + +// SSE2 is part of the x86-64 baseline, so comparing a group's sixteen fingerprints in one +// instruction is available on every x86-64 build without asking for it. Elsewhere, and in a build +// that defines this to 0, they are compared eight per machine word with ordinary arithmetic. +// +// This picks the code, never the layout: a group is sixteen slots either way, so two translation +// units that disagree about this macro -- which they may, it is documented as a per-target switch +// -- still agree about every byte of the index they share. +#if !defined(ANKERL_UNORDERED_DENSE_HAS_SSE2) +# if defined(__SSE2__) || (defined(_MSC_VER) && (defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))) +# define ANKERL_UNORDERED_DENSE_HAS_SSE2 1 // NOLINT(cppcoreguidelines-macro-usage) +# else +# define ANKERL_UNORDERED_DENSE_HAS_SSE2 0 // NOLINT(cppcoreguidelines-macro-usage) +# endif +#endif + +// The same sixteen-at-once compare on AArch64, which is the other baseline worth having: NEON is +// mandatory there, so this needs no runtime dispatch either. Restricted to little endian because +// the mask below reads the comparison result as one 64 bit word, and to AArch64 because 32 bit ARM +// lacks the horizontal ops -- both fall back to SWAR, which is correct everywhere. +#if !defined(ANKERL_UNORDERED_DENSE_HAS_NEON) +# if defined(__ARM_NEON) && defined(__aarch64__) && \ + (!defined(__BYTE_ORDER__) || !defined(__ORDER_BIG_ENDIAN__) || (__BYTE_ORDER__ != __ORDER_BIG_ENDIAN__)) +# define ANKERL_UNORDERED_DENSE_HAS_NEON 1 // NOLINT(cppcoreguidelines-macro-usage) +# else +# define ANKERL_UNORDERED_DENSE_HAS_NEON 0 // NOLINT(cppcoreguidelines-macro-usage) +# endif +#endif +#if ANKERL_UNORDERED_DENSE_HAS_SSE2 && ANKERL_UNORDERED_DENSE_HAS_NEON +# error "ANKERL_UNORDERED_DENSE_HAS_SSE2 and ANKERL_UNORDERED_DENSE_HAS_NEON cannot both be on" +#endif + +#if defined(__clang__) && defined(__has_attribute) +# if __has_attribute(__no_sanitize__) +# define ANKERL_UNORDERED_DENSE_DISABLE_UBSAN_UNSIGNED_INTEGER_CHECK \ + __attribute__((__no_sanitize__("unsigned-integer-overflow"))) +# endif +#endif + +#if !defined(ANKERL_UNORDERED_DENSE_DISABLE_UBSAN_UNSIGNED_INTEGER_CHECK) +# define ANKERL_UNORDERED_DENSE_DISABLE_UBSAN_UNSIGNED_INTEGER_CHECK +#endif + +#if ANKERL_UNORDERED_DENSE_CPP_VERSION < 201703L +# error ankerl::unordered_dense requires C++17 or higher +#else + +# if !defined(ANKERL_UNORDERED_DENSE_STD_MODULE) +// NOLINTNEXTLINE(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_STD_MODULE 0 +# endif + +# if !ANKERL_UNORDERED_DENSE_STD_MODULE +# include "stl.h" +# endif +# if ANKERL_UNORDERED_DENSE_HAS_SSE2 +# include // for _mm_loadu_si128, _mm_cmpeq_epi8, ... +# endif +# if ANKERL_UNORDERED_DENSE_HAS_NEON +# include // for vld1q_u8, vceqq_u8, vshrn_n_u16, ... +# endif +# if defined(_MSC_VER) +# include // for _BitScanForward +# endif + +# if __has_cpp_attribute(likely) && __has_cpp_attribute(unlikely) && ANKERL_UNORDERED_DENSE_CPP_VERSION >= 202002L +# define ANKERL_UNORDERED_DENSE_LIKELY_ATTR [[likely]] // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_UNLIKELY_ATTR [[unlikely]] // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_LIKELY(x) (x) // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_UNLIKELY(x) (x) // NOLINT(cppcoreguidelines-macro-usage) +# else +# define ANKERL_UNORDERED_DENSE_LIKELY_ATTR // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_UNLIKELY_ATTR // NOLINT(cppcoreguidelines-macro-usage) + +# if defined(__GNUC__) || defined(__INTEL_COMPILER) || defined(__clang__) +# define ANKERL_UNORDERED_DENSE_LIKELY(x) __builtin_expect(x, 1) // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_UNLIKELY(x) __builtin_expect(x, 0) // NOLINT(cppcoreguidelines-macro-usage) +# else +# define ANKERL_UNORDERED_DENSE_LIKELY(x) (x) // NOLINT(cppcoreguidelines-macro-usage) +# define ANKERL_UNORDERED_DENSE_UNLIKELY(x) (x) // NOLINT(cppcoreguidelines-macro-usage) +# endif + +# endif + +namespace ankerl::unordered_dense { +inline namespace ANKERL_UNORDERED_DENSE_NAMESPACE { + +namespace detail { + +# if ANKERL_UNORDERED_DENSE_HAS_EXCEPTIONS() + +// make sure this is not inlined as it is slow and dramatically enlarges code, thus making other +// inlinings more difficult. Throws are also generally the slow path. +[[noreturn]] inline ANKERL_UNORDERED_DENSE_NOINLINE void on_error_key_not_found() { + throw std::out_of_range("ankerl::unordered_dense::map::at(): key not found"); +} +[[noreturn]] inline ANKERL_UNORDERED_DENSE_NOINLINE void on_error_bucket_overflow() { + throw std::overflow_error("ankerl::unordered_dense: reached max bucket size, cannot increase size"); +} +[[noreturn]] inline ANKERL_UNORDERED_DENSE_NOINLINE void on_error_too_many_elements() { + throw std::out_of_range("ankerl::unordered_dense::map::replace(): too many elements"); +} +[[noreturn]] inline ANKERL_UNORDERED_DENSE_NOINLINE void on_error_key_changed() { + throw std::logic_error("ankerl::unordered_dense: an element's key changed after it was inserted; use replace_key()"); +} + +# else + +[[noreturn]] inline void on_error_key_not_found() { + abort(); +} +[[noreturn]] inline void on_error_bucket_overflow() { + abort(); +} +[[noreturn]] inline void on_error_too_many_elements() { + abort(); +} +[[noreturn]] inline void on_error_key_changed() { + abort(); +} + +# endif + +// Index of the lowest set bit, for the lane mask a group compare produces. x +// must not be zero. +[[nodiscard]] inline auto countr_zero(std::uint32_t x) -> unsigned { +# if defined(_MSC_VER) + unsigned long idx{}; + _BitScanForward(&idx, x); + return static_cast(idx); +# else + return static_cast(__builtin_ctz(x)); +# endif +} + +# if ANKERL_UNORDERED_DENSE_HAS_NEON +// NEON's match mask is one bit per lane four bits apart, so it needs the whole word. +[[nodiscard]] inline auto countr_zero(std::uint64_t x) -> unsigned { +# if defined(_MSC_VER) + unsigned long idx{}; + _BitScanForward64(&idx, x); + return static_cast(idx); +# else + return static_cast(__builtin_ctzll(x)); +# endif +} +# endif + +} // namespace detail + +// hash /////////////////////////////////////////////////////////////////////// + +// This is no longer wyhash and does not produce wyhash's values, so it is not named after it. It is +// descended from it: https://github.com/wangyi-fudan/wyhash gives the reads, the multiply-and-xor +// mix, the short path and the chained lanes for long keys. What changed is the middle lengths, +// restructured into independent blocks, which the comment on hash_bytes() explains. If it ever +// leaves this header as something callers can use on its own it will be called `ankerlhash`; until +// then the entry points are `detail::hash_bytes` and `detail::hash_int` and the name is not needed. +// +// No big-endian support, because different values on different machines do not matter here. +// The seed and the secret are hardcoded, so there is nothing to salt a table with. +namespace detail::hash_impl { + +inline void mum(std::uint64_t* a, std::uint64_t* b) { +# if defined(__SIZEOF_INT128__) + __uint128_t r = *a; + r *= *b; + *a = static_cast(r); + *b = static_cast(r >> 64U); +# elif defined(_MSC_VER) && defined(_M_X64) + *a = _umul128(*a, *b, b); +# else + std::uint64_t ha = *a >> 32U; + std::uint64_t hb = *b >> 32U; + std::uint64_t la = static_cast(*a); + std::uint64_t lb = static_cast(*b); + std::uint64_t hi{}; + std::uint64_t lo{}; + std::uint64_t rh = ha * hb; + std::uint64_t rm0 = ha * lb; + std::uint64_t rm1 = hb * la; + std::uint64_t rl = la * lb; + std::uint64_t t = rl + (rm0 << 32U); + auto c = static_cast(t < rl); + lo = t + (rm1 << 32U); + c += static_cast(lo < t); + hi = rh + (rm0 >> 32U) + (rm1 >> 32U) + c; + *a = lo; + *b = hi; +# endif +} + +// multiply and xor mix function, aka MUM +[[nodiscard]] inline auto mix(std::uint64_t a, std::uint64_t b) -> std::uint64_t { + mum(&a, &b); + return a ^ b; +} + +// read functions. WARNING: we don't care about endianness, so results are different on big endian! +[[nodiscard]] inline auto r8(const std::uint8_t* p) -> std::uint64_t { + std::uint64_t v{}; + std::memcpy(&v, p, 8U); + return v; +} + +[[nodiscard]] inline auto r4(const std::uint8_t* p) -> std::uint64_t { + std::uint32_t v{}; + std::memcpy(&v, p, 4); + return v; +} + +// reads 1, 2, or 3 bytes +[[nodiscard]] inline auto r3(const std::uint8_t* p, std::size_t k) -> std::uint64_t { + return (static_cast(p[0]) << 16U) | (static_cast(p[k >> 1U]) << 8U) | p[k - 1]; +} + +// The shape of this is wyhash's up to 16 bytes and for anything past 144, and in between it is +// not: every 16 byte block is mixed on its own, with its own pair of secrets, and the results are +// xor-folded into one finalizer. wyhash chains the blocks through `seed`, so a 48 byte key is +// three multiplies one after another and then the finalizer, and a map lookup waits for all of +// them before it can so much as form the group address. Here the block multiplies are independent, +// so the latency of any key up to 144 bytes is one multiply plus the finalizer, and the block +// loop's trip count -- a data-dependent branch that mispredicts whenever lengths vary -- is a +// short chain of compares that the predictor learns from the top. The last sixteen bytes are +// always a block of their own, wherever they fall, so every byte is read at least once and nothing +// is read past the end. +// +// Measured on the scored benchmark's own keys (8 to 135 bytes, skewed short), one function per +// binary, ns per hash: throughput 2.52 to 2.00 under clang and 2.18 to 2.05 under gcc, latency +// 8.64 to 7.69 and 8.47 to 7.81, with fewer branch misses on both. Two shapes measured and +// rejected on the way: making the block range branchless by always mixing three (17-48) or six +// (49-96) overlapping blocks, which costs more in redundant multiplies than it saves in +// mispredictions; and a one multiply short path, which fails an avalanche test outright at 8 bytes +// (output bits that never flip for some input bits), as does dropping the finalizer in the block +// range. Both multiplies stay. +// +// Independent blocks need distinct secrets: with a shared one, swapping two blocks gives the same +// hash. Sixteen pairs cover 144 bytes, and past that the chained lanes take over, where reuse is +// harmless because the chain carries the position. The secrets have wyhash's property, every +// byte with four bits set, odd, and were drawn once from a fixed seed. +[[maybe_unused]] [[nodiscard]] inline auto hash_bytes(void const* key, std::size_t len) -> std::uint64_t { + static constexpr auto secret = std::array{ + UINT64_C(0xa0761d6478bd642f), UINT64_C(0xe7037ed1a0b428db), UINT64_C(0x8ebc6af09c88c6e3), UINT64_C(0x589965cc75374cc3), + UINT64_C(0x2d358dccaa6c78a5), UINT64_C(0x8bb84b93962eacc9), UINT64_C(0x4b33a62ed433d4a3), UINT64_C(0xa693c93927d87217), + UINT64_C(0x2b63728e53473c2b), UINT64_C(0x696cb2a95635a3c5), UINT64_C(0xa9ccd81ed1b29359), UINT64_C(0x5c2d66ace48db84d), + UINT64_C(0x69a99c5c53b4ca2d), UINT64_C(0x9a9c5a1b27d10f69), UINT64_C(0x2b27f02dc3d4360f), UINT64_C(0x2b39665c8d2d5553), + UINT64_C(0x966cd8878bb4b187), UINT64_C(0xc6351e99932b1ee1), UINT64_C(0xd1c5d24d63c959c9), UINT64_C(0x56c54d9c955aca2b), + UINT64_C(0xd136d27872563559)}; + + auto const* p = static_cast(key); + std::uint64_t seed = secret[0]; + std::uint64_t a{}; + std::uint64_t b{}; + if (ANKERL_UNORDERED_DENSE_LIKELY(len <= 16)) + ANKERL_UNORDERED_DENSE_LIKELY_ATTR { + if (ANKERL_UNORDERED_DENSE_LIKELY(len >= 8)) + ANKERL_UNORDERED_DENSE_LIKELY_ATTR { + // two (potentially overlapping) 8 byte reads cover the whole input + a = r8(p); + b = r8(p + len - 8); + } + else if (len >= 4) { + a = r4(p); + b = r4(p + len - 4); + } else if (ANKERL_UNORDERED_DENSE_LIKELY(len > 0)) + ANKERL_UNORDERED_DENSE_LIKELY_ATTR { + // b stays zero: r3 packs all len bytes it is given into a, and there are at + // most three of them. + a = r3(p, len); + } + // ... and an empty input needs no branch of its own: it hashes whatever a and b were + // declared with, which is the zero it has to be. Assigning it again here is what a + // deletion sweep of this file kept pointing at. + + // Return, rather than falling through to the same expression at the end of the + // function. Falling through makes seed, a and b values of two paths at once, and then + // the compiler cannot fold the constant seed of this one into the mix: measured, the + // short path costs 36 instructions that way and 24 this way. + return mix(secret[1] ^ len, mix(a ^ secret[1], b ^ seed)); + } + + if (ANKERL_UNORDERED_DENSE_LIKELY(len <= 144)) + ANKERL_UNORDERED_DENSE_LIKELY_ATTR { + // The first block and the last sixteen bytes, then whole blocks from the front for as + // long as there are any: a key of 17 to 32 bytes is two multiplies, one of 129 to 144 + // is nine, all of them independent. + auto x = + mix(r8(p) ^ secret[1], r8(p + 8) ^ secret[2]) ^ mix(r8(p + len - 16) ^ secret[3], r8(p + len - 8) ^ secret[4]); + if (len > 32) { + x ^= mix(r8(p + 16) ^ secret[5], r8(p + 24) ^ secret[6]); + if (len > 48) { + x ^= mix(r8(p + 32) ^ secret[7], r8(p + 40) ^ secret[8]); + if (len > 64) { + x ^= mix(r8(p + 48) ^ secret[9], r8(p + 56) ^ secret[10]); + if (len > 80) { + x ^= mix(r8(p + 64) ^ secret[11], r8(p + 72) ^ secret[12]); + if (len > 96) { + x ^= mix(r8(p + 80) ^ secret[13], r8(p + 88) ^ secret[14]); + if (len > 112) { + x ^= mix(r8(p + 96) ^ secret[15], r8(p + 104) ^ secret[16]); + if (len > 128) { + x ^= mix(r8(p + 112) ^ secret[17], r8(p + 120) ^ secret[18]); + } + } + } + } + } + } + } + return mix(secret[1] ^ len, x); + } + + // Anything longer, in chained lanes of 16 bytes, ending on the same expression as above. + std::size_t i = len; + std::uint64_t see1 = seed; + std::uint64_t see2 = seed; + // Six lanes cost three more accumulators to set up and fold back in, so the block has to + // run more than once to pay for them. Entering it at 96 meant exactly one iteration for + // everything from 97 to 192 bytes, which never can: measured, 23.4 cycles for a 100 byte key + // against 21.8 when it takes the 48 byte loop instead, and 29.3 against 28.2 at 150. Above + // 192 the block runs at least twice and wins again -- 143.5 cycles against 147.1 at 1000 + // bytes -- so it keeps those. + if (i > 192) { + // 6 independent lanes: twice the instruction level parallelism of the 48 byte loop below + std::uint64_t see3 = seed; + std::uint64_t see4 = seed; + std::uint64_t see5 = seed; + do { + seed = mix(r8(p) ^ secret[1], r8(p + 8) ^ seed); + see1 = mix(r8(p + 16) ^ secret[2], r8(p + 24) ^ see1); + see2 = mix(r8(p + 32) ^ secret[3], r8(p + 40) ^ see2); + see3 = mix(r8(p + 48) ^ secret[4], r8(p + 56) ^ see3); + see4 = mix(r8(p + 64) ^ secret[5], r8(p + 72) ^ see4); + see5 = mix(r8(p + 80) ^ secret[6], r8(p + 88) ^ see5); + p += 96; + i -= 96; + } while (ANKERL_UNORDERED_DENSE_LIKELY(i > 96)); + seed ^= see3 ^ see4 ^ see5; + } + while (i > 48) { + seed = mix(r8(p) ^ secret[1], r8(p + 8) ^ seed); + see1 = mix(r8(p + 16) ^ secret[2], r8(p + 24) ^ see1); + see2 = mix(r8(p + 32) ^ secret[3], r8(p + 40) ^ see2); + p += 48; + i -= 48; + } + seed ^= see1 ^ see2; + while (i > 16) { + seed = mix(r8(p) ^ secret[1], r8(p + 8) ^ seed); + i -= 16; + p += 16; + } + + // the tail lane only depends on the input, not on seed, so it can execute in parallel + // with the lane loops above, and a single dependent mix finishes the hash + auto tail = mix(r8(p + i - 16) ^ secret[2], r8(p + i - 8) ^ secret[3]); + return mix(secret[1] ^ len, seed ^ tail); +} + +[[nodiscard]] inline auto hash_int(std::uint64_t x) -> std::uint64_t { + return mix(x, UINT64_C(0x9E3779B97F4A7C15)); +} + +} // namespace detail::hash_impl + +namespace detail { + +// The two entry points, at `detail` scope because that is where callers writing their own hash for +// their own type reach for them, and doc/usage.md shows exactly that. +using hash_impl::hash_bytes; +using hash_impl::hash_int; + +} // namespace detail + +namespace detail { + +struct nonesuch {}; + +template class Op, class... Args> +struct detector { + using value_t = std::false_type; + using type = Default; +}; + +template class Op, class... Args> +struct detector>, Op, Args...> { + using value_t = std::true_type; + using type = Op; +}; + +template