| Line | Branch | Exec | Source |
|---|---|---|---|
| 1 | /* | ||
| 2 | * Copyright (c) 2026 Tiger Data, Inc. | ||
| 3 | * Licensed under the PostgreSQL License. See LICENSE for details. | ||
| 4 | * | ||
| 5 | * platform.h - Platform abstraction and SIMD capability detection | ||
| 6 | * | ||
| 7 | * Provides runtime CPU feature detection and compiler intrinsics wrappers | ||
| 8 | * for portable SIMD code. | ||
| 9 | */ | ||
| 10 | |||
| 11 | #ifndef VS_PLATFORM_H | ||
| 12 | #define VS_PLATFORM_H | ||
| 13 | |||
| 14 | #include <stdint.h> | ||
| 15 | |||
| 16 | /* | ||
| 17 | * SIMD capability flags detected at runtime. | ||
| 18 | * Multiple flags may be set (e.g., AVX512F implies AVX2 implies SSE4.1). | ||
| 19 | */ | ||
| 20 | typedef enum | ||
| 21 | { | ||
| 22 | SIMD_NONE = 0, | ||
| 23 | SIMD_SSE2 = 1 << 0, | ||
| 24 | SIMD_SSE4_1 = 1 << 1, | ||
| 25 | SIMD_AVX2 = 1 << 2, | ||
| 26 | SIMD_AVX512F = 1 << 3, | ||
| 27 | SIMD_NEON = 1 << 4, | ||
| 28 | SIMD_AVX512_VPOPCNTDQ = 1 << 5, | ||
| 29 | SIMD_AVX512DQ = 1 << 6, | ||
| 30 | SIMD_AVX512BW = 1 << 7, | ||
| 31 | } SimdCapability; | ||
| 32 | |||
| 33 | /* | ||
| 34 | * The AVX-512 kernels are compiled for specific sub-extensions (see the | ||
| 35 | * VS_TARGET_AVX512* attributes in simd_utils.h and the target attributes in | ||
| 36 | * the *_avx512.c files), not plain AVX512F. Dispatch -- and the test/bench | ||
| 37 | * overrides that force a path -- must require every bit the kernels use: a | ||
| 38 | * CPU can implement F without BW or DQ (e.g. Knights Landing), and running | ||
| 39 | * the compiled kernels there faults with SIGILL. VPOPCNTDQ kernels are gated | ||
| 40 | * separately on SIMD_AVX512_VPOPCNTDQ (which already implies F). | ||
| 41 | */ | ||
| 42 | #define VS_SIMD_AVX512_DQ \ | ||
| 43 | (SIMD_AVX512F | SIMD_AVX512DQ) /* distance, rabitq \ | ||
| 44 | */ | ||
| 45 | #define VS_SIMD_AVX512_BW (SIMD_AVX512F | SIMD_AVX512BW) /* fastscan */ | ||
| 46 | |||
| 47 | /* | ||
| 48 | * Detect CPU SIMD capabilities at runtime. | ||
| 49 | * | ||
| 50 | * On x86/x64: Uses CPUID to detect SSE2, SSE4.1, AVX2, AVX512F. | ||
| 51 | * On ARM: Returns SIMD_NEON if compiled with NEON support. | ||
| 52 | * | ||
| 53 | * Result is cached after first call. | ||
| 54 | */ | ||
| 55 | SimdCapability vs_detect_simd(void); | ||
| 56 | |||
| 57 | /* | ||
| 58 | * Check if specific SIMD capability is available. | ||
| 59 | */ | ||
| 60 | static inline int | ||
| 61 | 198 | vs_has_simd(SimdCapability cap) | |
| 62 | { | ||
| 63 |
2/4✓ Branch 1 taken 86 times.
✗ Branch 2 not taken.
✓ Branch 4 taken 86 times.
✗ Branch 5 not taken.
|
112 | return (vs_detect_simd() & cap) != 0; |
| 64 | } | ||
| 65 | |||
| 66 | /* | ||
| 67 | * Check that ALL bits in mask are available. Use with the VS_SIMD_AVX512_* | ||
| 68 | * masks so a multi-bit requirement (F + BW, F + DQ) is tested as a unit. | ||
| 69 | */ | ||
| 70 | static inline int | ||
| 71 | 243 | vs_has_all_simd(uint32_t mask) | |
| 72 | { | ||
| 73 | 243 | return ((uint32_t)vs_detect_simd() & mask) == mask; | |
| 74 | } | ||
| 75 | |||
| 76 | /* | ||
| 77 | * Override SIMD capability detection (for testing and benchmarking). | ||
| 78 | * | ||
| 79 | * Forces all SIMD-using code to use a specific instruction set by masking | ||
| 80 | * detected CPU capabilities. This allows: | ||
| 81 | * - Testing all SIMD code paths, including fallbacks | ||
| 82 | * - Benchmarking different SIMD implementations | ||
| 83 | * - Debugging SIMD-specific issues | ||
| 84 | * | ||
| 85 | * Parameters: | ||
| 86 | * mask: Bitwise OR of SimdCapability flags | ||
| 87 | * - 0 or SIMD_NONE: Force scalar implementation | ||
| 88 | * - SIMD_AVX2: Force AVX2 (on x86-64) | ||
| 89 | * - SIMD_AVX512F: Force AVX-512 (on x86-64) | ||
| 90 | * - SIMD_NEON: Force NEON (on ARM) | ||
| 91 | * - 0xFFFFFFFF: Auto-detect (default, clears override) | ||
| 92 | * | ||
| 93 | * Note: Call this BEFORE any SIMD detection occurs. Calling after | ||
| 94 | * detection may require re-initialization of SIMD-using modules. | ||
| 95 | * | ||
| 96 | * Example usage: | ||
| 97 | * // Test AVX2 code path even on AVX-512 CPU | ||
| 98 | * vs_simd_set_override(SIMD_AVX2); | ||
| 99 | * // ... run tests or benchmarks ... | ||
| 100 | * vs_simd_set_override(0xFFFFFFFF); // Reset to auto-detect | ||
| 101 | */ | ||
| 102 | void vs_simd_set_override(uint32_t mask); | ||
| 103 | |||
| 104 | /* | ||
| 105 | * Clear SIMD detection cache, forcing re-detection. | ||
| 106 | * | ||
| 107 | * Useful when testing different SIMD overrides. Call this after | ||
| 108 | * vs_simd_set_override() to ensure the new mask takes effect. | ||
| 109 | */ | ||
| 110 | void vs_simd_reset_cache(void); | ||
| 111 | |||
| 112 | /* Cache line size (typical for modern CPUs) */ | ||
| 113 | #define VS_CACHE_LINE 64 | ||
| 114 | |||
| 115 | /* | ||
| 116 | * Prefetch hints for memory access optimization. | ||
| 117 | * | ||
| 118 | * vs_prefetch_read: Prefetch for reading (non-temporal, keep in all caches) | ||
| 119 | * vs_prefetch_write: Prefetch for writing (exclusive access) | ||
| 120 | */ | ||
| 121 | #define vs_prefetch_read(addr) __builtin_prefetch((addr), 0, 3) | ||
| 122 | #define vs_prefetch_write(addr) __builtin_prefetch((addr), 1, 3) | ||
| 123 | |||
| 124 | /* | ||
| 125 | * Branch prediction hints. | ||
| 126 | * | ||
| 127 | * Use sparingly - modern CPUs have good branch predictors. | ||
| 128 | * Most useful for error paths that are rarely taken. | ||
| 129 | */ | ||
| 130 | /* Time-unit conversion factors for nanosecond-based instrumentation | ||
| 131 | * (cf. PostgreSQL's NS_PER_S family in portability/instr_time.h; defined | ||
| 132 | * here so shared, non-PG code can use them too). */ | ||
| 133 | #define VS_NS_PER_SEC 1000000000ULL | ||
| 134 | #define VS_NS_PER_MS 1000000ULL | ||
| 135 | #define VS_NS_PER_US 1000ULL | ||
| 136 | |||
| 137 | #define vs_likely(x) __builtin_expect(!!(x), 1) | ||
| 138 | #define vs_unlikely(x) __builtin_expect(!!(x), 0) | ||
| 139 | |||
| 140 | /* | ||
| 141 | * Compiler memory barrier. | ||
| 142 | * | ||
| 143 | * Prevents compiler from reordering memory accesses across this point. | ||
| 144 | * Does NOT generate CPU fence instructions (use atomics for that). | ||
| 145 | */ | ||
| 146 | #define vs_compiler_barrier() __asm__ __volatile__("" ::: "memory") | ||
| 147 | |||
| 148 | /* | ||
| 149 | * Alignment helpers. | ||
| 150 | */ | ||
| 151 | #define VS_ALIGN(x, a) (((x) + ((a) - 1)) & ~((a) - 1)) | ||
| 152 | #define VS_IS_ALIGNED(x, a) (((uintptr_t)(x) & ((a) - 1)) == 0) | ||
| 153 | |||
| 154 | /* | ||
| 155 | * SIMD vector alignment (64 bytes for AVX-512). | ||
| 156 | */ | ||
| 157 | #define VS_SIMD_ALIGN 64 | ||
| 158 | |||
| 159 | #endif /* VS_PLATFORM_H */ | ||
| 160 |