GCC Code Coverage Report


Directory: src/
File: src/core/platform.h
Date: 2026-09-30 11:11:31
Exec Total Coverage
Lines: 4 4 100.0%
Functions: 2 2 100.0%
Branches: 2 4 50.0%

Line Branch Exec Source
1 /*
2 * Copyright (c) 2026 Tiger Data, Inc.
3 * Licensed under the PostgreSQL License. See LICENSE for details.
4 *
5 * platform.h - Platform abstraction and SIMD capability detection
6 *
7 * Provides runtime CPU feature detection and compiler intrinsics wrappers
8 * for portable SIMD code.
9 */
10
11 #ifndef VS_PLATFORM_H
12 #define VS_PLATFORM_H
13
14 #include <stdint.h>
15
16 /*
17 * SIMD capability flags detected at runtime.
18 * Multiple flags may be set (e.g., AVX512F implies AVX2 implies SSE4.1).
19 */
20 typedef enum
21 {
22 SIMD_NONE = 0,
23 SIMD_SSE2 = 1 << 0,
24 SIMD_SSE4_1 = 1 << 1,
25 SIMD_AVX2 = 1 << 2,
26 SIMD_AVX512F = 1 << 3,
27 SIMD_NEON = 1 << 4,
28 SIMD_AVX512_VPOPCNTDQ = 1 << 5,
29 SIMD_AVX512DQ = 1 << 6,
30 SIMD_AVX512BW = 1 << 7,
31 } SimdCapability;
32
33 /*
34 * The AVX-512 kernels are compiled for specific sub-extensions (see the
35 * VS_TARGET_AVX512* attributes in simd_utils.h and the target attributes in
36 * the *_avx512.c files), not plain AVX512F. Dispatch -- and the test/bench
37 * overrides that force a path -- must require every bit the kernels use: a
38 * CPU can implement F without BW or DQ (e.g. Knights Landing), and running
39 * the compiled kernels there faults with SIGILL. VPOPCNTDQ kernels are gated
40 * separately on SIMD_AVX512_VPOPCNTDQ (which already implies F).
41 */
42 #define VS_SIMD_AVX512_DQ \
43 (SIMD_AVX512F | SIMD_AVX512DQ) /* distance, rabitq \
44 */
45 #define VS_SIMD_AVX512_BW (SIMD_AVX512F | SIMD_AVX512BW) /* fastscan */
46
47 /*
48 * Detect CPU SIMD capabilities at runtime.
49 *
50 * On x86/x64: Uses CPUID to detect SSE2, SSE4.1, AVX2, AVX512F.
51 * On ARM: Returns SIMD_NEON if compiled with NEON support.
52 *
53 * Result is cached after first call.
54 */
55 SimdCapability vs_detect_simd(void);
56
57 /*
58 * Check if specific SIMD capability is available.
59 */
60 static inline int
61 198 vs_has_simd(SimdCapability cap)
62 {
63
2/4
✓ Branch 1 taken 86 times.
✗ Branch 2 not taken.
✓ Branch 4 taken 86 times.
✗ Branch 5 not taken.
112 return (vs_detect_simd() & cap) != 0;
64 }
65
66 /*
67 * Check that ALL bits in mask are available. Use with the VS_SIMD_AVX512_*
68 * masks so a multi-bit requirement (F + BW, F + DQ) is tested as a unit.
69 */
70 static inline int
71 243 vs_has_all_simd(uint32_t mask)
72 {
73 243 return ((uint32_t)vs_detect_simd() & mask) == mask;
74 }
75
76 /*
77 * Override SIMD capability detection (for testing and benchmarking).
78 *
79 * Forces all SIMD-using code to use a specific instruction set by masking
80 * detected CPU capabilities. This allows:
81 * - Testing all SIMD code paths, including fallbacks
82 * - Benchmarking different SIMD implementations
83 * - Debugging SIMD-specific issues
84 *
85 * Parameters:
86 * mask: Bitwise OR of SimdCapability flags
87 * - 0 or SIMD_NONE: Force scalar implementation
88 * - SIMD_AVX2: Force AVX2 (on x86-64)
89 * - SIMD_AVX512F: Force AVX-512 (on x86-64)
90 * - SIMD_NEON: Force NEON (on ARM)
91 * - 0xFFFFFFFF: Auto-detect (default, clears override)
92 *
93 * Note: Call this BEFORE any SIMD detection occurs. Calling after
94 * detection may require re-initialization of SIMD-using modules.
95 *
96 * Example usage:
97 * // Test AVX2 code path even on AVX-512 CPU
98 * vs_simd_set_override(SIMD_AVX2);
99 * // ... run tests or benchmarks ...
100 * vs_simd_set_override(0xFFFFFFFF); // Reset to auto-detect
101 */
102 void vs_simd_set_override(uint32_t mask);
103
104 /*
105 * Clear SIMD detection cache, forcing re-detection.
106 *
107 * Useful when testing different SIMD overrides. Call this after
108 * vs_simd_set_override() to ensure the new mask takes effect.
109 */
110 void vs_simd_reset_cache(void);
111
112 /* Cache line size (typical for modern CPUs) */
113 #define VS_CACHE_LINE 64
114
115 /*
116 * Prefetch hints for memory access optimization.
117 *
118 * vs_prefetch_read: Prefetch for reading (non-temporal, keep in all caches)
119 * vs_prefetch_write: Prefetch for writing (exclusive access)
120 */
121 #define vs_prefetch_read(addr) __builtin_prefetch((addr), 0, 3)
122 #define vs_prefetch_write(addr) __builtin_prefetch((addr), 1, 3)
123
124 /*
125 * Branch prediction hints.
126 *
127 * Use sparingly - modern CPUs have good branch predictors.
128 * Most useful for error paths that are rarely taken.
129 */
130 /* Time-unit conversion factors for nanosecond-based instrumentation
131 * (cf. PostgreSQL's NS_PER_S family in portability/instr_time.h; defined
132 * here so shared, non-PG code can use them too). */
133 #define VS_NS_PER_SEC 1000000000ULL
134 #define VS_NS_PER_MS 1000000ULL
135 #define VS_NS_PER_US 1000ULL
136
137 #define vs_likely(x) __builtin_expect(!!(x), 1)
138 #define vs_unlikely(x) __builtin_expect(!!(x), 0)
139
140 /*
141 * Compiler memory barrier.
142 *
143 * Prevents compiler from reordering memory accesses across this point.
144 * Does NOT generate CPU fence instructions (use atomics for that).
145 */
146 #define vs_compiler_barrier() __asm__ __volatile__("" ::: "memory")
147
148 /*
149 * Alignment helpers.
150 */
151 #define VS_ALIGN(x, a) (((x) + ((a) - 1)) & ~((a) - 1))
152 #define VS_IS_ALIGNED(x, a) (((uintptr_t)(x) & ((a) - 1)) == 0)
153
154 /*
155 * SIMD vector alignment (64 bytes for AVX-512).
156 */
157 #define VS_SIMD_ALIGN 64
158
159 #endif /* VS_PLATFORM_H */
160