GCC Code Coverage Report


Directory: src/
File: src/pg/typeinfo.h
Date: 2026-09-30 11:11:31
Exec Total Coverage
Lines: 13 14 92.9%
Functions: 2 2 100.0%
Branches: 3 6 50.0%

Line Branch Exec Source
1 /*
2 * Copyright (c) 2026 Tiger Data, Inc.
3 * Licensed under the PostgreSQL License. See LICENSE for details.
4 *
5 * typeinfo.h - per-type behaviour for a prism-indexed column
6 *
7 * The access method indexes more than one vector type, so every path that
8 * reads an indexed value needs to know how to reach the float32 arrays the
9 * distance and RaBitQ kernels work on. That knowledge is one descriptor per
10 * type, and the descriptor comes from the *opclass* -- support function
11 * PRISM_TYPE_INFO_PROC -- not from inspecting the column.
12 *
13 * Asking the opclass matters beyond tidiness. It is the same authority the
14 * planner consulted when it decided the index applied, so the access method
15 * cannot disagree with the planner about a column an index was built on.
16 * Deriving the type from the column means answering "which type is this?" by
17 * name or by OID, and both answers are wrong somewhere: PostgreSQL restricts
18 * the search path around CREATE INDEX and index_build, so a name lookup fails
19 * during a build -- exactly where a wrong answer corrupts an index -- and OID
20 * equality rejects a binary-coercible column the opclass itself accepted, a
21 * pgvector `halfvec` column being the case in point. It also makes a new type
22 * an opclass plus a descriptor, with no new branch in the access method.
23 *
24 * The contract is deliberately narrower than pgvector's, which keeps values as
25 * opaque Datums and reaches them only through opclass functions. That approach
26 * would work here too -- quantization could be a support function taking a
27 * Datum, the shape pgvector already uses for normalize -- so this is a choice
28 * rather than a limit. Two reasons for it:
29 *
30 * - The quantization and distance kernels live in src/index, src/algo and
31 * src/quant and are compiled into the standalone build, which has no fmgr,
32 * no Datum and no type OIDs. A Datum-typed seam would have to be shimmed or
33 * duplicated there, and the point of standalone is to measure the same
34 * code.
35 *
36 * - The conversion would not disappear, only move. The kernels are element-
37 * wise SIMD over contiguous float32, so a Datum-typed quantize would widen
38 * to float32 inside the callback -- the same work, done per call instead of
39 * once at the boundary, and without the shared Vec32TypeOps to do it.
40 *
41 * Where this contract would need revisiting is a type for which "widen to
42 * dense float32" is the wrong shape -- a sparse or bit-packed vector, both of
43 * which pgvector's hnsw does support. Dense float types fit it exactly.
44 *
45 * The conversion itself is not new here: Vec32TypeOps already covers
46 * float32 and float16 for k-means and quantization, and is shared with
47 * standalone. This descriptor binds an opclass to one of those vtables and
48 * adds the two things only PostgreSQL needs -- how to unwrap a Datum, and
49 * which centroid format the type implies. Per docs/development.md, the
50 * conversion runs at the "Specialized" tier: one vtable call per vector to
51 * obtain a float32 view, then inlined f32 kernels over it.
52 *
53 * A descriptor describes its opclass's own input type. A cross-type ordering
54 * operator whose argument had a different layout would break that; every
55 * family member prism declares or adds shares its opclass's layout.
56 */
57
58 #ifndef VS_TYPEINFO_H
59 #define VS_TYPEINFO_H
60
61 #include <postgres.h>
62
63 #include <utils/rel.h>
64
65 #include "core/types.h"
66 #include "index/centroid_page.h"
67
68 typedef struct PrismIndexTypeInfo
69 {
70 /* Type name, for error messages. */
71 const char *name;
72
73 /* Widest dimension the type carries, before the index's own limit. */
74 Dimension max_dimensions;
75
76 /*
77 * Centroid storage the type implies. Centroids are drawn from the indexed
78 * data, so a half-precision column has no use for full-precision
79 * centroids.
80 */
81 PrismCentroidFormat centroid_format;
82
83 /*
84 * Element kernels and float32 conversion, shared with standalone. f32
85 * returns its input pointer from to_float_block; f16 widens into the
86 * caller's buffer.
87 */
88 const Vec32TypeOps *ops;
89
90 /*
91 * Datum -> raw element block, plus the value's dimension. The only part
92 * that touches PostgreSQL's representation; deliberately per-type rather
93 * than one function assuming a shared header layout.
94 */
95 const void *(*unwrap)(Datum d, Dimension *dim);
96 } PrismIndexTypeInfo;
97
98 /*
99 * Descriptor for an index's column type. The support function is optional:
100 * an opclass that declares none indexes `vec32`, which is both the original
101 * behaviour and what an index built before the function existed still needs.
102 */
103 const PrismIndexTypeInfo *prism_index_type_info(Relation index);
104
105 /* Does a float32 view of this type need a caller-provided buffer? */
106 static inline bool
107 18461 vec32_needs_buffer(const PrismIndexTypeInfo *ti)
108 {
109 18461 return ti->ops->element_size != sizeof(float);
110 }
111
112 /*
113 * An index's binding of a type descriptor.
114 *
115 * PrismIndexTypeInfo is a shared static -- one instance per type for the whole
116 * backend -- so it holds nothing specific to an index. This is the per-index
117 * half: the dimension its column is pinned to, and the buffer a float32 view
118 * needs at that dimension. Binding them together means a read cannot be handed
119 * a dimension the buffer was not sized for.
120 *
121 * Resolve once per build / scan / rerank and read many times.
122 */
123 typedef struct Vec32Access
124 {
125 const PrismIndexTypeInfo *ti; /* shared, per type */
126 Dimension dim; /* this index's column */
127 float *buf; /* NULL when no conversion is needed */
128 } Vec32Access;
129
130 /*
131 * Bind a descriptor to a dimension, allocating the conversion buffer if the
132 * type needs one.
133 *
134 * The context is the caller's because the useful lifetime differs per path:
135 * the whole scan for a query vector, the whole build for the scan callbacks,
136 * one rerank call for the heap fetches. Pass CurrentMemoryContext for plain
137 * palloc semantics.
138 */
139 static inline Vec32Access
140 18461 vec32_access(const PrismIndexTypeInfo *ti, Dimension dim, MemoryContext ctx)
141 {
142 36922 return (Vec32Access){
143 .ti = ti,
144 .dim = dim,
145 18461 .buf = vec32_needs_buffer(ti)
146 ? (float *)
147 247 MemoryContextAlloc(ctx, sizeof(float) * dim)
148
2/2
✓ Branch 0 taken 247 times.
✓ Branch 1 taken 18214 times.
18461 : NULL,
149 };
150 }
151
152 /*
153 * Float32 view of one indexed value.
154 *
155 * The dimension is checked before any conversion, so a value wider than the
156 * index cannot be converted into a buffer sized for the index. A column's
157 * typmod pins its dimension, so a mismatch should be unreachable -- but this
158 * reads a varlena into a fixed buffer.
159 */
160 static inline Vec32Ref
161 793440 vec32_read(const Vec32Access *a, Datum d)
162 {
163 793440 Dimension dim;
164 793440 const void *raw = a->ti->unwrap(d, &dim);
165
166
1/2
✗ Branch 0 not taken.
✓ Branch 1 taken 793440 times.
793440 if (dim != a->dim)
167 ✗ ereport(ERROR,
168 (errcode(ERRCODE_DATA_EXCEPTION),
169 errmsg("%s dimension %u does not match index dimension %u",
170 a->ti->name,
171 dim,
172 a->dim)));
173
174 1586880 return (Vec32Ref){
175 793440 .data = a->ti->ops->to_float_block(raw, a->buf, 1, dim),
176 .dim = dim,
177 };
178 }
179
180 #endif /* VS_TYPEINFO_H */
181