Compare commits

..

38 Commits

Author SHA1 Message Date
Andrew Kane
721fcf4d2c Improved tests for rescans / nested loop joins [skip ci] 2026-07-28 02:49:06 -07:00
Andrew Kane
2f99911554 Fixed tests for Postgres < 16 2026-07-28 00:49:43 -07:00
Andrew Kane
982593ef01 Added tests for rescans / nested loop joins - #1002
Co-authored-by: zthhhhh <97zthhhh@gmail.com>
2026-07-28 00:44:44 -07:00
Andrew Kane
2a7b954beb Fixed version check for palloc variants, take 2 2026-07-27 22:30:23 -07:00
Andrew Kane
0b2d92bc4f Fixed version check for palloc variants 2026-07-27 22:25:21 -07:00
Andrew Kane
e6bd4c8acf Improved CreateStateDatums [skip ci] 2026-07-27 22:09:33 -07:00
Andrew Kane
1be6148eca Switched to palloc_array and palloc0_array where possible 2026-07-27 21:51:15 -07:00
Andrew Kane
bc4e6a13a2 Updated changelog [skip ci] 2026-07-27 21:47:57 -07:00
Andrew Kane
c267348f21 Updated changelog [skip ci] 2026-07-27 21:39:46 -07:00
zthhhhh
93a5b4a16d Fix memory leak in ivfrescan 2026-07-27 21:22:17 -07:00
Andrew Kane
c0919bc26a Added nnz check for vector/halfvec to sparsevec casts (in case limits change in future) [skip ci]
Co-authored-by: 0xJi3F
2026-07-27 20:38:17 -07:00
Andrew Kane
3566276bd4 Fixed array to sparsevec cast not limiting to 16,000 non-zero elements [skip ci]
Co-authored-by: 0xJi3F
2026-07-27 20:33:15 -07:00
Andrew Kane
3e523434f4 Added test for array to halfvec cast [skip ci] 2026-07-27 20:30:04 -07:00
Andrew Kane
d2e2aa0e63 Switched to palloc_object and palloc0_object when possible 2026-07-27 19:26:50 -07:00
Andrew Kane
573040d3a2 Hardened VectorArrayGet [skip ci] 2026-07-27 16:14:25 -07:00
Andrew Kane
a6420355c5 Updated FreeBSD package name in readme [skip ci] 2026-07-10 22:17:27 -07:00
Andrew Kane
73356ecfa7 Improved check for VectorArrayInit [skip ci] 2026-07-10 16:30:35 -07:00
Andrew Kane
0e557b1d18 Updated readme [skip ci] 2026-07-10 15:16:36 -07:00
Andrew Kane
769a60884c Updated readme [skip ci] 2026-07-10 14:20:26 -07:00
Andrew Kane
8711840058 Added section on multitenancy [skip ci] 2026-07-10 14:13:13 -07:00
Andrew Kane
159b79aaad Version bump to 0.8.5 [skip ci] 2026-07-08 16:03:40 -07:00
Andrew Kane
c5a277a975 Hardened VectorArrayInit 2026-07-08 10:15:21 -07:00
Andrew Kane
124f6c61a4 Added Postgres 20 to CI [skip ci] 2026-07-07 16:15:39 -07:00
Andrew Kane
5ca52d12b6 Updated checkout action [skip ci] 2026-07-07 15:46:41 -07:00
Andrew Kane
1f61d1111b Added ubuntu-26.04 to CI [skip ci] 2026-07-07 15:46:06 -07:00
Andrew Kane
fb1b8966eb Added comment to IVFFlat vacuuming test [skip ci] 2026-07-01 17:21:26 -07:00
Andrew Kane
71ce9d3311 Added test for concurrent INSERTs, DELETEs, SELECTs, and VACUUM with IVFFlat [skip ci] 2026-07-01 17:18:34 -07:00
Andrew Kane
d19cc0d371 Improved HNSW vacuuming test [skip ci] 2026-07-01 17:14:53 -07:00
Andrew Kane
89fda3e100 Removed slow tests [skip ci] 2026-07-01 16:05:53 -07:00
Andrew Kane
30d8654b47 Added tests for HNSW index dimensions for bit type [skip ci] 2026-07-01 15:49:07 -07:00
Andrew Kane
a846385cc7 Added tests for index dimensions for bit type [skip ci] 2026-07-01 15:47:05 -07:00
Andrew Kane
98d7c4124e Added IVFFlat memory tests for bit [skip ci] 2026-07-01 15:45:18 -07:00
Andrew Kane
2a2b4a0b58 Updated changelog [skip ci] 2026-07-01 12:35:46 -07:00
Andrew Kane
f15a50387f Moved logic for calculating number of samples [skip ci] 2026-07-01 12:27:03 -07:00
Andrew Kane
971b7d7fd6 Added IVFFlat memory tests for halfvec [skip ci] 2026-07-01 12:21:08 -07:00
Andrew Kane
f51d8ed989 Improved readability of options tests [skip ci] 2026-07-01 12:17:50 -07:00
Andrew Kane
a76a18d526 Added tests for index dimensions [skip ci] 2026-07-01 12:15:27 -07:00
Andrew Kane
b383e4d191 Reduced memory usage for small tables for IVFFlat index builds - resolves #995 and resolves #996
Co-authored-by: Itai Spiegel <itai@mave.com>
2026-07-01 11:54:21 -07:00
39 changed files with 415 additions and 75 deletions

View File

@@ -8,10 +8,12 @@ jobs:
fail-fast: false fail-fast: false
matrix: matrix:
include: include:
- postgres: 20
os: ubuntu-26.04
- postgres: 19 - postgres: 19
os: ubuntu-24.04 os: ubuntu-26.04
- postgres: 18 - postgres: 18
os: ubuntu-24.04 os: ubuntu-26.04-arm
- postgres: 17 - postgres: 17
os: ubuntu-24.04 os: ubuntu-24.04
- postgres: 16 - postgres: 16
@@ -23,7 +25,7 @@ jobs:
- postgres: 13 - postgres: 13
os: ubuntu-22.04 os: ubuntu-22.04
steps: steps:
- uses: actions/checkout@v6 - uses: actions/checkout@v7
- uses: ankane/setup-postgres@v1 - uses: ankane/setup-postgres@v1
with: with:
postgres-version: ${{ matrix.postgres }} postgres-version: ${{ matrix.postgres }}
@@ -53,7 +55,7 @@ jobs:
- postgres: 14 - postgres: 14
os: macos-15-intel os: macos-15-intel
steps: steps:
- uses: actions/checkout@v6 - uses: actions/checkout@v7
- uses: ankane/setup-postgres@v1 - uses: ankane/setup-postgres@v1
with: with:
postgres-version: ${{ matrix.postgres }} postgres-version: ${{ matrix.postgres }}
@@ -92,7 +94,7 @@ jobs:
- postgres: 14 - postgres: 14
os: windows-2022 os: windows-2022
steps: steps:
- uses: actions/checkout@v6 - uses: actions/checkout@v7
- uses: ankane/setup-postgres@v1 - uses: ankane/setup-postgres@v1
with: with:
postgres-version: ${{ matrix.postgres }} postgres-version: ${{ matrix.postgres }}
@@ -133,7 +135,7 @@ jobs:
if: ${{ !startsWith(github.ref_name, 'mac') && !startsWith(github.ref_name, 'windows') }} if: ${{ !startsWith(github.ref_name, 'mac') && !startsWith(github.ref_name, 'windows') }}
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- uses: actions/checkout@v6 - uses: actions/checkout@v7
- uses: ankane/setup-postgres-valgrind@v1 - uses: ankane/setup-postgres-valgrind@v1
with: with:
postgres-version: 18 postgres-version: 18

View File

@@ -1,7 +1,17 @@
## 0.8.6 (unreleased)
- Fixed array to `sparsevec` cast not limiting non-zero elements
- Fixed memory usage for IVFFlat index scans with nested loop joins
## 0.8.5 (2026-07-08)
- Reduced memory usage for small tables for IVFFlat index builds
## 0.8.4 (2026-06-30) ## 0.8.4 (2026-06-30)
- Fixed `hnsw graph not repaired` error with HNSW vacuuming - Fixed `hnsw graph not repaired` error with HNSW vacuuming
- Fixed possible error with inserts during HNSW vacuuming - Fixed possible error with inserts during HNSW vacuuming
- Fixed memory exceeding `maintenance_work_mem` with IVFFlat index builds
## 0.8.3 (2026-06-17) ## 0.8.3 (2026-06-17)

View File

@@ -5,7 +5,7 @@ ARG DEBIAN_CODENAME=bookworm
FROM postgres:$PG_MAJOR-$DEBIAN_CODENAME FROM postgres:$PG_MAJOR-$DEBIAN_CODENAME
ARG PG_MAJOR ARG PG_MAJOR
ADD https://github.com/pgvector/pgvector.git#v0.8.4 /tmp/pgvector ADD https://github.com/pgvector/pgvector.git#v0.8.5 /tmp/pgvector
RUN apt-get update && \ RUN apt-get update && \
apt-mark hold locales && \ apt-mark hold locales && \

View File

@@ -2,7 +2,7 @@
"name": "vector", "name": "vector",
"abstract": "Open-source vector similarity search for Postgres", "abstract": "Open-source vector similarity search for Postgres",
"description": "Supports L2 distance, inner product, and cosine distance", "description": "Supports L2 distance, inner product, and cosine distance",
"version": "0.8.4", "version": "0.8.5",
"maintainer": [ "maintainer": [
"Andrew Kane <andrew@ankane.org>" "Andrew Kane <andrew@ankane.org>"
], ],
@@ -20,7 +20,7 @@
"vector": { "vector": {
"file": "sql/vector.sql", "file": "sql/vector.sql",
"docfile": "README.md", "docfile": "README.md",
"version": "0.8.4", "version": "0.8.5",
"abstract": "Open-source vector similarity search for Postgres" "abstract": "Open-source vector similarity search for Postgres"
} }
}, },

View File

@@ -1,5 +1,5 @@
EXTENSION = vector EXTENSION = vector
EXTVERSION = 0.8.4 EXTVERSION = 0.8.5
MODULE_big = vector MODULE_big = vector
DATA = $(wildcard sql/*--*--*.sql) DATA = $(wildcard sql/*--*--*.sql)

View File

@@ -1,5 +1,5 @@
EXTENSION = vector EXTENSION = vector
EXTVERSION = 0.8.4 EXTVERSION = 0.8.5
DATA_built = sql\$(EXTENSION)--$(EXTVERSION).sql DATA_built = sql\$(EXTENSION)--$(EXTVERSION).sql
OBJS = src\bitutils.obj src\bitvec.obj src\halfutils.obj src\halfvec.obj src\hnsw.obj src\hnswbuild.obj src\hnswinsert.obj src\hnswscan.obj src\hnswutils.obj src\hnswvacuum.obj src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\sparsevec.obj src\vector.obj OBJS = src\bitutils.obj src\bitvec.obj src\halfutils.obj src\halfvec.obj src\hnsw.obj src\hnswbuild.obj src\hnswinsert.obj src\hnswscan.obj src\hnswutils.obj src\hnswvacuum.obj src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\sparsevec.obj src\vector.obj

View File

@@ -23,7 +23,7 @@ Compile and install the extension (supports Postgres 13+)
```sh ```sh
cd /tmp cd /tmp
git clone --branch v0.8.4 https://github.com/pgvector/pgvector.git git clone --branch v0.8.5 https://github.com/pgvector/pgvector.git
cd pgvector cd pgvector
make make
make install # may need sudo make install # may need sudo
@@ -40,7 +40,7 @@ Ensure [C++ support in Visual Studio](https://learn.microsoft.com/en-us/cpp/buil
```cmd ```cmd
set "PGROOT=C:\Program Files\PostgreSQL\18" set "PGROOT=C:\Program Files\PostgreSQL\18"
cd %TEMP% cd %TEMP%
git clone --branch v0.8.4 https://github.com/pgvector/pgvector.git git clone --branch v0.8.5 https://github.com/pgvector/pgvector.git
cd pgvector cd pgvector
nmake /F Makefile.win nmake /F Makefile.win
nmake /F Makefile.win install nmake /F Makefile.win install
@@ -465,6 +465,16 @@ If filtering by many different values, consider [partitioning](https://www.postg
CREATE TABLE items (embedding vector(3), category_id int) PARTITION BY LIST(category_id); CREATE TABLE items (embedding vector(3), category_id int) PARTITION BY LIST(category_id);
``` ```
## Multitenancy
For applications with multiple tenants, sharing an approximate index between tenants means vectors from one tenant can affect recall (and speed) for other tenants.
For tenant isolation, use [list partitioning](https://www.postgresql.org/docs/current/ddl-partitioning.html) or separate tables.
```sql
CREATE TABLE items (customer_id int, embedding vector(3)) PARTITION BY LIST(customer_id);
```
## Iterative Index Scans ## Iterative Index Scans
With approximate indexes, queries with filtering can return less results since filtering is applied *after* the index is scanned. Starting with 0.8.0, you can enable iterative index scans, which will automatically scan more of the index until enough results are found (or it reaches `hnsw.max_scan_tuples` or `ivfflat.max_probes`). With approximate indexes, queries with filtering can return less results since filtering is applied *after* the index is scanned. Starting with 0.8.0, you can enable iterative index scans, which will automatically scan more of the index until enough results are found (or it reaches `hnsw.max_scan_tuples` or `ivfflat.max_probes`).
@@ -1151,23 +1161,23 @@ This adds pgvector to the [Postgres image](https://hub.docker.com/_/postgres) (r
Supported tags are: Supported tags are:
- `pg18-trixie`, `0.8.4-pg18-trixie` - `pg18-trixie`, `0.8.5-pg18-trixie`
- `pg18-bookworm`, `0.8.4-pg18-bookworm`, `pg18`, `0.8.4-pg18` - `pg18-bookworm`, `0.8.5-pg18-bookworm`, `pg18`, `0.8.5-pg18`
- `pg17-trixie`, `0.8.4-pg17-trixie` - `pg17-trixie`, `0.8.5-pg17-trixie`
- `pg17-bookworm`, `0.8.4-pg17-bookworm`, `pg17`, `0.8.4-pg17` - `pg17-bookworm`, `0.8.5-pg17-bookworm`, `pg17`, `0.8.5-pg17`
- `pg16-trixie`, `0.8.4-pg16-trixie` - `pg16-trixie`, `0.8.5-pg16-trixie`
- `pg16-bookworm`, `0.8.4-pg16-bookworm`, `pg16`, `0.8.4-pg16` - `pg16-bookworm`, `0.8.5-pg16-bookworm`, `pg16`, `0.8.5-pg16`
- `pg15-trixie`, `0.8.4-pg15-trixie` - `pg15-trixie`, `0.8.5-pg15-trixie`
- `pg15-bookworm`, `0.8.4-pg15-bookworm`, `pg15`, `0.8.4-pg15` - `pg15-bookworm`, `0.8.5-pg15-bookworm`, `pg15`, `0.8.5-pg15`
- `pg14-trixie`, `0.8.4-pg14-trixie` - `pg14-trixie`, `0.8.5-pg14-trixie`
- `pg14-bookworm`, `0.8.4-pg14-bookworm`, `pg14`, `0.8.4-pg14` - `pg14-bookworm`, `0.8.5-pg14-bookworm`, `pg14`, `0.8.5-pg14`
- `pg13-trixie`, `0.8.4-pg13-trixie` - `pg13-trixie`, `0.8.5-pg13-trixie`
- `pg13-bookworm`, `0.8.4-pg13-bookworm`, `pg13`, `0.8.4-pg13` - `pg13-bookworm`, `0.8.5-pg13-bookworm`, `pg13`, `0.8.5-pg13`
You can also build the image manually: You can also build the image manually:
```sh ```sh
git clone --branch v0.8.4 https://github.com/pgvector/pgvector.git git clone --branch v0.8.5 https://github.com/pgvector/pgvector.git
cd pgvector cd pgvector
docker build --pull --build-arg PG_MAJOR=18 -t myuser/pgvector . docker build --pull --build-arg PG_MAJOR=18 -t myuser/pgvector .
``` ```
@@ -1223,7 +1233,7 @@ Note: Replace `18` with your Postgres server version
Install the FreeBSD package with: Install the FreeBSD package with:
```sh ```sh
pkg install postgresql17-pgvector pkg install postgresql18-pgvector
``` ```
or the port with: or the port with:

View File

@@ -0,0 +1,2 @@
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
\echo Use "ALTER EXTENSION vector UPDATE TO '0.8.5'" to load this file. \quit

View File

@@ -27,8 +27,12 @@
#include "parser/scansup.h" #include "parser/scansup.h"
#endif #endif
#if PG_VERSION_NUM < 140006
#define palloc_array(type, count) ((type *) palloc(sizeof(type) * (count)))
#endif
#define STATE_DIMS(x) (ARR_DIMS(x)[0] - 1) #define STATE_DIMS(x) (ARR_DIMS(x)[0] - 1)
#define CreateStateDatums(dim) palloc(sizeof(Datum) * (dim + 1)) #define CreateStateDatums(dim) palloc_array(Datum, (dim) + 1)
/* /*
* Get a half from a message buffer * Get a half from a message buffer
@@ -513,7 +517,7 @@ halfvec_to_float4(PG_FUNCTION_ARGS)
Datum *datums; Datum *datums;
ArrayType *result; ArrayType *result;
datums = (Datum *) palloc(sizeof(Datum) * vec->dim); datums = palloc_array(Datum, vec->dim);
for (int i = 0; i < vec->dim; i++) for (int i = 0; i < vec->dim; i++)
datums[i] = Float4GetDatum(HalfToFloat4(vec->x[i])); datums[i] = Float4GetDatum(HalfToFloat4(vec->x[i]));

View File

@@ -105,6 +105,12 @@ typedef Pointer Item;
#define SeedRandom(seed) srandom(seed) #define SeedRandom(seed) srandom(seed)
#endif #endif
#if PG_VERSION_NUM < 140006
#define palloc_object(type) ((type *) palloc(sizeof(type)))
#define palloc0_object(type) ((type *) palloc0(sizeof(type)))
#define palloc_array(type, count) ((type *) palloc(sizeof(type) * (count)))
#endif
#define HnswIsElementTuple(tup) ((tup)->type == HNSW_ELEMENT_TUPLE_TYPE) #define HnswIsElementTuple(tup) ((tup)->type == HNSW_ELEMENT_TUPLE_TYPE)
#define HnswIsNeighborTuple(tup) ((tup)->type == HNSW_NEIGHBOR_TUPLE_TYPE) #define HnswIsNeighborTuple(tup) ((tup)->type == HNSW_NEIGHBOR_TUPLE_TYPE)

View File

@@ -930,7 +930,7 @@ HnswBeginParallel(HnswBuildState * buildstate, bool isconcurrent, int request)
Size estother; Size estother;
HnswShared *hnswshared; HnswShared *hnswshared;
char *hnswarea; char *hnswarea;
HnswLeader *hnswleader = (HnswLeader *) palloc0(sizeof(HnswLeader)); HnswLeader *hnswleader = palloc0_object(HnswLeader);
bool leaderparticipates = true; bool leaderparticipates = true;
int querylen; int querylen;
@@ -1151,7 +1151,7 @@ hnswbuild(Relation heap, Relation index, IndexInfo *indexInfo)
BuildIndex(heap, index, indexInfo, &buildstate, MAIN_FORKNUM); BuildIndex(heap, index, indexInfo, &buildstate, MAIN_FORKNUM);
result = (IndexBuildResult *) palloc(sizeof(IndexBuildResult)); result = palloc_object(IndexBuildResult);
result->heap_tuples = buildstate.reltuples; result->heap_tuples = buildstate.reltuples;
result->index_tuples = buildstate.indtuples; result->index_tuples = buildstate.indtuples;

View File

@@ -136,7 +136,7 @@ hnswbeginscan(Relation index, int nkeys, int norderbys)
scan = RelationGetIndexScan(index, nkeys, norderbys); scan = RelationGetIndexScan(index, nkeys, norderbys);
so = (HnswScanOpaque) palloc(sizeof(HnswScanOpaqueData)); so = palloc_object(HnswScanOpaqueData);
so->typeInfo = HnswGetTypeInfo(index); so->typeInfo = HnswGetTypeInfo(index);
/* Set support functions */ /* Set support functions */

View File

@@ -282,7 +282,7 @@ HnswAddHeapTid(HnswElement element, ItemPointer heaptid)
HnswElement HnswElement
HnswInitElementFromBlock(BlockNumber blkno, OffsetNumber offno) HnswInitElementFromBlock(BlockNumber blkno, OffsetNumber offno)
{ {
HnswElement element = palloc(sizeof(HnswElementData)); HnswElement element = palloc_object(HnswElementData);
char *base = NULL; char *base = NULL;
element->blkno = blkno; element->blkno = blkno;
@@ -596,7 +596,7 @@ GetElementDistance(char *base, HnswElement element, HnswQuery * q, HnswSupport *
static HnswSearchCandidate * static HnswSearchCandidate *
HnswInitSearchCandidate(char *base, HnswElement element, double distance) HnswInitSearchCandidate(char *base, HnswElement element, double distance)
{ {
HnswSearchCandidate *sc = palloc(sizeof(HnswSearchCandidate)); HnswSearchCandidate *sc = palloc_object(HnswSearchCandidate);
HnswPtrStore(base, sc->element, element); HnswPtrStore(base, sc->element, element);
sc->distance = distance; sc->distance = distance;
@@ -831,7 +831,7 @@ HnswSearchLayer(char *base, HnswQuery * q, List *ep, int ef, int lc, Relation in
HnswNeighborArray *localNeighborhood = NULL; HnswNeighborArray *localNeighborhood = NULL;
Size neighborhoodSize = 0; Size neighborhoodSize = 0;
int lm = HnswGetLayerM(m, lc); int lm = HnswGetLayerM(m, lc);
HnswUnvisited *unvisited = palloc(lm * sizeof(HnswUnvisited)); HnswUnvisited *unvisited = palloc_array(HnswUnvisited, lm);
int unvisitedLength; int unvisitedLength;
bool inMemory = index == NULL; bool inMemory = index == NULL;
@@ -1074,7 +1074,7 @@ SelectNeighbors(char *base, List *c, int lm, HnswSupport * support, bool *closer
if (list_length(w) <= lm) if (list_length(w) <= lm)
return w; return w;
wd = palloc(sizeof(HnswCandidate *) * list_length(w)); wd = palloc_array(HnswCandidate *, list_length(w));
/* Ensure order of candidates is deterministic for closer caching */ /* Ensure order of candidates is deterministic for closer caching */
if (sortCandidates) if (sortCandidates)
@@ -1328,7 +1328,7 @@ HnswFindElementNeighbors(char *base, HnswElement element, HnswElement entryPoint
foreach(lc2, w) foreach(lc2, w)
{ {
HnswSearchCandidate *sc = lfirst(lc2); HnswSearchCandidate *sc = lfirst(lc2);
HnswCandidate *hc = palloc(sizeof(HnswCandidate)); HnswCandidate *hc = palloc_object(HnswCandidate);
hc->element = sc->element; hc->element = sc->element;
hc->distance = sc->distance; hc->distance = sc->distance;

View File

@@ -737,7 +737,7 @@ InitVacuumState(HnswVacuumState * vacuumstate, IndexVacuumInfo *info, IndexBulkD
Relation index = info->index; Relation index = info->index;
if (stats == NULL) if (stats == NULL)
stats = (IndexBulkDeleteResult *) palloc0(sizeof(IndexBulkDeleteResult)); stats = palloc0_object(IndexBulkDeleteResult);
vacuumstate->index = index; vacuumstate->index = index;
vacuumstate->stats = stats; vacuumstate->stats = stats;

View File

@@ -396,7 +396,7 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
buildstate->centers = VectorArrayInit(buildstate->lists, buildstate->dimensions, buildstate->itemsize); buildstate->centers = VectorArrayInit(buildstate->lists, buildstate->dimensions, buildstate->itemsize);
/* TODO Move allocation to page creation */ /* TODO Move allocation to page creation */
buildstate->listInfo = palloc(sizeof(ListInfo) * buildstate->lists); buildstate->listInfo = palloc_array(ListInfo, buildstate->lists);
buildstate->tmpCtx = AllocSetContextCreate(CurrentMemoryContext, buildstate->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
"Ivfflat build temporary context", "Ivfflat build temporary context",
@@ -404,8 +404,8 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
#ifdef IVFFLAT_KMEANS_DEBUG #ifdef IVFFLAT_KMEANS_DEBUG
buildstate->inertia = 0; buildstate->inertia = 0;
buildstate->listSums = palloc0(sizeof(double) * buildstate->lists); buildstate->listSums = palloc0_array(double, buildstate->lists);
buildstate->listCounts = palloc0(sizeof(int) * buildstate->lists); buildstate->listCounts = palloc0_array(int, buildstate->lists);
#endif #endif
buildstate->ivfleader = NULL; buildstate->ivfleader = NULL;
@@ -438,15 +438,22 @@ ComputeCenters(IvfflatBuildState * buildstate)
pgstat_progress_update_param(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS); pgstat_progress_update_param(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS);
/* Skip samples for unlogged table */
if (buildstate->heap == NULL)
numSamples = 1;
else
{
int64 maxTuples = (int64) RelationGetNumberOfBlocks(buildstate->heap) * MaxHeapTuplesPerPage;
/* Target 50 samples per list, with at least 10000 samples */ /* Target 50 samples per list, with at least 10000 samples */
/* The number of samples has a large effect on index build time */ /* The number of samples has a large effect on index build time */
numSamples = buildstate->lists * 50; numSamples = buildstate->lists * 50;
if (numSamples < 10000) if (numSamples < 10000)
numSamples = 10000; numSamples = 10000;
/* Skip samples for unlogged table */ /* Save memory since will not have more than max tuples */
if (buildstate->heap == NULL) numSamples = Max(Min(numSamples, maxTuples), 1);
numSamples = 1; }
/* Sample rows */ /* Sample rows */
buildstate->memoryUsed += VECTOR_ARRAY_SIZE(numSamples, buildstate->itemsize); buildstate->memoryUsed += VECTOR_ARRAY_SIZE(numSamples, buildstate->itemsize);
@@ -655,7 +662,7 @@ IvfflatParallelScanAndSort(IvfflatSpool * ivfspool, IvfflatShared * ivfshared, S
IndexInfo *indexInfo; IndexInfo *indexInfo;
/* Initialize local tuplesort coordination state */ /* Initialize local tuplesort coordination state */
coordinate = palloc0(sizeof(SortCoordinateData)); coordinate = palloc0_object(SortCoordinateData);
coordinate->isWorker = true; coordinate->isWorker = true;
coordinate->nParticipants = -1; coordinate->nParticipants = -1;
coordinate->sharedsort = sharedsort; coordinate->sharedsort = sharedsort;
@@ -750,7 +757,7 @@ IvfflatParallelBuildMain(dsm_segment *seg, shm_toc *toc)
indexRel = index_open(ivfshared->indexrelid, indexLockmode); indexRel = index_open(ivfshared->indexrelid, indexLockmode);
/* Initialize worker's own spool */ /* Initialize worker's own spool */
ivfspool = (IvfflatSpool *) palloc0(sizeof(IvfflatSpool)); ivfspool = palloc0_object(IvfflatSpool);
ivfspool->heap = heapRel; ivfspool->heap = heapRel;
ivfspool->index = indexRel; ivfspool->index = indexRel;
@@ -805,7 +812,7 @@ IvfflatLeaderParticipateAsWorker(IvfflatBuildState * buildstate)
int sortmem; int sortmem;
/* Allocate memory and initialize private spool */ /* Allocate memory and initialize private spool */
leaderworker = (IvfflatSpool *) palloc0(sizeof(IvfflatSpool)); leaderworker = palloc0_object(IvfflatSpool);
leaderworker->heap = buildstate->heap; leaderworker->heap = buildstate->heap;
leaderworker->index = buildstate->index; leaderworker->index = buildstate->index;
@@ -831,7 +838,7 @@ IvfflatBeginParallel(IvfflatBuildState * buildstate, bool isconcurrent, int requ
IvfflatShared *ivfshared; IvfflatShared *ivfshared;
Sharedsort *sharedsort; Sharedsort *sharedsort;
char *ivfcenters; char *ivfcenters;
IvfflatLeader *ivfleader = (IvfflatLeader *) palloc0(sizeof(IvfflatLeader)); IvfflatLeader *ivfleader = palloc0_object(IvfflatLeader);
bool leaderparticipates = true; bool leaderparticipates = true;
int querylen; int querylen;
@@ -980,7 +987,7 @@ AssignTuples(IvfflatBuildState * buildstate)
/* Set up coordination state if at least one worker launched */ /* Set up coordination state if at least one worker launched */
if (buildstate->ivfleader) if (buildstate->ivfleader)
{ {
coordinate = (SortCoordinate) palloc0(sizeof(SortCoordinateData)); coordinate = palloc0_object(SortCoordinateData);
coordinate->isWorker = false; coordinate->isWorker = false;
coordinate->nParticipants = buildstate->ivfleader->nparticipanttuplesorts; coordinate->nParticipants = buildstate->ivfleader->nparticipanttuplesorts;
coordinate->sharedsort = buildstate->ivfleader->sharedsort; coordinate->sharedsort = buildstate->ivfleader->sharedsort;
@@ -1065,7 +1072,7 @@ ivfflatbuild(Relation heap, Relation index, IndexInfo *indexInfo)
BuildIndex(heap, index, indexInfo, &buildstate, MAIN_FORKNUM); BuildIndex(heap, index, indexInfo, &buildstate, MAIN_FORKNUM);
result = (IndexBuildResult *) palloc(sizeof(IndexBuildResult)); result = palloc_object(IndexBuildResult);
result->heap_tuples = buildstate.reltuples; result->heap_tuples = buildstate.reltuples;
result->index_tuples = buildstate.indtuples; result->index_tuples = buildstate.indtuples;

View File

@@ -89,6 +89,13 @@ typedef Pointer Item;
#define SeedRandom(seed) srandom(seed) #define SeedRandom(seed) srandom(seed)
#endif #endif
#if PG_VERSION_NUM < 140006
#define palloc_object(type) ((type *) palloc(sizeof(type)))
#define palloc0_object(type) ((type *) palloc0(sizeof(type)))
#define palloc_array(type, count) ((type *) palloc(sizeof(type) * (count)))
#define palloc0_array(type, count) ((type *) palloc0(sizeof(type) * (count)))
#endif
/* Variables */ /* Variables */
extern int ivfflat_probes; extern int ivfflat_probes;
extern int ivfflat_iterative_scan; extern int ivfflat_iterative_scan;
@@ -305,7 +312,7 @@ typedef IvfflatScanOpaqueData * IvfflatScanOpaque;
static inline Pointer static inline Pointer
VectorArrayGet(VectorArray arr, int offset) VectorArrayGet(VectorArray arr, int offset)
{ {
if (offset >= arr->maxlen) if (offset < 0 || offset >= arr->maxlen)
elog(ERROR, "safety check failed"); elog(ERROR, "safety check failed");
return ((char *) arr->items) + (offset * arr->itemsize); return ((char *) arr->items) + (offset * arr->itemsize);

View File

@@ -26,7 +26,7 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers, float *low
FmgrInfo *procinfo; FmgrInfo *procinfo;
Oid collation; Oid collation;
int64 j; int64 j;
float *weight = palloc(samples->length * sizeof(float)); float *weight = palloc_array(float, samples->length);
int numCenters = centers->maxlen; int numCenters = centers->maxlen;
int numSamples = samples->length; int numSamples = samples->length;
@@ -113,7 +113,7 @@ RandomCenters(Relation index, VectorArray centers, const IvfflatTypeInfo * typeI
int dimensions = centers->dim; int dimensions = centers->dim;
FmgrInfo *normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_KMEANS_NORM_PROC); FmgrInfo *normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_KMEANS_NORM_PROC);
Oid collation = index->rd_indcollation[0]; Oid collation = index->rd_indcollation[0];
float *x = (float *) palloc(sizeof(float) * dimensions); float *x = palloc_array(float, dimensions);
/* Fill with random data */ /* Fill with random data */
while (centers->length < centers->maxlen) while (centers->length < centers->maxlen)
@@ -480,7 +480,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers, const Ivff
static void static void
CheckElements(VectorArray centers, const IvfflatTypeInfo * typeInfo) CheckElements(VectorArray centers, const IvfflatTypeInfo * typeInfo)
{ {
float *scratch = palloc(sizeof(float) * centers->dim); float *scratch = palloc_array(float, centers->dim);
for (int i = 0; i < centers->length; i++) for (int i = 0; i < centers->length; i++)
{ {

View File

@@ -276,12 +276,13 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
if (maxProbes > lists) if (maxProbes > lists)
maxProbes = lists; maxProbes = lists;
so = (IvfflatScanOpaque) palloc(sizeof(IvfflatScanOpaqueData)); so = palloc_object(IvfflatScanOpaqueData);
so->typeInfo = IvfflatGetTypeInfo(index); so->typeInfo = IvfflatGetTypeInfo(index);
so->first = true; so->first = true;
so->probes = probes; so->probes = probes;
so->maxProbes = maxProbes; so->maxProbes = maxProbes;
so->dimensions = dimensions; so->dimensions = dimensions;
so->value = PointerGetDatum(NULL);
/* Set support functions */ /* Set support functions */
so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC); so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
@@ -317,9 +318,9 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
so->bas = GetAccessStrategy(BAS_BULKREAD); so->bas = GetAccessStrategy(BAS_BULKREAD);
so->listQueue = pairingheap_allocate(CompareLists, scan); so->listQueue = pairingheap_allocate(CompareLists, scan);
so->listPages = palloc(maxProbes * sizeof(BlockNumber)); so->listPages = palloc_array(BlockNumber, maxProbes);
so->listIndex = 0; so->listIndex = 0;
so->lists = palloc(maxProbes * sizeof(IvfflatScanList)); so->lists = palloc_array(IvfflatScanList, maxProbes);
MemoryContextSwitchTo(oldCtx); MemoryContextSwitchTo(oldCtx);
@@ -340,6 +341,12 @@ ivfflatrescan(IndexScanDesc scan, ScanKey keys, int nkeys, ScanKey orderbys, int
pairingheap_reset(so->listQueue); pairingheap_reset(so->listQueue);
so->listIndex = 0; so->listIndex = 0;
if (so->normprocinfo != NULL && DatumGetPointer(so->value) != NULL)
{
pfree(DatumGetPointer(so->value));
so->value = PointerGetDatum(NULL);
}
if (keys && scan->numberOfKeys > 0) if (keys && scan->numberOfKeys > 0)
memmove(scan->keyData, keys, scan->numberOfKeys * sizeof(ScanKeyData)); memmove(scan->keyData, keys, scan->numberOfKeys * sizeof(ScanKeyData));

View File

@@ -23,11 +23,15 @@
VectorArray VectorArray
VectorArrayInit(int maxlen, int dimensions, Size itemsize) VectorArrayInit(int maxlen, int dimensions, Size itemsize)
{ {
VectorArray res = palloc(sizeof(VectorArrayData)); VectorArray res;
if (maxlen < 1 || dimensions < 1 || itemsize == 0)
elog(ERROR, "cannot create vector array");
/* Ensure items are aligned to prevent UB */ /* Ensure items are aligned to prevent UB */
itemsize = MAXALIGN(itemsize); itemsize = MAXALIGN(itemsize);
res = palloc_object(VectorArrayData);
res->length = 0; res->length = 0;
res->maxlen = maxlen; res->maxlen = maxlen;
res->dim = dimensions; res->dim = dimensions;

View File

@@ -24,7 +24,7 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD); BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
if (stats == NULL) if (stats == NULL)
stats = (IndexBulkDeleteResult *) palloc0(sizeof(IndexBulkDeleteResult)); stats = palloc0_object(IndexBulkDeleteResult);
/* Iterate over list pages */ /* Iterate over list pages */
while (BlockNumberIsValid(blkno)) while (BlockNumberIsValid(blkno))

View File

@@ -26,6 +26,10 @@
#include "parser/scansup.h" #include "parser/scansup.h"
#endif #endif
#if PG_VERSION_NUM < 140006
#define palloc_array(type, count) ((type *) palloc(sizeof(type) * (count)))
#endif
typedef struct SparseInputElement typedef struct SparseInputElement
{ {
int32 index; int32 index;
@@ -223,7 +227,7 @@ sparsevec_in(PG_FUNCTION_ARGS)
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED), (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
errmsg("sparsevec cannot have more than %d non-zero elements", SPARSEVEC_MAX_NNZ))); errmsg("sparsevec cannot have more than %d non-zero elements", SPARSEVEC_MAX_NNZ)));
elements = palloc(maxNnz * sizeof(SparseInputElement)); elements = palloc_array(SparseInputElement, maxNnz);
pt = lit; pt = lit;
@@ -614,6 +618,7 @@ vector_to_sparsevec(PG_FUNCTION_ARGS)
nnz++; nnz++;
} }
CheckNnz(nnz, dim);
result = InitSparseVector(dim, nnz); result = InitSparseVector(dim, nnz);
values = SPARSEVEC_VALUES(result); values = SPARSEVEC_VALUES(result);
for (int i = 0; i < dim; i++) for (int i = 0; i < dim; i++)
@@ -657,6 +662,7 @@ halfvec_to_sparsevec(PG_FUNCTION_ARGS)
nnz++; nnz++;
} }
CheckNnz(nnz, dim);
result = InitSparseVector(dim, nnz); result = InitSparseVector(dim, nnz);
values = SPARSEVEC_VALUES(result); values = SPARSEVEC_VALUES(result);
for (int i = 0; i < dim; i++) for (int i = 0; i < dim; i++)
@@ -745,6 +751,7 @@ array_to_sparsevec(PG_FUNCTION_ARGS)
errmsg("unsupported array type"))); errmsg("unsupported array type")));
} }
CheckNnz(nnz, nelemsp);
result = InitSparseVector(nelemsp, nnz); result = InitSparseVector(nelemsp, nnz);
values = SPARSEVEC_VALUES(result); values = SPARSEVEC_VALUES(result);

View File

@@ -30,8 +30,12 @@
#include "parser/scansup.h" #include "parser/scansup.h"
#endif #endif
#if PG_VERSION_NUM < 140006
#define palloc_array(type, count) ((type *) palloc(sizeof(type) * (count)))
#endif
#define STATE_DIMS(x) (ARR_DIMS(x)[0] - 1) #define STATE_DIMS(x) (ARR_DIMS(x)[0] - 1)
#define CreateStateDatums(dim) palloc(sizeof(Datum) * (dim + 1)) #define CreateStateDatums(dim) palloc_array(Datum, (dim) + 1)
#if defined(USE_TARGET_CLONES) && !defined(__FMA__) #if defined(USE_TARGET_CLONES) && !defined(__FMA__)
#define VECTOR_TARGET_CLONES __attribute__((target_clones("default", "fma"))) #define VECTOR_TARGET_CLONES __attribute__((target_clones("default", "fma")))
@@ -40,7 +44,7 @@
#endif #endif
#if PG_VERSION_NUM >= 180000 #if PG_VERSION_NUM >= 180000
PG_MODULE_MAGIC_EXT(.name = "vector", .version = "0.8.4"); PG_MODULE_MAGIC_EXT(.name = "vector", .version = "0.8.5");
#else #else
PG_MODULE_MAGIC; PG_MODULE_MAGIC;
#endif #endif
@@ -516,7 +520,7 @@ vector_to_float4(PG_FUNCTION_ARGS)
Datum *datums; Datum *datums;
ArrayType *result; ArrayType *result;
datums = (Datum *) palloc(sizeof(Datum) * vec->dim); datums = palloc_array(Datum, vec->dim);
for (int i = 0; i < vec->dim; i++) for (int i = 0; i < vec->dim; i++)
datums[i] = Float4GetDatum(vec->x[i]); datums[i] = Float4GetDatum(vec->x[i]);

View File

@@ -268,6 +268,10 @@ SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
ERROR: vector cannot have more than 16000 dimensions ERROR: vector cannot have more than 16000 dimensions
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n; SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
ERROR: vector cannot have more than 16000 dimensions ERROR: vector cannot have more than 16000 dimensions
SELECT array_agg(n)::halfvec FROM generate_series(1, 16001) n;
ERROR: halfvec cannot have more than 16000 dimensions
SELECT array_agg(n)::sparsevec FROM generate_series(1, 16001) n;
ERROR: sparsevec cannot have more than 16000 non-zero elements
-- ensure no error -- ensure no error
SELECT ARRAY[1,2,3] = ARRAY[1,2,3]; SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
?column? ?column?

View File

@@ -49,3 +49,11 @@ CREATE INDEX ON t USING hnsw ((val::bit(3)) bit_hamming_ops);
CREATE INDEX ON t USING hnsw ((val::bit(64001)) bit_hamming_ops); CREATE INDEX ON t USING hnsw ((val::bit(64001)) bit_hamming_ops);
ERROR: column cannot have more than 64000 dimensions for hnsw index ERROR: column cannot have more than 64000 dimensions for hnsw index
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING hnsw (val bit_hamming_ops);
DROP TABLE t;
CREATE TABLE t (val bit(64001));
CREATE INDEX ON t USING hnsw (val bit_hamming_ops);
ERROR: column cannot have more than 64000 dimensions for hnsw index
DROP TABLE t;

View File

@@ -100,3 +100,11 @@ SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <+> (SELECT NULL::halfvec)) t
(1 row) (1 row)
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING hnsw (val halfvec_l2_ops);
DROP TABLE t;
CREATE TABLE t (val halfvec(4001));
CREATE INDEX ON t USING hnsw (val halfvec_l2_ops);
ERROR: column cannot have more than 4000 dimensions for hnsw index
DROP TABLE t;

View File

@@ -78,6 +78,14 @@ SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2
3 3
(1 row) (1 row)
SELECT * FROM t CROSS JOIN LATERAL (SELECT * FROM t t2 ORDER BY val <=> t.val LIMIT 1) t2 WHERE t.val != '[0,0,0]' ORDER BY t.val;
val | val
---------+---------
[1,1,1] | [1,1,1]
[1,2,3] | [1,2,3]
[1,2,4] | [1,2,4]
(3 rows)
DROP TABLE t; DROP TABLE t;
-- L1 -- L1
CREATE TABLE t (val vector(3)); CREATE TABLE t (val vector(3));
@@ -161,6 +169,7 @@ ERROR: value 1001 out of bounds for option "ef_construction"
DETAIL: Valid values are between "4" and "1000". DETAIL: Valid values are between "4" and "1000".
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31); CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31);
ERROR: ef_construction must be greater than or equal to 2 * m ERROR: ef_construction must be greater than or equal to 2 * m
DROP TABLE t;
SHOW hnsw.ef_search; SHOW hnsw.ef_search;
hnsw.ef_search hnsw.ef_search
---------------- ----------------
@@ -198,4 +207,11 @@ SET hnsw.scan_mem_multiplier = 0;
ERROR: 0 is outside the valid range for parameter "hnsw.scan_mem_multiplier" (1 .. 1000) ERROR: 0 is outside the valid range for parameter "hnsw.scan_mem_multiplier" (1 .. 1000)
SET hnsw.scan_mem_multiplier = 1001; SET hnsw.scan_mem_multiplier = 1001;
ERROR: 1001 is outside the valid range for parameter "hnsw.scan_mem_multiplier" (1 .. 1000) ERROR: 1001 is outside the valid range for parameter "hnsw.scan_mem_multiplier" (1 .. 1000)
-- dimensions
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
DROP TABLE t;
CREATE TABLE t (val vector(2001));
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
ERROR: column cannot have more than 2000 dimensions for hnsw index
DROP TABLE t; DROP TABLE t;

View File

@@ -35,3 +35,32 @@ NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall. DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data. HINT: Drop the index until the table has more data.
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
CREATE TABLE t (val bit(64001));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
ERROR: column cannot have more than 64000 dimensions for ivfflat index
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '29MB';
CREATE TABLE t (val bit(64000));
INSERT INTO t (val) VALUES (B'0'::bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -82,3 +82,32 @@ SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::halfvec)) t
(1 row) (1 row)
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
CREATE TABLE t (val halfvec(4001));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
ERROR: column cannot have more than 4000 dimensions for ivfflat index
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '6MB';
CREATE TABLE t (val halfvec(4000));
INSERT INTO t (val) VALUES (array_fill(0, ARRAY[4000]));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -81,6 +81,14 @@ SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2
3 3
(1 row) (1 row)
SELECT * FROM t CROSS JOIN LATERAL (SELECT * FROM t t2 ORDER BY val <=> t.val LIMIT 1) t2 WHERE t.val != '[0,0,0]' ORDER BY t.val;
val | val
---------+---------
[1,1,1] | [1,1,1]
[1,2,3] | [1,2,3]
[1,2,4] | [1,2,4]
(3 rows)
DROP TABLE t; DROP TABLE t;
-- iterative -- iterative
CREATE TABLE t (val vector(3)); CREATE TABLE t (val vector(3));
@@ -143,6 +151,7 @@ DETAIL: Valid values are between "1" and "32768".
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769); CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769);
ERROR: value 32769 out of bounds for option "lists" ERROR: value 32769 out of bounds for option "lists"
DETAIL: Valid values are between "1" and "32768". DETAIL: Valid values are between "1" and "32768".
DROP TABLE t;
SHOW ivfflat.probes; SHOW ivfflat.probes;
ivfflat.probes ivfflat.probes
---------------- ----------------
@@ -172,4 +181,32 @@ SET ivfflat.max_probes = 0;
ERROR: 0 is outside the valid range for parameter "ivfflat.max_probes" (1 .. 32768) ERROR: 0 is outside the valid range for parameter "ivfflat.max_probes" (1 .. 32768)
SET ivfflat.max_probes = 32769; SET ivfflat.max_probes = 32769;
ERROR: 32769 is outside the valid range for parameter "ivfflat.max_probes" (1 .. 32768) ERROR: 32769 is outside the valid range for parameter "ivfflat.max_probes" (1 .. 32768)
-- dimensions
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t; DROP TABLE t;
CREATE TABLE t (val vector(2001));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
ERROR: column cannot have more than 2000 dimensions for ivfflat index
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '5MB';
CREATE TABLE t (val vector(2000));
INSERT INTO t (val) VALUES (array_fill(0, ARRAY[2000]));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
NOTICE: ivfflat index created with little data
DETAIL: This will cause low recall.
HINT: Drop the index until the table has more data.
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -76,6 +76,8 @@ SELECT '{{1}}'::real[]::sparsevec;
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n; SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n; SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
SELECT array_agg(n)::halfvec FROM generate_series(1, 16001) n;
SELECT array_agg(n)::sparsevec FROM generate_series(1, 16001) n;
-- ensure no error -- ensure no error
SELECT ARRAY[1,2,3] = ARRAY[1,2,3]; SELECT ARRAY[1,2,3] = ARRAY[1,2,3];

View File

@@ -33,3 +33,13 @@ CREATE INDEX ON t USING hnsw (val bit_hamming_ops);
CREATE INDEX ON t USING hnsw ((val::bit(3)) bit_hamming_ops); CREATE INDEX ON t USING hnsw ((val::bit(3)) bit_hamming_ops);
CREATE INDEX ON t USING hnsw ((val::bit(64001)) bit_hamming_ops); CREATE INDEX ON t USING hnsw ((val::bit(64001)) bit_hamming_ops);
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING hnsw (val bit_hamming_ops);
DROP TABLE t;
CREATE TABLE t (val bit(64001));
CREATE INDEX ON t USING hnsw (val bit_hamming_ops);
DROP TABLE t;

View File

@@ -56,3 +56,13 @@ SELECT * FROM t ORDER BY val <+> '[3,3,3]';
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <+> (SELECT NULL::halfvec)) t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <+> (SELECT NULL::halfvec)) t2;
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING hnsw (val halfvec_l2_ops);
DROP TABLE t;
CREATE TABLE t (val halfvec(4001));
CREATE INDEX ON t USING hnsw (val halfvec_l2_ops);
DROP TABLE t;

View File

@@ -41,6 +41,7 @@ INSERT INTO t (val) VALUES ('[1,2,4]');
SELECT * FROM t ORDER BY val <=> '[3,3,3]'; SELECT * FROM t ORDER BY val <=> '[3,3,3]';
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
SELECT * FROM t CROSS JOIN LATERAL (SELECT * FROM t t2 ORDER BY val <=> t.val LIMIT 1) t2 WHERE t.val != '[0,0,0]' ORDER BY t.val;
DROP TABLE t; DROP TABLE t;
@@ -95,23 +96,28 @@ CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 101);
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 3); CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 3);
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 1001); CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 1001);
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31); CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31);
DROP TABLE t;
SHOW hnsw.ef_search; SHOW hnsw.ef_search;
SET hnsw.ef_search = 0; SET hnsw.ef_search = 0;
SET hnsw.ef_search = 1001; SET hnsw.ef_search = 1001;
SHOW hnsw.iterative_scan; SHOW hnsw.iterative_scan;
SET hnsw.iterative_scan = on; SET hnsw.iterative_scan = on;
SHOW hnsw.max_scan_tuples; SHOW hnsw.max_scan_tuples;
SET hnsw.max_scan_tuples = 0; SET hnsw.max_scan_tuples = 0;
SHOW hnsw.scan_mem_multiplier; SHOW hnsw.scan_mem_multiplier;
SET hnsw.scan_mem_multiplier = 0; SET hnsw.scan_mem_multiplier = 0;
SET hnsw.scan_mem_multiplier = 1001; SET hnsw.scan_mem_multiplier = 1001;
-- dimensions
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
DROP TABLE t;
CREATE TABLE t (val vector(2001));
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
DROP TABLE t; DROP TABLE t;

View File

@@ -21,3 +21,28 @@ CREATE INDEX ON t USING ivfflat ((val::bit(3)) bit_hamming_ops) WITH (lists = 1)
CREATE INDEX ON t USING ivfflat ((val::bit(64001)) bit_hamming_ops) WITH (lists = 1); CREATE INDEX ON t USING ivfflat ((val::bit(64001)) bit_hamming_ops) WITH (lists = 1);
CREATE INDEX ON t USING ivfflat ((val::bit(2)) bit_hamming_ops) WITH (lists = 5); CREATE INDEX ON t USING ivfflat ((val::bit(2)) bit_hamming_ops) WITH (lists = 5);
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
DROP TABLE t;
CREATE TABLE t (val bit(64001));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '29MB';
CREATE TABLE t (val bit(64000));
INSERT INTO t (val) VALUES (B'0'::bit(64000));
CREATE INDEX ON t USING ivfflat (val bit_hamming_ops);
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -43,3 +43,28 @@ SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::halfvec)) t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::halfvec)) t2;
DROP TABLE t; DROP TABLE t;
-- dimensions
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
DROP TABLE t;
CREATE TABLE t (val halfvec(4001));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val halfvec(4000));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '6MB';
CREATE TABLE t (val halfvec(4000));
INSERT INTO t (val) VALUES (array_fill(0, ARRAY[4000]));
CREATE INDEX ON t USING ivfflat (val halfvec_l2_ops);
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -41,6 +41,7 @@ INSERT INTO t (val) VALUES ('[1,2,4]');
SELECT * FROM t ORDER BY val <=> '[3,3,3]'; SELECT * FROM t ORDER BY val <=> '[3,3,3]';
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2; SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
SELECT * FROM t CROSS JOIN LATERAL (SELECT * FROM t t2 ORDER BY val <=> t.val LIMIT 1) t2 WHERE t.val != '[0,0,0]' ORDER BY t.val;
DROP TABLE t; DROP TABLE t;
@@ -81,19 +82,40 @@ DROP TABLE t;
CREATE TABLE t (val vector(3)); CREATE TABLE t (val vector(3));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 0); CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 0);
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769); CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769);
DROP TABLE t;
SHOW ivfflat.probes; SHOW ivfflat.probes;
SET ivfflat.probes = 0; SET ivfflat.probes = 0;
SET ivfflat.probes = 32769; SET ivfflat.probes = 32769;
SHOW ivfflat.iterative_scan; SHOW ivfflat.iterative_scan;
SET ivfflat.iterative_scan = on; SET ivfflat.iterative_scan = on;
SHOW ivfflat.max_probes; SHOW ivfflat.max_probes;
SET ivfflat.max_probes = 0; SET ivfflat.max_probes = 0;
SET ivfflat.max_probes = 32769; SET ivfflat.max_probes = 32769;
-- dimensions
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
DROP TABLE t; DROP TABLE t;
CREATE TABLE t (val vector(2001));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
DROP TABLE t;
-- memory
SET maintenance_work_mem = '1MB';
CREATE TABLE t (val vector(2000));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
DROP TABLE t;
RESET maintenance_work_mem;
SET maintenance_work_mem = '5MB';
CREATE TABLE t (val vector(2000));
INSERT INTO t (val) VALUES (array_fill(0, ARRAY[2000]));
CREATE INDEX ON t USING ivfflat (val vector_l2_ops);
DROP TABLE t;
RESET maintenance_work_mem;

View File

@@ -31,7 +31,7 @@ $node->pgbench(
{ {
"047_hnsw_vacuum_insert_insert\@500" => "INSERT INTO tst (v) VALUES (ARRAY[$array_sql]);", "047_hnsw_vacuum_insert_insert\@500" => "INSERT INTO tst (v) VALUES (ARRAY[$array_sql]);",
"047_hnsw_vacuum_insert_delete\@500" => "DELETE FROM tst WHERE i = (SELECT i FROM tst LIMIT 1);", "047_hnsw_vacuum_insert_delete\@500" => "DELETE FROM tst WHERE i = (SELECT i FROM tst LIMIT 1);",
"047_hnsw_vacuum_insert_select\@20" => "SELECT i FROM tst ORDER BY v <-> '[0,0,0]' LIMIT 10;", "047_hnsw_vacuum_insert_select\@20" => "SELECT i FROM tst ORDER BY v <-> (SELECT ARRAY[$array_sql]::vector) LIMIT 10;",
"047_hnsw_vacuum_insert_vacuum\@1" => "VACUUM tst;" "047_hnsw_vacuum_insert_vacuum\@1" => "VACUUM tst;"
} }
); );

View File

@@ -0,0 +1,39 @@
use strict;
use warnings FATAL => 'all';
use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
my $dim = 3;
my $array_sql = join(",", ('random()') x $dim);
# Initialize node
my $node = PostgreSQL::Test::Cluster->new('node');
$node->init;
$node->start;
# Create table and index
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
$node->safe_psql("postgres", "CREATE TABLE tst (i serial, v vector($dim));");
$node->safe_psql("postgres", "ALTER TABLE tst SET (autovacuum_enabled = false);");
$node->safe_psql("postgres",
"INSERT INTO tst (v) SELECT ARRAY[$array_sql] FROM generate_series(1, 1000) i;"
);
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 10);");
# Test no errors
$node->pgbench(
"--no-vacuum --client=5 --transactions=1500",
0,
[qr{actually processed}],
[qr{^$}],
"concurrent INSERTs, DELETEs, SELECTs, and VACUUM",
{
"048_ivfflat_vacuum_insert_insert\@500" => "INSERT INTO tst (v) VALUES (ARRAY[$array_sql]);",
"048_ivfflat_vacuum_insert_delete\@500" => "DELETE FROM tst WHERE i = (SELECT i FROM tst LIMIT 1);",
"048_ivfflat_vacuum_insert_select\@500" => "SELECT i FROM tst ORDER BY v <-> (SELECT ARRAY[$array_sql]::vector) LIMIT 10;",
"048_ivfflat_vacuum_insert_vacuum\@1" => "VACUUM tst;"
}
);
done_testing();

View File

@@ -1,4 +1,4 @@
comment = 'vector data type and ivfflat and hnsw access methods' comment = 'vector data type and ivfflat and hnsw access methods'
default_version = '0.8.4' default_version = '0.8.5'
module_pathname = '$libdir/vector' module_pathname = '$libdir/vector'
relocatable = true relocatable = true