mirror of
https://github.com/pgvector/pgvector.git
synced 2026-07-22 12:07:34 +08:00
Compare commits
2 Commits
subscript
...
parallel-i
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
54870b9bdc | ||
|
|
0950968c11 |
58
.github/workflows/build.yml
vendored
58
.github/workflows/build.yml
vendored
@@ -8,10 +8,6 @@ jobs:
|
|||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- postgres: 17
|
|
||||||
os: ubuntu-22.04
|
|
||||||
- postgres: 16
|
|
||||||
os: ubuntu-22.04
|
|
||||||
- postgres: 15
|
- postgres: 15
|
||||||
os: ubuntu-22.04
|
os: ubuntu-22.04
|
||||||
- postgres: 14
|
- postgres: 14
|
||||||
@@ -20,15 +16,15 @@ jobs:
|
|||||||
os: ubuntu-20.04
|
os: ubuntu-20.04
|
||||||
- postgres: 12
|
- postgres: 12
|
||||||
os: ubuntu-20.04
|
os: ubuntu-20.04
|
||||||
|
- postgres: 11
|
||||||
|
os: ubuntu-18.04
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v3
|
||||||
- uses: ankane/setup-postgres@v1
|
- uses: ankane/setup-postgres@v1
|
||||||
with:
|
with:
|
||||||
postgres-version: ${{ matrix.postgres }}
|
postgres-version: ${{ matrix.postgres }}
|
||||||
dev-files: true
|
dev-files: true
|
||||||
- run: make
|
- run: make
|
||||||
env:
|
|
||||||
PG_CFLAGS: -DUSE_ASSERT_CHECKING -Wall -Wextra -Werror -Wno-unused-parameter -Wno-sign-compare
|
|
||||||
- run: |
|
- run: |
|
||||||
export PG_CONFIG=`which pg_config`
|
export PG_CONFIG=`which pg_config`
|
||||||
sudo --preserve-env=PG_CONFIG make install
|
sudo --preserve-env=PG_CONFIG make install
|
||||||
@@ -43,13 +39,11 @@ jobs:
|
|||||||
runs-on: macos-latest
|
runs-on: macos-latest
|
||||||
if: ${{ !startsWith(github.ref_name, 'windows') }}
|
if: ${{ !startsWith(github.ref_name, 'windows') }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v3
|
||||||
- uses: ankane/setup-postgres@v1
|
- uses: ankane/setup-postgres@v1
|
||||||
with:
|
with:
|
||||||
postgres-version: 14
|
postgres-version: 14
|
||||||
- run: make
|
- run: make
|
||||||
env:
|
|
||||||
PG_CFLAGS: -DUSE_ASSERT_CHECKING -Wall -Wextra -Werror -Wno-unused-parameter
|
|
||||||
- run: make install
|
- run: make install
|
||||||
- run: make installcheck
|
- run: make installcheck
|
||||||
- if: ${{ failure() }}
|
- if: ${{ failure() }}
|
||||||
@@ -57,58 +51,22 @@ jobs:
|
|||||||
- run: |
|
- run: |
|
||||||
brew install cpanm
|
brew install cpanm
|
||||||
cpanm --notest IPC::Run
|
cpanm --notest IPC::Run
|
||||||
wget -q https://github.com/postgres/postgres/archive/refs/tags/REL_14_10.tar.gz
|
wget -q https://github.com/postgres/postgres/archive/refs/tags/REL_14_5.tar.gz
|
||||||
tar xf REL_14_10.tar.gz
|
tar xf REL_14_5.tar.gz
|
||||||
- run: make prove_installcheck PROVE_FLAGS="-I ./postgres-REL_14_10/src/test/perl" PERL5LIB="/Users/runner/perl5/lib/perl5"
|
- run: make prove_installcheck PROVE_FLAGS="-I ./postgres-REL_14_5/src/test/perl" PERL5LIB="/Users/runner/perl5/lib/perl5"
|
||||||
- run: make clean && /usr/local/opt/llvm@15/bin/scan-build --status-bugs make
|
|
||||||
env:
|
|
||||||
PG_CFLAGS: -DUSE_ASSERT_CHECKING
|
|
||||||
windows:
|
windows:
|
||||||
runs-on: windows-latest
|
runs-on: windows-latest
|
||||||
if: ${{ !startsWith(github.ref_name, 'mac') }}
|
if: ${{ !startsWith(github.ref_name, 'mac') }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v3
|
||||||
- uses: ankane/setup-postgres@v1
|
- uses: ankane/setup-postgres@v1
|
||||||
with:
|
with:
|
||||||
postgres-version: 14
|
postgres-version: 14
|
||||||
- run: |
|
- run: |
|
||||||
call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvars64.bat" && ^
|
call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvars64.bat" && ^
|
||||||
cd %TEMP% && ^
|
|
||||||
nmake /NOLOGO /F Makefile.win && ^
|
nmake /NOLOGO /F Makefile.win && ^
|
||||||
nmake /NOLOGO /F Makefile.win install && ^
|
nmake /NOLOGO /F Makefile.win install && ^
|
||||||
nmake /NOLOGO /F Makefile.win installcheck && ^
|
nmake /NOLOGO /F Makefile.win installcheck && ^
|
||||||
nmake /NOLOGO /F Makefile.win clean && ^
|
nmake /NOLOGO /F Makefile.win clean && ^
|
||||||
nmake /NOLOGO /F Makefile.win uninstall
|
nmake /NOLOGO /F Makefile.win uninstall
|
||||||
shell: cmd
|
shell: cmd
|
||||||
i386:
|
|
||||||
if: ${{ !startsWith(github.ref_name, 'mac') && !startsWith(github.ref_name, 'windows') }}
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
container:
|
|
||||||
image: debian:12
|
|
||||||
options: --platform linux/386
|
|
||||||
steps:
|
|
||||||
- run: apt-get update && apt-get install -y build-essential git libipc-run-perl postgresql-15 postgresql-server-dev-15 sudo
|
|
||||||
- run: service postgresql start
|
|
||||||
- run: |
|
|
||||||
git clone https://github.com/${{ github.repository }}.git pgvector
|
|
||||||
cd pgvector
|
|
||||||
git fetch origin ${{ github.ref }}
|
|
||||||
git reset --hard FETCH_HEAD
|
|
||||||
make
|
|
||||||
make install
|
|
||||||
chown -R postgres .
|
|
||||||
sudo -u postgres make installcheck
|
|
||||||
sudo -u postgres make prove_installcheck
|
|
||||||
env:
|
|
||||||
PG_CFLAGS: -DUSE_ASSERT_CHECKING -Wall -Wextra -Werror -Wno-unused-parameter -Wno-sign-compare
|
|
||||||
valgrind:
|
|
||||||
if: ${{ !startsWith(github.ref_name, 'mac') && !startsWith(github.ref_name, 'windows') }}
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: ankane/setup-postgres-valgrind@v1
|
|
||||||
with:
|
|
||||||
postgres-version: 16
|
|
||||||
- run: make
|
|
||||||
- run: sudo --preserve-env=PG_CONFIG make install
|
|
||||||
- run: make installcheck
|
|
||||||
|
|||||||
73
CHANGELOG.md
73
CHANGELOG.md
@@ -1,72 +1,3 @@
|
|||||||
## 0.7.0 (unreleased)
|
|
||||||
|
|
||||||
- Added subscript function for vectors
|
|
||||||
|
|
||||||
## 0.6.2 (unreleased)
|
|
||||||
|
|
||||||
- Reduced lock contention with parallel HNSW index builds
|
|
||||||
|
|
||||||
## 0.6.1 (2024-03-04)
|
|
||||||
|
|
||||||
- Fixed error with `ANALYZE` and vectors with different dimensions
|
|
||||||
- Fixed error with `shared_preload_libraries`
|
|
||||||
- Fixed vector subtraction being marked as commutative
|
|
||||||
|
|
||||||
## 0.6.0 (2024-01-29)
|
|
||||||
|
|
||||||
If upgrading with Postgres 12 or Docker, see [these notes](https://github.com/pgvector/pgvector#060).
|
|
||||||
|
|
||||||
- Added support for parallel index builds for HNSW
|
|
||||||
- Added validation for GUC parameters
|
|
||||||
- Changed storage for vector from `extended` to `external`
|
|
||||||
- Improved performance of HNSW
|
|
||||||
- Reduced memory usage for HNSW index builds
|
|
||||||
- Reduced WAL generation for HNSW index builds
|
|
||||||
- Fixed error with logical replication
|
|
||||||
- Fixed `invalid memory alloc request size` error with HNSW index builds
|
|
||||||
- Moved Docker image to `pgvector` org
|
|
||||||
- Added Docker tags for each supported version of Postgres
|
|
||||||
- Dropped support for Postgres 11
|
|
||||||
|
|
||||||
## 0.5.1 (2023-10-10)
|
|
||||||
|
|
||||||
- Improved performance of HNSW index builds
|
|
||||||
- Added check for MVCC-compliant snapshot for index scans
|
|
||||||
|
|
||||||
## 0.5.0 (2023-08-28)
|
|
||||||
|
|
||||||
- Added HNSW index type
|
|
||||||
- Added support for parallel index builds for IVFFlat
|
|
||||||
- Added `l1_distance` function
|
|
||||||
- Added element-wise multiplication for vectors
|
|
||||||
- Added `sum` aggregate
|
|
||||||
- Improved performance of distance functions
|
|
||||||
- Fixed out of range results for cosine distance
|
|
||||||
- Fixed results for NULL and NaN distances for IVFFlat
|
|
||||||
|
|
||||||
## 0.4.4 (2023-06-12)
|
|
||||||
|
|
||||||
- Improved error message for malformed vector literal
|
|
||||||
- Fixed segmentation fault with text input
|
|
||||||
- Fixed consecutive delimiters with text input
|
|
||||||
|
|
||||||
## 0.4.3 (2023-06-10)
|
|
||||||
|
|
||||||
- Improved cost estimation
|
|
||||||
- Improved support for spaces with text input
|
|
||||||
- Fixed infinite and NaN values with binary input
|
|
||||||
- Fixed infinite values with vector addition and subtraction
|
|
||||||
- Fixed infinite values with list centers
|
|
||||||
- Fixed compilation error when `float8` is pass by reference
|
|
||||||
- Fixed compilation error on PowerPC
|
|
||||||
- Fixed segmentation fault with index creation on i386
|
|
||||||
|
|
||||||
## 0.4.2 (2023-05-13)
|
|
||||||
|
|
||||||
- Added notice when index created with little data
|
|
||||||
- Fixed dimensions check for some direct function calls
|
|
||||||
- Fixed installation error with Postgres 12.0-12.2
|
|
||||||
|
|
||||||
## 0.4.1 (2023-03-21)
|
## 0.4.1 (2023-03-21)
|
||||||
|
|
||||||
- Improved performance of cosine distance
|
- Improved performance of cosine distance
|
||||||
@@ -74,7 +5,7 @@ If upgrading with Postgres 12 or Docker, see [these notes](https://github.com/pg
|
|||||||
|
|
||||||
## 0.4.0 (2023-01-11)
|
## 0.4.0 (2023-01-11)
|
||||||
|
|
||||||
If upgrading with Postgres < 13, see [this note](https://github.com/pgvector/pgvector/blob/v0.4.0/README.md#040).
|
If upgrading with Postgres < 13, see [this note](https://github.com/pgvector/pgvector#040).
|
||||||
|
|
||||||
- Changed text representation for vector elements to match `real`
|
- Changed text representation for vector elements to match `real`
|
||||||
- Changed storage for vector from `plain` to `extended`
|
- Changed storage for vector from `plain` to `extended`
|
||||||
@@ -91,7 +22,7 @@ If upgrading with Postgres < 13, see [this note](https://github.com/pgvector/pgv
|
|||||||
|
|
||||||
## 0.3.1 (2022-11-02)
|
## 0.3.1 (2022-11-02)
|
||||||
|
|
||||||
If upgrading from 0.2.7 or 0.3.0, [recreate](https://github.com/pgvector/pgvector/blob/v0.3.1/README.md#031) all `ivfflat` indexes after upgrading to ensure all data is indexed.
|
If upgrading from 0.2.7 or 0.3.0, [recreate](https://github.com/pgvector/pgvector#031) all `ivfflat` indexes after upgrading to ensure all data is indexed.
|
||||||
|
|
||||||
- Fixed issue with inserts silently corrupting `ivfflat` indexes (introduced in 0.2.7)
|
- Fixed issue with inserts silently corrupting `ivfflat` indexes (introduced in 0.2.7)
|
||||||
- Fixed segmentation fault with index creation when lists > 6500
|
- Fixed segmentation fault with index creation when lists > 6500
|
||||||
|
|||||||
10
Dockerfile
10
Dockerfile
@@ -1,12 +1,9 @@
|
|||||||
ARG PG_MAJOR=16
|
FROM postgres:15
|
||||||
FROM postgres:$PG_MAJOR
|
|
||||||
ARG PG_MAJOR
|
|
||||||
|
|
||||||
COPY . /tmp/pgvector
|
COPY . /tmp/pgvector
|
||||||
|
|
||||||
RUN apt-get update && \
|
RUN apt-get update && \
|
||||||
apt-mark hold locales && \
|
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-15 && \
|
||||||
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-$PG_MAJOR && \
|
|
||||||
cd /tmp/pgvector && \
|
cd /tmp/pgvector && \
|
||||||
make clean && \
|
make clean && \
|
||||||
make OPTFLAGS="" && \
|
make OPTFLAGS="" && \
|
||||||
@@ -14,7 +11,6 @@ RUN apt-get update && \
|
|||||||
mkdir /usr/share/doc/pgvector && \
|
mkdir /usr/share/doc/pgvector && \
|
||||||
cp LICENSE README.md /usr/share/doc/pgvector && \
|
cp LICENSE README.md /usr/share/doc/pgvector && \
|
||||||
rm -r /tmp/pgvector && \
|
rm -r /tmp/pgvector && \
|
||||||
apt-get remove -y build-essential postgresql-server-dev-$PG_MAJOR && \
|
apt-get remove -y build-essential postgresql-server-dev-15 && \
|
||||||
apt-get autoremove -y && \
|
apt-get autoremove -y && \
|
||||||
apt-mark unhold locales && \
|
|
||||||
rm -rf /var/lib/apt/lists/*
|
rm -rf /var/lib/apt/lists/*
|
||||||
|
|||||||
2
LICENSE
2
LICENSE
@@ -1,4 +1,4 @@
|
|||||||
Portions Copyright (c) 1996-2024, PostgreSQL Global Development Group
|
Portions Copyright (c) 1996-2022, PostgreSQL Global Development Group
|
||||||
|
|
||||||
Portions Copyright (c) 1994, The Regents of the University of California
|
Portions Copyright (c) 1994, The Regents of the University of California
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
"name": "vector",
|
"name": "vector",
|
||||||
"abstract": "Open-source vector similarity search for Postgres",
|
"abstract": "Open-source vector similarity search for Postgres",
|
||||||
"description": "Supports L2 distance, inner product, and cosine distance",
|
"description": "Supports L2 distance, inner product, and cosine distance",
|
||||||
"version": "0.6.1",
|
"version": "0.4.1",
|
||||||
"maintainer": [
|
"maintainer": [
|
||||||
"Andrew Kane <andrew@ankane.org>"
|
"Andrew Kane <andrew@ankane.org>"
|
||||||
],
|
],
|
||||||
@@ -12,7 +12,7 @@
|
|||||||
"prereqs": {
|
"prereqs": {
|
||||||
"runtime": {
|
"runtime": {
|
||||||
"requires": {
|
"requires": {
|
||||||
"PostgreSQL": "12.0.0"
|
"PostgreSQL": "11.0.0"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -20,7 +20,7 @@
|
|||||||
"vector": {
|
"vector": {
|
||||||
"file": "sql/vector.sql",
|
"file": "sql/vector.sql",
|
||||||
"docfile": "README.md",
|
"docfile": "README.md",
|
||||||
"version": "0.6.1",
|
"version": "0.4.1",
|
||||||
"abstract": "Open-source vector similarity search for Postgres"
|
"abstract": "Open-source vector similarity search for Postgres"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|||||||
23
Makefile
23
Makefile
@@ -1,30 +1,23 @@
|
|||||||
EXTENSION = vector
|
EXTENSION = vector
|
||||||
EXTVERSION = 0.6.1
|
EXTVERSION = 0.4.1
|
||||||
|
|
||||||
MODULE_big = vector
|
MODULE_big = vector
|
||||||
DATA = $(wildcard sql/*--*.sql)
|
DATA = $(wildcard sql/*--*.sql)
|
||||||
OBJS = src/hnsw.o src/hnswbuild.o src/hnswinsert.o src/hnswscan.o src/hnswutils.o src/hnswvacuum.o src/ivfbuild.o src/ivfflat.o src/ivfinsert.o src/ivfkmeans.o src/ivfscan.o src/ivfutils.o src/ivfvacuum.o src/vector.o
|
OBJS = src/ivfbuild.o src/ivfflat.o src/ivfinsert.o src/ivfkmeans.o src/ivfscan.o src/ivfutils.o src/ivfvacuum.o src/vector.o
|
||||||
HEADERS = src/vector.h
|
|
||||||
|
|
||||||
TESTS = $(wildcard test/sql/*.sql)
|
TESTS = $(wildcard test/sql/*.sql)
|
||||||
REGRESS = $(patsubst test/sql/%.sql,%,$(TESTS))
|
REGRESS = $(patsubst test/sql/%.sql,%,$(TESTS))
|
||||||
REGRESS_OPTS = --inputdir=test --load-extension=$(EXTENSION)
|
REGRESS_OPTS = --inputdir=test --load-extension=vector
|
||||||
|
|
||||||
OPTFLAGS = -march=native
|
OPTFLAGS = -march=native
|
||||||
|
|
||||||
# Mac ARM doesn't support -march=native
|
# Mac ARM doesn't support -march=native
|
||||||
ifeq ($(shell uname -s), Darwin)
|
ifeq ($(shell uname -s), Darwin)
|
||||||
ifeq ($(shell uname -p), arm)
|
ifeq ($(shell uname -p), arm)
|
||||||
# no difference with -march=armv8.5-a
|
|
||||||
OPTFLAGS =
|
OPTFLAGS =
|
||||||
endif
|
endif
|
||||||
endif
|
endif
|
||||||
|
|
||||||
# PowerPC doesn't support -march=native
|
|
||||||
ifneq ($(filter ppc64%, $(shell uname -m)), )
|
|
||||||
OPTFLAGS =
|
|
||||||
endif
|
|
||||||
|
|
||||||
# For auto-vectorization:
|
# For auto-vectorization:
|
||||||
# - GCC (needs -ftree-vectorize OR -O3) - https://gcc.gnu.org/projects/tree-ssa/vectorization.html
|
# - GCC (needs -ftree-vectorize OR -O3) - https://gcc.gnu.org/projects/tree-ssa/vectorization.html
|
||||||
# - Clang (could use pragma instead) - https://llvm.org/docs/Vectorizers.html
|
# - Clang (could use pragma instead) - https://llvm.org/docs/Vectorizers.html
|
||||||
@@ -65,15 +58,7 @@ dist:
|
|||||||
mkdir -p dist
|
mkdir -p dist
|
||||||
git archive --format zip --prefix=$(EXTENSION)-$(EXTVERSION)/ --output dist/$(EXTENSION)-$(EXTVERSION).zip master
|
git archive --format zip --prefix=$(EXTENSION)-$(EXTVERSION)/ --output dist/$(EXTENSION)-$(EXTVERSION).zip master
|
||||||
|
|
||||||
# for Docker
|
|
||||||
PG_MAJOR ?= 16
|
|
||||||
|
|
||||||
.PHONY: docker
|
.PHONY: docker
|
||||||
|
|
||||||
docker:
|
docker:
|
||||||
docker build --pull --no-cache --build-arg PG_MAJOR=$(PG_MAJOR) -t pgvector/pgvector:pg$(PG_MAJOR) -t pgvector/pgvector:$(EXTVERSION)-pg$(PG_MAJOR) .
|
docker build --pull --no-cache --platform linux/amd64 -t ankane/pgvector:latest .
|
||||||
|
|
||||||
.PHONY: docker-release
|
|
||||||
|
|
||||||
docker-release:
|
|
||||||
docker buildx build --push --pull --no-cache --platform linux/amd64,linux/arm64 --build-arg PG_MAJOR=$(PG_MAJOR) -t pgvector/pgvector:pg$(PG_MAJOR) -t pgvector/pgvector:$(EXTVERSION)-pg$(PG_MAJOR) .
|
|
||||||
|
|||||||
13
Makefile.win
13
Makefile.win
@@ -1,11 +1,10 @@
|
|||||||
EXTENSION = vector
|
EXTENSION = vector
|
||||||
EXTVERSION = 0.6.1
|
EXTVERSION = 0.4.1
|
||||||
|
|
||||||
OBJS = src\hnsw.obj src\hnswbuild.obj src\hnswinsert.obj src\hnswscan.obj src\hnswutils.obj src\hnswvacuum.obj src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\vector.obj
|
OBJS = src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\vector.obj
|
||||||
HEADERS = src\vector.h
|
|
||||||
|
|
||||||
REGRESS = btree cast copy functions input ivfflat_cosine ivfflat_ip ivfflat_l2 ivfflat_options ivfflat_unlogged
|
REGRESS = btree cast copy functions input ivfflat_cosine ivfflat_ip ivfflat_l2 ivfflat_options ivfflat_unlogged
|
||||||
REGRESS_OPTS = --inputdir=test --load-extension=$(EXTENSION)
|
REGRESS_OPTS = --inputdir=test --load-extension=vector
|
||||||
|
|
||||||
# For /arch flags
|
# For /arch flags
|
||||||
# https://learn.microsoft.com/en-us/cpp/build/reference/arch-minimum-cpu-architecture
|
# https://learn.microsoft.com/en-us/cpp/build/reference/arch-minimum-cpu-architecture
|
||||||
@@ -55,8 +54,6 @@ install:
|
|||||||
copy $(SHLIB) "$(PKGLIBDIR)"
|
copy $(SHLIB) "$(PKGLIBDIR)"
|
||||||
copy $(EXTENSION).control "$(SHAREDIR)\extension"
|
copy $(EXTENSION).control "$(SHAREDIR)\extension"
|
||||||
copy sql\$(EXTENSION)--*.sql "$(SHAREDIR)\extension"
|
copy sql\$(EXTENSION)--*.sql "$(SHAREDIR)\extension"
|
||||||
mkdir "$(INCLUDEDIR_SERVER)\extension\$(EXTENSION)"
|
|
||||||
for %f in ($(HEADERS)) do copy %f "$(INCLUDEDIR_SERVER)\extension\$(EXTENSION)"
|
|
||||||
|
|
||||||
installcheck:
|
installcheck:
|
||||||
"$(BINDIR)\pg_regress" --bindir="$(BINDIR)" $(REGRESS_OPTS) $(REGRESS)
|
"$(BINDIR)\pg_regress" --bindir="$(BINDIR)" $(REGRESS_OPTS) $(REGRESS)
|
||||||
@@ -64,9 +61,7 @@ installcheck:
|
|||||||
uninstall:
|
uninstall:
|
||||||
del /f "$(PKGLIBDIR)\$(SHLIB)"
|
del /f "$(PKGLIBDIR)\$(SHLIB)"
|
||||||
del /f "$(SHAREDIR)\extension\$(EXTENSION).control"
|
del /f "$(SHAREDIR)\extension\$(EXTENSION).control"
|
||||||
del /f "$(SHAREDIR)\extension\$(EXTENSION)--*.sql"
|
del /f "$(SHAREDIR)\extension\vector--*.sql"
|
||||||
del /f "$(INCLUDEDIR_SERVER)\extension\$(EXTENSION)\*.h"
|
|
||||||
rmdir "$(INCLUDEDIR_SERVER)\extension\$(EXTENSION)"
|
|
||||||
|
|
||||||
clean:
|
clean:
|
||||||
del /f $(SHLIB) $(EXTENSION).lib $(EXTENSION).exp
|
del /f $(SHLIB) $(EXTENSION).lib $(EXTENSION).exp
|
||||||
|
|||||||
@@ -1,2 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.2'" to load this file. \quit
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.3'" to load this file. \quit
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.4'" to load this file. \quit
|
|
||||||
@@ -1,43 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.5.0'" to load this file. \quit
|
|
||||||
|
|
||||||
CREATE FUNCTION l1_distance(vector, vector) RETURNS float8
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
CREATE FUNCTION vector_mul(vector, vector) RETURNS vector
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
CREATE OPERATOR * (
|
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_mul,
|
|
||||||
COMMUTATOR = *
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE AGGREGATE sum(vector) (
|
|
||||||
SFUNC = vector_add,
|
|
||||||
STYPE = vector,
|
|
||||||
COMBINEFUNC = vector_add,
|
|
||||||
PARALLEL = SAFE
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE FUNCTION hnswhandler(internal) RETURNS index_am_handler
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C;
|
|
||||||
|
|
||||||
CREATE ACCESS METHOD hnsw TYPE INDEX HANDLER hnswhandler;
|
|
||||||
|
|
||||||
COMMENT ON ACCESS METHOD hnsw IS 'hnsw index access method';
|
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_l2_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <-> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_l2_squared_distance(vector, vector);
|
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_ip_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <#> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_negative_inner_product(vector, vector);
|
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_cosine_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <=> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_negative_inner_product(vector, vector),
|
|
||||||
FUNCTION 2 vector_norm(vector);
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.5.1'" to load this file. \quit
|
|
||||||
@@ -1,5 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.6.0'" to load this file. \quit
|
|
||||||
|
|
||||||
-- remove this single line for Postgres < 13
|
|
||||||
ALTER TYPE vector SET (STORAGE = external);
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.6.1'" to load this file. \quit
|
|
||||||
|
|
||||||
DROP OPERATOR - (vector, vector);
|
|
||||||
|
|
||||||
CREATE OPERATOR - (
|
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_sub
|
|
||||||
);
|
|
||||||
|
|
||||||
ALTER OPERATOR <= (vector, vector) SET (
|
|
||||||
RESTRICT = scalarlesel, JOIN = scalarlejoinsel
|
|
||||||
);
|
|
||||||
|
|
||||||
ALTER OPERATOR >= (vector, vector) SET (
|
|
||||||
RESTRICT = scalargesel, JOIN = scalargejoinsel
|
|
||||||
);
|
|
||||||
@@ -1,7 +0,0 @@
|
|||||||
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
|
||||||
\echo Use "ALTER EXTENSION vector UPDATE TO '0.7.0'" to load this file. \quit
|
|
||||||
|
|
||||||
CREATE FUNCTION vector_subscript(internal) RETURNS internal
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
ALTER TYPE vector SET (SUBSCRIPT = vector_subscript);
|
|
||||||
@@ -20,17 +20,13 @@ CREATE FUNCTION vector_recv(internal, oid, integer) RETURNS vector
|
|||||||
CREATE FUNCTION vector_send(vector) RETURNS bytea
|
CREATE FUNCTION vector_send(vector) RETURNS bytea
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
||||||
|
|
||||||
CREATE FUNCTION vector_subscript(internal) RETURNS internal
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
CREATE TYPE vector (
|
CREATE TYPE vector (
|
||||||
INPUT = vector_in,
|
INPUT = vector_in,
|
||||||
OUTPUT = vector_out,
|
OUTPUT = vector_out,
|
||||||
TYPMOD_IN = vector_typmod_in,
|
TYPMOD_IN = vector_typmod_in,
|
||||||
RECEIVE = vector_recv,
|
RECEIVE = vector_recv,
|
||||||
SEND = vector_send,
|
SEND = vector_send,
|
||||||
SUBSCRIPT = vector_subscript,
|
STORAGE = extended
|
||||||
STORAGE = external
|
|
||||||
);
|
);
|
||||||
|
|
||||||
-- functions
|
-- functions
|
||||||
@@ -44,9 +40,6 @@ CREATE FUNCTION inner_product(vector, vector) RETURNS float8
|
|||||||
CREATE FUNCTION cosine_distance(vector, vector) RETURNS float8
|
CREATE FUNCTION cosine_distance(vector, vector) RETURNS float8
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
||||||
|
|
||||||
CREATE FUNCTION l1_distance(vector, vector) RETURNS float8
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
CREATE FUNCTION vector_dims(vector) RETURNS integer
|
CREATE FUNCTION vector_dims(vector) RETURNS integer
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
||||||
|
|
||||||
@@ -59,9 +52,6 @@ CREATE FUNCTION vector_add(vector, vector) RETURNS vector
|
|||||||
CREATE FUNCTION vector_sub(vector, vector) RETURNS vector
|
CREATE FUNCTION vector_sub(vector, vector) RETURNS vector
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
||||||
|
|
||||||
CREATE FUNCTION vector_mul(vector, vector) RETURNS vector
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C IMMUTABLE STRICT PARALLEL SAFE;
|
|
||||||
|
|
||||||
-- private functions
|
-- private functions
|
||||||
|
|
||||||
CREATE FUNCTION vector_lt(vector, vector) RETURNS bool
|
CREATE FUNCTION vector_lt(vector, vector) RETURNS bool
|
||||||
@@ -114,13 +104,6 @@ CREATE AGGREGATE avg(vector) (
|
|||||||
PARALLEL = SAFE
|
PARALLEL = SAFE
|
||||||
);
|
);
|
||||||
|
|
||||||
CREATE AGGREGATE sum(vector) (
|
|
||||||
SFUNC = vector_add,
|
|
||||||
STYPE = vector,
|
|
||||||
COMBINEFUNC = vector_add,
|
|
||||||
PARALLEL = SAFE
|
|
||||||
);
|
|
||||||
|
|
||||||
-- cast functions
|
-- cast functions
|
||||||
|
|
||||||
CREATE FUNCTION vector(vector, integer, boolean) RETURNS vector
|
CREATE FUNCTION vector(vector, integer, boolean) RETURNS vector
|
||||||
@@ -184,12 +167,8 @@ CREATE OPERATOR + (
|
|||||||
);
|
);
|
||||||
|
|
||||||
CREATE OPERATOR - (
|
CREATE OPERATOR - (
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_sub
|
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_sub,
|
||||||
);
|
COMMUTATOR = -
|
||||||
|
|
||||||
CREATE OPERATOR * (
|
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_mul,
|
|
||||||
COMMUTATOR = *
|
|
||||||
);
|
);
|
||||||
|
|
||||||
CREATE OPERATOR < (
|
CREATE OPERATOR < (
|
||||||
@@ -198,10 +177,11 @@ CREATE OPERATOR < (
|
|||||||
RESTRICT = scalarltsel, JOIN = scalarltjoinsel
|
RESTRICT = scalarltsel, JOIN = scalarltjoinsel
|
||||||
);
|
);
|
||||||
|
|
||||||
|
-- should use scalarlesel and scalarlejoinsel, but not supported in Postgres < 11
|
||||||
CREATE OPERATOR <= (
|
CREATE OPERATOR <= (
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_le,
|
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_le,
|
||||||
COMMUTATOR = >= , NEGATOR = > ,
|
COMMUTATOR = >= , NEGATOR = > ,
|
||||||
RESTRICT = scalarlesel, JOIN = scalarlejoinsel
|
RESTRICT = scalarltsel, JOIN = scalarltjoinsel
|
||||||
);
|
);
|
||||||
|
|
||||||
CREATE OPERATOR = (
|
CREATE OPERATOR = (
|
||||||
@@ -216,10 +196,11 @@ CREATE OPERATOR <> (
|
|||||||
RESTRICT = eqsel, JOIN = eqjoinsel
|
RESTRICT = eqsel, JOIN = eqjoinsel
|
||||||
);
|
);
|
||||||
|
|
||||||
|
-- should use scalargesel and scalargejoinsel, but not supported in Postgres < 11
|
||||||
CREATE OPERATOR >= (
|
CREATE OPERATOR >= (
|
||||||
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_ge,
|
LEFTARG = vector, RIGHTARG = vector, PROCEDURE = vector_ge,
|
||||||
COMMUTATOR = <= , NEGATOR = < ,
|
COMMUTATOR = <= , NEGATOR = < ,
|
||||||
RESTRICT = scalargesel, JOIN = scalargejoinsel
|
RESTRICT = scalargtsel, JOIN = scalargtjoinsel
|
||||||
);
|
);
|
||||||
|
|
||||||
CREATE OPERATOR > (
|
CREATE OPERATOR > (
|
||||||
@@ -228,7 +209,7 @@ CREATE OPERATOR > (
|
|||||||
RESTRICT = scalargtsel, JOIN = scalargtjoinsel
|
RESTRICT = scalargtsel, JOIN = scalargtjoinsel
|
||||||
);
|
);
|
||||||
|
|
||||||
-- access methods
|
-- access method
|
||||||
|
|
||||||
CREATE FUNCTION ivfflathandler(internal) RETURNS index_am_handler
|
CREATE FUNCTION ivfflathandler(internal) RETURNS index_am_handler
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C;
|
AS 'MODULE_PATHNAME' LANGUAGE C;
|
||||||
@@ -237,13 +218,6 @@ CREATE ACCESS METHOD ivfflat TYPE INDEX HANDLER ivfflathandler;
|
|||||||
|
|
||||||
COMMENT ON ACCESS METHOD ivfflat IS 'ivfflat index access method';
|
COMMENT ON ACCESS METHOD ivfflat IS 'ivfflat index access method';
|
||||||
|
|
||||||
CREATE FUNCTION hnswhandler(internal) RETURNS index_am_handler
|
|
||||||
AS 'MODULE_PATHNAME' LANGUAGE C;
|
|
||||||
|
|
||||||
CREATE ACCESS METHOD hnsw TYPE INDEX HANDLER hnswhandler;
|
|
||||||
|
|
||||||
COMMENT ON ACCESS METHOD hnsw IS 'hnsw index access method';
|
|
||||||
|
|
||||||
-- opclasses
|
-- opclasses
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_ops
|
CREATE OPERATOR CLASS vector_ops
|
||||||
@@ -275,19 +249,3 @@ CREATE OPERATOR CLASS vector_cosine_ops
|
|||||||
FUNCTION 2 vector_norm(vector),
|
FUNCTION 2 vector_norm(vector),
|
||||||
FUNCTION 3 vector_spherical_distance(vector, vector),
|
FUNCTION 3 vector_spherical_distance(vector, vector),
|
||||||
FUNCTION 4 vector_norm(vector);
|
FUNCTION 4 vector_norm(vector);
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_l2_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <-> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_l2_squared_distance(vector, vector);
|
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_ip_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <#> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_negative_inner_product(vector, vector);
|
|
||||||
|
|
||||||
CREATE OPERATOR CLASS vector_cosine_ops
|
|
||||||
FOR TYPE vector USING hnsw AS
|
|
||||||
OPERATOR 1 <=> (vector, vector) FOR ORDER BY float_ops,
|
|
||||||
FUNCTION 1 vector_negative_inner_product(vector, vector),
|
|
||||||
FUNCTION 2 vector_norm(vector);
|
|
||||||
|
|||||||
249
src/hnsw.c
249
src/hnsw.c
@@ -1,249 +0,0 @@
|
|||||||
#include "postgres.h"
|
|
||||||
|
|
||||||
#include <float.h>
|
|
||||||
#include <math.h>
|
|
||||||
|
|
||||||
#include "access/amapi.h"
|
|
||||||
#include "access/reloptions.h"
|
|
||||||
#include "commands/progress.h"
|
|
||||||
#include "commands/vacuum.h"
|
|
||||||
#include "hnsw.h"
|
|
||||||
#include "miscadmin.h"
|
|
||||||
#include "utils/guc.h"
|
|
||||||
#include "utils/selfuncs.h"
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 150000
|
|
||||||
#define MarkGUCPrefixReserved(x) EmitWarningsOnPlaceholders(x)
|
|
||||||
#endif
|
|
||||||
|
|
||||||
int hnsw_ef_search;
|
|
||||||
int hnsw_lock_tranche_id;
|
|
||||||
static relopt_kind hnsw_relopt_kind;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Assign a tranche ID for our LWLocks. This only needs to be done by one
|
|
||||||
* backend, as the tranche ID is remembered in shared memory.
|
|
||||||
*
|
|
||||||
* This shared memory area is very small, so we just allocate it from the
|
|
||||||
* "slop" that PostgreSQL reserves for small allocations like this. If
|
|
||||||
* this grows bigger, we should use a shmem_request_hook and
|
|
||||||
* RequestAddinShmemSpace() to pre-reserve space for this.
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
HnswInitLockTranche(void)
|
|
||||||
{
|
|
||||||
int *tranche_ids;
|
|
||||||
bool found;
|
|
||||||
|
|
||||||
LWLockAcquire(AddinShmemInitLock, LW_EXCLUSIVE);
|
|
||||||
tranche_ids = ShmemInitStruct("hnsw LWLock ids",
|
|
||||||
sizeof(int) * 1,
|
|
||||||
&found);
|
|
||||||
if (!found)
|
|
||||||
tranche_ids[0] = LWLockNewTrancheId();
|
|
||||||
hnsw_lock_tranche_id = tranche_ids[0];
|
|
||||||
LWLockRelease(AddinShmemInitLock);
|
|
||||||
|
|
||||||
/* Per-backend registration of the tranche ID */
|
|
||||||
LWLockRegisterTranche(hnsw_lock_tranche_id, "HnswBuild");
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Initialize index options and variables
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
HnswInit(void)
|
|
||||||
{
|
|
||||||
if (!process_shared_preload_libraries_in_progress)
|
|
||||||
HnswInitLockTranche();
|
|
||||||
|
|
||||||
hnsw_relopt_kind = add_reloption_kind();
|
|
||||||
add_int_reloption(hnsw_relopt_kind, "m", "Max number of connections",
|
|
||||||
HNSW_DEFAULT_M, HNSW_MIN_M, HNSW_MAX_M
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
,AccessExclusiveLock
|
|
||||||
#endif
|
|
||||||
);
|
|
||||||
add_int_reloption(hnsw_relopt_kind, "ef_construction", "Size of the dynamic candidate list for construction",
|
|
||||||
HNSW_DEFAULT_EF_CONSTRUCTION, HNSW_MIN_EF_CONSTRUCTION, HNSW_MAX_EF_CONSTRUCTION
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
,AccessExclusiveLock
|
|
||||||
#endif
|
|
||||||
);
|
|
||||||
|
|
||||||
DefineCustomIntVariable("hnsw.ef_search", "Sets the size of the dynamic candidate list for search",
|
|
||||||
"Valid range is 1..1000.", &hnsw_ef_search,
|
|
||||||
HNSW_DEFAULT_EF_SEARCH, HNSW_MIN_EF_SEARCH, HNSW_MAX_EF_SEARCH, PGC_USERSET, 0, NULL, NULL, NULL);
|
|
||||||
|
|
||||||
MarkGUCPrefixReserved("hnsw");
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get the name of index build phase
|
|
||||||
*/
|
|
||||||
static char *
|
|
||||||
hnswbuildphasename(int64 phasenum)
|
|
||||||
{
|
|
||||||
switch (phasenum)
|
|
||||||
{
|
|
||||||
case PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE:
|
|
||||||
return "initializing";
|
|
||||||
case PROGRESS_HNSW_PHASE_LOAD:
|
|
||||||
return "loading tuples";
|
|
||||||
default:
|
|
||||||
return NULL;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Estimate the cost of an index scan
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
hnswcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
|
||||||
Cost *indexStartupCost, Cost *indexTotalCost,
|
|
||||||
Selectivity *indexSelectivity, double *indexCorrelation,
|
|
||||||
double *indexPages)
|
|
||||||
{
|
|
||||||
GenericCosts costs;
|
|
||||||
int m;
|
|
||||||
int entryLevel;
|
|
||||||
Relation index;
|
|
||||||
|
|
||||||
/* Never use index without order */
|
|
||||||
if (path->indexorderbys == NULL)
|
|
||||||
{
|
|
||||||
*indexStartupCost = DBL_MAX;
|
|
||||||
*indexTotalCost = DBL_MAX;
|
|
||||||
*indexSelectivity = 0;
|
|
||||||
*indexCorrelation = 0;
|
|
||||||
*indexPages = 0;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
MemSet(&costs, 0, sizeof(costs));
|
|
||||||
|
|
||||||
index = index_open(path->indexinfo->indexoid, NoLock);
|
|
||||||
HnswGetMetaPageInfo(index, &m, NULL);
|
|
||||||
index_close(index, NoLock);
|
|
||||||
|
|
||||||
/* Approximate entry level */
|
|
||||||
entryLevel = (int) -log(1.0 / path->indexinfo->tuples) * HnswGetMl(m);
|
|
||||||
|
|
||||||
/* TODO Improve estimate of visited tuples (currently underestimates) */
|
|
||||||
/* Account for number of tuples (or entry level), m, and ef_search */
|
|
||||||
costs.numIndexTuples = (entryLevel + 2) * m;
|
|
||||||
|
|
||||||
genericcostestimate(root, path, loop_count, &costs);
|
|
||||||
|
|
||||||
/* Use total cost since most work happens before first tuple is returned */
|
|
||||||
*indexStartupCost = costs.indexTotalCost;
|
|
||||||
*indexTotalCost = costs.indexTotalCost;
|
|
||||||
*indexSelectivity = costs.indexSelectivity;
|
|
||||||
*indexCorrelation = costs.indexCorrelation;
|
|
||||||
*indexPages = costs.numIndexPages;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Parse and validate the reloptions
|
|
||||||
*/
|
|
||||||
static bytea *
|
|
||||||
hnswoptions(Datum reloptions, bool validate)
|
|
||||||
{
|
|
||||||
static const relopt_parse_elt tab[] = {
|
|
||||||
{"m", RELOPT_TYPE_INT, offsetof(HnswOptions, m)},
|
|
||||||
{"ef_construction", RELOPT_TYPE_INT, offsetof(HnswOptions, efConstruction)},
|
|
||||||
};
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
return (bytea *) build_reloptions(reloptions, validate,
|
|
||||||
hnsw_relopt_kind,
|
|
||||||
sizeof(HnswOptions),
|
|
||||||
tab, lengthof(tab));
|
|
||||||
#else
|
|
||||||
relopt_value *options;
|
|
||||||
int numoptions;
|
|
||||||
HnswOptions *rdopts;
|
|
||||||
|
|
||||||
options = parseRelOptions(reloptions, validate, hnsw_relopt_kind, &numoptions);
|
|
||||||
rdopts = allocateReloptStruct(sizeof(HnswOptions), options, numoptions);
|
|
||||||
fillRelOptions((void *) rdopts, sizeof(HnswOptions), options, numoptions,
|
|
||||||
validate, tab, lengthof(tab));
|
|
||||||
|
|
||||||
return (bytea *) rdopts;
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Validate catalog entries for the specified operator class
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
hnswvalidate(Oid opclassoid)
|
|
||||||
{
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Define index handler
|
|
||||||
*
|
|
||||||
* See https://www.postgresql.org/docs/current/index-api.html
|
|
||||||
*/
|
|
||||||
PGDLLEXPORT PG_FUNCTION_INFO_V1(hnswhandler);
|
|
||||||
Datum
|
|
||||||
hnswhandler(PG_FUNCTION_ARGS)
|
|
||||||
{
|
|
||||||
IndexAmRoutine *amroutine = makeNode(IndexAmRoutine);
|
|
||||||
|
|
||||||
amroutine->amstrategies = 0;
|
|
||||||
amroutine->amsupport = 2;
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
amroutine->amoptsprocnum = 0;
|
|
||||||
#endif
|
|
||||||
amroutine->amcanorder = false;
|
|
||||||
amroutine->amcanorderbyop = true;
|
|
||||||
amroutine->amcanbackward = false; /* can change direction mid-scan */
|
|
||||||
amroutine->amcanunique = false;
|
|
||||||
amroutine->amcanmulticol = false;
|
|
||||||
amroutine->amoptionalkey = true;
|
|
||||||
amroutine->amsearcharray = false;
|
|
||||||
amroutine->amsearchnulls = false;
|
|
||||||
amroutine->amstorage = false;
|
|
||||||
amroutine->amclusterable = false;
|
|
||||||
amroutine->ampredlocks = false;
|
|
||||||
amroutine->amcanparallel = false;
|
|
||||||
amroutine->amcaninclude = false;
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
|
|
||||||
amroutine->amparallelvacuumoptions = VACUUM_OPTION_PARALLEL_BULKDEL;
|
|
||||||
#endif
|
|
||||||
amroutine->amkeytype = InvalidOid;
|
|
||||||
|
|
||||||
/* Interface functions */
|
|
||||||
amroutine->ambuild = hnswbuild;
|
|
||||||
amroutine->ambuildempty = hnswbuildempty;
|
|
||||||
amroutine->aminsert = hnswinsert;
|
|
||||||
amroutine->ambulkdelete = hnswbulkdelete;
|
|
||||||
amroutine->amvacuumcleanup = hnswvacuumcleanup;
|
|
||||||
amroutine->amcanreturn = NULL;
|
|
||||||
amroutine->amcostestimate = hnswcostestimate;
|
|
||||||
amroutine->amoptions = hnswoptions;
|
|
||||||
amroutine->amproperty = NULL; /* TODO AMPROP_DISTANCE_ORDERABLE */
|
|
||||||
amroutine->ambuildphasename = hnswbuildphasename;
|
|
||||||
amroutine->amvalidate = hnswvalidate;
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
|
||||||
amroutine->amadjustmembers = NULL;
|
|
||||||
#endif
|
|
||||||
amroutine->ambeginscan = hnswbeginscan;
|
|
||||||
amroutine->amrescan = hnswrescan;
|
|
||||||
amroutine->amgettuple = hnswgettuple;
|
|
||||||
amroutine->amgetbitmap = NULL;
|
|
||||||
amroutine->amendscan = hnswendscan;
|
|
||||||
amroutine->ammarkpos = NULL;
|
|
||||||
amroutine->amrestrpos = NULL;
|
|
||||||
|
|
||||||
/* Interface functions to support parallel index scans */
|
|
||||||
amroutine->amestimateparallelscan = NULL;
|
|
||||||
amroutine->aminitparallelscan = NULL;
|
|
||||||
amroutine->amparallelrescan = NULL;
|
|
||||||
|
|
||||||
PG_RETURN_POINTER(amroutine);
|
|
||||||
}
|
|
||||||
462
src/hnsw.h
462
src/hnsw.h
@@ -1,462 +0,0 @@
|
|||||||
#ifndef HNSW_H
|
|
||||||
#define HNSW_H
|
|
||||||
|
|
||||||
#include "postgres.h"
|
|
||||||
|
|
||||||
#include "access/genam.h"
|
|
||||||
#include "access/parallel.h"
|
|
||||||
#include "lib/pairingheap.h"
|
|
||||||
#include "nodes/execnodes.h"
|
|
||||||
#include "port.h" /* for random() */
|
|
||||||
#include "utils/relptr.h"
|
|
||||||
#include "utils/sampling.h"
|
|
||||||
#include "vector.h"
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 120000
|
|
||||||
#error "Requires PostgreSQL 12+"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define HNSW_MAX_DIM 2000
|
|
||||||
|
|
||||||
/* Support functions */
|
|
||||||
#define HNSW_DISTANCE_PROC 1
|
|
||||||
#define HNSW_NORM_PROC 2
|
|
||||||
|
|
||||||
#define HNSW_VERSION 1
|
|
||||||
#define HNSW_MAGIC_NUMBER 0xA953A953
|
|
||||||
#define HNSW_PAGE_ID 0xFF90
|
|
||||||
|
|
||||||
/* Preserved page numbers */
|
|
||||||
#define HNSW_METAPAGE_BLKNO 0
|
|
||||||
#define HNSW_HEAD_BLKNO 1 /* first element page */
|
|
||||||
|
|
||||||
/* Must correspond to page numbers since page lock is used */
|
|
||||||
#define HNSW_UPDATE_LOCK 0
|
|
||||||
#define HNSW_SCAN_LOCK 1
|
|
||||||
|
|
||||||
/* HNSW parameters */
|
|
||||||
#define HNSW_DEFAULT_M 16
|
|
||||||
#define HNSW_MIN_M 2
|
|
||||||
#define HNSW_MAX_M 100
|
|
||||||
#define HNSW_DEFAULT_EF_CONSTRUCTION 64
|
|
||||||
#define HNSW_MIN_EF_CONSTRUCTION 4
|
|
||||||
#define HNSW_MAX_EF_CONSTRUCTION 1000
|
|
||||||
#define HNSW_DEFAULT_EF_SEARCH 40
|
|
||||||
#define HNSW_MIN_EF_SEARCH 1
|
|
||||||
#define HNSW_MAX_EF_SEARCH 1000
|
|
||||||
|
|
||||||
/* Tuple types */
|
|
||||||
#define HNSW_ELEMENT_TUPLE_TYPE 1
|
|
||||||
#define HNSW_NEIGHBOR_TUPLE_TYPE 2
|
|
||||||
|
|
||||||
/* Make graph robust against non-HOT updates */
|
|
||||||
#define HNSW_HEAPTIDS 10
|
|
||||||
|
|
||||||
#define HNSW_UPDATE_ENTRY_GREATER 1
|
|
||||||
#define HNSW_UPDATE_ENTRY_ALWAYS 2
|
|
||||||
|
|
||||||
/* Build phases */
|
|
||||||
/* PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE is 1 */
|
|
||||||
#define PROGRESS_HNSW_PHASE_LOAD 2
|
|
||||||
|
|
||||||
#define HNSW_MAX_SIZE (BLCKSZ - MAXALIGN(SizeOfPageHeaderData) - MAXALIGN(sizeof(HnswPageOpaqueData)) - sizeof(ItemIdData))
|
|
||||||
#define HNSW_TUPLE_ALLOC_SIZE BLCKSZ
|
|
||||||
|
|
||||||
#define HNSW_ELEMENT_TUPLE_SIZE(size) MAXALIGN(offsetof(HnswElementTupleData, data) + (size))
|
|
||||||
#define HNSW_NEIGHBOR_TUPLE_SIZE(level, m) MAXALIGN(offsetof(HnswNeighborTupleData, indextids) + ((level) + 2) * (m) * sizeof(ItemPointerData))
|
|
||||||
|
|
||||||
#define HNSW_NEIGHBOR_ARRAY_SIZE(lm) (offsetof(HnswNeighborArray, items) + sizeof(HnswCandidate) * (lm))
|
|
||||||
|
|
||||||
#define HnswPageGetOpaque(page) ((HnswPageOpaque) PageGetSpecialPointer(page))
|
|
||||||
#define HnswPageGetMeta(page) ((HnswMetaPageData *) PageGetContents(page))
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 150000
|
|
||||||
#define RandomDouble() pg_prng_double(&pg_global_prng_state)
|
|
||||||
#define SeedRandom(seed) pg_prng_seed(&pg_global_prng_state, seed)
|
|
||||||
#else
|
|
||||||
#define RandomDouble() (((double) random()) / MAX_RANDOM_VALUE)
|
|
||||||
#define SeedRandom(seed) srandom(seed)
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 130000
|
|
||||||
#define list_delete_last(list) list_truncate(list, list_length(list) - 1)
|
|
||||||
#define list_sort(list, cmp) ((list) = list_qsort(list, cmp))
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define HnswIsElementTuple(tup) ((tup)->type == HNSW_ELEMENT_TUPLE_TYPE)
|
|
||||||
#define HnswIsNeighborTuple(tup) ((tup)->type == HNSW_NEIGHBOR_TUPLE_TYPE)
|
|
||||||
|
|
||||||
/* 2 * M connections for ground layer */
|
|
||||||
#define HnswGetLayerM(m, layer) (layer == 0 ? (m) * 2 : (m))
|
|
||||||
|
|
||||||
/* Optimal ML from paper */
|
|
||||||
#define HnswGetMl(m) (1 / log(m))
|
|
||||||
|
|
||||||
/* Ensure fits on page and in uint8 */
|
|
||||||
#define HnswGetMaxLevel(m) Min(((BLCKSZ - MAXALIGN(SizeOfPageHeaderData) - MAXALIGN(sizeof(HnswPageOpaqueData)) - offsetof(HnswNeighborTupleData, indextids) - sizeof(ItemIdData)) / (sizeof(ItemPointerData)) / (m)) - 2, 255)
|
|
||||||
|
|
||||||
#define HnswGetValue(base, element) PointerGetDatum(HnswPtrAccess(base, (element)->value))
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 140005
|
|
||||||
#define relptr_offset(rp) ((rp).relptr_off - 1)
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Pointer macros */
|
|
||||||
#define HnswPtrAccess(base, hp) ((base) == NULL ? (hp).ptr : relptr_access(base, (hp).relptr))
|
|
||||||
#define HnswPtrStore(base, hp, value) ((base) == NULL ? (void) ((hp).ptr = (value)) : (void) relptr_store(base, (hp).relptr, value))
|
|
||||||
#define HnswPtrIsNull(base, hp) ((base) == NULL ? (hp).ptr == NULL : relptr_is_null((hp).relptr))
|
|
||||||
#define HnswPtrEqual(base, hp1, hp2) ((base) == NULL ? (hp1).ptr == (hp2).ptr : relptr_offset((hp1).relptr) == relptr_offset((hp2).relptr))
|
|
||||||
|
|
||||||
/* For code paths dedicated to each type */
|
|
||||||
#define HnswPtrPointer(hp) (hp).ptr
|
|
||||||
#define HnswPtrOffset(hp) relptr_offset((hp).relptr)
|
|
||||||
|
|
||||||
/* Variables */
|
|
||||||
extern int hnsw_ef_search;
|
|
||||||
extern int hnsw_lock_tranche_id;
|
|
||||||
|
|
||||||
typedef struct HnswElementData HnswElementData;
|
|
||||||
typedef struct HnswNeighborArray HnswNeighborArray;
|
|
||||||
|
|
||||||
#define HnswPtrDeclare(type, relptrtype, ptrtype) \
|
|
||||||
relptr_declare(type, relptrtype); \
|
|
||||||
typedef union { type *ptr; relptrtype relptr; } ptrtype;
|
|
||||||
|
|
||||||
/* Pointers that can be absolute or relative */
|
|
||||||
/* Use char for DatumPtr so works with Pointer */
|
|
||||||
HnswPtrDeclare(HnswElementData, HnswElementRelptr, HnswElementPtr);
|
|
||||||
HnswPtrDeclare(HnswNeighborArray, HnswNeighborArrayRelptr, HnswNeighborArrayPtr);
|
|
||||||
HnswPtrDeclare(HnswNeighborArrayPtr, HnswNeighborsRelptr, HnswNeighborsPtr);
|
|
||||||
HnswPtrDeclare(char, DatumRelptr, DatumPtr);
|
|
||||||
|
|
||||||
struct HnswElementData
|
|
||||||
{
|
|
||||||
HnswElementPtr next;
|
|
||||||
ItemPointerData heaptids[HNSW_HEAPTIDS];
|
|
||||||
uint8 heaptidsLength;
|
|
||||||
uint8 level;
|
|
||||||
uint8 deleted;
|
|
||||||
uint32 hash;
|
|
||||||
HnswNeighborsPtr neighbors;
|
|
||||||
BlockNumber blkno;
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber neighborOffno;
|
|
||||||
BlockNumber neighborPage;
|
|
||||||
DatumPtr value;
|
|
||||||
LWLock lock;
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef HnswElementData * HnswElement;
|
|
||||||
|
|
||||||
typedef struct HnswCandidate
|
|
||||||
{
|
|
||||||
HnswElementPtr element;
|
|
||||||
float distance;
|
|
||||||
bool closer;
|
|
||||||
} HnswCandidate;
|
|
||||||
|
|
||||||
struct HnswNeighborArray
|
|
||||||
{
|
|
||||||
int length;
|
|
||||||
bool closerSet;
|
|
||||||
HnswCandidate items[FLEXIBLE_ARRAY_MEMBER];
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef struct HnswPairingHeapNode
|
|
||||||
{
|
|
||||||
pairingheap_node ph_node;
|
|
||||||
HnswCandidate *inner;
|
|
||||||
} HnswPairingHeapNode;
|
|
||||||
|
|
||||||
/* HNSW index options */
|
|
||||||
typedef struct HnswOptions
|
|
||||||
{
|
|
||||||
int32 vl_len_; /* varlena header (do not touch directly!) */
|
|
||||||
int m; /* number of connections */
|
|
||||||
int efConstruction; /* size of dynamic candidate list */
|
|
||||||
} HnswOptions;
|
|
||||||
|
|
||||||
typedef struct HnswGraph
|
|
||||||
{
|
|
||||||
/* Graph state */
|
|
||||||
slock_t lock;
|
|
||||||
HnswElementPtr head;
|
|
||||||
double indtuples;
|
|
||||||
|
|
||||||
/* Entry state */
|
|
||||||
LWLock entryLock;
|
|
||||||
LWLock entryWaitLock;
|
|
||||||
HnswElementPtr entryPoint;
|
|
||||||
|
|
||||||
/* Allocations state */
|
|
||||||
LWLock allocatorLock;
|
|
||||||
long memoryUsed;
|
|
||||||
long memoryTotal;
|
|
||||||
|
|
||||||
/* Flushed state */
|
|
||||||
LWLock flushLock;
|
|
||||||
bool flushed;
|
|
||||||
} HnswGraph;
|
|
||||||
|
|
||||||
typedef struct HnswShared
|
|
||||||
{
|
|
||||||
/* Immutable state */
|
|
||||||
Oid heaprelid;
|
|
||||||
Oid indexrelid;
|
|
||||||
bool isconcurrent;
|
|
||||||
|
|
||||||
/* Worker progress */
|
|
||||||
ConditionVariable workersdonecv;
|
|
||||||
|
|
||||||
/* Mutex for mutable state */
|
|
||||||
slock_t mutex;
|
|
||||||
|
|
||||||
/* Mutable state */
|
|
||||||
int nparticipantsdone;
|
|
||||||
double reltuples;
|
|
||||||
HnswGraph graphData;
|
|
||||||
} HnswShared;
|
|
||||||
|
|
||||||
#define ParallelTableScanFromHnswShared(shared) \
|
|
||||||
(ParallelTableScanDesc) ((char *) (shared) + BUFFERALIGN(sizeof(HnswShared)))
|
|
||||||
|
|
||||||
typedef struct HnswLeader
|
|
||||||
{
|
|
||||||
ParallelContext *pcxt;
|
|
||||||
int nparticipanttuplesorts;
|
|
||||||
HnswShared *hnswshared;
|
|
||||||
Snapshot snapshot;
|
|
||||||
char *hnswarea;
|
|
||||||
} HnswLeader;
|
|
||||||
|
|
||||||
typedef struct HnswAllocator
|
|
||||||
{
|
|
||||||
void *(*alloc) (Size size, void *state);
|
|
||||||
void *state;
|
|
||||||
} HnswAllocator;
|
|
||||||
|
|
||||||
typedef struct HnswBuildState
|
|
||||||
{
|
|
||||||
/* Info */
|
|
||||||
Relation heap;
|
|
||||||
Relation index;
|
|
||||||
IndexInfo *indexInfo;
|
|
||||||
ForkNumber forkNum;
|
|
||||||
|
|
||||||
/* Settings */
|
|
||||||
int dimensions;
|
|
||||||
int m;
|
|
||||||
int efConstruction;
|
|
||||||
|
|
||||||
/* Statistics */
|
|
||||||
double indtuples;
|
|
||||||
double reltuples;
|
|
||||||
|
|
||||||
/* Support functions */
|
|
||||||
FmgrInfo *procinfo;
|
|
||||||
FmgrInfo *normprocinfo;
|
|
||||||
Oid collation;
|
|
||||||
|
|
||||||
/* Variables */
|
|
||||||
HnswGraph graphData;
|
|
||||||
HnswGraph *graph;
|
|
||||||
double ml;
|
|
||||||
int maxLevel;
|
|
||||||
Vector *normvec;
|
|
||||||
|
|
||||||
/* Memory */
|
|
||||||
MemoryContext graphCtx;
|
|
||||||
MemoryContext tmpCtx;
|
|
||||||
HnswAllocator allocator;
|
|
||||||
|
|
||||||
/* Parallel builds */
|
|
||||||
HnswLeader *hnswleader;
|
|
||||||
HnswShared *hnswshared;
|
|
||||||
char *hnswarea;
|
|
||||||
} HnswBuildState;
|
|
||||||
|
|
||||||
typedef struct HnswMetaPageData
|
|
||||||
{
|
|
||||||
uint32 magicNumber;
|
|
||||||
uint32 version;
|
|
||||||
uint32 dimensions;
|
|
||||||
uint16 m;
|
|
||||||
uint16 efConstruction;
|
|
||||||
BlockNumber entryBlkno;
|
|
||||||
OffsetNumber entryOffno;
|
|
||||||
int16 entryLevel;
|
|
||||||
BlockNumber insertPage;
|
|
||||||
} HnswMetaPageData;
|
|
||||||
|
|
||||||
typedef HnswMetaPageData * HnswMetaPage;
|
|
||||||
|
|
||||||
typedef struct HnswPageOpaqueData
|
|
||||||
{
|
|
||||||
BlockNumber nextblkno;
|
|
||||||
uint16 unused;
|
|
||||||
uint16 page_id; /* for identification of HNSW indexes */
|
|
||||||
} HnswPageOpaqueData;
|
|
||||||
|
|
||||||
typedef HnswPageOpaqueData * HnswPageOpaque;
|
|
||||||
|
|
||||||
typedef struct HnswElementTupleData
|
|
||||||
{
|
|
||||||
uint8 type;
|
|
||||||
uint8 level;
|
|
||||||
uint8 deleted;
|
|
||||||
uint8 unused;
|
|
||||||
ItemPointerData heaptids[HNSW_HEAPTIDS];
|
|
||||||
ItemPointerData neighbortid;
|
|
||||||
uint16 unused2;
|
|
||||||
Vector data;
|
|
||||||
} HnswElementTupleData;
|
|
||||||
|
|
||||||
typedef HnswElementTupleData * HnswElementTuple;
|
|
||||||
|
|
||||||
typedef struct HnswNeighborTupleData
|
|
||||||
{
|
|
||||||
uint8 type;
|
|
||||||
uint8 unused;
|
|
||||||
uint16 count;
|
|
||||||
ItemPointerData indextids[FLEXIBLE_ARRAY_MEMBER];
|
|
||||||
} HnswNeighborTupleData;
|
|
||||||
|
|
||||||
typedef HnswNeighborTupleData * HnswNeighborTuple;
|
|
||||||
|
|
||||||
typedef struct HnswScanOpaqueData
|
|
||||||
{
|
|
||||||
bool first;
|
|
||||||
List *w;
|
|
||||||
MemoryContext tmpCtx;
|
|
||||||
|
|
||||||
/* Support functions */
|
|
||||||
FmgrInfo *procinfo;
|
|
||||||
FmgrInfo *normprocinfo;
|
|
||||||
Oid collation;
|
|
||||||
} HnswScanOpaqueData;
|
|
||||||
|
|
||||||
typedef HnswScanOpaqueData * HnswScanOpaque;
|
|
||||||
|
|
||||||
typedef struct HnswVacuumState
|
|
||||||
{
|
|
||||||
/* Info */
|
|
||||||
Relation index;
|
|
||||||
IndexBulkDeleteResult *stats;
|
|
||||||
IndexBulkDeleteCallback callback;
|
|
||||||
void *callback_state;
|
|
||||||
|
|
||||||
/* Settings */
|
|
||||||
int m;
|
|
||||||
int efConstruction;
|
|
||||||
|
|
||||||
/* Support functions */
|
|
||||||
FmgrInfo *procinfo;
|
|
||||||
Oid collation;
|
|
||||||
|
|
||||||
/* Variables */
|
|
||||||
struct tidhash_hash *deleted;
|
|
||||||
BufferAccessStrategy bas;
|
|
||||||
HnswNeighborTuple ntup;
|
|
||||||
HnswElementData highestPoint;
|
|
||||||
|
|
||||||
/* Memory */
|
|
||||||
MemoryContext tmpCtx;
|
|
||||||
} HnswVacuumState;
|
|
||||||
|
|
||||||
/* Methods */
|
|
||||||
int HnswGetM(Relation index);
|
|
||||||
int HnswGetEfConstruction(Relation index);
|
|
||||||
FmgrInfo *HnswOptionalProcInfo(Relation index, uint16 procnum);
|
|
||||||
bool HnswNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result);
|
|
||||||
Buffer HnswNewBuffer(Relation index, ForkNumber forkNum);
|
|
||||||
void HnswInitPage(Buffer buf, Page page);
|
|
||||||
void HnswInit(void);
|
|
||||||
List *HnswSearchLayer(char *base, Datum q, List *ep, int ef, int lc, Relation index, FmgrInfo *procinfo, Oid collation, int m, bool inserting, HnswElement skipElement);
|
|
||||||
HnswElement HnswGetEntryPoint(Relation index);
|
|
||||||
void HnswGetMetaPageInfo(Relation index, int *m, HnswElement * entryPoint);
|
|
||||||
void *HnswAlloc(HnswAllocator * allocator, Size size);
|
|
||||||
HnswElement HnswInitElement(char *base, ItemPointer tid, int m, double ml, int maxLevel, HnswAllocator * alloc);
|
|
||||||
HnswElement HnswInitElementFromBlock(BlockNumber blkno, OffsetNumber offno);
|
|
||||||
void HnswFindElementNeighbors(char *base, HnswElement element, HnswElement entryPoint, Relation index, FmgrInfo *procinfo, Oid collation, int m, int efConstruction, bool existing);
|
|
||||||
HnswCandidate *HnswEntryCandidate(char *base, HnswElement em, Datum q, Relation rel, FmgrInfo *procinfo, Oid collation, bool loadVec);
|
|
||||||
void HnswUpdateMetaPage(Relation index, int updateEntry, HnswElement entryPoint, BlockNumber insertPage, ForkNumber forkNum, bool building);
|
|
||||||
void HnswSetNeighborTuple(char *base, HnswNeighborTuple ntup, HnswElement e, int m);
|
|
||||||
void HnswAddHeapTid(HnswElement element, ItemPointer heaptid);
|
|
||||||
void HnswInitNeighbors(char *base, HnswElement element, int m, HnswAllocator * alloc);
|
|
||||||
bool HnswInsertTupleOnDisk(Relation index, Datum value, Datum *values, bool *isnull, ItemPointer heap_tid, bool building);
|
|
||||||
void HnswUpdateNeighborsOnDisk(Relation index, FmgrInfo *procinfo, Oid collation, HnswElement e, int m, bool checkExisting, bool building);
|
|
||||||
void HnswLoadElementFromTuple(HnswElement element, HnswElementTuple etup, bool loadHeaptids, bool loadVec);
|
|
||||||
void HnswLoadElement(HnswElement element, float *distance, Datum *q, Relation index, FmgrInfo *procinfo, Oid collation, bool loadVec);
|
|
||||||
void HnswSetElementTuple(char *base, HnswElementTuple etup, HnswElement element);
|
|
||||||
void HnswUpdateConnection(char *base, HnswElement element, HnswCandidate * hc, int lm, int lc, int *updateIdx, Relation index, FmgrInfo *procinfo, Oid collation);
|
|
||||||
void HnswLoadNeighbors(HnswElement element, Relation index, int m);
|
|
||||||
void HnswInitLockTranche(void);
|
|
||||||
PGDLLEXPORT void HnswParallelBuildMain(dsm_segment *seg, shm_toc *toc);
|
|
||||||
|
|
||||||
/* Index access methods */
|
|
||||||
IndexBuildResult *hnswbuild(Relation heap, Relation index, IndexInfo *indexInfo);
|
|
||||||
void hnswbuildempty(Relation index);
|
|
||||||
bool hnswinsert(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, Relation heap, IndexUniqueCheck checkUnique
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
|
||||||
,bool indexUnchanged
|
|
||||||
#endif
|
|
||||||
,IndexInfo *indexInfo
|
|
||||||
);
|
|
||||||
IndexBulkDeleteResult *hnswbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, IndexBulkDeleteCallback callback, void *callback_state);
|
|
||||||
IndexBulkDeleteResult *hnswvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats);
|
|
||||||
IndexScanDesc hnswbeginscan(Relation index, int nkeys, int norderbys);
|
|
||||||
void hnswrescan(IndexScanDesc scan, ScanKey keys, int nkeys, ScanKey orderbys, int norderbys);
|
|
||||||
bool hnswgettuple(IndexScanDesc scan, ScanDirection dir);
|
|
||||||
void hnswendscan(IndexScanDesc scan);
|
|
||||||
|
|
||||||
static inline HnswNeighborArray *
|
|
||||||
HnswGetNeighbors(char *base, HnswElement element, int lc)
|
|
||||||
{
|
|
||||||
HnswNeighborArrayPtr *neighborList = HnswPtrAccess(base, element->neighbors);
|
|
||||||
|
|
||||||
Assert(element->level >= lc);
|
|
||||||
|
|
||||||
return HnswPtrAccess(base, neighborList[lc]);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Hash tables */
|
|
||||||
typedef struct TidHashEntry
|
|
||||||
{
|
|
||||||
ItemPointerData tid;
|
|
||||||
char status;
|
|
||||||
} TidHashEntry;
|
|
||||||
|
|
||||||
#define SH_PREFIX tidhash
|
|
||||||
#define SH_ELEMENT_TYPE TidHashEntry
|
|
||||||
#define SH_KEY_TYPE ItemPointerData
|
|
||||||
#define SH_SCOPE extern
|
|
||||||
#define SH_DECLARE
|
|
||||||
#include "lib/simplehash.h"
|
|
||||||
|
|
||||||
typedef struct PointerHashEntry
|
|
||||||
{
|
|
||||||
uintptr_t ptr;
|
|
||||||
char status;
|
|
||||||
} PointerHashEntry;
|
|
||||||
|
|
||||||
#define SH_PREFIX pointerhash
|
|
||||||
#define SH_ELEMENT_TYPE PointerHashEntry
|
|
||||||
#define SH_KEY_TYPE uintptr_t
|
|
||||||
#define SH_SCOPE extern
|
|
||||||
#define SH_DECLARE
|
|
||||||
#include "lib/simplehash.h"
|
|
||||||
|
|
||||||
typedef struct OffsetHashEntry
|
|
||||||
{
|
|
||||||
Size offset;
|
|
||||||
char status;
|
|
||||||
} OffsetHashEntry;
|
|
||||||
|
|
||||||
#define SH_PREFIX offsethash
|
|
||||||
#define SH_ELEMENT_TYPE OffsetHashEntry
|
|
||||||
#define SH_KEY_TYPE Size
|
|
||||||
#define SH_SCOPE extern
|
|
||||||
#define SH_DECLARE
|
|
||||||
#include "lib/simplehash.h"
|
|
||||||
|
|
||||||
#endif
|
|
||||||
1149
src/hnswbuild.c
1149
src/hnswbuild.c
File diff suppressed because it is too large
Load Diff
665
src/hnswinsert.c
665
src/hnswinsert.c
@@ -1,665 +0,0 @@
|
|||||||
#include "postgres.h"
|
|
||||||
|
|
||||||
#include <math.h>
|
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
|
||||||
#include "hnsw.h"
|
|
||||||
#include "storage/bufmgr.h"
|
|
||||||
#include "storage/lmgr.h"
|
|
||||||
#include "utils/datum.h"
|
|
||||||
#include "utils/memutils.h"
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get the insert page
|
|
||||||
*/
|
|
||||||
static BlockNumber
|
|
||||||
GetInsertPage(Relation index)
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
HnswMetaPage metap;
|
|
||||||
BlockNumber insertPage;
|
|
||||||
|
|
||||||
buf = ReadBuffer(index, HNSW_METAPAGE_BLKNO);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
metap = HnswPageGetMeta(page);
|
|
||||||
|
|
||||||
insertPage = metap->insertPage;
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
return insertPage;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Check for a free offset
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
HnswFreeOffset(Relation index, Buffer buf, Page page, HnswElement element, Size ntupSize, Buffer *nbuf, Page *npage, OffsetNumber *freeOffno, OffsetNumber *freeNeighborOffno, BlockNumber *newInsertPage)
|
|
||||||
{
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber maxoffno = PageGetMaxOffsetNumber(page);
|
|
||||||
|
|
||||||
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
|
||||||
{
|
|
||||||
HnswElementTuple etup = (HnswElementTuple) PageGetItem(page, PageGetItemId(page, offno));
|
|
||||||
|
|
||||||
/* Skip neighbor tuples */
|
|
||||||
if (!HnswIsElementTuple(etup))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
if (etup->deleted)
|
|
||||||
{
|
|
||||||
BlockNumber elementPage = BufferGetBlockNumber(buf);
|
|
||||||
BlockNumber neighborPage = ItemPointerGetBlockNumber(&etup->neighbortid);
|
|
||||||
OffsetNumber neighborOffno = ItemPointerGetOffsetNumber(&etup->neighbortid);
|
|
||||||
ItemId itemid;
|
|
||||||
|
|
||||||
if (!BlockNumberIsValid(*newInsertPage))
|
|
||||||
*newInsertPage = elementPage;
|
|
||||||
|
|
||||||
if (neighborPage == elementPage)
|
|
||||||
{
|
|
||||||
*nbuf = buf;
|
|
||||||
*npage = page;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
*nbuf = ReadBuffer(index, neighborPage);
|
|
||||||
LockBuffer(*nbuf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
|
|
||||||
/* Skip WAL for now */
|
|
||||||
*npage = BufferGetPage(*nbuf);
|
|
||||||
}
|
|
||||||
|
|
||||||
itemid = PageGetItemId(*npage, neighborOffno);
|
|
||||||
|
|
||||||
/* Check for space on neighbor tuple page */
|
|
||||||
if (PageGetFreeSpace(*npage) + ItemIdGetLength(itemid) - sizeof(ItemIdData) >= ntupSize)
|
|
||||||
{
|
|
||||||
*freeOffno = offno;
|
|
||||||
*freeNeighborOffno = neighborOffno;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
else if (*nbuf != buf)
|
|
||||||
UnlockReleaseBuffer(*nbuf);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Add a new page
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswInsertAppendPage(Relation index, Buffer *nbuf, Page *npage, GenericXLogState *state, Page page, bool building)
|
|
||||||
{
|
|
||||||
/* Add a new page */
|
|
||||||
LockRelationForExtension(index, ExclusiveLock);
|
|
||||||
*nbuf = HnswNewBuffer(index, MAIN_FORKNUM);
|
|
||||||
UnlockRelationForExtension(index, ExclusiveLock);
|
|
||||||
|
|
||||||
/* Init new page */
|
|
||||||
if (building)
|
|
||||||
*npage = BufferGetPage(*nbuf);
|
|
||||||
else
|
|
||||||
*npage = GenericXLogRegisterBuffer(state, *nbuf, GENERIC_XLOG_FULL_IMAGE);
|
|
||||||
|
|
||||||
HnswInitPage(*nbuf, *npage);
|
|
||||||
|
|
||||||
/* Update previous buffer */
|
|
||||||
HnswPageGetOpaque(page)->nextblkno = BufferGetBlockNumber(*nbuf);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Add to element and neighbor pages
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
AddElementOnDisk(Relation index, HnswElement e, int m, BlockNumber insertPage, BlockNumber *updatedInsertPage, bool building)
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
Size etupSize;
|
|
||||||
Size ntupSize;
|
|
||||||
Size combinedSize;
|
|
||||||
Size maxSize;
|
|
||||||
Size minCombinedSize;
|
|
||||||
HnswElementTuple etup;
|
|
||||||
BlockNumber currentPage = insertPage;
|
|
||||||
HnswNeighborTuple ntup;
|
|
||||||
Buffer nbuf;
|
|
||||||
Page npage;
|
|
||||||
OffsetNumber freeOffno = InvalidOffsetNumber;
|
|
||||||
OffsetNumber freeNeighborOffno = InvalidOffsetNumber;
|
|
||||||
BlockNumber newInsertPage = InvalidBlockNumber;
|
|
||||||
char *base = NULL;
|
|
||||||
|
|
||||||
/* Calculate sizes */
|
|
||||||
etupSize = HNSW_ELEMENT_TUPLE_SIZE(VARSIZE_ANY(HnswPtrAccess(base, e->value)));
|
|
||||||
ntupSize = HNSW_NEIGHBOR_TUPLE_SIZE(e->level, m);
|
|
||||||
combinedSize = etupSize + ntupSize + sizeof(ItemIdData);
|
|
||||||
maxSize = HNSW_MAX_SIZE;
|
|
||||||
minCombinedSize = etupSize + HNSW_NEIGHBOR_TUPLE_SIZE(0, m) + sizeof(ItemIdData);
|
|
||||||
|
|
||||||
/* Prepare element tuple */
|
|
||||||
etup = palloc0(etupSize);
|
|
||||||
HnswSetElementTuple(base, etup, e);
|
|
||||||
|
|
||||||
/* Prepare neighbor tuple */
|
|
||||||
ntup = palloc0(ntupSize);
|
|
||||||
HnswSetNeighborTuple(base, ntup, e, m);
|
|
||||||
|
|
||||||
/* Find a page (or two if needed) to insert the tuples */
|
|
||||||
for (;;)
|
|
||||||
{
|
|
||||||
buf = ReadBuffer(index, currentPage);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
|
|
||||||
if (building)
|
|
||||||
{
|
|
||||||
state = NULL;
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Keep track of first page where element at level 0 can fit */
|
|
||||||
if (!BlockNumberIsValid(newInsertPage) && PageGetFreeSpace(page) >= minCombinedSize)
|
|
||||||
newInsertPage = currentPage;
|
|
||||||
|
|
||||||
/* First, try the fastest path */
|
|
||||||
/* Space for both tuples on the current page */
|
|
||||||
/* This can split existing tuples in rare cases */
|
|
||||||
if (PageGetFreeSpace(page) >= combinedSize)
|
|
||||||
{
|
|
||||||
nbuf = buf;
|
|
||||||
npage = page;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Next, try space from a deleted element */
|
|
||||||
if (HnswFreeOffset(index, buf, page, e, ntupSize, &nbuf, &npage, &freeOffno, &freeNeighborOffno, &newInsertPage))
|
|
||||||
{
|
|
||||||
if (nbuf != buf)
|
|
||||||
{
|
|
||||||
if (building)
|
|
||||||
npage = BufferGetPage(nbuf);
|
|
||||||
else
|
|
||||||
npage = GenericXLogRegisterBuffer(state, nbuf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Finally, try space for element only if last page */
|
|
||||||
/* Skip if both tuples can fit on the same page */
|
|
||||||
if (combinedSize > maxSize && PageGetFreeSpace(page) >= etupSize && !BlockNumberIsValid(HnswPageGetOpaque(page)->nextblkno))
|
|
||||||
{
|
|
||||||
HnswInsertAppendPage(index, &nbuf, &npage, state, page, building);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
currentPage = HnswPageGetOpaque(page)->nextblkno;
|
|
||||||
|
|
||||||
if (BlockNumberIsValid(currentPage))
|
|
||||||
{
|
|
||||||
/* Move to next page */
|
|
||||||
if (!building)
|
|
||||||
GenericXLogAbort(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
Buffer newbuf;
|
|
||||||
Page newpage;
|
|
||||||
|
|
||||||
HnswInsertAppendPage(index, &newbuf, &newpage, state, page, building);
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
if (building)
|
|
||||||
MarkBufferDirty(buf);
|
|
||||||
else
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
|
|
||||||
/* Unlock previous buffer */
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
/* Prepare new buffer */
|
|
||||||
buf = newbuf;
|
|
||||||
if (building)
|
|
||||||
{
|
|
||||||
state = NULL;
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Create new page for neighbors if needed */
|
|
||||||
if (PageGetFreeSpace(page) < combinedSize)
|
|
||||||
HnswInsertAppendPage(index, &nbuf, &npage, state, page, building);
|
|
||||||
else
|
|
||||||
{
|
|
||||||
nbuf = buf;
|
|
||||||
npage = page;
|
|
||||||
}
|
|
||||||
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
e->blkno = BufferGetBlockNumber(buf);
|
|
||||||
e->neighborPage = BufferGetBlockNumber(nbuf);
|
|
||||||
|
|
||||||
/* Added tuple to new page if newInsertPage is not set */
|
|
||||||
/* So can set to neighbor page instead of element page */
|
|
||||||
if (!BlockNumberIsValid(newInsertPage))
|
|
||||||
newInsertPage = e->neighborPage;
|
|
||||||
|
|
||||||
if (OffsetNumberIsValid(freeOffno))
|
|
||||||
{
|
|
||||||
e->offno = freeOffno;
|
|
||||||
e->neighborOffno = freeNeighborOffno;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
e->offno = OffsetNumberNext(PageGetMaxOffsetNumber(page));
|
|
||||||
if (nbuf == buf)
|
|
||||||
e->neighborOffno = OffsetNumberNext(e->offno);
|
|
||||||
else
|
|
||||||
e->neighborOffno = FirstOffsetNumber;
|
|
||||||
}
|
|
||||||
|
|
||||||
ItemPointerSet(&etup->neighbortid, e->neighborPage, e->neighborOffno);
|
|
||||||
|
|
||||||
/* Add element and neighbors */
|
|
||||||
if (OffsetNumberIsValid(freeOffno))
|
|
||||||
{
|
|
||||||
if (!PageIndexTupleOverwrite(page, e->offno, (Item) etup, etupSize))
|
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
|
||||||
|
|
||||||
if (!PageIndexTupleOverwrite(npage, e->neighborOffno, (Item) ntup, ntupSize))
|
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (PageAddItem(page, (Item) etup, etupSize, InvalidOffsetNumber, false, false) != e->offno)
|
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
|
||||||
|
|
||||||
if (PageAddItem(npage, (Item) ntup, ntupSize, InvalidOffsetNumber, false, false) != e->neighborOffno)
|
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
if (building)
|
|
||||||
{
|
|
||||||
MarkBufferDirty(buf);
|
|
||||||
if (nbuf != buf)
|
|
||||||
MarkBufferDirty(nbuf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
if (nbuf != buf)
|
|
||||||
UnlockReleaseBuffer(nbuf);
|
|
||||||
|
|
||||||
/* Update the insert page */
|
|
||||||
if (BlockNumberIsValid(newInsertPage) && newInsertPage != insertPage)
|
|
||||||
*updatedInsertPage = newInsertPage;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Check if connection already exists
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
ConnectionExists(HnswElement e, HnswNeighborTuple ntup, int startIdx, int lm)
|
|
||||||
{
|
|
||||||
for (int i = 0; i < lm; i++)
|
|
||||||
{
|
|
||||||
ItemPointer indextid = &ntup->indextids[startIdx + i];
|
|
||||||
|
|
||||||
if (!ItemPointerIsValid(indextid))
|
|
||||||
break;
|
|
||||||
|
|
||||||
if (ItemPointerGetBlockNumber(indextid) == e->blkno && ItemPointerGetOffsetNumber(indextid) == e->offno)
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Update neighbors
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
HnswUpdateNeighborsOnDisk(Relation index, FmgrInfo *procinfo, Oid collation, HnswElement e, int m, bool checkExisting, bool building)
|
|
||||||
{
|
|
||||||
char *base = NULL;
|
|
||||||
|
|
||||||
for (int lc = e->level; lc >= 0; lc--)
|
|
||||||
{
|
|
||||||
int lm = HnswGetLayerM(m, lc);
|
|
||||||
HnswNeighborArray *neighbors = HnswGetNeighbors(base, e, lc);
|
|
||||||
|
|
||||||
for (int i = 0; i < neighbors->length; i++)
|
|
||||||
{
|
|
||||||
HnswCandidate *hc = &neighbors->items[i];
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
HnswNeighborTuple ntup;
|
|
||||||
int idx = -1;
|
|
||||||
int startIdx;
|
|
||||||
HnswElement neighborElement = HnswPtrAccess(base, hc->element);
|
|
||||||
OffsetNumber offno = neighborElement->neighborOffno;
|
|
||||||
|
|
||||||
/* Get latest neighbors since they may have changed */
|
|
||||||
/* Do not lock yet since selecting neighbors can take time */
|
|
||||||
HnswLoadNeighbors(neighborElement, index, m);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Could improve performance for vacuuming by checking neighbors
|
|
||||||
* against list of elements being deleted to find index. It's
|
|
||||||
* important to exclude already deleted elements for this since
|
|
||||||
* they can be replaced at any time.
|
|
||||||
*/
|
|
||||||
|
|
||||||
/* Select neighbors */
|
|
||||||
HnswUpdateConnection(NULL, e, hc, lm, lc, &idx, index, procinfo, collation);
|
|
||||||
|
|
||||||
/* New element was not selected as a neighbor */
|
|
||||||
if (idx == -1)
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Register page */
|
|
||||||
buf = ReadBuffer(index, neighborElement->neighborPage);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
if (building)
|
|
||||||
{
|
|
||||||
state = NULL;
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Get tuple */
|
|
||||||
ntup = (HnswNeighborTuple) PageGetItem(page, PageGetItemId(page, offno));
|
|
||||||
|
|
||||||
/* Calculate index for update */
|
|
||||||
startIdx = (neighborElement->level - lc) * m;
|
|
||||||
|
|
||||||
/* Check for existing connection */
|
|
||||||
if (checkExisting && ConnectionExists(e, ntup, startIdx, lm))
|
|
||||||
idx = -1;
|
|
||||||
else if (idx == -2)
|
|
||||||
{
|
|
||||||
/* Find free offset if still exists */
|
|
||||||
/* TODO Retry updating connections if not */
|
|
||||||
for (int j = 0; j < lm; j++)
|
|
||||||
{
|
|
||||||
if (!ItemPointerIsValid(&ntup->indextids[startIdx + j]))
|
|
||||||
{
|
|
||||||
idx = startIdx + j;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
idx += startIdx;
|
|
||||||
|
|
||||||
/* Make robust to issues */
|
|
||||||
if (idx >= 0 && idx < ntup->count)
|
|
||||||
{
|
|
||||||
ItemPointer indextid = &ntup->indextids[idx];
|
|
||||||
|
|
||||||
/* Update neighbor on the buffer */
|
|
||||||
ItemPointerSet(indextid, e->blkno, e->offno);
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
if (building)
|
|
||||||
MarkBufferDirty(buf);
|
|
||||||
else
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
}
|
|
||||||
else if (!building)
|
|
||||||
GenericXLogAbort(state);
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Add a heap TID to an existing element
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
AddDuplicateOnDisk(Relation index, HnswElement element, HnswElement dup, bool building)
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
HnswElementTuple etup;
|
|
||||||
int i;
|
|
||||||
|
|
||||||
/* Read page */
|
|
||||||
buf = ReadBuffer(index, dup->blkno);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
if (building)
|
|
||||||
{
|
|
||||||
state = NULL;
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Find space */
|
|
||||||
etup = (HnswElementTuple) PageGetItem(page, PageGetItemId(page, dup->offno));
|
|
||||||
for (i = 0; i < HNSW_HEAPTIDS; i++)
|
|
||||||
{
|
|
||||||
if (!ItemPointerIsValid(&etup->heaptids[i]))
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Either being deleted or we lost our chance to another backend */
|
|
||||||
if (i == 0 || i == HNSW_HEAPTIDS)
|
|
||||||
{
|
|
||||||
if (!building)
|
|
||||||
GenericXLogAbort(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Add heap TID, modifying the tuple on the page directly */
|
|
||||||
etup->heaptids[i] = element->heaptids[0];
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
if (building)
|
|
||||||
MarkBufferDirty(buf);
|
|
||||||
else
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Find duplicate element
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
FindDuplicateOnDisk(Relation index, HnswElement element, bool building)
|
|
||||||
{
|
|
||||||
char *base = NULL;
|
|
||||||
HnswNeighborArray *neighbors = HnswGetNeighbors(base, element, 0);
|
|
||||||
Datum value = HnswGetValue(base, element);
|
|
||||||
|
|
||||||
for (int i = 0; i < neighbors->length; i++)
|
|
||||||
{
|
|
||||||
HnswCandidate *neighbor = &neighbors->items[i];
|
|
||||||
HnswElement neighborElement = HnswPtrAccess(base, neighbor->element);
|
|
||||||
Datum neighborValue = HnswGetValue(base, neighborElement);
|
|
||||||
|
|
||||||
/* Exit early since ordered by distance */
|
|
||||||
if (!datumIsEqual(value, neighborValue, false, -1))
|
|
||||||
return false;
|
|
||||||
|
|
||||||
if (AddDuplicateOnDisk(index, element, neighborElement, building))
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Update graph on disk
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
UpdateGraphOnDisk(Relation index, FmgrInfo *procinfo, Oid collation, HnswElement element, int m, int efConstruction, HnswElement entryPoint, bool building)
|
|
||||||
{
|
|
||||||
BlockNumber newInsertPage = InvalidBlockNumber;
|
|
||||||
|
|
||||||
/* Look for duplicate */
|
|
||||||
if (FindDuplicateOnDisk(index, element, building))
|
|
||||||
return;
|
|
||||||
|
|
||||||
/* Add element */
|
|
||||||
AddElementOnDisk(index, element, m, GetInsertPage(index), &newInsertPage, building);
|
|
||||||
|
|
||||||
/* Update insert page if needed */
|
|
||||||
if (BlockNumberIsValid(newInsertPage))
|
|
||||||
HnswUpdateMetaPage(index, 0, NULL, newInsertPage, MAIN_FORKNUM, building);
|
|
||||||
|
|
||||||
/* Update neighbors */
|
|
||||||
HnswUpdateNeighborsOnDisk(index, procinfo, collation, element, m, false, building);
|
|
||||||
|
|
||||||
/* Update entry point if needed */
|
|
||||||
if (entryPoint == NULL || element->level > entryPoint->level)
|
|
||||||
HnswUpdateMetaPage(index, HNSW_UPDATE_ENTRY_GREATER, element, InvalidBlockNumber, MAIN_FORKNUM, building);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Insert a tuple into the index
|
|
||||||
*/
|
|
||||||
bool
|
|
||||||
HnswInsertTupleOnDisk(Relation index, Datum value, Datum *values, bool *isnull, ItemPointer heap_tid, bool building)
|
|
||||||
{
|
|
||||||
HnswElement entryPoint;
|
|
||||||
HnswElement element;
|
|
||||||
int m;
|
|
||||||
int efConstruction = HnswGetEfConstruction(index);
|
|
||||||
FmgrInfo *procinfo = index_getprocinfo(index, 1, HNSW_DISTANCE_PROC);
|
|
||||||
Oid collation = index->rd_indcollation[0];
|
|
||||||
LOCKMODE lockmode = ShareLock;
|
|
||||||
char *base = NULL;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get a shared lock. This allows vacuum to ensure no in-flight inserts
|
|
||||||
* before repairing graph. Use a page lock so it does not interfere with
|
|
||||||
* buffer lock (or reads when vacuuming).
|
|
||||||
*/
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Get m and entry point */
|
|
||||||
HnswGetMetaPageInfo(index, &m, &entryPoint);
|
|
||||||
|
|
||||||
/* Create an element */
|
|
||||||
element = HnswInitElement(base, heap_tid, m, HnswGetMl(m), HnswGetMaxLevel(m), NULL);
|
|
||||||
HnswPtrStore(base, element->value, DatumGetPointer(value));
|
|
||||||
|
|
||||||
/* Prevent concurrent inserts when likely updating entry point */
|
|
||||||
if (entryPoint == NULL || element->level > entryPoint->level)
|
|
||||||
{
|
|
||||||
/* Release shared lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Get exclusive lock */
|
|
||||||
lockmode = ExclusiveLock;
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Get latest entry point after lock is acquired */
|
|
||||||
entryPoint = HnswGetEntryPoint(index);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Find neighbors for element */
|
|
||||||
HnswFindElementNeighbors(base, element, entryPoint, index, procinfo, collation, m, efConstruction, false);
|
|
||||||
|
|
||||||
/* Update graph on disk */
|
|
||||||
UpdateGraphOnDisk(index, procinfo, collation, element, m, efConstruction, entryPoint, building);
|
|
||||||
|
|
||||||
/* Release lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Insert a tuple into the index
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswInsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid)
|
|
||||||
{
|
|
||||||
Datum value;
|
|
||||||
FmgrInfo *normprocinfo;
|
|
||||||
Oid collation = index->rd_indcollation[0];
|
|
||||||
|
|
||||||
/* Detoast once for all calls */
|
|
||||||
value = PointerGetDatum(PG_DETOAST_DATUM(values[0]));
|
|
||||||
|
|
||||||
/* Normalize if needed */
|
|
||||||
normprocinfo = HnswOptionalProcInfo(index, HNSW_NORM_PROC);
|
|
||||||
if (normprocinfo != NULL)
|
|
||||||
{
|
|
||||||
if (!HnswNormValue(normprocinfo, collation, &value, NULL))
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
HnswInsertTupleOnDisk(index, value, values, isnull, heap_tid, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Insert a tuple into the index
|
|
||||||
*/
|
|
||||||
bool
|
|
||||||
hnswinsert(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid,
|
|
||||||
Relation heap, IndexUniqueCheck checkUnique
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
|
||||||
,bool indexUnchanged
|
|
||||||
#endif
|
|
||||||
,IndexInfo *indexInfo
|
|
||||||
)
|
|
||||||
{
|
|
||||||
MemoryContext oldCtx;
|
|
||||||
MemoryContext insertCtx;
|
|
||||||
|
|
||||||
/* Skip nulls */
|
|
||||||
if (isnull[0])
|
|
||||||
return false;
|
|
||||||
|
|
||||||
/* Create memory context */
|
|
||||||
insertCtx = AllocSetContextCreate(CurrentMemoryContext,
|
|
||||||
"Hnsw insert temporary context",
|
|
||||||
ALLOCSET_DEFAULT_SIZES);
|
|
||||||
oldCtx = MemoryContextSwitchTo(insertCtx);
|
|
||||||
|
|
||||||
/* Insert tuple */
|
|
||||||
HnswInsertTuple(index, values, isnull, heap_tid);
|
|
||||||
|
|
||||||
/* Delete memory context */
|
|
||||||
MemoryContextSwitchTo(oldCtx);
|
|
||||||
MemoryContextDelete(insertCtx);
|
|
||||||
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
226
src/hnswscan.c
226
src/hnswscan.c
@@ -1,226 +0,0 @@
|
|||||||
#include "postgres.h"
|
|
||||||
|
|
||||||
#include "access/relscan.h"
|
|
||||||
#include "hnsw.h"
|
|
||||||
#include "pgstat.h"
|
|
||||||
#include "storage/bufmgr.h"
|
|
||||||
#include "storage/lmgr.h"
|
|
||||||
#include "utils/memutils.h"
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Algorithm 5 from paper
|
|
||||||
*/
|
|
||||||
static List *
|
|
||||||
GetScanItems(IndexScanDesc scan, Datum q)
|
|
||||||
{
|
|
||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
|
||||||
Relation index = scan->indexRelation;
|
|
||||||
FmgrInfo *procinfo = so->procinfo;
|
|
||||||
Oid collation = so->collation;
|
|
||||||
List *ep;
|
|
||||||
List *w;
|
|
||||||
int m;
|
|
||||||
HnswElement entryPoint;
|
|
||||||
char *base = NULL;
|
|
||||||
|
|
||||||
/* Get m and entry point */
|
|
||||||
HnswGetMetaPageInfo(index, &m, &entryPoint);
|
|
||||||
|
|
||||||
if (entryPoint == NULL)
|
|
||||||
return NIL;
|
|
||||||
|
|
||||||
ep = list_make1(HnswEntryCandidate(base, entryPoint, q, index, procinfo, collation, false));
|
|
||||||
|
|
||||||
for (int lc = entryPoint->level; lc >= 1; lc--)
|
|
||||||
{
|
|
||||||
w = HnswSearchLayer(base, q, ep, 1, lc, index, procinfo, collation, m, false, NULL);
|
|
||||||
ep = w;
|
|
||||||
}
|
|
||||||
|
|
||||||
return HnswSearchLayer(base, q, ep, hnsw_ef_search, 0, index, procinfo, collation, m, false, NULL);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get dimensions from metapage
|
|
||||||
*/
|
|
||||||
static int
|
|
||||||
GetDimensions(Relation index)
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
HnswMetaPage metap;
|
|
||||||
int dimensions;
|
|
||||||
|
|
||||||
buf = ReadBuffer(index, HNSW_METAPAGE_BLKNO);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
metap = HnswPageGetMeta(page);
|
|
||||||
|
|
||||||
dimensions = metap->dimensions;
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
return dimensions;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get scan value
|
|
||||||
*/
|
|
||||||
static Datum
|
|
||||||
GetScanValue(IndexScanDesc scan)
|
|
||||||
{
|
|
||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
|
||||||
Datum value;
|
|
||||||
|
|
||||||
if (scan->orderByData->sk_flags & SK_ISNULL)
|
|
||||||
value = PointerGetDatum(InitVector(GetDimensions(scan->indexRelation)));
|
|
||||||
else
|
|
||||||
{
|
|
||||||
value = scan->orderByData->sk_argument;
|
|
||||||
|
|
||||||
/* Value should not be compressed or toasted */
|
|
||||||
Assert(!VARATT_IS_COMPRESSED(DatumGetPointer(value)));
|
|
||||||
Assert(!VARATT_IS_EXTENDED(DatumGetPointer(value)));
|
|
||||||
|
|
||||||
/* Fine if normalization fails */
|
|
||||||
if (so->normprocinfo != NULL)
|
|
||||||
HnswNormValue(so->normprocinfo, so->collation, &value, NULL);
|
|
||||||
}
|
|
||||||
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Prepare for an index scan
|
|
||||||
*/
|
|
||||||
IndexScanDesc
|
|
||||||
hnswbeginscan(Relation index, int nkeys, int norderbys)
|
|
||||||
{
|
|
||||||
IndexScanDesc scan;
|
|
||||||
HnswScanOpaque so;
|
|
||||||
|
|
||||||
scan = RelationGetIndexScan(index, nkeys, norderbys);
|
|
||||||
|
|
||||||
so = (HnswScanOpaque) palloc(sizeof(HnswScanOpaqueData));
|
|
||||||
so->first = true;
|
|
||||||
so->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
|
||||||
"Hnsw scan temporary context",
|
|
||||||
ALLOCSET_DEFAULT_SIZES);
|
|
||||||
|
|
||||||
/* Set support functions */
|
|
||||||
so->procinfo = index_getprocinfo(index, 1, HNSW_DISTANCE_PROC);
|
|
||||||
so->normprocinfo = HnswOptionalProcInfo(index, HNSW_NORM_PROC);
|
|
||||||
so->collation = index->rd_indcollation[0];
|
|
||||||
|
|
||||||
scan->opaque = so;
|
|
||||||
|
|
||||||
return scan;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Start or restart an index scan
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
hnswrescan(IndexScanDesc scan, ScanKey keys, int nkeys, ScanKey orderbys, int norderbys)
|
|
||||||
{
|
|
||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
|
||||||
|
|
||||||
so->first = true;
|
|
||||||
MemoryContextReset(so->tmpCtx);
|
|
||||||
|
|
||||||
if (keys && scan->numberOfKeys > 0)
|
|
||||||
memmove(scan->keyData, keys, scan->numberOfKeys * sizeof(ScanKeyData));
|
|
||||||
|
|
||||||
if (orderbys && scan->numberOfOrderBys > 0)
|
|
||||||
memmove(scan->orderByData, orderbys, scan->numberOfOrderBys * sizeof(ScanKeyData));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Fetch the next tuple in the given scan
|
|
||||||
*/
|
|
||||||
bool
|
|
||||||
hnswgettuple(IndexScanDesc scan, ScanDirection dir)
|
|
||||||
{
|
|
||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
|
||||||
MemoryContext oldCtx = MemoryContextSwitchTo(so->tmpCtx);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Index can be used to scan backward, but Postgres doesn't support
|
|
||||||
* backward scan on operators
|
|
||||||
*/
|
|
||||||
Assert(ScanDirectionIsForward(dir));
|
|
||||||
|
|
||||||
if (so->first)
|
|
||||||
{
|
|
||||||
Datum value;
|
|
||||||
|
|
||||||
/* Count index scan for stats */
|
|
||||||
pgstat_count_index_scan(scan->indexRelation);
|
|
||||||
|
|
||||||
/* Safety check */
|
|
||||||
if (scan->orderByData == NULL)
|
|
||||||
elog(ERROR, "cannot scan hnsw index without order");
|
|
||||||
|
|
||||||
/* Requires MVCC-compliant snapshot as not able to maintain a pin */
|
|
||||||
/* https://www.postgresql.org/docs/current/index-locking.html */
|
|
||||||
if (!IsMVCCSnapshot(scan->xs_snapshot))
|
|
||||||
elog(ERROR, "non-MVCC snapshots are not supported with hnsw");
|
|
||||||
|
|
||||||
/* Get scan value */
|
|
||||||
value = GetScanValue(scan);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Get a shared lock. This allows vacuum to ensure no in-flight scans
|
|
||||||
* before marking tuples as deleted.
|
|
||||||
*/
|
|
||||||
LockPage(scan->indexRelation, HNSW_SCAN_LOCK, ShareLock);
|
|
||||||
|
|
||||||
so->w = GetScanItems(scan, value);
|
|
||||||
|
|
||||||
/* Release shared lock */
|
|
||||||
UnlockPage(scan->indexRelation, HNSW_SCAN_LOCK, ShareLock);
|
|
||||||
|
|
||||||
so->first = false;
|
|
||||||
}
|
|
||||||
|
|
||||||
while (list_length(so->w) > 0)
|
|
||||||
{
|
|
||||||
char *base = NULL;
|
|
||||||
HnswCandidate *hc = llast(so->w);
|
|
||||||
HnswElement element = HnswPtrAccess(base, hc->element);
|
|
||||||
ItemPointer heaptid;
|
|
||||||
|
|
||||||
/* Move to next element if no valid heap TIDs */
|
|
||||||
if (element->heaptidsLength == 0)
|
|
||||||
{
|
|
||||||
so->w = list_delete_last(so->w);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
heaptid = &element->heaptids[--element->heaptidsLength];
|
|
||||||
|
|
||||||
MemoryContextSwitchTo(oldCtx);
|
|
||||||
|
|
||||||
scan->xs_heaptid = *heaptid;
|
|
||||||
scan->xs_recheck = false;
|
|
||||||
scan->xs_recheckorderby = false;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
MemoryContextSwitchTo(oldCtx);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* End a scan and release resources
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
hnswendscan(IndexScanDesc scan)
|
|
||||||
{
|
|
||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
|
||||||
|
|
||||||
MemoryContextDelete(so->tmpCtx);
|
|
||||||
|
|
||||||
pfree(so);
|
|
||||||
scan->opaque = NULL;
|
|
||||||
}
|
|
||||||
1270
src/hnswutils.c
1270
src/hnswutils.c
File diff suppressed because it is too large
Load Diff
646
src/hnswvacuum.c
646
src/hnswvacuum.c
@@ -1,646 +0,0 @@
|
|||||||
#include "postgres.h"
|
|
||||||
|
|
||||||
#include <math.h>
|
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
|
||||||
#include "commands/vacuum.h"
|
|
||||||
#include "hnsw.h"
|
|
||||||
#include "storage/bufmgr.h"
|
|
||||||
#include "storage/lmgr.h"
|
|
||||||
#include "utils/memutils.h"
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Check if deleted list contains an index TID
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
DeletedContains(tidhash_hash * deleted, ItemPointer indextid)
|
|
||||||
{
|
|
||||||
return tidhash_lookup(deleted, *indextid) != NULL;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Remove deleted heap TIDs
|
|
||||||
*
|
|
||||||
* OK to remove for entry point, since always considered for searches and inserts
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
RemoveHeapTids(HnswVacuumState * vacuumstate)
|
|
||||||
{
|
|
||||||
BlockNumber blkno = HNSW_HEAD_BLKNO;
|
|
||||||
HnswElement highestPoint = &vacuumstate->highestPoint;
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
BufferAccessStrategy bas = vacuumstate->bas;
|
|
||||||
HnswElement entryPoint = HnswGetEntryPoint(vacuumstate->index);
|
|
||||||
IndexBulkDeleteResult *stats = vacuumstate->stats;
|
|
||||||
|
|
||||||
/* Store separately since highestPoint.level is uint8 */
|
|
||||||
int highestLevel = -1;
|
|
||||||
|
|
||||||
/* Initialize highest point */
|
|
||||||
highestPoint->blkno = InvalidBlockNumber;
|
|
||||||
highestPoint->offno = InvalidOffsetNumber;
|
|
||||||
|
|
||||||
while (BlockNumberIsValid(blkno))
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
bool updated = false;
|
|
||||||
|
|
||||||
vacuum_delay_point();
|
|
||||||
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, bas);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
maxoffno = PageGetMaxOffsetNumber(page);
|
|
||||||
|
|
||||||
/* Iterate over nodes */
|
|
||||||
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
|
||||||
{
|
|
||||||
HnswElementTuple etup = (HnswElementTuple) PageGetItem(page, PageGetItemId(page, offno));
|
|
||||||
int idx = 0;
|
|
||||||
bool itemUpdated = false;
|
|
||||||
|
|
||||||
/* Skip neighbor tuples */
|
|
||||||
if (!HnswIsElementTuple(etup))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
if (ItemPointerIsValid(&etup->heaptids[0]))
|
|
||||||
{
|
|
||||||
for (int i = 0; i < HNSW_HEAPTIDS; i++)
|
|
||||||
{
|
|
||||||
/* Stop at first unused */
|
|
||||||
if (!ItemPointerIsValid(&etup->heaptids[i]))
|
|
||||||
break;
|
|
||||||
|
|
||||||
if (vacuumstate->callback(&etup->heaptids[i], vacuumstate->callback_state))
|
|
||||||
{
|
|
||||||
itemUpdated = true;
|
|
||||||
stats->tuples_removed++;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
/* Move to front of list */
|
|
||||||
etup->heaptids[idx++] = etup->heaptids[i];
|
|
||||||
stats->num_index_tuples++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (itemUpdated)
|
|
||||||
{
|
|
||||||
/* Mark rest as invalid */
|
|
||||||
for (int i = idx; i < HNSW_HEAPTIDS; i++)
|
|
||||||
ItemPointerSetInvalid(&etup->heaptids[i]);
|
|
||||||
|
|
||||||
updated = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!ItemPointerIsValid(&etup->heaptids[0]))
|
|
||||||
{
|
|
||||||
ItemPointerData ip;
|
|
||||||
bool found;
|
|
||||||
|
|
||||||
/* Add to deleted list */
|
|
||||||
ItemPointerSet(&ip, blkno, offno);
|
|
||||||
|
|
||||||
tidhash_insert(vacuumstate->deleted, ip, &found);
|
|
||||||
Assert(!found);
|
|
||||||
}
|
|
||||||
else if (etup->level > highestLevel && !(entryPoint != NULL && blkno == entryPoint->blkno && offno == entryPoint->offno))
|
|
||||||
{
|
|
||||||
/* Keep track of highest non-entry point */
|
|
||||||
highestPoint->blkno = blkno;
|
|
||||||
highestPoint->offno = offno;
|
|
||||||
highestPoint->level = etup->level;
|
|
||||||
highestLevel = etup->level;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
blkno = HnswPageGetOpaque(page)->nextblkno;
|
|
||||||
|
|
||||||
if (updated)
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
else
|
|
||||||
GenericXLogAbort(state);
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Check for deleted neighbors
|
|
||||||
*/
|
|
||||||
static bool
|
|
||||||
NeedsUpdated(HnswVacuumState * vacuumstate, HnswElement element)
|
|
||||||
{
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
BufferAccessStrategy bas = vacuumstate->bas;
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
HnswNeighborTuple ntup;
|
|
||||||
bool needsUpdated = false;
|
|
||||||
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, element->neighborPage, RBM_NORMAL, bas);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
ntup = (HnswNeighborTuple) PageGetItem(page, PageGetItemId(page, element->neighborOffno));
|
|
||||||
|
|
||||||
Assert(HnswIsNeighborTuple(ntup));
|
|
||||||
|
|
||||||
/* Check neighbors */
|
|
||||||
for (int i = 0; i < ntup->count; i++)
|
|
||||||
{
|
|
||||||
ItemPointer indextid = &ntup->indextids[i];
|
|
||||||
|
|
||||||
if (!ItemPointerIsValid(indextid))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Check if in deleted list */
|
|
||||||
if (DeletedContains(vacuumstate->deleted, indextid))
|
|
||||||
{
|
|
||||||
needsUpdated = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Also update if layer 0 is not full */
|
|
||||||
/* This could indicate too many candidates being deleted during insert */
|
|
||||||
if (!needsUpdated)
|
|
||||||
needsUpdated = !ItemPointerIsValid(&ntup->indextids[ntup->count - 1]);
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
return needsUpdated;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Repair graph for a single element
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
RepairGraphElement(HnswVacuumState * vacuumstate, HnswElement element, HnswElement entryPoint)
|
|
||||||
{
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
int m = vacuumstate->m;
|
|
||||||
int efConstruction = vacuumstate->efConstruction;
|
|
||||||
FmgrInfo *procinfo = vacuumstate->procinfo;
|
|
||||||
Oid collation = vacuumstate->collation;
|
|
||||||
BufferAccessStrategy bas = vacuumstate->bas;
|
|
||||||
HnswNeighborTuple ntup = vacuumstate->ntup;
|
|
||||||
Size ntupSize = HNSW_NEIGHBOR_TUPLE_SIZE(element->level, m);
|
|
||||||
char *base = NULL;
|
|
||||||
|
|
||||||
/* Skip if element is entry point */
|
|
||||||
if (entryPoint != NULL && element->blkno == entryPoint->blkno && element->offno == entryPoint->offno)
|
|
||||||
return;
|
|
||||||
|
|
||||||
/* Init fields */
|
|
||||||
HnswInitNeighbors(base, element, m, NULL);
|
|
||||||
element->heaptidsLength = 0;
|
|
||||||
|
|
||||||
/* Find neighbors for element, skipping itself */
|
|
||||||
HnswFindElementNeighbors(base, element, entryPoint, index, procinfo, collation, m, efConstruction, true);
|
|
||||||
|
|
||||||
/* Zero memory for each element */
|
|
||||||
MemSet(ntup, 0, HNSW_TUPLE_ALLOC_SIZE);
|
|
||||||
|
|
||||||
/* Update neighbor tuple */
|
|
||||||
/* Do this before getting page to minimize locking */
|
|
||||||
HnswSetNeighborTuple(base, ntup, element, m);
|
|
||||||
|
|
||||||
/* Get neighbor page */
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, element->neighborPage, RBM_NORMAL, bas);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
|
|
||||||
/* Overwrite tuple */
|
|
||||||
if (!PageIndexTupleOverwrite(page, element->neighborOffno, (Item) ntup, ntupSize))
|
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
/* Update neighbors */
|
|
||||||
HnswUpdateNeighborsOnDisk(index, procinfo, collation, element, m, true, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Repair graph entry point
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
RepairGraphEntryPoint(HnswVacuumState * vacuumstate)
|
|
||||||
{
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
HnswElement highestPoint = &vacuumstate->highestPoint;
|
|
||||||
HnswElement entryPoint;
|
|
||||||
MemoryContext oldCtx = MemoryContextSwitchTo(vacuumstate->tmpCtx);
|
|
||||||
|
|
||||||
if (!BlockNumberIsValid(highestPoint->blkno))
|
|
||||||
highestPoint = NULL;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Repair graph for highest non-entry point. Highest point may be outdated
|
|
||||||
* due to inserts that happen during and after RemoveHeapTids.
|
|
||||||
*/
|
|
||||||
if (highestPoint != NULL)
|
|
||||||
{
|
|
||||||
/* Get a shared lock */
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, ShareLock);
|
|
||||||
|
|
||||||
/* Load element */
|
|
||||||
HnswLoadElement(highestPoint, NULL, NULL, index, vacuumstate->procinfo, vacuumstate->collation, true);
|
|
||||||
|
|
||||||
/* Repair if needed */
|
|
||||||
if (NeedsUpdated(vacuumstate, highestPoint))
|
|
||||||
RepairGraphElement(vacuumstate, highestPoint, HnswGetEntryPoint(index));
|
|
||||||
|
|
||||||
/* Release lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, ShareLock);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Prevent concurrent inserts when possibly updating entry point */
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, ExclusiveLock);
|
|
||||||
|
|
||||||
/* Get latest entry point */
|
|
||||||
entryPoint = HnswGetEntryPoint(index);
|
|
||||||
|
|
||||||
if (entryPoint != NULL)
|
|
||||||
{
|
|
||||||
ItemPointerData epData;
|
|
||||||
|
|
||||||
ItemPointerSet(&epData, entryPoint->blkno, entryPoint->offno);
|
|
||||||
|
|
||||||
if (DeletedContains(vacuumstate->deleted, &epData))
|
|
||||||
{
|
|
||||||
/*
|
|
||||||
* Replace the entry point with the highest point. If highest
|
|
||||||
* point is outdated and empty, the entry point will be empty
|
|
||||||
* until an element is repaired.
|
|
||||||
*/
|
|
||||||
HnswUpdateMetaPage(index, HNSW_UPDATE_ENTRY_ALWAYS, highestPoint, InvalidBlockNumber, MAIN_FORKNUM, false);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
/*
|
|
||||||
* Repair the entry point with the highest point. If highest point
|
|
||||||
* is outdated, this can remove connections at higher levels in
|
|
||||||
* the graph until they are repaired, but this should be fine.
|
|
||||||
*/
|
|
||||||
HnswLoadElement(entryPoint, NULL, NULL, index, vacuumstate->procinfo, vacuumstate->collation, true);
|
|
||||||
|
|
||||||
if (NeedsUpdated(vacuumstate, entryPoint))
|
|
||||||
{
|
|
||||||
/* Reset neighbors from previous update */
|
|
||||||
if (highestPoint != NULL)
|
|
||||||
HnswPtrStore((char *) NULL, highestPoint->neighbors, (HnswNeighborArrayPtr *) NULL);
|
|
||||||
|
|
||||||
RepairGraphElement(vacuumstate, entryPoint, highestPoint);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Release lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, ExclusiveLock);
|
|
||||||
|
|
||||||
/* Reset memory context */
|
|
||||||
MemoryContextSwitchTo(oldCtx);
|
|
||||||
MemoryContextReset(vacuumstate->tmpCtx);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Repair graph for all elements
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
RepairGraph(HnswVacuumState * vacuumstate)
|
|
||||||
{
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
BufferAccessStrategy bas = vacuumstate->bas;
|
|
||||||
BlockNumber blkno = HNSW_HEAD_BLKNO;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Wait for inserts to complete. Inserts before this point may have
|
|
||||||
* neighbors about to be deleted. Inserts after this point will not.
|
|
||||||
*/
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, ExclusiveLock);
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, ExclusiveLock);
|
|
||||||
|
|
||||||
/* Repair entry point first */
|
|
||||||
RepairGraphEntryPoint(vacuumstate);
|
|
||||||
|
|
||||||
while (BlockNumberIsValid(blkno))
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
List *elements = NIL;
|
|
||||||
ListCell *lc2;
|
|
||||||
MemoryContext oldCtx;
|
|
||||||
|
|
||||||
vacuum_delay_point();
|
|
||||||
|
|
||||||
oldCtx = MemoryContextSwitchTo(vacuumstate->tmpCtx);
|
|
||||||
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, bas);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
maxoffno = PageGetMaxOffsetNumber(page);
|
|
||||||
|
|
||||||
/* Load items into memory to minimize locking */
|
|
||||||
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
|
||||||
{
|
|
||||||
HnswElementTuple etup = (HnswElementTuple) PageGetItem(page, PageGetItemId(page, offno));
|
|
||||||
HnswElement element;
|
|
||||||
|
|
||||||
/* Skip neighbor tuples */
|
|
||||||
if (!HnswIsElementTuple(etup))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Skip updating neighbors if being deleted */
|
|
||||||
if (!ItemPointerIsValid(&etup->heaptids[0]))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Create an element */
|
|
||||||
element = HnswInitElementFromBlock(blkno, offno);
|
|
||||||
HnswLoadElementFromTuple(element, etup, false, true);
|
|
||||||
|
|
||||||
elements = lappend(elements, element);
|
|
||||||
}
|
|
||||||
|
|
||||||
blkno = HnswPageGetOpaque(page)->nextblkno;
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
|
|
||||||
/* Update neighbor pages */
|
|
||||||
foreach(lc2, elements)
|
|
||||||
{
|
|
||||||
HnswElement element = (HnswElement) lfirst(lc2);
|
|
||||||
HnswElement entryPoint;
|
|
||||||
LOCKMODE lockmode = ShareLock;
|
|
||||||
|
|
||||||
/* Check if any neighbors point to deleted values */
|
|
||||||
if (!NeedsUpdated(vacuumstate, element))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Get a shared lock */
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Refresh entry point for each element */
|
|
||||||
entryPoint = HnswGetEntryPoint(index);
|
|
||||||
|
|
||||||
/* Prevent concurrent inserts when likely updating entry point */
|
|
||||||
if (entryPoint == NULL || element->level > entryPoint->level)
|
|
||||||
{
|
|
||||||
/* Release shared lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Get exclusive lock */
|
|
||||||
lockmode = ExclusiveLock;
|
|
||||||
LockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
|
|
||||||
/* Get latest entry point after lock is acquired */
|
|
||||||
entryPoint = HnswGetEntryPoint(index);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Repair connections */
|
|
||||||
RepairGraphElement(vacuumstate, element, entryPoint);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Update metapage if needed. Should only happen if entry point
|
|
||||||
* was replaced and highest point was outdated.
|
|
||||||
*/
|
|
||||||
if (entryPoint == NULL || element->level > entryPoint->level)
|
|
||||||
HnswUpdateMetaPage(index, HNSW_UPDATE_ENTRY_GREATER, element, InvalidBlockNumber, MAIN_FORKNUM, false);
|
|
||||||
|
|
||||||
/* Release lock */
|
|
||||||
UnlockPage(index, HNSW_UPDATE_LOCK, lockmode);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Reset memory context */
|
|
||||||
MemoryContextSwitchTo(oldCtx);
|
|
||||||
MemoryContextReset(vacuumstate->tmpCtx);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Mark items as deleted
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
MarkDeleted(HnswVacuumState * vacuumstate)
|
|
||||||
{
|
|
||||||
BlockNumber blkno = HNSW_HEAD_BLKNO;
|
|
||||||
BlockNumber insertPage = InvalidBlockNumber;
|
|
||||||
Relation index = vacuumstate->index;
|
|
||||||
BufferAccessStrategy bas = vacuumstate->bas;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Wait for index scans to complete. Scans before this point may contain
|
|
||||||
* tuples about to be deleted. Scans after this point will not, since the
|
|
||||||
* graph has been repaired.
|
|
||||||
*/
|
|
||||||
LockPage(index, HNSW_SCAN_LOCK, ExclusiveLock);
|
|
||||||
UnlockPage(index, HNSW_SCAN_LOCK, ExclusiveLock);
|
|
||||||
|
|
||||||
while (BlockNumberIsValid(blkno))
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
|
|
||||||
vacuum_delay_point();
|
|
||||||
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, bas);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* ambulkdelete cannot delete entries from pages that are pinned by
|
|
||||||
* other backends
|
|
||||||
*
|
|
||||||
* https://www.postgresql.org/docs/current/index-locking.html
|
|
||||||
*/
|
|
||||||
LockBufferForCleanup(buf);
|
|
||||||
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
maxoffno = PageGetMaxOffsetNumber(page);
|
|
||||||
|
|
||||||
/* Update element and neighbors together */
|
|
||||||
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
|
||||||
{
|
|
||||||
HnswElementTuple etup = (HnswElementTuple) PageGetItem(page, PageGetItemId(page, offno));
|
|
||||||
HnswNeighborTuple ntup;
|
|
||||||
Buffer nbuf;
|
|
||||||
Page npage;
|
|
||||||
BlockNumber neighborPage;
|
|
||||||
OffsetNumber neighborOffno;
|
|
||||||
|
|
||||||
/* Skip neighbor tuples */
|
|
||||||
if (!HnswIsElementTuple(etup))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Skip deleted tuples */
|
|
||||||
if (etup->deleted)
|
|
||||||
{
|
|
||||||
/* Set to first free page */
|
|
||||||
if (!BlockNumberIsValid(insertPage))
|
|
||||||
insertPage = blkno;
|
|
||||||
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Skip live tuples */
|
|
||||||
if (ItemPointerIsValid(&etup->heaptids[0]))
|
|
||||||
continue;
|
|
||||||
|
|
||||||
/* Get neighbor page */
|
|
||||||
neighborPage = ItemPointerGetBlockNumber(&etup->neighbortid);
|
|
||||||
neighborOffno = ItemPointerGetOffsetNumber(&etup->neighbortid);
|
|
||||||
|
|
||||||
if (neighborPage == blkno)
|
|
||||||
{
|
|
||||||
nbuf = buf;
|
|
||||||
npage = page;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
nbuf = ReadBufferExtended(index, MAIN_FORKNUM, neighborPage, RBM_NORMAL, bas);
|
|
||||||
LockBuffer(nbuf, BUFFER_LOCK_EXCLUSIVE);
|
|
||||||
npage = GenericXLogRegisterBuffer(state, nbuf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
ntup = (HnswNeighborTuple) PageGetItem(npage, PageGetItemId(npage, neighborOffno));
|
|
||||||
|
|
||||||
/* Overwrite element */
|
|
||||||
etup->deleted = 1;
|
|
||||||
MemSet(&etup->data, 0, VARSIZE_ANY(&etup->data));
|
|
||||||
|
|
||||||
/* Overwrite neighbors */
|
|
||||||
for (int i = 0; i < ntup->count; i++)
|
|
||||||
ItemPointerSetInvalid(&ntup->indextids[i]);
|
|
||||||
|
|
||||||
/*
|
|
||||||
* We modified the tuples in place, no need to call
|
|
||||||
* PageIndexTupleOverwrite
|
|
||||||
*/
|
|
||||||
|
|
||||||
/* Commit */
|
|
||||||
GenericXLogFinish(state);
|
|
||||||
if (nbuf != buf)
|
|
||||||
UnlockReleaseBuffer(nbuf);
|
|
||||||
|
|
||||||
/* Set to first free page */
|
|
||||||
if (!BlockNumberIsValid(insertPage))
|
|
||||||
insertPage = blkno;
|
|
||||||
|
|
||||||
/* Prepare new xlog */
|
|
||||||
state = GenericXLogStart(index);
|
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
blkno = HnswPageGetOpaque(page)->nextblkno;
|
|
||||||
|
|
||||||
GenericXLogAbort(state);
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Update insert page last, after everything has been marked as deleted */
|
|
||||||
HnswUpdateMetaPage(index, 0, NULL, insertPage, MAIN_FORKNUM, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Initialize the vacuum state
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
InitVacuumState(HnswVacuumState * vacuumstate, IndexVacuumInfo *info, IndexBulkDeleteResult *stats, IndexBulkDeleteCallback callback, void *callback_state)
|
|
||||||
{
|
|
||||||
Relation index = info->index;
|
|
||||||
|
|
||||||
if (stats == NULL)
|
|
||||||
stats = (IndexBulkDeleteResult *) palloc0(sizeof(IndexBulkDeleteResult));
|
|
||||||
|
|
||||||
vacuumstate->index = index;
|
|
||||||
vacuumstate->stats = stats;
|
|
||||||
vacuumstate->callback = callback;
|
|
||||||
vacuumstate->callback_state = callback_state;
|
|
||||||
vacuumstate->efConstruction = HnswGetEfConstruction(index);
|
|
||||||
vacuumstate->bas = GetAccessStrategy(BAS_BULKREAD);
|
|
||||||
vacuumstate->procinfo = index_getprocinfo(index, 1, HNSW_DISTANCE_PROC);
|
|
||||||
vacuumstate->collation = index->rd_indcollation[0];
|
|
||||||
vacuumstate->ntup = palloc0(HNSW_TUPLE_ALLOC_SIZE);
|
|
||||||
vacuumstate->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
|
||||||
"Hnsw vacuum temporary context",
|
|
||||||
ALLOCSET_DEFAULT_SIZES);
|
|
||||||
|
|
||||||
/* Get m from metapage */
|
|
||||||
HnswGetMetaPageInfo(index, &vacuumstate->m, NULL);
|
|
||||||
|
|
||||||
/* Create hash table */
|
|
||||||
vacuumstate->deleted = tidhash_create(CurrentMemoryContext, 256, NULL);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Free resources
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
FreeVacuumState(HnswVacuumState * vacuumstate)
|
|
||||||
{
|
|
||||||
tidhash_destroy(vacuumstate->deleted);
|
|
||||||
FreeAccessStrategy(vacuumstate->bas);
|
|
||||||
pfree(vacuumstate->ntup);
|
|
||||||
MemoryContextDelete(vacuumstate->tmpCtx);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Bulk delete tuples from the index
|
|
||||||
*/
|
|
||||||
IndexBulkDeleteResult *
|
|
||||||
hnswbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|
||||||
IndexBulkDeleteCallback callback, void *callback_state)
|
|
||||||
{
|
|
||||||
HnswVacuumState vacuumstate;
|
|
||||||
|
|
||||||
InitVacuumState(&vacuumstate, info, stats, callback, callback_state);
|
|
||||||
|
|
||||||
/* Pass 1: Remove heap TIDs */
|
|
||||||
RemoveHeapTids(&vacuumstate);
|
|
||||||
|
|
||||||
/* Pass 2: Repair graph */
|
|
||||||
RepairGraph(&vacuumstate);
|
|
||||||
|
|
||||||
/* Pass 3: Mark as deleted */
|
|
||||||
MarkDeleted(&vacuumstate);
|
|
||||||
|
|
||||||
FreeVacuumState(&vacuumstate);
|
|
||||||
|
|
||||||
return vacuumstate.stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Clean up after a VACUUM operation
|
|
||||||
*/
|
|
||||||
IndexBulkDeleteResult *
|
|
||||||
hnswvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats)
|
|
||||||
{
|
|
||||||
Relation rel = info->index;
|
|
||||||
|
|
||||||
if (info->analyze_only)
|
|
||||||
return stats;
|
|
||||||
|
|
||||||
/* stats is NULL if ambulkdelete not called */
|
|
||||||
/* OK to return NULL if index not changed */
|
|
||||||
if (stats == NULL)
|
|
||||||
return NULL;
|
|
||||||
|
|
||||||
stats->num_pages = RelationGetNumberOfBlocks(rel);
|
|
||||||
|
|
||||||
return stats;
|
|
||||||
}
|
|
||||||
564
src/ivfbuild.c
564
src/ivfbuild.c
@@ -2,43 +2,42 @@
|
|||||||
|
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
|
||||||
#include "access/table.h"
|
|
||||||
#include "access/tableam.h"
|
|
||||||
#include "access/parallel.h"
|
|
||||||
#include "access/xact.h"
|
|
||||||
#include "catalog/index.h"
|
#include "catalog/index.h"
|
||||||
#include "catalog/pg_operator_d.h"
|
|
||||||
#include "catalog/pg_type_d.h"
|
|
||||||
#include "commands/progress.h"
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "optimizer/optimizer.h"
|
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
#include "tcop/tcopprot.h"
|
|
||||||
#include "utils/memutils.h"
|
#include "utils/memutils.h"
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
#if PG_VERSION_NUM >= 140000
|
||||||
#include "utils/backend_progress.h"
|
#include "utils/backend_progress.h"
|
||||||
#else
|
#elif PG_VERSION_NUM >= 120000
|
||||||
#include "pgstat.h"
|
#include "pgstat.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
#include "access/tableam.h"
|
||||||
|
#include "commands/progress.h"
|
||||||
|
#else
|
||||||
|
#define PROGRESS_CREATEIDX_SUBPHASE 0
|
||||||
|
#define PROGRESS_CREATEIDX_TUPLES_TOTAL 0
|
||||||
|
#define PROGRESS_CREATEIDX_TUPLES_DONE 0
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#include "catalog/pg_operator_d.h"
|
||||||
|
#include "catalog/pg_type_d.h"
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
#if PG_VERSION_NUM >= 130000
|
||||||
#define CALLBACK_ITEM_POINTER ItemPointer tid
|
#define CALLBACK_ITEM_POINTER ItemPointer tid
|
||||||
#else
|
#else
|
||||||
#define CALLBACK_ITEM_POINTER HeapTuple hup
|
#define CALLBACK_ITEM_POINTER HeapTuple hup
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
#if PG_VERSION_NUM >= 120000
|
||||||
#include "utils/backend_status.h"
|
#define UpdateProgress(index, val) pgstat_progress_update_param(index, val)
|
||||||
#include "utils/wait_event.h"
|
#else
|
||||||
|
#define UpdateProgress(index, val) ((void)val)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define PARALLEL_KEY_IVFFLAT_SHARED UINT64CONST(0xA000000000000001)
|
|
||||||
#define PARALLEL_KEY_TUPLESORT UINT64CONST(0xA000000000000002)
|
|
||||||
#define PARALLEL_KEY_IVFFLAT_CENTERS UINT64CONST(0xA000000000000003)
|
|
||||||
#define PARALLEL_KEY_QUERY_TEXT UINT64CONST(0xA000000000000004)
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Add sample
|
* Add sample
|
||||||
*/
|
*/
|
||||||
@@ -130,8 +129,13 @@ SampleRows(IvfflatBuildState * buildstate)
|
|||||||
{
|
{
|
||||||
BlockNumber targblock = BlockSampler_Next(&buildstate->bs);
|
BlockNumber targblock = BlockSampler_Next(&buildstate->bs);
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
table_index_build_range_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
table_index_build_range_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
false, true, false, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
false, true, false, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
||||||
|
#else
|
||||||
|
IndexBuildHeapRangeScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
false, true, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -143,9 +147,10 @@ AddTupleToSort(Relation index, ItemPointer tid, Datum *values, IvfflatBuildState
|
|||||||
{
|
{
|
||||||
double distance;
|
double distance;
|
||||||
double minDistance = DBL_MAX;
|
double minDistance = DBL_MAX;
|
||||||
int closestCenter = 0;
|
int closestCenter = -1;
|
||||||
VectorArray centers = buildstate->centers;
|
VectorArray centers = buildstate->centers;
|
||||||
TupleTableSlot *slot = buildstate->slot;
|
TupleTableSlot *slot = buildstate->slot;
|
||||||
|
int i;
|
||||||
|
|
||||||
/* Detoast once for all calls */
|
/* Detoast once for all calls */
|
||||||
Datum value = PointerGetDatum(PG_DETOAST_DATUM(values[0]));
|
Datum value = PointerGetDatum(PG_DETOAST_DATUM(values[0]));
|
||||||
@@ -158,7 +163,7 @@ AddTupleToSort(Relation index, ItemPointer tid, Datum *values, IvfflatBuildState
|
|||||||
}
|
}
|
||||||
|
|
||||||
/* Find the list that minimizes the distance */
|
/* Find the list that minimizes the distance */
|
||||||
for (int i = 0; i < centers->length; i++)
|
for (i = 0; i < centers->length; i++)
|
||||||
{
|
{
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(buildstate->procinfo, buildstate->collation, value, PointerGetDatum(VectorArrayGet(centers, i))));
|
distance = DatumGetFloat8(FunctionCall2Coll(buildstate->procinfo, buildstate->collation, value, PointerGetDatum(VectorArrayGet(centers, i))));
|
||||||
|
|
||||||
@@ -253,27 +258,32 @@ GetNextTuple(Tuplesortstate *sortstate, TupleDesc tupdesc, TupleTableSlot *slot,
|
|||||||
static void
|
static void
|
||||||
InsertTuples(Relation index, IvfflatBuildState * buildstate, ForkNumber forkNum)
|
InsertTuples(Relation index, IvfflatBuildState * buildstate, ForkNumber forkNum)
|
||||||
{
|
{
|
||||||
|
Buffer buf;
|
||||||
|
Page page;
|
||||||
|
GenericXLogState *state;
|
||||||
int list;
|
int list;
|
||||||
IndexTuple itup = NULL; /* silence compiler warning */
|
IndexTuple itup = NULL; /* silence compiler warning */
|
||||||
|
BlockNumber startPage;
|
||||||
|
BlockNumber insertPage;
|
||||||
|
Size itemsz;
|
||||||
|
int i;
|
||||||
int64 inserted = 0;
|
int64 inserted = 0;
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
TupleTableSlot *slot = MakeSingleTupleTableSlot(buildstate->tupdesc, &TTSOpsMinimalTuple);
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(buildstate->tupdesc, &TTSOpsMinimalTuple);
|
||||||
|
#else
|
||||||
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(buildstate->tupdesc);
|
||||||
|
#endif
|
||||||
TupleDesc tupdesc = RelationGetDescr(index);
|
TupleDesc tupdesc = RelationGetDescr(index);
|
||||||
|
|
||||||
pgstat_progress_update_param(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_LOAD);
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_LOAD);
|
||||||
|
|
||||||
pgstat_progress_update_param(PROGRESS_CREATEIDX_TUPLES_TOTAL, buildstate->indtuples);
|
UpdateProgress(PROGRESS_CREATEIDX_TUPLES_TOTAL, buildstate->indtuples);
|
||||||
|
|
||||||
GetNextTuple(buildstate->sortstate, tupdesc, slot, &itup, &list);
|
GetNextTuple(buildstate->sortstate, tupdesc, slot, &itup, &list);
|
||||||
|
|
||||||
for (int i = 0; i < buildstate->centers->length; i++)
|
for (i = 0; i < buildstate->centers->length; i++)
|
||||||
{
|
{
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
BlockNumber startPage;
|
|
||||||
BlockNumber insertPage;
|
|
||||||
|
|
||||||
/* Can take a while, so ensure we can interrupt */
|
/* Can take a while, so ensure we can interrupt */
|
||||||
/* Needs to be called when no buffer locks are held */
|
/* Needs to be called when no buffer locks are held */
|
||||||
CHECK_FOR_INTERRUPTS();
|
CHECK_FOR_INTERRUPTS();
|
||||||
@@ -287,8 +297,7 @@ InsertTuples(Relation index, IvfflatBuildState * buildstate, ForkNumber forkNum)
|
|||||||
while (list == i)
|
while (list == i)
|
||||||
{
|
{
|
||||||
/* Check for free space */
|
/* Check for free space */
|
||||||
Size itemsz = MAXALIGN(IndexTupleSize(itup));
|
itemsz = MAXALIGN(IndexTupleSize(itup));
|
||||||
|
|
||||||
if (PageGetFreeSpace(page) < itemsz)
|
if (PageGetFreeSpace(page) < itemsz)
|
||||||
IvfflatAppendPage(index, &buf, &page, &state, forkNum);
|
IvfflatAppendPage(index, &buf, &page, &state, forkNum);
|
||||||
|
|
||||||
@@ -298,7 +307,7 @@ InsertTuples(Relation index, IvfflatBuildState * buildstate, ForkNumber forkNum)
|
|||||||
|
|
||||||
pfree(itup);
|
pfree(itup);
|
||||||
|
|
||||||
pgstat_progress_update_param(PROGRESS_CREATEIDX_TUPLES_DONE, ++inserted);
|
UpdateProgress(PROGRESS_CREATEIDX_TUPLES_DONE, ++inserted);
|
||||||
|
|
||||||
GetNextTuple(buildstate->sortstate, tupdesc, slot, &itup, &list);
|
GetNextTuple(buildstate->sortstate, tupdesc, slot, &itup, &list);
|
||||||
}
|
}
|
||||||
@@ -308,7 +317,7 @@ InsertTuples(Relation index, IvfflatBuildState * buildstate, ForkNumber forkNum)
|
|||||||
IvfflatCommitBuffer(buf, state);
|
IvfflatCommitBuffer(buf, state);
|
||||||
|
|
||||||
/* Set the start and insert pages */
|
/* Set the start and insert pages */
|
||||||
IvfflatUpdateList(index, buildstate->listInfo[i], insertPage, InvalidBlockNumber, startPage, forkNum);
|
IvfflatUpdateList(index, state, buildstate->listInfo[i], insertPage, InvalidBlockNumber, startPage, forkNum);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -342,16 +351,26 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
|
|||||||
buildstate->collation = index->rd_indcollation[0];
|
buildstate->collation = index->rd_indcollation[0];
|
||||||
|
|
||||||
/* Require more than one dimension for spherical k-means */
|
/* Require more than one dimension for spherical k-means */
|
||||||
if (buildstate->kmeansnormprocinfo != NULL && buildstate->dimensions == 1)
|
/* Lists check for backwards compatibility */
|
||||||
|
/* TODO Remove lists check in 0.3.0 */
|
||||||
|
if (buildstate->kmeansnormprocinfo != NULL && buildstate->dimensions == 1 && buildstate->lists > 1)
|
||||||
elog(ERROR, "dimensions must be greater than one for this opclass");
|
elog(ERROR, "dimensions must be greater than one for this opclass");
|
||||||
|
|
||||||
/* Create tuple description for sorting */
|
/* Create tuple description for sorting */
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
buildstate->tupdesc = CreateTemplateTupleDesc(3);
|
buildstate->tupdesc = CreateTemplateTupleDesc(3);
|
||||||
|
#else
|
||||||
|
buildstate->tupdesc = CreateTemplateTupleDesc(3, false);
|
||||||
|
#endif
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 1, "list", INT4OID, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 1, "list", INT4OID, -1, 0);
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 2, "tid", TIDOID, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 2, "tid", TIDOID, -1, 0);
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 3, "vector", RelationGetDescr(index)->attrs[0].atttypid, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 3, "vector", RelationGetDescr(index)->attrs[0].atttypid, -1, 0);
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
buildstate->slot = MakeSingleTupleTableSlot(buildstate->tupdesc, &TTSOpsVirtual);
|
buildstate->slot = MakeSingleTupleTableSlot(buildstate->tupdesc, &TTSOpsVirtual);
|
||||||
|
#else
|
||||||
|
buildstate->slot = MakeSingleTupleTableSlot(buildstate->tupdesc);
|
||||||
|
#endif
|
||||||
|
|
||||||
buildstate->centers = VectorArrayInit(buildstate->lists, buildstate->dimensions);
|
buildstate->centers = VectorArrayInit(buildstate->lists, buildstate->dimensions);
|
||||||
buildstate->listInfo = palloc(sizeof(ListInfo) * buildstate->lists);
|
buildstate->listInfo = palloc(sizeof(ListInfo) * buildstate->lists);
|
||||||
@@ -368,8 +387,6 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
|
|||||||
buildstate->listSums = palloc0(sizeof(double) * buildstate->lists);
|
buildstate->listSums = palloc0(sizeof(double) * buildstate->lists);
|
||||||
buildstate->listCounts = palloc0(sizeof(int) * buildstate->lists);
|
buildstate->listCounts = palloc0(sizeof(int) * buildstate->lists);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
buildstate->ivfleader = NULL;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -398,7 +415,7 @@ ComputeCenters(IvfflatBuildState * buildstate)
|
|||||||
{
|
{
|
||||||
int numSamples;
|
int numSamples;
|
||||||
|
|
||||||
pgstat_progress_update_param(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS);
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS);
|
||||||
|
|
||||||
/* Target 50 samples per list, with at least 10000 samples */
|
/* Target 50 samples per list, with at least 10000 samples */
|
||||||
/* The number of samples has a large effect on index build time */
|
/* The number of samples has a large effect on index build time */
|
||||||
@@ -414,18 +431,8 @@ ComputeCenters(IvfflatBuildState * buildstate)
|
|||||||
/* TODO Ensure within maintenance_work_mem */
|
/* TODO Ensure within maintenance_work_mem */
|
||||||
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
||||||
if (buildstate->heap != NULL)
|
if (buildstate->heap != NULL)
|
||||||
{
|
|
||||||
SampleRows(buildstate);
|
SampleRows(buildstate);
|
||||||
|
|
||||||
if (buildstate->samples->length < buildstate->lists)
|
|
||||||
{
|
|
||||||
ereport(NOTICE,
|
|
||||||
(errmsg("ivfflat index created with little data"),
|
|
||||||
errdetail("This will cause low recall."),
|
|
||||||
errhint("Drop the index until the table has more data.")));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Calculate centers */
|
/* Calculate centers */
|
||||||
IvfflatBench("k-means", IvfflatKmeans(buildstate->index, buildstate->samples, buildstate->centers));
|
IvfflatBench("k-means", IvfflatKmeans(buildstate->index, buildstate->samples, buildstate->centers));
|
||||||
|
|
||||||
@@ -466,33 +473,33 @@ static void
|
|||||||
CreateListPages(Relation index, VectorArray centers, int dimensions,
|
CreateListPages(Relation index, VectorArray centers, int dimensions,
|
||||||
int lists, ForkNumber forkNum, ListInfo * *listInfo)
|
int lists, ForkNumber forkNum, ListInfo * *listInfo)
|
||||||
{
|
{
|
||||||
|
int i;
|
||||||
Buffer buf;
|
Buffer buf;
|
||||||
Page page;
|
Page page;
|
||||||
GenericXLogState *state;
|
GenericXLogState *state;
|
||||||
Size listSize;
|
OffsetNumber offno;
|
||||||
|
Size itemsz;
|
||||||
IvfflatList list;
|
IvfflatList list;
|
||||||
|
|
||||||
listSize = MAXALIGN(IVFFLAT_LIST_SIZE(dimensions));
|
itemsz = MAXALIGN(IVFFLAT_LIST_SIZE(dimensions));
|
||||||
list = palloc0(listSize);
|
list = palloc(itemsz);
|
||||||
|
|
||||||
buf = IvfflatNewBuffer(index, forkNum);
|
buf = IvfflatNewBuffer(index, forkNum);
|
||||||
IvfflatInitRegisterPage(index, &buf, &page, &state);
|
IvfflatInitRegisterPage(index, &buf, &page, &state);
|
||||||
|
|
||||||
for (int i = 0; i < lists; i++)
|
for (i = 0; i < lists; i++)
|
||||||
{
|
{
|
||||||
OffsetNumber offno;
|
|
||||||
|
|
||||||
/* Load list */
|
/* Load list */
|
||||||
list->startPage = InvalidBlockNumber;
|
list->startPage = InvalidBlockNumber;
|
||||||
list->insertPage = InvalidBlockNumber;
|
list->insertPage = InvalidBlockNumber;
|
||||||
memcpy(&list->center, VectorArrayGet(centers, i), VECTOR_SIZE(dimensions));
|
memcpy(&list->center, VectorArrayGet(centers, i), VECTOR_SIZE(dimensions));
|
||||||
|
|
||||||
/* Ensure free space */
|
/* Ensure free space */
|
||||||
if (PageGetFreeSpace(page) < listSize)
|
if (PageGetFreeSpace(page) < itemsz)
|
||||||
IvfflatAppendPage(index, &buf, &page, &state, forkNum);
|
IvfflatAppendPage(index, &buf, &page, &state, forkNum);
|
||||||
|
|
||||||
/* Add the item */
|
/* Add the item */
|
||||||
offno = PageAddItem(page, (Item) list, listSize, InvalidOffsetNumber, false, false);
|
offno = PageAddItem(page, (Item) list, itemsz, InvalidOffsetNumber, false, false);
|
||||||
if (offno == InvalidOffsetNumber)
|
if (offno == InvalidOffsetNumber)
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
||||||
|
|
||||||
@@ -506,17 +513,17 @@ CreateListPages(Relation index, VectorArray centers, int dimensions,
|
|||||||
pfree(list);
|
pfree(list);
|
||||||
}
|
}
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
/*
|
/*
|
||||||
* Print k-means metrics
|
* Print k-means metrics
|
||||||
*/
|
*/
|
||||||
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
static void
|
static void
|
||||||
PrintKmeansMetrics(IvfflatBuildState * buildstate)
|
PrintKmeansMetrics(IvfflatBuildState * buildstate)
|
||||||
{
|
{
|
||||||
elog(INFO, "inertia: %.3e", buildstate->inertia);
|
elog(INFO, "inertia: %.3e", buildstate->inertia);
|
||||||
|
|
||||||
/* Calculate Davies-Bouldin index */
|
/* Calculate Davies-Bouldin index */
|
||||||
if (buildstate->lists > 1 && !buildstate->ivfleader)
|
if (buildstate->lists > 1)
|
||||||
{
|
{
|
||||||
double db = 0.0;
|
double db = 0.0;
|
||||||
|
|
||||||
@@ -551,432 +558,43 @@ PrintKmeansMetrics(IvfflatBuildState * buildstate)
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/*
|
|
||||||
* Within leader, wait for end of heap scan
|
|
||||||
*/
|
|
||||||
static double
|
|
||||||
ParallelHeapScan(IvfflatBuildState * buildstate)
|
|
||||||
{
|
|
||||||
IvfflatShared *ivfshared = buildstate->ivfleader->ivfshared;
|
|
||||||
int nparticipanttuplesorts;
|
|
||||||
double reltuples;
|
|
||||||
|
|
||||||
nparticipanttuplesorts = buildstate->ivfleader->nparticipanttuplesorts;
|
|
||||||
for (;;)
|
|
||||||
{
|
|
||||||
SpinLockAcquire(&ivfshared->mutex);
|
|
||||||
if (ivfshared->nparticipantsdone == nparticipanttuplesorts)
|
|
||||||
{
|
|
||||||
buildstate->indtuples = ivfshared->indtuples;
|
|
||||||
reltuples = ivfshared->reltuples;
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
buildstate->inertia = ivfshared->inertia;
|
|
||||||
#endif
|
|
||||||
SpinLockRelease(&ivfshared->mutex);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
SpinLockRelease(&ivfshared->mutex);
|
|
||||||
|
|
||||||
ConditionVariableSleep(&ivfshared->workersdonecv,
|
|
||||||
WAIT_EVENT_PARALLEL_CREATE_INDEX_SCAN);
|
|
||||||
}
|
|
||||||
|
|
||||||
ConditionVariableCancelSleep();
|
|
||||||
|
|
||||||
return reltuples;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Perform a worker's portion of a parallel sort
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
IvfflatParallelScanAndSort(IvfflatSpool * ivfspool, IvfflatShared * ivfshared, Sharedsort *sharedsort, Vector * ivfcenters, int sortmem, bool progress)
|
|
||||||
{
|
|
||||||
SortCoordinate coordinate;
|
|
||||||
IvfflatBuildState buildstate;
|
|
||||||
TableScanDesc scan;
|
|
||||||
double reltuples;
|
|
||||||
IndexInfo *indexInfo;
|
|
||||||
|
|
||||||
/* Sort options, which must match AssignTuples */
|
|
||||||
AttrNumber attNums[] = {1};
|
|
||||||
Oid sortOperators[] = {Int4LessOperator};
|
|
||||||
Oid sortCollations[] = {InvalidOid};
|
|
||||||
bool nullsFirstFlags[] = {false};
|
|
||||||
|
|
||||||
/* Initialize local tuplesort coordination state */
|
|
||||||
coordinate = palloc0(sizeof(SortCoordinateData));
|
|
||||||
coordinate->isWorker = true;
|
|
||||||
coordinate->nParticipants = -1;
|
|
||||||
coordinate->sharedsort = sharedsort;
|
|
||||||
|
|
||||||
/* Join parallel scan */
|
|
||||||
indexInfo = BuildIndexInfo(ivfspool->index);
|
|
||||||
indexInfo->ii_Concurrent = ivfshared->isconcurrent;
|
|
||||||
InitBuildState(&buildstate, ivfspool->heap, ivfspool->index, indexInfo);
|
|
||||||
memcpy(buildstate.centers->items, ivfcenters, VECTOR_SIZE(buildstate.centers->dim) * buildstate.centers->maxlen);
|
|
||||||
buildstate.centers->length = buildstate.centers->maxlen;
|
|
||||||
ivfspool->sortstate = tuplesort_begin_heap(buildstate.tupdesc, 1, attNums, sortOperators, sortCollations, nullsFirstFlags, sortmem, coordinate, false);
|
|
||||||
buildstate.sortstate = ivfspool->sortstate;
|
|
||||||
scan = table_beginscan_parallel(ivfspool->heap,
|
|
||||||
ParallelTableScanFromIvfflatShared(ivfshared));
|
|
||||||
reltuples = table_index_build_scan(ivfspool->heap, ivfspool->index, indexInfo,
|
|
||||||
true, progress, BuildCallback,
|
|
||||||
(void *) &buildstate, scan);
|
|
||||||
|
|
||||||
/* Execute this worker's part of the sort */
|
|
||||||
tuplesort_performsort(ivfspool->sortstate);
|
|
||||||
|
|
||||||
/* Record statistics */
|
|
||||||
SpinLockAcquire(&ivfshared->mutex);
|
|
||||||
ivfshared->nparticipantsdone++;
|
|
||||||
ivfshared->reltuples += reltuples;
|
|
||||||
ivfshared->indtuples += buildstate.indtuples;
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
ivfshared->inertia += buildstate.inertia;
|
|
||||||
#endif
|
|
||||||
SpinLockRelease(&ivfshared->mutex);
|
|
||||||
|
|
||||||
/* Log statistics */
|
|
||||||
if (progress)
|
|
||||||
ereport(DEBUG1, (errmsg("leader processed " INT64_FORMAT " tuples", (int64) reltuples)));
|
|
||||||
else
|
|
||||||
ereport(DEBUG1, (errmsg("worker processed " INT64_FORMAT " tuples", (int64) reltuples)));
|
|
||||||
|
|
||||||
/* Notify leader */
|
|
||||||
ConditionVariableSignal(&ivfshared->workersdonecv);
|
|
||||||
|
|
||||||
/* We can end tuplesorts immediately */
|
|
||||||
tuplesort_end(ivfspool->sortstate);
|
|
||||||
|
|
||||||
FreeBuildState(&buildstate);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Perform work within a launched parallel process
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
IvfflatParallelBuildMain(dsm_segment *seg, shm_toc *toc)
|
|
||||||
{
|
|
||||||
char *sharedquery;
|
|
||||||
IvfflatSpool *ivfspool;
|
|
||||||
IvfflatShared *ivfshared;
|
|
||||||
Sharedsort *sharedsort;
|
|
||||||
Vector *ivfcenters;
|
|
||||||
Relation heapRel;
|
|
||||||
Relation indexRel;
|
|
||||||
LOCKMODE heapLockmode;
|
|
||||||
LOCKMODE indexLockmode;
|
|
||||||
int sortmem;
|
|
||||||
|
|
||||||
/* Set debug_query_string for individual workers first */
|
|
||||||
sharedquery = shm_toc_lookup(toc, PARALLEL_KEY_QUERY_TEXT, true);
|
|
||||||
debug_query_string = sharedquery;
|
|
||||||
|
|
||||||
/* Report the query string from leader */
|
|
||||||
pgstat_report_activity(STATE_RUNNING, debug_query_string);
|
|
||||||
|
|
||||||
/* Look up shared state */
|
|
||||||
ivfshared = shm_toc_lookup(toc, PARALLEL_KEY_IVFFLAT_SHARED, false);
|
|
||||||
|
|
||||||
/* Open relations using lock modes known to be obtained by index.c */
|
|
||||||
if (!ivfshared->isconcurrent)
|
|
||||||
{
|
|
||||||
heapLockmode = ShareLock;
|
|
||||||
indexLockmode = AccessExclusiveLock;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
heapLockmode = ShareUpdateExclusiveLock;
|
|
||||||
indexLockmode = RowExclusiveLock;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Open relations within worker */
|
|
||||||
heapRel = table_open(ivfshared->heaprelid, heapLockmode);
|
|
||||||
indexRel = index_open(ivfshared->indexrelid, indexLockmode);
|
|
||||||
|
|
||||||
/* Initialize worker's own spool */
|
|
||||||
ivfspool = (IvfflatSpool *) palloc0(sizeof(IvfflatSpool));
|
|
||||||
ivfspool->heap = heapRel;
|
|
||||||
ivfspool->index = indexRel;
|
|
||||||
|
|
||||||
/* Look up shared state private to tuplesort.c */
|
|
||||||
sharedsort = shm_toc_lookup(toc, PARALLEL_KEY_TUPLESORT, false);
|
|
||||||
tuplesort_attach_shared(sharedsort, seg);
|
|
||||||
|
|
||||||
ivfcenters = shm_toc_lookup(toc, PARALLEL_KEY_IVFFLAT_CENTERS, false);
|
|
||||||
|
|
||||||
/* Perform sorting */
|
|
||||||
sortmem = maintenance_work_mem / ivfshared->scantuplesortstates;
|
|
||||||
IvfflatParallelScanAndSort(ivfspool, ivfshared, sharedsort, ivfcenters, sortmem, false);
|
|
||||||
|
|
||||||
/* Close relations within worker */
|
|
||||||
index_close(indexRel, indexLockmode);
|
|
||||||
table_close(heapRel, heapLockmode);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* End parallel build
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
IvfflatEndParallel(IvfflatLeader * ivfleader)
|
|
||||||
{
|
|
||||||
/* Shutdown worker processes */
|
|
||||||
WaitForParallelWorkersToFinish(ivfleader->pcxt);
|
|
||||||
|
|
||||||
/* Free last reference to MVCC snapshot, if one was used */
|
|
||||||
if (IsMVCCSnapshot(ivfleader->snapshot))
|
|
||||||
UnregisterSnapshot(ivfleader->snapshot);
|
|
||||||
DestroyParallelContext(ivfleader->pcxt);
|
|
||||||
ExitParallelMode();
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Return size of shared memory required for parallel index build
|
|
||||||
*/
|
|
||||||
static Size
|
|
||||||
ParallelEstimateShared(Relation heap, Snapshot snapshot)
|
|
||||||
{
|
|
||||||
return add_size(BUFFERALIGN(sizeof(IvfflatShared)), table_parallelscan_estimate(heap, snapshot));
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Within leader, participate as a parallel worker
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
IvfflatLeaderParticipateAsWorker(IvfflatBuildState * buildstate)
|
|
||||||
{
|
|
||||||
IvfflatLeader *ivfleader = buildstate->ivfleader;
|
|
||||||
IvfflatSpool *leaderworker;
|
|
||||||
int sortmem;
|
|
||||||
|
|
||||||
/* Allocate memory and initialize private spool */
|
|
||||||
leaderworker = (IvfflatSpool *) palloc0(sizeof(IvfflatSpool));
|
|
||||||
leaderworker->heap = buildstate->heap;
|
|
||||||
leaderworker->index = buildstate->index;
|
|
||||||
|
|
||||||
/* Perform work common to all participants */
|
|
||||||
sortmem = maintenance_work_mem / ivfleader->nparticipanttuplesorts;
|
|
||||||
IvfflatParallelScanAndSort(leaderworker, ivfleader->ivfshared,
|
|
||||||
ivfleader->sharedsort, ivfleader->ivfcenters,
|
|
||||||
sortmem, true);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Begin parallel build
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
IvfflatBeginParallel(IvfflatBuildState * buildstate, bool isconcurrent, int request)
|
|
||||||
{
|
|
||||||
ParallelContext *pcxt;
|
|
||||||
int scantuplesortstates;
|
|
||||||
Snapshot snapshot;
|
|
||||||
Size estivfshared;
|
|
||||||
Size estsort;
|
|
||||||
Size estcenters;
|
|
||||||
IvfflatShared *ivfshared;
|
|
||||||
Sharedsort *sharedsort;
|
|
||||||
Vector *ivfcenters;
|
|
||||||
IvfflatLeader *ivfleader = (IvfflatLeader *) palloc0(sizeof(IvfflatLeader));
|
|
||||||
bool leaderparticipates = true;
|
|
||||||
int querylen;
|
|
||||||
|
|
||||||
#ifdef DISABLE_LEADER_PARTICIPATION
|
|
||||||
leaderparticipates = false;
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Enter parallel mode and create context */
|
|
||||||
EnterParallelMode();
|
|
||||||
Assert(request > 0);
|
|
||||||
pcxt = CreateParallelContext("vector", "IvfflatParallelBuildMain", request);
|
|
||||||
|
|
||||||
scantuplesortstates = leaderparticipates ? request + 1 : request;
|
|
||||||
|
|
||||||
/* Get snapshot for table scan */
|
|
||||||
if (!isconcurrent)
|
|
||||||
snapshot = SnapshotAny;
|
|
||||||
else
|
|
||||||
snapshot = RegisterSnapshot(GetTransactionSnapshot());
|
|
||||||
|
|
||||||
/* Estimate size of workspaces */
|
|
||||||
estivfshared = ParallelEstimateShared(buildstate->heap, snapshot);
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, estivfshared);
|
|
||||||
estsort = tuplesort_estimate_shared(scantuplesortstates);
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, estsort);
|
|
||||||
estcenters = VECTOR_SIZE(buildstate->dimensions) * buildstate->lists;
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, estcenters);
|
|
||||||
shm_toc_estimate_keys(&pcxt->estimator, 3);
|
|
||||||
|
|
||||||
/* Finally, estimate PARALLEL_KEY_QUERY_TEXT space */
|
|
||||||
if (debug_query_string)
|
|
||||||
{
|
|
||||||
querylen = strlen(debug_query_string);
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, querylen + 1);
|
|
||||||
shm_toc_estimate_keys(&pcxt->estimator, 1);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
querylen = 0; /* keep compiler quiet */
|
|
||||||
|
|
||||||
/* Everyone's had a chance to ask for space, so now create the DSM */
|
|
||||||
InitializeParallelDSM(pcxt);
|
|
||||||
|
|
||||||
/* If no DSM segment was available, back out (do serial build) */
|
|
||||||
if (pcxt->seg == NULL)
|
|
||||||
{
|
|
||||||
if (IsMVCCSnapshot(snapshot))
|
|
||||||
UnregisterSnapshot(snapshot);
|
|
||||||
DestroyParallelContext(pcxt);
|
|
||||||
ExitParallelMode();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Store shared build state, for which we reserved space */
|
|
||||||
ivfshared = (IvfflatShared *) shm_toc_allocate(pcxt->toc, estivfshared);
|
|
||||||
/* Initialize immutable state */
|
|
||||||
ivfshared->heaprelid = RelationGetRelid(buildstate->heap);
|
|
||||||
ivfshared->indexrelid = RelationGetRelid(buildstate->index);
|
|
||||||
ivfshared->isconcurrent = isconcurrent;
|
|
||||||
ivfshared->scantuplesortstates = scantuplesortstates;
|
|
||||||
ConditionVariableInit(&ivfshared->workersdonecv);
|
|
||||||
SpinLockInit(&ivfshared->mutex);
|
|
||||||
/* Initialize mutable state */
|
|
||||||
ivfshared->nparticipantsdone = 0;
|
|
||||||
ivfshared->reltuples = 0;
|
|
||||||
ivfshared->indtuples = 0;
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
ivfshared->inertia = 0;
|
|
||||||
#endif
|
|
||||||
table_parallelscan_initialize(buildstate->heap,
|
|
||||||
ParallelTableScanFromIvfflatShared(ivfshared),
|
|
||||||
snapshot);
|
|
||||||
|
|
||||||
/* Store shared tuplesort-private state, for which we reserved space */
|
|
||||||
sharedsort = (Sharedsort *) shm_toc_allocate(pcxt->toc, estsort);
|
|
||||||
tuplesort_initialize_shared(sharedsort, scantuplesortstates,
|
|
||||||
pcxt->seg);
|
|
||||||
|
|
||||||
ivfcenters = (Vector *) shm_toc_allocate(pcxt->toc, estcenters);
|
|
||||||
memcpy(ivfcenters, buildstate->centers->items, estcenters);
|
|
||||||
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_IVFFLAT_SHARED, ivfshared);
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_TUPLESORT, sharedsort);
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_IVFFLAT_CENTERS, ivfcenters);
|
|
||||||
|
|
||||||
/* Store query string for workers */
|
|
||||||
if (debug_query_string)
|
|
||||||
{
|
|
||||||
char *sharedquery;
|
|
||||||
|
|
||||||
sharedquery = (char *) shm_toc_allocate(pcxt->toc, querylen + 1);
|
|
||||||
memcpy(sharedquery, debug_query_string, querylen + 1);
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_QUERY_TEXT, sharedquery);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Launch workers, saving status for leader/caller */
|
|
||||||
LaunchParallelWorkers(pcxt);
|
|
||||||
ivfleader->pcxt = pcxt;
|
|
||||||
ivfleader->nparticipanttuplesorts = pcxt->nworkers_launched;
|
|
||||||
if (leaderparticipates)
|
|
||||||
ivfleader->nparticipanttuplesorts++;
|
|
||||||
ivfleader->ivfshared = ivfshared;
|
|
||||||
ivfleader->sharedsort = sharedsort;
|
|
||||||
ivfleader->snapshot = snapshot;
|
|
||||||
ivfleader->ivfcenters = ivfcenters;
|
|
||||||
|
|
||||||
/* If no workers were successfully launched, back out (do serial build) */
|
|
||||||
if (pcxt->nworkers_launched == 0)
|
|
||||||
{
|
|
||||||
IvfflatEndParallel(ivfleader);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Log participants */
|
|
||||||
ereport(DEBUG1, (errmsg("using %d parallel workers", pcxt->nworkers_launched)));
|
|
||||||
|
|
||||||
/* Save leader state now that it's clear build will be parallel */
|
|
||||||
buildstate->ivfleader = ivfleader;
|
|
||||||
|
|
||||||
/* Join heap scan ourselves */
|
|
||||||
if (leaderparticipates)
|
|
||||||
IvfflatLeaderParticipateAsWorker(buildstate);
|
|
||||||
|
|
||||||
/* Wait for all launched workers */
|
|
||||||
WaitForParallelWorkersToAttach(pcxt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Scan table for tuples to index
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
AssignTuples(IvfflatBuildState * buildstate)
|
|
||||||
{
|
|
||||||
int parallel_workers = 0;
|
|
||||||
SortCoordinate coordinate = NULL;
|
|
||||||
|
|
||||||
/* Sort options, which must match IvfflatParallelScanAndSort */
|
|
||||||
AttrNumber attNums[] = {1};
|
|
||||||
Oid sortOperators[] = {Int4LessOperator};
|
|
||||||
Oid sortCollations[] = {InvalidOid};
|
|
||||||
bool nullsFirstFlags[] = {false};
|
|
||||||
|
|
||||||
pgstat_progress_update_param(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_ASSIGN);
|
|
||||||
|
|
||||||
/* Calculate parallel workers */
|
|
||||||
if (buildstate->heap != NULL)
|
|
||||||
parallel_workers = plan_create_index_workers(RelationGetRelid(buildstate->heap), RelationGetRelid(buildstate->index));
|
|
||||||
|
|
||||||
/* Attempt to launch parallel worker scan when required */
|
|
||||||
if (parallel_workers > 0)
|
|
||||||
IvfflatBeginParallel(buildstate, buildstate->indexInfo->ii_Concurrent, parallel_workers);
|
|
||||||
|
|
||||||
/* Set up coordination state if at least one worker launched */
|
|
||||||
if (buildstate->ivfleader)
|
|
||||||
{
|
|
||||||
coordinate = (SortCoordinate) palloc0(sizeof(SortCoordinateData));
|
|
||||||
coordinate->isWorker = false;
|
|
||||||
coordinate->nParticipants = buildstate->ivfleader->nparticipanttuplesorts;
|
|
||||||
coordinate->sharedsort = buildstate->ivfleader->sharedsort;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Begin serial/leader tuplesort */
|
|
||||||
buildstate->sortstate = tuplesort_begin_heap(buildstate->tupdesc, 1, attNums, sortOperators, sortCollations, nullsFirstFlags, maintenance_work_mem, coordinate, false);
|
|
||||||
|
|
||||||
/* Add tuples to sort */
|
|
||||||
if (buildstate->heap != NULL)
|
|
||||||
{
|
|
||||||
if (buildstate->ivfleader)
|
|
||||||
buildstate->reltuples = ParallelHeapScan(buildstate);
|
|
||||||
else
|
|
||||||
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
true, true, BuildCallback, (void *) buildstate, NULL);
|
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
PrintKmeansMetrics(buildstate);
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Create entry pages
|
* Create entry pages
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
||||||
{
|
{
|
||||||
/* Assign */
|
AttrNumber attNums[] = {1};
|
||||||
IvfflatBench("assign tuples", AssignTuples(buildstate));
|
Oid sortOperators[] = {Float8LessOperator};
|
||||||
|
Oid sortCollations[] = {InvalidOid};
|
||||||
|
bool nullsFirstFlags[] = {false};
|
||||||
|
|
||||||
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_SORT);
|
||||||
|
|
||||||
|
buildstate->sortstate = tuplesort_begin_heap(buildstate->tupdesc, 1, attNums, sortOperators, sortCollations, nullsFirstFlags, maintenance_work_mem, NULL, false);
|
||||||
|
|
||||||
|
/* Add tuples to sort */
|
||||||
|
if (buildstate->heap != NULL)
|
||||||
|
{
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, true, BuildCallback, (void *) buildstate, NULL);
|
||||||
|
#else
|
||||||
|
buildstate->reltuples = IndexBuildHeapScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, BuildCallback, (void *) buildstate, NULL);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
/* Sort */
|
/* Sort */
|
||||||
IvfflatBench("sort tuples", tuplesort_performsort(buildstate->sortstate));
|
tuplesort_performsort(buildstate->sortstate);
|
||||||
|
|
||||||
/* Load */
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
IvfflatBench("load tuples", InsertTuples(buildstate->index, buildstate, forkNum));
|
PrintKmeansMetrics(buildstate);
|
||||||
|
#endif
|
||||||
|
|
||||||
/* End sort */
|
/* Insert */
|
||||||
|
InsertTuples(buildstate->index, buildstate, forkNum);
|
||||||
tuplesort_end(buildstate->sortstate);
|
tuplesort_end(buildstate->sortstate);
|
||||||
|
|
||||||
/* End parallel build */
|
|
||||||
if (buildstate->ivfleader)
|
|
||||||
IvfflatEndParallel(buildstate->ivfleader);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -993,7 +611,7 @@ BuildIndex(Relation heap, Relation index, IndexInfo *indexInfo,
|
|||||||
/* Create pages */
|
/* Create pages */
|
||||||
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
||||||
CreateListPages(index, buildstate->centers, buildstate->dimensions, buildstate->lists, forkNum, &buildstate->listInfo);
|
CreateListPages(index, buildstate->centers, buildstate->dimensions, buildstate->lists, forkNum, &buildstate->listInfo);
|
||||||
CreateEntryPages(buildstate, forkNum);
|
IvfflatBench("CreateEntryPages", CreateEntryPages(buildstate, forkNum));
|
||||||
|
|
||||||
FreeBuildState(buildstate);
|
FreeBuildState(buildstate);
|
||||||
}
|
}
|
||||||
|
|||||||
116
src/ivfflat.c
116
src/ivfflat.c
@@ -3,16 +3,13 @@
|
|||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
|
||||||
#include "access/amapi.h"
|
#include "access/amapi.h"
|
||||||
#include "access/reloptions.h"
|
|
||||||
#include "commands/progress.h"
|
|
||||||
#include "commands/vacuum.h"
|
#include "commands/vacuum.h"
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "utils/guc.h"
|
#include "utils/guc.h"
|
||||||
#include "utils/selfuncs.h"
|
#include "utils/selfuncs.h"
|
||||||
#include "utils/spccache.h"
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 150000
|
#if PG_VERSION_NUM >= 120000
|
||||||
#define MarkGUCPrefixReserved(x) EmitWarningsOnPlaceholders(x)
|
#include "commands/progress.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
int ivfflat_probes;
|
int ivfflat_probes;
|
||||||
@@ -22,11 +19,11 @@ static relopt_kind ivfflat_relopt_kind;
|
|||||||
* Initialize index options and variables
|
* Initialize index options and variables
|
||||||
*/
|
*/
|
||||||
void
|
void
|
||||||
IvfflatInit(void)
|
_PG_init(void)
|
||||||
{
|
{
|
||||||
ivfflat_relopt_kind = add_reloption_kind();
|
ivfflat_relopt_kind = add_reloption_kind();
|
||||||
add_int_reloption(ivfflat_relopt_kind, "lists", "Number of inverted lists",
|
add_int_reloption(ivfflat_relopt_kind, "lists", "Number of inverted lists",
|
||||||
IVFFLAT_DEFAULT_LISTS, IVFFLAT_MIN_LISTS, IVFFLAT_MAX_LISTS
|
IVFFLAT_DEFAULT_LISTS, 1, IVFFLAT_MAX_LISTS
|
||||||
#if PG_VERSION_NUM >= 130000
|
#if PG_VERSION_NUM >= 130000
|
||||||
,AccessExclusiveLock
|
,AccessExclusiveLock
|
||||||
#endif
|
#endif
|
||||||
@@ -34,14 +31,13 @@ IvfflatInit(void)
|
|||||||
|
|
||||||
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
||||||
"Valid range is 1..lists.", &ivfflat_probes,
|
"Valid range is 1..lists.", &ivfflat_probes,
|
||||||
IVFFLAT_DEFAULT_PROBES, IVFFLAT_MIN_LISTS, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
1, 1, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
||||||
|
|
||||||
MarkGUCPrefixReserved("ivfflat");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Get the name of index build phase
|
* Get the name of index build phase
|
||||||
*/
|
*/
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
static char *
|
static char *
|
||||||
ivfflatbuildphasename(int64 phasenum)
|
ivfflatbuildphasename(int64 phasenum)
|
||||||
{
|
{
|
||||||
@@ -51,14 +47,15 @@ ivfflatbuildphasename(int64 phasenum)
|
|||||||
return "initializing";
|
return "initializing";
|
||||||
case PROGRESS_IVFFLAT_PHASE_KMEANS:
|
case PROGRESS_IVFFLAT_PHASE_KMEANS:
|
||||||
return "performing k-means";
|
return "performing k-means";
|
||||||
case PROGRESS_IVFFLAT_PHASE_ASSIGN:
|
case PROGRESS_IVFFLAT_PHASE_SORT:
|
||||||
return "assigning tuples";
|
return "sorting tuples";
|
||||||
case PROGRESS_IVFFLAT_PHASE_LOAD:
|
case PROGRESS_IVFFLAT_PHASE_LOAD:
|
||||||
return "loading tuples";
|
return "loading tuples";
|
||||||
default:
|
default:
|
||||||
return NULL;
|
return NULL;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Estimate the cost of an index scan
|
* Estimate the cost of an index scan
|
||||||
@@ -66,14 +63,17 @@ ivfflatbuildphasename(int64 phasenum)
|
|||||||
static void
|
static void
|
||||||
ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
||||||
Cost *indexStartupCost, Cost *indexTotalCost,
|
Cost *indexStartupCost, Cost *indexTotalCost,
|
||||||
Selectivity *indexSelectivity, double *indexCorrelation,
|
Selectivity *indexSelectivity, double *indexCorrelation
|
||||||
double *indexPages)
|
,double *indexPages
|
||||||
|
)
|
||||||
{
|
{
|
||||||
GenericCosts costs;
|
GenericCosts costs;
|
||||||
int lists;
|
int lists;
|
||||||
double ratio;
|
double ratio;
|
||||||
double spc_seq_page_cost;
|
Relation indexRel;
|
||||||
Relation index;
|
#if PG_VERSION_NUM < 120000
|
||||||
|
List *qinfos;
|
||||||
|
#endif
|
||||||
|
|
||||||
/* Never use index without order */
|
/* Never use index without order */
|
||||||
if (path->indexorderbys == NULL)
|
if (path->indexorderbys == NULL)
|
||||||
@@ -88,49 +88,30 @@ ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
|||||||
|
|
||||||
MemSet(&costs, 0, sizeof(costs));
|
MemSet(&costs, 0, sizeof(costs));
|
||||||
|
|
||||||
index = index_open(path->indexinfo->indexoid, NoLock);
|
#if PG_VERSION_NUM >= 120000
|
||||||
IvfflatGetMetaPageInfo(index, &lists, NULL);
|
|
||||||
index_close(index, NoLock);
|
|
||||||
|
|
||||||
/* Get the ratio of lists that we need to visit */
|
|
||||||
ratio = ((double) ivfflat_probes) / lists;
|
|
||||||
if (ratio > 1.0)
|
|
||||||
ratio = 1.0;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* This gives us the subset of tuples to visit. This value is passed into
|
|
||||||
* the generic cost estimator to determine the number of pages to visit
|
|
||||||
* during the index scan.
|
|
||||||
*/
|
|
||||||
costs.numIndexTuples = path->indexinfo->tuples * ratio;
|
|
||||||
|
|
||||||
genericcostestimate(root, path, loop_count, &costs);
|
genericcostestimate(root, path, loop_count, &costs);
|
||||||
|
#else
|
||||||
|
qinfos = deconstruct_indexquals(path);
|
||||||
|
genericcostestimate(root, path, loop_count, qinfos, &costs);
|
||||||
|
#endif
|
||||||
|
|
||||||
get_tablespace_page_costs(path->indexinfo->reltablespace, NULL, &spc_seq_page_cost);
|
indexRel = index_open(path->indexinfo->indexoid, NoLock);
|
||||||
|
lists = IvfflatGetLists(indexRel);
|
||||||
|
index_close(indexRel, NoLock);
|
||||||
|
|
||||||
/* Adjust cost if needed since TOAST not included in seq scan cost */
|
ratio = ((double) ivfflat_probes) / lists;
|
||||||
if (costs.numIndexPages > path->indexinfo->rel->pages && ratio < 0.5)
|
if (ratio > 1)
|
||||||
{
|
ratio = 1;
|
||||||
/* Change all page cost from random to sequential */
|
|
||||||
costs.indexTotalCost -= costs.numIndexPages * (costs.spc_random_page_cost - spc_seq_page_cost);
|
|
||||||
|
|
||||||
/* Remove cost of extra pages */
|
// cost estimates for parallel workers applied outside of amcostestimate
|
||||||
costs.indexTotalCost -= (costs.numIndexPages - path->indexinfo->rel->pages) * spc_seq_page_cost;
|
elog(INFO, "parallel_workers = %d, parallel aware = %d", path->path.parallel_workers, path->path.parallel_aware);
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
/* Change some page cost from random to sequential */
|
|
||||||
costs.indexTotalCost -= 0.5 * costs.numIndexPages * (costs.spc_random_page_cost - spc_seq_page_cost);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
costs.indexTotalCost *= ratio;
|
||||||
* If the list selectivity is lower than what is returned from the generic
|
costs.numIndexPages *= ratio;
|
||||||
* cost estimator, use that.
|
|
||||||
*/
|
|
||||||
if (ratio < costs.indexSelectivity)
|
|
||||||
costs.indexSelectivity = ratio;
|
|
||||||
|
|
||||||
/* Use total cost since most work happens before first tuple is returned */
|
elog(INFO, "ivfflatcostestimate = %f", costs.indexTotalCost);
|
||||||
|
|
||||||
|
/* Startup cost and total cost are same */
|
||||||
*indexStartupCost = costs.indexTotalCost;
|
*indexStartupCost = costs.indexTotalCost;
|
||||||
*indexTotalCost = costs.indexTotalCost;
|
*indexTotalCost = costs.indexTotalCost;
|
||||||
*indexSelectivity = costs.indexSelectivity;
|
*indexSelectivity = costs.indexSelectivity;
|
||||||
@@ -176,6 +157,25 @@ ivfflatvalidate(Oid opclassoid)
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static Size
|
||||||
|
ivfflatestimateparallelscan()
|
||||||
|
{
|
||||||
|
elog(INFO, "ivfflatestimateparallelscan");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void
|
||||||
|
ivfflatinitparallelscan(void *target)
|
||||||
|
{
|
||||||
|
elog(INFO, "ivfflatinitparallelscan");
|
||||||
|
}
|
||||||
|
|
||||||
|
static void
|
||||||
|
ivfflatparallelrescan(IndexScanDesc scan)
|
||||||
|
{
|
||||||
|
elog(INFO, "ivfflatparallelrescan");
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Define index handler
|
* Define index handler
|
||||||
*
|
*
|
||||||
@@ -203,7 +203,7 @@ ivfflathandler(PG_FUNCTION_ARGS)
|
|||||||
amroutine->amstorage = false;
|
amroutine->amstorage = false;
|
||||||
amroutine->amclusterable = false;
|
amroutine->amclusterable = false;
|
||||||
amroutine->ampredlocks = false;
|
amroutine->ampredlocks = false;
|
||||||
amroutine->amcanparallel = false;
|
amroutine->amcanparallel = true;
|
||||||
amroutine->amcaninclude = false;
|
amroutine->amcaninclude = false;
|
||||||
#if PG_VERSION_NUM >= 130000
|
#if PG_VERSION_NUM >= 130000
|
||||||
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
|
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
|
||||||
@@ -221,7 +221,9 @@ ivfflathandler(PG_FUNCTION_ARGS)
|
|||||||
amroutine->amcostestimate = ivfflatcostestimate;
|
amroutine->amcostestimate = ivfflatcostestimate;
|
||||||
amroutine->amoptions = ivfflatoptions;
|
amroutine->amoptions = ivfflatoptions;
|
||||||
amroutine->amproperty = NULL; /* TODO AMPROP_DISTANCE_ORDERABLE */
|
amroutine->amproperty = NULL; /* TODO AMPROP_DISTANCE_ORDERABLE */
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
amroutine->ambuildphasename = ivfflatbuildphasename;
|
amroutine->ambuildphasename = ivfflatbuildphasename;
|
||||||
|
#endif
|
||||||
amroutine->amvalidate = ivfflatvalidate;
|
amroutine->amvalidate = ivfflatvalidate;
|
||||||
#if PG_VERSION_NUM >= 140000
|
#if PG_VERSION_NUM >= 140000
|
||||||
amroutine->amadjustmembers = NULL;
|
amroutine->amadjustmembers = NULL;
|
||||||
@@ -235,9 +237,9 @@ ivfflathandler(PG_FUNCTION_ARGS)
|
|||||||
amroutine->amrestrpos = NULL;
|
amroutine->amrestrpos = NULL;
|
||||||
|
|
||||||
/* Interface functions to support parallel index scans */
|
/* Interface functions to support parallel index scans */
|
||||||
amroutine->amestimateparallelscan = NULL;
|
amroutine->amestimateparallelscan = ivfflatestimateparallelscan;
|
||||||
amroutine->aminitparallelscan = NULL;
|
amroutine->aminitparallelscan = ivfflatinitparallelscan;
|
||||||
amroutine->amparallelrescan = NULL;
|
amroutine->amparallelrescan = ivfflatparallelrescan;
|
||||||
|
|
||||||
PG_RETURN_POINTER(amroutine);
|
PG_RETURN_POINTER(amroutine);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,12 +3,14 @@
|
|||||||
|
|
||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include "access/genam.h"
|
#if PG_VERSION_NUM < 110000
|
||||||
|
#error "Requires PostgreSQL 11+"
|
||||||
|
#endif
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
#include "access/generic_xlog.h"
|
||||||
#include "access/parallel.h"
|
#include "access/reloptions.h"
|
||||||
#include "lib/pairingheap.h"
|
|
||||||
#include "nodes/execnodes.h"
|
#include "nodes/execnodes.h"
|
||||||
#include "port.h" /* for random() */
|
#include "port.h" /* for strtof() and random() */
|
||||||
#include "utils/sampling.h"
|
#include "utils/sampling.h"
|
||||||
#include "utils/tuplesort.h"
|
#include "utils/tuplesort.h"
|
||||||
#include "vector.h"
|
#include "vector.h"
|
||||||
@@ -37,16 +39,13 @@
|
|||||||
#define IVFFLAT_METAPAGE_BLKNO 0
|
#define IVFFLAT_METAPAGE_BLKNO 0
|
||||||
#define IVFFLAT_HEAD_BLKNO 1 /* first list page */
|
#define IVFFLAT_HEAD_BLKNO 1 /* first list page */
|
||||||
|
|
||||||
/* IVFFlat parameters */
|
|
||||||
#define IVFFLAT_DEFAULT_LISTS 100
|
#define IVFFLAT_DEFAULT_LISTS 100
|
||||||
#define IVFFLAT_MIN_LISTS 1
|
|
||||||
#define IVFFLAT_MAX_LISTS 32768
|
#define IVFFLAT_MAX_LISTS 32768
|
||||||
#define IVFFLAT_DEFAULT_PROBES 1
|
|
||||||
|
|
||||||
/* Build phases */
|
/* Build phases */
|
||||||
/* PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE is 1 */
|
/* PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE is 1 */
|
||||||
#define PROGRESS_IVFFLAT_PHASE_KMEANS 2
|
#define PROGRESS_IVFFLAT_PHASE_KMEANS 2
|
||||||
#define PROGRESS_IVFFLAT_PHASE_ASSIGN 3
|
#define PROGRESS_IVFFLAT_PHASE_SORT 3
|
||||||
#define PROGRESS_IVFFLAT_PHASE_LOAD 4
|
#define PROGRESS_IVFFLAT_PHASE_LOAD 4
|
||||||
|
|
||||||
#define IVFFLAT_LIST_SIZE(_dim) (offsetof(IvfflatListData, center) + VECTOR_SIZE(_dim))
|
#define IVFFLAT_LIST_SIZE(_dim) (offsetof(IvfflatListData, center) + VECTOR_SIZE(_dim))
|
||||||
@@ -80,6 +79,9 @@
|
|||||||
/* Variables */
|
/* Variables */
|
||||||
extern int ivfflat_probes;
|
extern int ivfflat_probes;
|
||||||
|
|
||||||
|
/* Exported functions */
|
||||||
|
PGDLLEXPORT void _PG_init(void);
|
||||||
|
|
||||||
typedef struct VectorArrayData
|
typedef struct VectorArrayData
|
||||||
{
|
{
|
||||||
int length;
|
int length;
|
||||||
@@ -103,50 +105,6 @@ typedef struct IvfflatOptions
|
|||||||
int lists; /* number of lists */
|
int lists; /* number of lists */
|
||||||
} IvfflatOptions;
|
} IvfflatOptions;
|
||||||
|
|
||||||
typedef struct IvfflatSpool
|
|
||||||
{
|
|
||||||
Tuplesortstate *sortstate;
|
|
||||||
Relation heap;
|
|
||||||
Relation index;
|
|
||||||
} IvfflatSpool;
|
|
||||||
|
|
||||||
typedef struct IvfflatShared
|
|
||||||
{
|
|
||||||
/* Immutable state */
|
|
||||||
Oid heaprelid;
|
|
||||||
Oid indexrelid;
|
|
||||||
bool isconcurrent;
|
|
||||||
int scantuplesortstates;
|
|
||||||
|
|
||||||
/* Worker progress */
|
|
||||||
ConditionVariable workersdonecv;
|
|
||||||
|
|
||||||
/* Mutex for mutable state */
|
|
||||||
slock_t mutex;
|
|
||||||
|
|
||||||
/* Mutable state */
|
|
||||||
int nparticipantsdone;
|
|
||||||
double reltuples;
|
|
||||||
double indtuples;
|
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
double inertia;
|
|
||||||
#endif
|
|
||||||
} IvfflatShared;
|
|
||||||
|
|
||||||
#define ParallelTableScanFromIvfflatShared(shared) \
|
|
||||||
(ParallelTableScanDesc) ((char *) (shared) + BUFFERALIGN(sizeof(IvfflatShared)))
|
|
||||||
|
|
||||||
typedef struct IvfflatLeader
|
|
||||||
{
|
|
||||||
ParallelContext *pcxt;
|
|
||||||
int nparticipanttuplesorts;
|
|
||||||
IvfflatShared *ivfshared;
|
|
||||||
Sharedsort *sharedsort;
|
|
||||||
Snapshot snapshot;
|
|
||||||
Vector *ivfcenters;
|
|
||||||
} IvfflatLeader;
|
|
||||||
|
|
||||||
typedef struct IvfflatBuildState
|
typedef struct IvfflatBuildState
|
||||||
{
|
{
|
||||||
/* Info */
|
/* Info */
|
||||||
@@ -192,9 +150,6 @@ typedef struct IvfflatBuildState
|
|||||||
|
|
||||||
/* Memory */
|
/* Memory */
|
||||||
MemoryContext tmpCtx;
|
MemoryContext tmpCtx;
|
||||||
|
|
||||||
/* Parallel builds */
|
|
||||||
IvfflatLeader *ivfleader;
|
|
||||||
} IvfflatBuildState;
|
} IvfflatBuildState;
|
||||||
|
|
||||||
typedef struct IvfflatMetaPageData
|
typedef struct IvfflatMetaPageData
|
||||||
@@ -235,8 +190,8 @@ typedef struct IvfflatScanList
|
|||||||
typedef struct IvfflatScanOpaqueData
|
typedef struct IvfflatScanOpaqueData
|
||||||
{
|
{
|
||||||
int probes;
|
int probes;
|
||||||
int dimensions;
|
|
||||||
bool first;
|
bool first;
|
||||||
|
Buffer buf;
|
||||||
|
|
||||||
/* Sorting */
|
/* Sorting */
|
||||||
Tuplesortstate *sortstate;
|
Tuplesortstate *sortstate;
|
||||||
@@ -266,18 +221,15 @@ VectorArray VectorArrayInit(int maxlen, int dimensions);
|
|||||||
void VectorArrayFree(VectorArray arr);
|
void VectorArrayFree(VectorArray arr);
|
||||||
void PrintVectorArray(char *msg, VectorArray arr);
|
void PrintVectorArray(char *msg, VectorArray arr);
|
||||||
void IvfflatKmeans(Relation index, VectorArray samples, VectorArray centers);
|
void IvfflatKmeans(Relation index, VectorArray samples, VectorArray centers);
|
||||||
FmgrInfo *IvfflatOptionalProcInfo(Relation index, uint16 procnum);
|
FmgrInfo *IvfflatOptionalProcInfo(Relation rel, uint16 procnum);
|
||||||
bool IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result);
|
bool IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result);
|
||||||
int IvfflatGetLists(Relation index);
|
int IvfflatGetLists(Relation index);
|
||||||
void IvfflatGetMetaPageInfo(Relation index, int *lists, int *dimensions);
|
void IvfflatUpdateList(Relation index, GenericXLogState *state, ListInfo listInfo, BlockNumber insertPage, BlockNumber originalInsertPage, BlockNumber startPage, ForkNumber forkNum);
|
||||||
void IvfflatUpdateList(Relation index, ListInfo listInfo, BlockNumber insertPage, BlockNumber originalInsertPage, BlockNumber startPage, ForkNumber forkNum);
|
|
||||||
void IvfflatCommitBuffer(Buffer buf, GenericXLogState *state);
|
void IvfflatCommitBuffer(Buffer buf, GenericXLogState *state);
|
||||||
void IvfflatAppendPage(Relation index, Buffer *buf, Page *page, GenericXLogState **state, ForkNumber forkNum);
|
void IvfflatAppendPage(Relation index, Buffer *buf, Page *page, GenericXLogState **state, ForkNumber forkNum);
|
||||||
Buffer IvfflatNewBuffer(Relation index, ForkNumber forkNum);
|
Buffer IvfflatNewBuffer(Relation index, ForkNumber forkNum);
|
||||||
void IvfflatInitPage(Buffer buf, Page page);
|
void IvfflatInitPage(Buffer buf, Page page);
|
||||||
void IvfflatInitRegisterPage(Relation index, Buffer *buf, Page *page, GenericXLogState **state);
|
void IvfflatInitRegisterPage(Relation index, Buffer *buf, Page *page, GenericXLogState **state);
|
||||||
void IvfflatInit(void);
|
|
||||||
PGDLLEXPORT void IvfflatParallelBuildMain(dsm_segment *seg, shm_toc *toc);
|
|
||||||
|
|
||||||
/* Index access methods */
|
/* Index access methods */
|
||||||
IndexBuildResult *ivfflatbuild(Relation heap, Relation index, IndexInfo *indexInfo);
|
IndexBuildResult *ivfflatbuild(Relation heap, Relation index, IndexInfo *indexInfo);
|
||||||
|
|||||||
@@ -2,51 +2,44 @@
|
|||||||
|
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
#include "storage/lmgr.h"
|
|
||||||
#include "utils/memutils.h"
|
#include "utils/memutils.h"
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Find the list that minimizes the distance function
|
* Find the list that minimizes the distance function
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
FindInsertPage(Relation index, Datum *values, BlockNumber *insertPage, ListInfo * listInfo)
|
FindInsertPage(Relation rel, Datum *values, BlockNumber *insertPage, ListInfo * listInfo)
|
||||||
{
|
{
|
||||||
|
Buffer cbuf;
|
||||||
|
Page cpage;
|
||||||
|
IvfflatList list;
|
||||||
|
double distance;
|
||||||
double minDistance = DBL_MAX;
|
double minDistance = DBL_MAX;
|
||||||
BlockNumber nextblkno = IVFFLAT_HEAD_BLKNO;
|
BlockNumber nextblkno = IVFFLAT_HEAD_BLKNO;
|
||||||
FmgrInfo *procinfo;
|
FmgrInfo *procinfo;
|
||||||
Oid collation;
|
Oid collation;
|
||||||
|
OffsetNumber offno;
|
||||||
|
OffsetNumber maxoffno;
|
||||||
|
|
||||||
/* Avoid compiler warning */
|
procinfo = index_getprocinfo(rel, 1, IVFFLAT_DISTANCE_PROC);
|
||||||
listInfo->blkno = nextblkno;
|
collation = rel->rd_indcollation[0];
|
||||||
listInfo->offno = FirstOffsetNumber;
|
|
||||||
|
|
||||||
procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
|
|
||||||
collation = index->rd_indcollation[0];
|
|
||||||
|
|
||||||
/* Search all list pages */
|
/* Search all list pages */
|
||||||
while (BlockNumberIsValid(nextblkno))
|
while (BlockNumberIsValid(nextblkno))
|
||||||
{
|
{
|
||||||
Buffer cbuf;
|
cbuf = ReadBuffer(rel, nextblkno);
|
||||||
Page cpage;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
|
|
||||||
cbuf = ReadBuffer(index, nextblkno);
|
|
||||||
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
||||||
cpage = BufferGetPage(cbuf);
|
cpage = BufferGetPage(cbuf);
|
||||||
maxoffno = PageGetMaxOffsetNumber(cpage);
|
maxoffno = PageGetMaxOffsetNumber(cpage);
|
||||||
|
|
||||||
for (OffsetNumber offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
||||||
{
|
{
|
||||||
IvfflatList list;
|
|
||||||
double distance;
|
|
||||||
|
|
||||||
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, values[0], PointerGetDatum(&list->center)));
|
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, values[0], PointerGetDatum(&list->center)));
|
||||||
|
|
||||||
if (distance < minDistance || !BlockNumberIsValid(*insertPage))
|
if (distance < minDistance)
|
||||||
{
|
{
|
||||||
*insertPage = list->insertPage;
|
*insertPage = list->insertPage;
|
||||||
listInfo->blkno = nextblkno;
|
listInfo->blkno = nextblkno;
|
||||||
@@ -65,7 +58,7 @@ FindInsertPage(Relation index, Datum *values, BlockNumber *insertPage, ListInfo
|
|||||||
* Insert a tuple into the index
|
* Insert a tuple into the index
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
InsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, Relation heapRel)
|
InsertTuple(Relation rel, Datum *values, bool *isnull, ItemPointer heap_tid, Relation heapRel)
|
||||||
{
|
{
|
||||||
IndexTuple itup;
|
IndexTuple itup;
|
||||||
Datum value;
|
Datum value;
|
||||||
@@ -82,33 +75,33 @@ InsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, R
|
|||||||
value = PointerGetDatum(PG_DETOAST_DATUM(values[0]));
|
value = PointerGetDatum(PG_DETOAST_DATUM(values[0]));
|
||||||
|
|
||||||
/* Normalize if needed */
|
/* Normalize if needed */
|
||||||
normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_NORM_PROC);
|
normprocinfo = IvfflatOptionalProcInfo(rel, IVFFLAT_NORM_PROC);
|
||||||
if (normprocinfo != NULL)
|
if (normprocinfo != NULL)
|
||||||
{
|
{
|
||||||
if (!IvfflatNormValue(normprocinfo, index->rd_indcollation[0], &value, NULL))
|
if (!IvfflatNormValue(normprocinfo, rel->rd_indcollation[0], &value, NULL))
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Find the insert page - sets the page and list info */
|
/* Find the insert page - sets the page and list info */
|
||||||
FindInsertPage(index, values, &insertPage, &listInfo);
|
FindInsertPage(rel, values, &insertPage, &listInfo);
|
||||||
Assert(BlockNumberIsValid(insertPage));
|
Assert(BlockNumberIsValid(insertPage));
|
||||||
originalInsertPage = insertPage;
|
originalInsertPage = insertPage;
|
||||||
|
|
||||||
/* Form tuple */
|
/* Form tuple */
|
||||||
itup = index_form_tuple(RelationGetDescr(index), &value, isnull);
|
itup = index_form_tuple(RelationGetDescr(rel), &value, isnull);
|
||||||
itup->t_tid = *heap_tid;
|
itup->t_tid = *heap_tid;
|
||||||
|
|
||||||
/* Get tuple size */
|
/* Get tuple size */
|
||||||
itemsz = MAXALIGN(IndexTupleSize(itup));
|
itemsz = MAXALIGN(IndexTupleSize(itup));
|
||||||
Assert(itemsz <= BLCKSZ - MAXALIGN(SizeOfPageHeaderData) - MAXALIGN(sizeof(IvfflatPageOpaqueData)) - sizeof(ItemIdData));
|
Assert(itemsz <= BLCKSZ - MAXALIGN(SizeOfPageHeaderData) - MAXALIGN(sizeof(IvfflatPageOpaqueData)));
|
||||||
|
|
||||||
/* Find a page to insert the item */
|
/* Find a page to insert the item */
|
||||||
for (;;)
|
for (;;)
|
||||||
{
|
{
|
||||||
buf = ReadBuffer(index, insertPage);
|
buf = ReadBuffer(rel, insertPage);
|
||||||
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
||||||
|
|
||||||
state = GenericXLogStart(index);
|
state = GenericXLogStart(rel);
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
page = GenericXLogRegisterBuffer(state, buf, 0);
|
||||||
|
|
||||||
if (PageGetFreeSpace(page) >= itemsz)
|
if (PageGetFreeSpace(page) >= itemsz)
|
||||||
@@ -124,16 +117,23 @@ InsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, R
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
Buffer metabuf;
|
||||||
Buffer newbuf;
|
Buffer newbuf;
|
||||||
Page newpage;
|
Page newpage;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* From ReadBufferExtended: Caller is responsible for ensuring
|
||||||
|
* that only one backend tries to extend a relation at the same
|
||||||
|
* time!
|
||||||
|
*/
|
||||||
|
metabuf = ReadBuffer(rel, IVFFLAT_METAPAGE_BLKNO);
|
||||||
|
LockBuffer(metabuf, BUFFER_LOCK_EXCLUSIVE);
|
||||||
|
|
||||||
/* Add a new page */
|
/* Add a new page */
|
||||||
LockRelationForExtension(index, ExclusiveLock);
|
newbuf = IvfflatNewBuffer(rel, MAIN_FORKNUM);
|
||||||
newbuf = IvfflatNewBuffer(index, MAIN_FORKNUM);
|
newpage = GenericXLogRegisterBuffer(state, newbuf, GENERIC_XLOG_FULL_IMAGE);
|
||||||
UnlockRelationForExtension(index, ExclusiveLock);
|
|
||||||
|
|
||||||
/* Init new page */
|
/* Init new page */
|
||||||
newpage = GenericXLogRegisterBuffer(state, newbuf, GENERIC_XLOG_FULL_IMAGE);
|
|
||||||
IvfflatInitPage(newbuf, newpage);
|
IvfflatInitPage(newbuf, newpage);
|
||||||
|
|
||||||
/* Update insert page */
|
/* Update insert page */
|
||||||
@@ -143,13 +143,18 @@ InsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, R
|
|||||||
IvfflatPageGetOpaque(page)->nextblkno = insertPage;
|
IvfflatPageGetOpaque(page)->nextblkno = insertPage;
|
||||||
|
|
||||||
/* Commit */
|
/* Commit */
|
||||||
|
MarkBufferDirty(newbuf);
|
||||||
|
MarkBufferDirty(buf);
|
||||||
GenericXLogFinish(state);
|
GenericXLogFinish(state);
|
||||||
|
|
||||||
|
/* Unlock extend relation lock as early as possible */
|
||||||
|
UnlockReleaseBuffer(metabuf);
|
||||||
|
|
||||||
/* Unlock previous buffer */
|
/* Unlock previous buffer */
|
||||||
UnlockReleaseBuffer(buf);
|
UnlockReleaseBuffer(buf);
|
||||||
|
|
||||||
/* Prepare new buffer */
|
/* Prepare new buffer */
|
||||||
state = GenericXLogStart(index);
|
state = GenericXLogStart(rel);
|
||||||
buf = newbuf;
|
buf = newbuf;
|
||||||
page = GenericXLogRegisterBuffer(state, buf, 0);
|
page = GenericXLogRegisterBuffer(state, buf, 0);
|
||||||
break;
|
break;
|
||||||
@@ -158,13 +163,13 @@ InsertTuple(Relation index, Datum *values, bool *isnull, ItemPointer heap_tid, R
|
|||||||
|
|
||||||
/* Add to next offset */
|
/* Add to next offset */
|
||||||
if (PageAddItem(page, (Item) itup, itemsz, InvalidOffsetNumber, false, false) == InvalidOffsetNumber)
|
if (PageAddItem(page, (Item) itup, itemsz, InvalidOffsetNumber, false, false) == InvalidOffsetNumber)
|
||||||
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(rel));
|
||||||
|
|
||||||
IvfflatCommitBuffer(buf, state);
|
IvfflatCommitBuffer(buf, state);
|
||||||
|
|
||||||
/* Update the insert page */
|
/* Update the insert page */
|
||||||
if (insertPage != originalInsertPage)
|
if (insertPage != originalInsertPage)
|
||||||
IvfflatUpdateList(index, listInfo, insertPage, originalInsertPage, InvalidBlockNumber, MAIN_FORKNUM);
|
IvfflatUpdateList(rel, state, listInfo, insertPage, originalInsertPage, InvalidBlockNumber, MAIN_FORKNUM);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
|
|||||||
134
src/ivfkmeans.c
134
src/ivfkmeans.c
@@ -1,15 +1,10 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
#include <math.h>
|
|
||||||
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
|
|
||||||
#ifdef IVFFLAT_MEMORY
|
|
||||||
#include "utils/memutils.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Initialize with kmeans++
|
* Initialize with kmeans++
|
||||||
*
|
*
|
||||||
@@ -20,7 +15,12 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers, float *low
|
|||||||
{
|
{
|
||||||
FmgrInfo *procinfo;
|
FmgrInfo *procinfo;
|
||||||
Oid collation;
|
Oid collation;
|
||||||
|
int i;
|
||||||
int64 j;
|
int64 j;
|
||||||
|
double distance;
|
||||||
|
double sum;
|
||||||
|
double choice;
|
||||||
|
Vector *vec;
|
||||||
float *weight = palloc(samples->length * sizeof(float));
|
float *weight = palloc(samples->length * sizeof(float));
|
||||||
int numCenters = centers->maxlen;
|
int numCenters = centers->maxlen;
|
||||||
int numSamples = samples->length;
|
int numSamples = samples->length;
|
||||||
@@ -33,21 +33,17 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers, float *low
|
|||||||
centers->length++;
|
centers->length++;
|
||||||
|
|
||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
weight[j] = FLT_MAX;
|
weight[j] = DBL_MAX;
|
||||||
|
|
||||||
for (int i = 0; i < numCenters; i++)
|
for (i = 0; i < numCenters; i++)
|
||||||
{
|
{
|
||||||
double sum;
|
|
||||||
double choice;
|
|
||||||
|
|
||||||
CHECK_FOR_INTERRUPTS();
|
CHECK_FOR_INTERRUPTS();
|
||||||
|
|
||||||
sum = 0.0;
|
sum = 0.0;
|
||||||
|
|
||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
Vector *vec = VectorArrayGet(samples, j);
|
vec = VectorArrayGet(samples, j);
|
||||||
double distance;
|
|
||||||
|
|
||||||
/* Only need to compute distance for new center */
|
/* Only need to compute distance for new center */
|
||||||
/* TODO Use triangle inequality to reduce distance calculations */
|
/* TODO Use triangle inequality to reduce distance calculations */
|
||||||
@@ -91,12 +87,13 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers, float *low
|
|||||||
static inline void
|
static inline void
|
||||||
ApplyNorm(FmgrInfo *normprocinfo, Oid collation, Vector * vec)
|
ApplyNorm(FmgrInfo *normprocinfo, Oid collation, Vector * vec)
|
||||||
{
|
{
|
||||||
|
int i;
|
||||||
double norm = DatumGetFloat8(FunctionCall1Coll(normprocinfo, collation, PointerGetDatum(vec)));
|
double norm = DatumGetFloat8(FunctionCall1Coll(normprocinfo, collation, PointerGetDatum(vec)));
|
||||||
|
|
||||||
/* TODO Handle zero norm */
|
/* TODO Handle zero norm */
|
||||||
if (norm > 0)
|
if (norm > 0)
|
||||||
{
|
{
|
||||||
for (int i = 0; i < vec->dim; i++)
|
for (i = 0; i < vec->dim; i++)
|
||||||
vec->x[i] /= norm;
|
vec->x[i] /= norm;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -116,6 +113,9 @@ CompareVectors(const void *a, const void *b)
|
|||||||
static void
|
static void
|
||||||
QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
||||||
{
|
{
|
||||||
|
int i;
|
||||||
|
int j;
|
||||||
|
Vector *vec;
|
||||||
int dimensions = centers->dim;
|
int dimensions = centers->dim;
|
||||||
Oid collation = index->rd_indcollation[0];
|
Oid collation = index->rd_indcollation[0];
|
||||||
FmgrInfo *normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_KMEANS_NORM_PROC);
|
FmgrInfo *normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_KMEANS_NORM_PROC);
|
||||||
@@ -124,9 +124,9 @@ QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
if (samples->length > 0)
|
if (samples->length > 0)
|
||||||
{
|
{
|
||||||
qsort(samples->items, samples->length, VECTOR_SIZE(samples->dim), CompareVectors);
|
qsort(samples->items, samples->length, VECTOR_SIZE(samples->dim), CompareVectors);
|
||||||
for (int i = 0; i < samples->length; i++)
|
for (i = 0; i < samples->length; i++)
|
||||||
{
|
{
|
||||||
Vector *vec = VectorArrayGet(samples, i);
|
vec = VectorArrayGet(samples, i);
|
||||||
|
|
||||||
if (i == 0 || CompareVectors(vec, VectorArrayGet(samples, i - 1)) != 0)
|
if (i == 0 || CompareVectors(vec, VectorArrayGet(samples, i - 1)) != 0)
|
||||||
{
|
{
|
||||||
@@ -139,12 +139,12 @@ QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
/* Fill remaining with random data */
|
/* Fill remaining with random data */
|
||||||
while (centers->length < centers->maxlen)
|
while (centers->length < centers->maxlen)
|
||||||
{
|
{
|
||||||
Vector *vec = VectorArrayGet(centers, centers->length);
|
vec = VectorArrayGet(centers, centers->length);
|
||||||
|
|
||||||
SET_VARSIZE(vec, VECTOR_SIZE(dimensions));
|
SET_VARSIZE(vec, VECTOR_SIZE(dimensions));
|
||||||
vec->dim = dimensions;
|
vec->dim = dimensions;
|
||||||
|
|
||||||
for (int j = 0; j < dimensions; j++)
|
for (j = 0; j < dimensions; j++)
|
||||||
vec->x[j] = RandomDouble();
|
vec->x[j] = RandomDouble();
|
||||||
|
|
||||||
/* Normalize if needed (only needed for random centers) */
|
/* Normalize if needed (only needed for random centers) */
|
||||||
@@ -155,23 +155,6 @@ QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#ifdef IVFFLAT_MEMORY
|
|
||||||
/*
|
|
||||||
* Show memory usage
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
ShowMemoryUsage(Size estimatedSize)
|
|
||||||
{
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
elog(INFO, "total memory: %zu MB",
|
|
||||||
MemoryContextMemAllocated(CurrentMemoryContext, true) / (1024 * 1024));
|
|
||||||
#else
|
|
||||||
MemoryContextStats(CurrentMemoryContext);
|
|
||||||
#endif
|
|
||||||
elog(INFO, "estimated memory: %zu MB", estimatedSize / (1024 * 1024));
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Use Elkan for performance. This requires distance function to satisfy triangle inequality.
|
* Use Elkan for performance. This requires distance function to satisfy triangle inequality.
|
||||||
*
|
*
|
||||||
@@ -188,6 +171,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
Oid collation;
|
Oid collation;
|
||||||
Vector *vec;
|
Vector *vec;
|
||||||
Vector *newCenter;
|
Vector *newCenter;
|
||||||
|
int iteration;
|
||||||
int64 j;
|
int64 j;
|
||||||
int64 k;
|
int64 k;
|
||||||
int dimensions = centers->dim;
|
int dimensions = centers->dim;
|
||||||
@@ -201,6 +185,14 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
float *s;
|
float *s;
|
||||||
float *halfcdist;
|
float *halfcdist;
|
||||||
float *newcdist;
|
float *newcdist;
|
||||||
|
int changes;
|
||||||
|
double minDistance;
|
||||||
|
int closestCenter;
|
||||||
|
double distance;
|
||||||
|
bool rj;
|
||||||
|
bool rjreset;
|
||||||
|
double dxcx;
|
||||||
|
double dxc;
|
||||||
|
|
||||||
/* Calculate allocation sizes */
|
/* Calculate allocation sizes */
|
||||||
Size samplesSize = VECTOR_ARRAY_SIZE(samples->maxlen, samples->dim);
|
Size samplesSize = VECTOR_ARRAY_SIZE(samples->maxlen, samples->dim);
|
||||||
@@ -219,7 +211,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
/* Check memory requirements */
|
/* Check memory requirements */
|
||||||
/* Add one to error message to ceil */
|
/* Add one to error message to ceil */
|
||||||
if (totalSize > (Size) maintenance_work_mem * 1024L)
|
if (totalSize / 1024 > maintenance_work_mem)
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
||||||
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
||||||
@@ -252,24 +244,20 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
vec->dim = dimensions;
|
vec->dim = dimensions;
|
||||||
}
|
}
|
||||||
|
|
||||||
#ifdef IVFFLAT_MEMORY
|
|
||||||
ShowMemoryUsage(totalSize);
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Pick initial centers */
|
/* Pick initial centers */
|
||||||
InitCenters(index, samples, centers, lowerBound);
|
InitCenters(index, samples, centers, lowerBound);
|
||||||
|
|
||||||
/* Assign each x to its closest initial center c(x) = argmin d(x,c) */
|
/* Assign each x to its closest initial center c(x) = argmin d(x,c) */
|
||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
float minDistance = FLT_MAX;
|
minDistance = DBL_MAX;
|
||||||
int closestCenter = 0;
|
closestCenter = -1;
|
||||||
|
|
||||||
/* Find closest center */
|
/* Find closest center */
|
||||||
for (k = 0; k < numCenters; k++)
|
for (k = 0; k < numCenters; k++)
|
||||||
{
|
{
|
||||||
/* TODO Use Lemma 1 in k-means++ initialization */
|
/* TODO Use Lemma 1 in k-means++ initialization */
|
||||||
float distance = lowerBound[j * numCenters + k];
|
distance = lowerBound[j * numCenters + k];
|
||||||
|
|
||||||
if (distance < minDistance)
|
if (distance < minDistance)
|
||||||
{
|
{
|
||||||
@@ -283,14 +271,13 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/* Give 500 iterations to converge */
|
/* Give 500 iterations to converge */
|
||||||
for (int iteration = 0; iteration < 500; iteration++)
|
for (iteration = 0; iteration < 500; iteration++)
|
||||||
{
|
{
|
||||||
int changes = 0;
|
|
||||||
bool rjreset;
|
|
||||||
|
|
||||||
/* Can take a while, so ensure we can interrupt */
|
/* Can take a while, so ensure we can interrupt */
|
||||||
CHECK_FOR_INTERRUPTS();
|
CHECK_FOR_INTERRUPTS();
|
||||||
|
|
||||||
|
changes = 0;
|
||||||
|
|
||||||
/* Step 1: For all centers, compute distance */
|
/* Step 1: For all centers, compute distance */
|
||||||
for (j = 0; j < numCenters; j++)
|
for (j = 0; j < numCenters; j++)
|
||||||
{
|
{
|
||||||
@@ -298,8 +285,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
for (k = j + 1; k < numCenters; k++)
|
for (k = j + 1; k < numCenters; k++)
|
||||||
{
|
{
|
||||||
float distance = 0.5 * DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
distance = 0.5 * DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
||||||
|
|
||||||
halfcdist[j * numCenters + k] = distance;
|
halfcdist[j * numCenters + k] = distance;
|
||||||
halfcdist[k * numCenters + j] = distance;
|
halfcdist[k * numCenters + j] = distance;
|
||||||
}
|
}
|
||||||
@@ -308,12 +294,10 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
/* For all centers c, compute s(c) */
|
/* For all centers c, compute s(c) */
|
||||||
for (j = 0; j < numCenters; j++)
|
for (j = 0; j < numCenters; j++)
|
||||||
{
|
{
|
||||||
float minDistance = FLT_MAX;
|
minDistance = DBL_MAX;
|
||||||
|
|
||||||
for (k = 0; k < numCenters; k++)
|
for (k = 0; k < numCenters; k++)
|
||||||
{
|
{
|
||||||
float distance;
|
|
||||||
|
|
||||||
if (j == k)
|
if (j == k)
|
||||||
continue;
|
continue;
|
||||||
|
|
||||||
@@ -329,8 +313,6 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
bool rj;
|
|
||||||
|
|
||||||
/* Step 2: Identify all points x such that u(x) <= s(c(x)) */
|
/* Step 2: Identify all points x such that u(x) <= s(c(x)) */
|
||||||
if (upperBound[j] <= s[closestCenters[j]])
|
if (upperBound[j] <= s[closestCenters[j]])
|
||||||
continue;
|
continue;
|
||||||
@@ -339,8 +321,6 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
for (k = 0; k < numCenters; k++)
|
for (k = 0; k < numCenters; k++)
|
||||||
{
|
{
|
||||||
float dxcx;
|
|
||||||
|
|
||||||
/* Step 3: For all remaining points x and centers c */
|
/* Step 3: For all remaining points x and centers c */
|
||||||
if (k == closestCenters[j])
|
if (k == closestCenters[j])
|
||||||
continue;
|
continue;
|
||||||
@@ -370,7 +350,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
/* Step 3b */
|
/* Step 3b */
|
||||||
if (dxcx > lowerBound[j * numCenters + k] || dxcx > halfcdist[closestCenters[j] * numCenters + k])
|
if (dxcx > lowerBound[j * numCenters + k] || dxcx > halfcdist[closestCenters[j] * numCenters + k])
|
||||||
{
|
{
|
||||||
float dxc = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
dxc = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
||||||
|
|
||||||
/* d(x,c) calculated */
|
/* d(x,c) calculated */
|
||||||
lowerBound[j * numCenters + k] = dxc;
|
lowerBound[j * numCenters + k] = dxc;
|
||||||
@@ -384,6 +364,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
changes++;
|
changes++;
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -400,8 +381,6 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
int closestCenter;
|
|
||||||
|
|
||||||
vec = VectorArrayGet(samples, j);
|
vec = VectorArrayGet(samples, j);
|
||||||
closestCenter = closestCenters[j];
|
closestCenter = closestCenters[j];
|
||||||
|
|
||||||
@@ -419,14 +398,6 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
if (centerCounts[j] > 0)
|
if (centerCounts[j] > 0)
|
||||||
{
|
{
|
||||||
/* Double avoids overflow, but requires more memory */
|
|
||||||
/* TODO Update bounds */
|
|
||||||
for (k = 0; k < dimensions; k++)
|
|
||||||
{
|
|
||||||
if (isinf(vec->x[k]))
|
|
||||||
vec->x[k] = vec->x[k] > 0 ? FLT_MAX : -FLT_MAX;
|
|
||||||
}
|
|
||||||
|
|
||||||
for (k = 0; k < dimensions; k++)
|
for (k = 0; k < dimensions; k++)
|
||||||
vec->x[k] /= centerCounts[j];
|
vec->x[k] /= centerCounts[j];
|
||||||
}
|
}
|
||||||
@@ -450,7 +421,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
{
|
{
|
||||||
for (k = 0; k < numCenters; k++)
|
for (k = 0; k < numCenters; k++)
|
||||||
{
|
{
|
||||||
float distance = lowerBound[j * numCenters + k] - newcdist[k];
|
distance = lowerBound[j * numCenters + k] - newcdist[k];
|
||||||
|
|
||||||
if (distance < 0)
|
if (distance < 0)
|
||||||
distance = 0;
|
distance = 0;
|
||||||
@@ -466,7 +437,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
/* Step 7 */
|
/* Step 7 */
|
||||||
for (j = 0; j < numCenters; j++)
|
for (j = 0; j < numCenters; j++)
|
||||||
VectorArraySet(centers, j, VectorArrayGet(newCenters, j));
|
memcpy(VectorArrayGet(centers, j), VectorArrayGet(newCenters, j), VECTOR_SIZE(dimensions));
|
||||||
|
|
||||||
if (changes == 0 && iteration != 0)
|
if (changes == 0 && iteration != 0)
|
||||||
break;
|
break;
|
||||||
@@ -489,29 +460,17 @@ static void
|
|||||||
CheckCenters(Relation index, VectorArray centers)
|
CheckCenters(Relation index, VectorArray centers)
|
||||||
{
|
{
|
||||||
FmgrInfo *normprocinfo;
|
FmgrInfo *normprocinfo;
|
||||||
|
Oid collation;
|
||||||
|
int i;
|
||||||
|
double norm;
|
||||||
|
|
||||||
if (centers->length != centers->maxlen)
|
if (centers->length != centers->maxlen)
|
||||||
elog(ERROR, "Not enough centers. Please report a bug.");
|
elog(ERROR, "Not enough centers. Please report a bug.");
|
||||||
|
|
||||||
/* Ensure no NaN or infinite values */
|
|
||||||
for (int i = 0; i < centers->length; i++)
|
|
||||||
{
|
|
||||||
Vector *vec = VectorArrayGet(centers, i);
|
|
||||||
|
|
||||||
for (int j = 0; j < vec->dim; j++)
|
|
||||||
{
|
|
||||||
if (isnan(vec->x[j]))
|
|
||||||
elog(ERROR, "NaN detected. Please report a bug.");
|
|
||||||
|
|
||||||
if (isinf(vec->x[j]))
|
|
||||||
elog(ERROR, "Infinite value detected. Please report a bug.");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Ensure no duplicate centers */
|
/* Ensure no duplicate centers */
|
||||||
/* Fine to sort in-place */
|
/* Fine to sort in-place */
|
||||||
qsort(centers->items, centers->length, VECTOR_SIZE(centers->dim), CompareVectors);
|
qsort(centers->items, centers->length, VECTOR_SIZE(centers->dim), CompareVectors);
|
||||||
for (int i = 1; i < centers->length; i++)
|
for (i = 1; i < centers->length; i++)
|
||||||
{
|
{
|
||||||
if (CompareVectors(VectorArrayGet(centers, i), VectorArrayGet(centers, i - 1)) == 0)
|
if (CompareVectors(VectorArrayGet(centers, i), VectorArrayGet(centers, i - 1)) == 0)
|
||||||
elog(ERROR, "Duplicate centers detected. Please report a bug.");
|
elog(ERROR, "Duplicate centers detected. Please report a bug.");
|
||||||
@@ -522,12 +481,11 @@ CheckCenters(Relation index, VectorArray centers)
|
|||||||
normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_NORM_PROC);
|
normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_NORM_PROC);
|
||||||
if (normprocinfo != NULL)
|
if (normprocinfo != NULL)
|
||||||
{
|
{
|
||||||
Oid collation = index->rd_indcollation[0];
|
collation = index->rd_indcollation[0];
|
||||||
|
|
||||||
for (int i = 0; i < centers->length; i++)
|
for (i = 0; i < centers->length; i++)
|
||||||
{
|
{
|
||||||
double norm = DatumGetFloat8(FunctionCall1Coll(normprocinfo, collation, PointerGetDatum(VectorArrayGet(centers, i))));
|
norm = DatumGetFloat8(FunctionCall1Coll(normprocinfo, collation, PointerGetDatum(VectorArrayGet(centers, i))));
|
||||||
|
|
||||||
if (norm == 0)
|
if (norm == 0)
|
||||||
elog(ERROR, "Zero norm detected. Please report a bug.");
|
elog(ERROR, "Zero norm detected. Please report a bug.");
|
||||||
}
|
}
|
||||||
|
|||||||
144
src/ivfscan.c
144
src/ivfscan.c
@@ -3,14 +3,14 @@
|
|||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
|
||||||
#include "access/relscan.h"
|
#include "access/relscan.h"
|
||||||
#include "catalog/pg_operator_d.h"
|
|
||||||
#include "catalog/pg_type_d.h"
|
|
||||||
#include "lib/pairingheap.h"
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "pgstat.h"
|
#include "pgstat.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
|
|
||||||
|
#include "catalog/pg_operator_d.h"
|
||||||
|
#include "catalog/pg_type_d.h"
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Compare list distances
|
* Compare list distances
|
||||||
*/
|
*/
|
||||||
@@ -32,36 +32,36 @@ CompareLists(const pairingheap_node *a, const pairingheap_node *b, void *arg)
|
|||||||
static void
|
static void
|
||||||
GetScanLists(IndexScanDesc scan, Datum value)
|
GetScanLists(IndexScanDesc scan, Datum value)
|
||||||
{
|
{
|
||||||
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
Buffer cbuf;
|
||||||
|
Page cpage;
|
||||||
|
IvfflatList list;
|
||||||
|
OffsetNumber offno;
|
||||||
|
OffsetNumber maxoffno;
|
||||||
BlockNumber nextblkno = IVFFLAT_HEAD_BLKNO;
|
BlockNumber nextblkno = IVFFLAT_HEAD_BLKNO;
|
||||||
int listCount = 0;
|
int listCount = 0;
|
||||||
|
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
||||||
|
double distance;
|
||||||
|
IvfflatScanList *scanlist;
|
||||||
double maxDistance = DBL_MAX;
|
double maxDistance = DBL_MAX;
|
||||||
|
|
||||||
/* Search all list pages */
|
/* Search all list pages */
|
||||||
while (BlockNumberIsValid(nextblkno))
|
while (BlockNumberIsValid(nextblkno))
|
||||||
{
|
{
|
||||||
Buffer cbuf;
|
|
||||||
Page cpage;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
|
|
||||||
cbuf = ReadBuffer(scan->indexRelation, nextblkno);
|
cbuf = ReadBuffer(scan->indexRelation, nextblkno);
|
||||||
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
||||||
cpage = BufferGetPage(cbuf);
|
cpage = BufferGetPage(cbuf);
|
||||||
|
|
||||||
maxoffno = PageGetMaxOffsetNumber(cpage);
|
maxoffno = PageGetMaxOffsetNumber(cpage);
|
||||||
|
|
||||||
for (OffsetNumber offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
||||||
{
|
{
|
||||||
IvfflatList list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
||||||
double distance;
|
|
||||||
|
|
||||||
/* Use procinfo from the index instead of scan key for performance */
|
/* Use procinfo from the index instead of scan key for performance */
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(so->procinfo, so->collation, PointerGetDatum(&list->center), value));
|
distance = DatumGetFloat8(FunctionCall2Coll(so->procinfo, so->collation, PointerGetDatum(&list->center), value));
|
||||||
|
|
||||||
if (listCount < so->probes)
|
if (listCount < so->probes)
|
||||||
{
|
{
|
||||||
IvfflatScanList *scanlist;
|
|
||||||
|
|
||||||
scanlist = &so->lists[listCount];
|
scanlist = &so->lists[listCount];
|
||||||
scanlist->startPage = list->startPage;
|
scanlist->startPage = list->startPage;
|
||||||
scanlist->distance = distance;
|
scanlist->distance = distance;
|
||||||
@@ -76,8 +76,6 @@ GetScanLists(IndexScanDesc scan, Datum value)
|
|||||||
}
|
}
|
||||||
else if (distance < maxDistance)
|
else if (distance < maxDistance)
|
||||||
{
|
{
|
||||||
IvfflatScanList *scanlist;
|
|
||||||
|
|
||||||
/* Remove */
|
/* Remove */
|
||||||
scanlist = (IvfflatScanList *) pairingheap_remove_first(so->listQueue);
|
scanlist = (IvfflatScanList *) pairingheap_remove_first(so->listQueue);
|
||||||
|
|
||||||
@@ -104,9 +102,21 @@ static void
|
|||||||
GetScanItems(IndexScanDesc scan, Datum value)
|
GetScanItems(IndexScanDesc scan, Datum value)
|
||||||
{
|
{
|
||||||
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
||||||
|
Buffer buf;
|
||||||
|
Page page;
|
||||||
|
IndexTuple itup;
|
||||||
|
BlockNumber searchPage;
|
||||||
|
OffsetNumber offno;
|
||||||
|
OffsetNumber maxoffno;
|
||||||
|
Datum datum;
|
||||||
|
bool isnull;
|
||||||
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
||||||
double tuples = 0;
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
||||||
|
#else
|
||||||
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc);
|
||||||
|
#endif
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Reuse same set of shared buffers for scan
|
* Reuse same set of shared buffers for scan
|
||||||
@@ -118,28 +128,19 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
/* Search closest probes lists */
|
/* Search closest probes lists */
|
||||||
while (!pairingheap_is_empty(so->listQueue))
|
while (!pairingheap_is_empty(so->listQueue))
|
||||||
{
|
{
|
||||||
BlockNumber searchPage = ((IvfflatScanList *) pairingheap_remove_first(so->listQueue))->startPage;
|
searchPage = ((IvfflatScanList *) pairingheap_remove_first(so->listQueue))->startPage;
|
||||||
|
|
||||||
/* Search all entry pages for list */
|
/* Search all entry pages for list */
|
||||||
while (BlockNumberIsValid(searchPage))
|
while (BlockNumberIsValid(searchPage))
|
||||||
{
|
{
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
|
|
||||||
buf = ReadBufferExtended(scan->indexRelation, MAIN_FORKNUM, searchPage, RBM_NORMAL, bas);
|
buf = ReadBufferExtended(scan->indexRelation, MAIN_FORKNUM, searchPage, RBM_NORMAL, bas);
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
||||||
page = BufferGetPage(buf);
|
page = BufferGetPage(buf);
|
||||||
maxoffno = PageGetMaxOffsetNumber(page);
|
maxoffno = PageGetMaxOffsetNumber(page);
|
||||||
|
|
||||||
for (OffsetNumber offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
||||||
{
|
{
|
||||||
IndexTuple itup;
|
itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, offno));
|
||||||
Datum datum;
|
|
||||||
bool isnull;
|
|
||||||
ItemId itemid = PageGetItemId(page, offno);
|
|
||||||
|
|
||||||
itup = (IndexTuple) PageGetItem(page, itemid);
|
|
||||||
datum = index_getattr(itup, 1, tupdesc, &isnull);
|
datum = index_getattr(itup, 1, tupdesc, &isnull);
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -153,11 +154,11 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
slot->tts_isnull[0] = false;
|
slot->tts_isnull[0] = false;
|
||||||
slot->tts_values[1] = PointerGetDatum(&itup->t_tid);
|
slot->tts_values[1] = PointerGetDatum(&itup->t_tid);
|
||||||
slot->tts_isnull[1] = false;
|
slot->tts_isnull[1] = false;
|
||||||
|
slot->tts_values[2] = Int32GetDatum((int) searchPage);
|
||||||
|
slot->tts_isnull[2] = false;
|
||||||
ExecStoreVirtualTuple(slot);
|
ExecStoreVirtualTuple(slot);
|
||||||
|
|
||||||
tuplesort_puttupleslot(so->sortstate, slot);
|
tuplesort_puttupleslot(so->sortstate, slot);
|
||||||
|
|
||||||
tuples++;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
||||||
@@ -166,14 +167,6 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
FreeAccessStrategy(bas);
|
|
||||||
|
|
||||||
if (tuples < 100)
|
|
||||||
ereport(DEBUG1,
|
|
||||||
(errmsg("index scan found few tuples"),
|
|
||||||
errdetail("Index may have been created with little data."),
|
|
||||||
errhint("Recreate the index and possibly decrease lists.")));
|
|
||||||
|
|
||||||
tuplesort_performsort(so->sortstate);
|
tuplesort_performsort(so->sortstate);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -186,7 +179,6 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
IndexScanDesc scan;
|
IndexScanDesc scan;
|
||||||
IvfflatScanOpaque so;
|
IvfflatScanOpaque so;
|
||||||
int lists;
|
int lists;
|
||||||
int dimensions;
|
|
||||||
AttrNumber attNums[] = {1};
|
AttrNumber attNums[] = {1};
|
||||||
Oid sortOperators[] = {Float8LessOperator};
|
Oid sortOperators[] = {Float8LessOperator};
|
||||||
Oid sortCollations[] = {InvalidOid};
|
Oid sortCollations[] = {InvalidOid};
|
||||||
@@ -194,17 +186,15 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
int probes = ivfflat_probes;
|
int probes = ivfflat_probes;
|
||||||
|
|
||||||
scan = RelationGetIndexScan(index, nkeys, norderbys);
|
scan = RelationGetIndexScan(index, nkeys, norderbys);
|
||||||
|
lists = IvfflatGetLists(scan->indexRelation);
|
||||||
/* Get lists and dimensions from metapage */
|
|
||||||
IvfflatGetMetaPageInfo(index, &lists, &dimensions);
|
|
||||||
|
|
||||||
if (probes > lists)
|
if (probes > lists)
|
||||||
probes = lists;
|
probes = lists;
|
||||||
|
|
||||||
so = (IvfflatScanOpaque) palloc(offsetof(IvfflatScanOpaqueData, lists) + probes * sizeof(IvfflatScanList));
|
so = (IvfflatScanOpaque) palloc(offsetof(IvfflatScanOpaqueData, lists) + probes * sizeof(IvfflatScanList));
|
||||||
|
so->buf = InvalidBuffer;
|
||||||
so->first = true;
|
so->first = true;
|
||||||
so->probes = probes;
|
so->probes = probes;
|
||||||
so->dimensions = dimensions;
|
|
||||||
|
|
||||||
/* Set support functions */
|
/* Set support functions */
|
||||||
so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
|
so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
|
||||||
@@ -212,14 +202,23 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
so->collation = index->rd_indcollation[0];
|
so->collation = index->rd_indcollation[0];
|
||||||
|
|
||||||
/* Create tuple description for sorting */
|
/* Create tuple description for sorting */
|
||||||
so->tupdesc = CreateTemplateTupleDesc(2);
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
so->tupdesc = CreateTemplateTupleDesc(3);
|
||||||
|
#else
|
||||||
|
so->tupdesc = CreateTemplateTupleDesc(3, false);
|
||||||
|
#endif
|
||||||
TupleDescInitEntry(so->tupdesc, (AttrNumber) 1, "distance", FLOAT8OID, -1, 0);
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 1, "distance", FLOAT8OID, -1, 0);
|
||||||
TupleDescInitEntry(so->tupdesc, (AttrNumber) 2, "heaptid", TIDOID, -1, 0);
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 2, "tid", TIDOID, -1, 0);
|
||||||
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 3, "indexblkno", INT4OID, -1, 0);
|
||||||
|
|
||||||
/* Prep sort */
|
/* Prep sort */
|
||||||
so->sortstate = tuplesort_begin_heap(so->tupdesc, 1, attNums, sortOperators, sortCollations, nullsFirstFlags, work_mem, NULL, false);
|
so->sortstate = tuplesort_begin_heap(so->tupdesc, 1, attNums, sortOperators, sortCollations, nullsFirstFlags, work_mem, NULL, false);
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
so->slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsMinimalTuple);
|
so->slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsMinimalTuple);
|
||||||
|
#else
|
||||||
|
so->slot = MakeSingleTupleTableSlot(so->tupdesc);
|
||||||
|
#endif
|
||||||
|
|
||||||
so->listQueue = pairingheap_allocate(CompareLists, scan);
|
so->listQueue = pairingheap_allocate(CompareLists, scan);
|
||||||
|
|
||||||
@@ -276,24 +275,21 @@ ivfflatgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
if (scan->orderByData == NULL)
|
if (scan->orderByData == NULL)
|
||||||
elog(ERROR, "cannot scan ivfflat index without order");
|
elog(ERROR, "cannot scan ivfflat index without order");
|
||||||
|
|
||||||
/* Requires MVCC-compliant snapshot as not able to pin during sorting */
|
/* No items will match if null */
|
||||||
/* https://www.postgresql.org/docs/current/index-locking.html */
|
|
||||||
if (!IsMVCCSnapshot(scan->xs_snapshot))
|
|
||||||
elog(ERROR, "non-MVCC snapshots are not supported with ivfflat");
|
|
||||||
|
|
||||||
if (scan->orderByData->sk_flags & SK_ISNULL)
|
if (scan->orderByData->sk_flags & SK_ISNULL)
|
||||||
value = PointerGetDatum(InitVector(so->dimensions));
|
return false;
|
||||||
else
|
|
||||||
|
value = scan->orderByData->sk_argument;
|
||||||
|
|
||||||
|
/* Value should not be compressed or toasted */
|
||||||
|
Assert(!VARATT_IS_COMPRESSED(DatumGetPointer(value)));
|
||||||
|
Assert(!VARATT_IS_EXTENDED(DatumGetPointer(value)));
|
||||||
|
|
||||||
|
if (so->normprocinfo != NULL)
|
||||||
{
|
{
|
||||||
value = scan->orderByData->sk_argument;
|
/* No items will match if normalization fails */
|
||||||
|
if (!IvfflatNormValue(so->normprocinfo, so->collation, &value, NULL))
|
||||||
/* Value should not be compressed or toasted */
|
return false;
|
||||||
Assert(!VARATT_IS_COMPRESSED(DatumGetPointer(value)));
|
|
||||||
Assert(!VARATT_IS_EXTENDED(DatumGetPointer(value)));
|
|
||||||
|
|
||||||
/* Fine if normalization fails */
|
|
||||||
if (so->normprocinfo != NULL)
|
|
||||||
IvfflatNormValue(so->normprocinfo, so->collation, &value, NULL);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
IvfflatBench("GetScanLists", GetScanLists(scan, value));
|
IvfflatBench("GetScanLists", GetScanLists(scan, value));
|
||||||
@@ -307,10 +303,26 @@ ivfflatgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
|
|
||||||
if (tuplesort_gettupleslot(so->sortstate, true, false, so->slot, NULL))
|
if (tuplesort_gettupleslot(so->sortstate, true, false, so->slot, NULL))
|
||||||
{
|
{
|
||||||
ItemPointer heaptid = (ItemPointer) DatumGetPointer(slot_getattr(so->slot, 2, &so->isnull));
|
ItemPointer tid = (ItemPointer) DatumGetPointer(slot_getattr(so->slot, 2, &so->isnull));
|
||||||
|
BlockNumber indexblkno = DatumGetInt32(slot_getattr(so->slot, 3, &so->isnull));
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
scan->xs_heaptid = *tid;
|
||||||
|
#else
|
||||||
|
scan->xs_ctup.t_self = *tid;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
if (BufferIsValid(so->buf))
|
||||||
|
ReleaseBuffer(so->buf);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* An index scan must maintain a pin on the index page holding the
|
||||||
|
* item last returned by amgettuple
|
||||||
|
*
|
||||||
|
* https://www.postgresql.org/docs/current/index-locking.html
|
||||||
|
*/
|
||||||
|
so->buf = ReadBuffer(scan->indexRelation, indexblkno);
|
||||||
|
|
||||||
scan->xs_heaptid = *heaptid;
|
|
||||||
scan->xs_recheck = false;
|
|
||||||
scan->xs_recheckorderby = false;
|
scan->xs_recheckorderby = false;
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -326,6 +338,10 @@ ivfflatendscan(IndexScanDesc scan)
|
|||||||
{
|
{
|
||||||
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
||||||
|
|
||||||
|
/* Release pin */
|
||||||
|
if (BufferIsValid(so->buf))
|
||||||
|
ReleaseBuffer(so->buf);
|
||||||
|
|
||||||
pairingheap_free(so->listQueue);
|
pairingheap_free(so->listQueue);
|
||||||
tuplesort_end(so->sortstate);
|
tuplesort_end(so->sortstate);
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
#include "vector.h"
|
#include "vector.h"
|
||||||
@@ -36,7 +35,9 @@ VectorArrayFree(VectorArray arr)
|
|||||||
void
|
void
|
||||||
PrintVectorArray(char *msg, VectorArray arr)
|
PrintVectorArray(char *msg, VectorArray arr)
|
||||||
{
|
{
|
||||||
for (int i = 0; i < arr->length; i++)
|
int i;
|
||||||
|
|
||||||
|
for (i = 0; i < arr->length; i++)
|
||||||
PrintVector(msg, VectorArrayGet(arr, i));
|
PrintVector(msg, VectorArrayGet(arr, i));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -58,12 +59,12 @@ IvfflatGetLists(Relation index)
|
|||||||
* Get proc
|
* Get proc
|
||||||
*/
|
*/
|
||||||
FmgrInfo *
|
FmgrInfo *
|
||||||
IvfflatOptionalProcInfo(Relation index, uint16 procnum)
|
IvfflatOptionalProcInfo(Relation rel, uint16 procnum)
|
||||||
{
|
{
|
||||||
if (!OidIsValid(index_getprocid(index, 1, procnum)))
|
if (!OidIsValid(index_getprocid(rel, 1, procnum)))
|
||||||
return NULL;
|
return NULL;
|
||||||
|
|
||||||
return index_getprocinfo(index, 1, procnum);
|
return index_getprocinfo(rel, 1, procnum);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -77,16 +78,20 @@ IvfflatOptionalProcInfo(Relation index, uint16 procnum)
|
|||||||
bool
|
bool
|
||||||
IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result)
|
IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result)
|
||||||
{
|
{
|
||||||
double norm = DatumGetFloat8(FunctionCall1Coll(procinfo, collation, *value));
|
Vector *v;
|
||||||
|
int i;
|
||||||
|
double norm;
|
||||||
|
|
||||||
|
norm = DatumGetFloat8(FunctionCall1Coll(procinfo, collation, *value));
|
||||||
|
|
||||||
if (norm > 0)
|
if (norm > 0)
|
||||||
{
|
{
|
||||||
Vector *v = DatumGetVector(*value);
|
v = DatumGetVector(*value);
|
||||||
|
|
||||||
if (result == NULL)
|
if (result == NULL)
|
||||||
result = InitVector(v->dim);
|
result = InitVector(v->dim);
|
||||||
|
|
||||||
for (int i = 0; i < v->dim; i++)
|
for (i = 0; i < v->dim; i++)
|
||||||
result->x[i] = v->x[i] / norm;
|
result->x[i] = v->x[i] / norm;
|
||||||
|
|
||||||
*value = PointerGetDatum(result);
|
*value = PointerGetDatum(result);
|
||||||
@@ -137,6 +142,7 @@ IvfflatInitRegisterPage(Relation index, Buffer *buf, Page *page, GenericXLogStat
|
|||||||
void
|
void
|
||||||
IvfflatCommitBuffer(Buffer buf, GenericXLogState *state)
|
IvfflatCommitBuffer(Buffer buf, GenericXLogState *state)
|
||||||
{
|
{
|
||||||
|
MarkBufferDirty(buf);
|
||||||
GenericXLogFinish(state);
|
GenericXLogFinish(state);
|
||||||
UnlockReleaseBuffer(buf);
|
UnlockReleaseBuffer(buf);
|
||||||
}
|
}
|
||||||
@@ -160,6 +166,8 @@ IvfflatAppendPage(Relation index, Buffer *buf, Page *page, GenericXLogState **st
|
|||||||
IvfflatInitPage(newbuf, newpage);
|
IvfflatInitPage(newbuf, newpage);
|
||||||
|
|
||||||
/* Commit */
|
/* Commit */
|
||||||
|
MarkBufferDirty(*buf);
|
||||||
|
MarkBufferDirty(newbuf);
|
||||||
GenericXLogFinish(*state);
|
GenericXLogFinish(*state);
|
||||||
|
|
||||||
/* Unlock */
|
/* Unlock */
|
||||||
@@ -170,40 +178,16 @@ IvfflatAppendPage(Relation index, Buffer *buf, Page *page, GenericXLogState **st
|
|||||||
*buf = newbuf;
|
*buf = newbuf;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
|
||||||
* Get the metapage info
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
IvfflatGetMetaPageInfo(Relation index, int *lists, int *dimensions)
|
|
||||||
{
|
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
IvfflatMetaPage metap;
|
|
||||||
|
|
||||||
buf = ReadBuffer(index, IVFFLAT_METAPAGE_BLKNO);
|
|
||||||
LockBuffer(buf, BUFFER_LOCK_SHARE);
|
|
||||||
page = BufferGetPage(buf);
|
|
||||||
metap = IvfflatPageGetMeta(page);
|
|
||||||
|
|
||||||
*lists = metap->lists;
|
|
||||||
|
|
||||||
if (dimensions != NULL)
|
|
||||||
*dimensions = metap->dimensions;
|
|
||||||
|
|
||||||
UnlockReleaseBuffer(buf);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Update the start or insert page of a list
|
* Update the start or insert page of a list
|
||||||
*/
|
*/
|
||||||
void
|
void
|
||||||
IvfflatUpdateList(Relation index, ListInfo listInfo,
|
IvfflatUpdateList(Relation index, GenericXLogState *state, ListInfo listInfo,
|
||||||
BlockNumber insertPage, BlockNumber originalInsertPage,
|
BlockNumber insertPage, BlockNumber originalInsertPage,
|
||||||
BlockNumber startPage, ForkNumber forkNum)
|
BlockNumber startPage, ForkNumber forkNum)
|
||||||
{
|
{
|
||||||
Buffer buf;
|
Buffer buf;
|
||||||
Page page;
|
Page page;
|
||||||
GenericXLogState *state;
|
|
||||||
IvfflatList list;
|
IvfflatList list;
|
||||||
bool changed = false;
|
bool changed = false;
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
|
||||||
#include "commands/vacuum.h"
|
#include "commands/vacuum.h"
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
@@ -13,23 +12,34 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
IndexBulkDeleteCallback callback, void *callback_state)
|
IndexBulkDeleteCallback callback, void *callback_state)
|
||||||
{
|
{
|
||||||
Relation index = info->index;
|
Relation index = info->index;
|
||||||
BlockNumber blkno = IVFFLAT_HEAD_BLKNO;
|
Buffer cbuf;
|
||||||
|
Page cpage;
|
||||||
|
Buffer buf;
|
||||||
|
Page page;
|
||||||
|
IvfflatList list;
|
||||||
|
IndexTuple itup;
|
||||||
|
ItemPointer htup;
|
||||||
|
OffsetNumber deletable[MaxOffsetNumber];
|
||||||
|
int ndeletable;
|
||||||
|
BlockNumber startPages[MaxOffsetNumber];
|
||||||
|
BlockNumber nextblkno = IVFFLAT_HEAD_BLKNO;
|
||||||
|
BlockNumber searchPage;
|
||||||
|
BlockNumber insertPage;
|
||||||
|
GenericXLogState *state;
|
||||||
|
OffsetNumber coffno;
|
||||||
|
OffsetNumber cmaxoffno;
|
||||||
|
OffsetNumber offno;
|
||||||
|
OffsetNumber maxoffno;
|
||||||
|
ListInfo listInfo;
|
||||||
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
||||||
|
|
||||||
if (stats == NULL)
|
if (stats == NULL)
|
||||||
stats = (IndexBulkDeleteResult *) palloc0(sizeof(IndexBulkDeleteResult));
|
stats = (IndexBulkDeleteResult *) palloc0(sizeof(IndexBulkDeleteResult));
|
||||||
|
|
||||||
/* Iterate over list pages */
|
/* Iterate over list pages */
|
||||||
while (BlockNumberIsValid(blkno))
|
while (BlockNumberIsValid(nextblkno))
|
||||||
{
|
{
|
||||||
Buffer cbuf;
|
cbuf = ReadBuffer(index, nextblkno);
|
||||||
Page cpage;
|
|
||||||
OffsetNumber coffno;
|
|
||||||
OffsetNumber cmaxoffno;
|
|
||||||
BlockNumber startPages[MaxOffsetNumber];
|
|
||||||
ListInfo listInfo;
|
|
||||||
|
|
||||||
cbuf = ReadBuffer(index, blkno);
|
|
||||||
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
LockBuffer(cbuf, BUFFER_LOCK_SHARE);
|
||||||
cpage = BufferGetPage(cbuf);
|
cpage = BufferGetPage(cbuf);
|
||||||
|
|
||||||
@@ -38,32 +48,23 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
/* Iterate over lists */
|
/* Iterate over lists */
|
||||||
for (coffno = FirstOffsetNumber; coffno <= cmaxoffno; coffno = OffsetNumberNext(coffno))
|
for (coffno = FirstOffsetNumber; coffno <= cmaxoffno; coffno = OffsetNumberNext(coffno))
|
||||||
{
|
{
|
||||||
IvfflatList list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, coffno));
|
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, coffno));
|
||||||
|
|
||||||
startPages[coffno - FirstOffsetNumber] = list->startPage;
|
startPages[coffno - FirstOffsetNumber] = list->startPage;
|
||||||
}
|
}
|
||||||
|
|
||||||
listInfo.blkno = blkno;
|
listInfo.blkno = nextblkno;
|
||||||
blkno = IvfflatPageGetOpaque(cpage)->nextblkno;
|
nextblkno = IvfflatPageGetOpaque(cpage)->nextblkno;
|
||||||
|
|
||||||
UnlockReleaseBuffer(cbuf);
|
UnlockReleaseBuffer(cbuf);
|
||||||
|
|
||||||
for (coffno = FirstOffsetNumber; coffno <= cmaxoffno; coffno = OffsetNumberNext(coffno))
|
for (coffno = FirstOffsetNumber; coffno <= cmaxoffno; coffno = OffsetNumberNext(coffno))
|
||||||
{
|
{
|
||||||
BlockNumber searchPage = startPages[coffno - FirstOffsetNumber];
|
searchPage = startPages[coffno - FirstOffsetNumber];
|
||||||
BlockNumber insertPage = InvalidBlockNumber;
|
insertPage = InvalidBlockNumber;
|
||||||
|
|
||||||
/* Iterate over entry pages */
|
/* Iterate over entry pages */
|
||||||
while (BlockNumberIsValid(searchPage))
|
while (BlockNumberIsValid(searchPage))
|
||||||
{
|
{
|
||||||
Buffer buf;
|
|
||||||
Page page;
|
|
||||||
GenericXLogState *state;
|
|
||||||
OffsetNumber offno;
|
|
||||||
OffsetNumber maxoffno;
|
|
||||||
OffsetNumber deletable[MaxOffsetNumber];
|
|
||||||
int ndeletable;
|
|
||||||
|
|
||||||
vacuum_delay_point();
|
vacuum_delay_point();
|
||||||
|
|
||||||
buf = ReadBufferExtended(index, MAIN_FORKNUM, searchPage, RBM_NORMAL, bas);
|
buf = ReadBufferExtended(index, MAIN_FORKNUM, searchPage, RBM_NORMAL, bas);
|
||||||
@@ -85,8 +86,8 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
/* Find deleted tuples */
|
/* Find deleted tuples */
|
||||||
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
for (offno = FirstOffsetNumber; offno <= maxoffno; offno = OffsetNumberNext(offno))
|
||||||
{
|
{
|
||||||
IndexTuple itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, offno));
|
itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, offno));
|
||||||
ItemPointer htup = &(itup->t_tid);
|
htup = &(itup->t_tid);
|
||||||
|
|
||||||
if (callback(htup, callback_state))
|
if (callback(htup, callback_state))
|
||||||
{
|
{
|
||||||
@@ -108,6 +109,7 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
{
|
{
|
||||||
/* Delete tuples */
|
/* Delete tuples */
|
||||||
PageIndexMultiDelete(page, deletable, ndeletable);
|
PageIndexMultiDelete(page, deletable, ndeletable);
|
||||||
|
MarkBufferDirty(buf);
|
||||||
GenericXLogFinish(state);
|
GenericXLogFinish(state);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
@@ -125,13 +127,11 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
if (BlockNumberIsValid(insertPage))
|
if (BlockNumberIsValid(insertPage))
|
||||||
{
|
{
|
||||||
listInfo.offno = coffno;
|
listInfo.offno = coffno;
|
||||||
IvfflatUpdateList(index, listInfo, insertPage, InvalidBlockNumber, InvalidBlockNumber, MAIN_FORKNUM);
|
IvfflatUpdateList(index, state, listInfo, insertPage, InvalidBlockNumber, InvalidBlockNumber, MAIN_FORKNUM);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
FreeAccessStrategy(bas);
|
|
||||||
|
|
||||||
return stats;
|
return stats;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
674
src/vector.c
674
src/vector.c
File diff suppressed because it is too large
Load Diff
24
src/vector.h
24
src/vector.h
@@ -1,6 +1,12 @@
|
|||||||
#ifndef VECTOR_H
|
#ifndef VECTOR_H
|
||||||
#define VECTOR_H
|
#define VECTOR_H
|
||||||
|
|
||||||
|
#include "postgres.h"
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 160000
|
||||||
|
#include "varatt.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
#define VECTOR_MAX_DIM 16000
|
#define VECTOR_MAX_DIM 16000
|
||||||
|
|
||||||
#define VECTOR_SIZE(_dim) (offsetof(Vector, x) + sizeof(float)*(_dim))
|
#define VECTOR_SIZE(_dim) (offsetof(Vector, x) + sizeof(float)*(_dim))
|
||||||
@@ -16,8 +22,24 @@ typedef struct Vector
|
|||||||
float x[FLEXIBLE_ARRAY_MEMBER];
|
float x[FLEXIBLE_ARRAY_MEMBER];
|
||||||
} Vector;
|
} Vector;
|
||||||
|
|
||||||
Vector *InitVector(int dim);
|
|
||||||
void PrintVector(char *msg, Vector * vector);
|
void PrintVector(char *msg, Vector * vector);
|
||||||
int vector_cmp_internal(Vector * a, Vector * b);
|
int vector_cmp_internal(Vector * a, Vector * b);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Allocate and initialize a new vector
|
||||||
|
*/
|
||||||
|
static inline Vector *
|
||||||
|
InitVector(int dim)
|
||||||
|
{
|
||||||
|
Vector *result;
|
||||||
|
int size;
|
||||||
|
|
||||||
|
size = VECTOR_SIZE(dim);
|
||||||
|
result = (Vector *) palloc0(size);
|
||||||
|
SET_VARSIZE(result, size);
|
||||||
|
result->dim = dim;
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ SELECT ARRAY[1,2,3]::numeric[]::vector;
|
|||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '{NULL}'::real[]::vector;
|
SELECT '{NULL}'::real[]::vector;
|
||||||
ERROR: array must not contain nulls
|
ERROR: array must not containing NULLs
|
||||||
SELECT '{NaN}'::real[]::vector;
|
SELECT '{NaN}'::real[]::vector;
|
||||||
ERROR: NaN not allowed in vector
|
ERROR: NaN not allowed in vector
|
||||||
SELECT '{Infinity}'::real[]::vector;
|
SELECT '{Infinity}'::real[]::vector;
|
||||||
@@ -38,8 +38,6 @@ SELECT '{-Infinity}'::real[]::vector;
|
|||||||
ERROR: infinite value not allowed in vector
|
ERROR: infinite value not allowed in vector
|
||||||
SELECT '{}'::real[]::vector;
|
SELECT '{}'::real[]::vector;
|
||||||
ERROR: vector must have at least 1 dimension
|
ERROR: vector must have at least 1 dimension
|
||||||
SELECT '{{1}}'::real[]::vector;
|
|
||||||
ERROR: array must be 1-D
|
|
||||||
SELECT '[1,2,3]'::vector::real[];
|
SELECT '[1,2,3]'::vector::real[];
|
||||||
float4
|
float4
|
||||||
---------
|
---------
|
||||||
@@ -48,8 +46,6 @@ SELECT '[1,2,3]'::vector::real[];
|
|||||||
|
|
||||||
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
||||||
ERROR: vector cannot have more than 16000 dimensions
|
ERROR: vector cannot have more than 16000 dimensions
|
||||||
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
|
|
||||||
ERROR: vector cannot have more than 16000 dimensions
|
|
||||||
-- ensure no error
|
-- ensure no error
|
||||||
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
||||||
?column?
|
?column?
|
||||||
|
|||||||
@@ -4,182 +4,12 @@ SELECT '[1,2,3]'::vector + '[4,5,6]';
|
|||||||
[5,7,9]
|
[5,7,9]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[3e38]'::vector + '[3e38]';
|
|
||||||
ERROR: value out of range: overflow
|
|
||||||
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
||||||
?column?
|
?column?
|
||||||
------------
|
------------
|
||||||
[-3,-3,-3]
|
[-3,-3,-3]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[-3e38]'::vector - '[3e38]';
|
|
||||||
ERROR: value out of range: overflow
|
|
||||||
SELECT '[1,2,3]'::vector * '[4,5,6]';
|
|
||||||
?column?
|
|
||||||
-----------
|
|
||||||
[4,10,18]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[1e37]'::vector * '[1e37]';
|
|
||||||
ERROR: value out of range: overflow
|
|
||||||
SELECT '[1e-37]'::vector * '[1e-37]';
|
|
||||||
ERROR: value out of range: underflow
|
|
||||||
SELECT ('[1,2,3]'::vector)[0];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[1];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[2];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
2
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[3];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
3
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[4];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[1:1];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[1]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[1:2];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[1,2]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:4];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[-2:2];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[1,2]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:1];
|
|
||||||
ERROR: vector must have at least 1 dimension
|
|
||||||
SELECT ('[1,2,3]'::vector)[:];
|
|
||||||
vector
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[:2];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[1,2]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
[2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[:4];
|
|
||||||
vector
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[-2:];
|
|
||||||
vector
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[NULL];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[NULL:2];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:NULL];
|
|
||||||
vector
|
|
||||||
--------
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[1][1];
|
|
||||||
ERROR: vector allows only one subscript
|
|
||||||
SELECT '[1,2,3]'::vector = '[1,2,3]';
|
|
||||||
?column?
|
|
||||||
----------
|
|
||||||
t
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[1,2,3]'::vector = '[1,2]';
|
|
||||||
ERROR: different vector dimensions 3 and 2
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[1,2,3]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
0
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[0,0,0]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[0,0,0]', '[1,2,3]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
-1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[1,2]', '[1,2,3]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
-1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[1,2]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[1,2]', '[2,3,4]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
-1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_cmp('[2,3]', '[1,2,3]');
|
|
||||||
vector_cmp
|
|
||||||
------------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_dims('[1,2,3]');
|
SELECT vector_dims('[1,2,3]');
|
||||||
vector_dims
|
vector_dims
|
||||||
-------------
|
-------------
|
||||||
@@ -204,12 +34,6 @@ SELECT vector_norm('[0,1]');
|
|||||||
1
|
1
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT vector_norm('[3e37,4e37]')::real;
|
|
||||||
vector_norm
|
|
||||||
-------------
|
|
||||||
5e+37
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT l2_distance('[0,0]', '[3,4]');
|
SELECT l2_distance('[0,0]', '[3,4]');
|
||||||
l2_distance
|
l2_distance
|
||||||
-------------
|
-------------
|
||||||
@@ -224,12 +48,6 @@ SELECT l2_distance('[0,0]', '[0,1]');
|
|||||||
|
|
||||||
SELECT l2_distance('[1,2]', '[3]');
|
SELECT l2_distance('[1,2]', '[3]');
|
||||||
ERROR: different vector dimensions 2 and 1
|
ERROR: different vector dimensions 2 and 1
|
||||||
SELECT l2_distance('[3e38]', '[-3e38]');
|
|
||||||
l2_distance
|
|
||||||
-------------
|
|
||||||
Infinity
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT inner_product('[1,2]', '[3,4]');
|
SELECT inner_product('[1,2]', '[3,4]');
|
||||||
inner_product
|
inner_product
|
||||||
---------------
|
---------------
|
||||||
@@ -238,12 +56,6 @@ SELECT inner_product('[1,2]', '[3,4]');
|
|||||||
|
|
||||||
SELECT inner_product('[1,2]', '[3]');
|
SELECT inner_product('[1,2]', '[3]');
|
||||||
ERROR: different vector dimensions 2 and 1
|
ERROR: different vector dimensions 2 and 1
|
||||||
SELECT inner_product('[3e38]', '[3e38]');
|
|
||||||
inner_product
|
|
||||||
---------------
|
|
||||||
Infinity
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT cosine_distance('[1,2]', '[2,4]');
|
SELECT cosine_distance('[1,2]', '[2,4]');
|
||||||
cosine_distance
|
cosine_distance
|
||||||
-----------------
|
-----------------
|
||||||
@@ -262,12 +74,6 @@ SELECT cosine_distance('[1,1]', '[1,1]');
|
|||||||
0
|
0
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT cosine_distance('[1,0]', '[0,2]');
|
|
||||||
cosine_distance
|
|
||||||
-----------------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT cosine_distance('[1,1]', '[-1,-1]');
|
SELECT cosine_distance('[1,1]', '[-1,-1]');
|
||||||
cosine_distance
|
cosine_distance
|
||||||
-----------------
|
-----------------
|
||||||
@@ -276,44 +82,6 @@ SELECT cosine_distance('[1,1]', '[-1,-1]');
|
|||||||
|
|
||||||
SELECT cosine_distance('[1,2]', '[3]');
|
SELECT cosine_distance('[1,2]', '[3]');
|
||||||
ERROR: different vector dimensions 2 and 1
|
ERROR: different vector dimensions 2 and 1
|
||||||
SELECT cosine_distance('[1,1]', '[1.1,1.1]');
|
|
||||||
cosine_distance
|
|
||||||
-----------------
|
|
||||||
0
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT cosine_distance('[1,1]', '[-1.1,-1.1]');
|
|
||||||
cosine_distance
|
|
||||||
-----------------
|
|
||||||
2
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT cosine_distance('[3e38]', '[3e38]');
|
|
||||||
cosine_distance
|
|
||||||
-----------------
|
|
||||||
NaN
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT l1_distance('[0,0]', '[3,4]');
|
|
||||||
l1_distance
|
|
||||||
-------------
|
|
||||||
7
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT l1_distance('[0,0]', '[0,1]');
|
|
||||||
l1_distance
|
|
||||||
-------------
|
|
||||||
1
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT l1_distance('[1,2]', '[3]');
|
|
||||||
ERROR: different vector dimensions 2 and 1
|
|
||||||
SELECT l1_distance('[3e38]', '[-3e38]');
|
|
||||||
l1_distance
|
|
||||||
-------------
|
|
||||||
Infinity
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
||||||
avg
|
avg
|
||||||
-----------
|
-----------
|
||||||
@@ -334,33 +102,3 @@ SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
|||||||
|
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
||||||
ERROR: expected 2 dimensions, not 1
|
ERROR: expected 2 dimensions, not 1
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[3e38]'::vector, '[3e38]']) v;
|
|
||||||
avg
|
|
||||||
---------
|
|
||||||
[3e+38]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT vector_avg(array_agg(n)) FROM generate_series(1, 16002) n;
|
|
||||||
ERROR: vector cannot have more than 16000 dimensions
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
|
||||||
sum
|
|
||||||
----------
|
|
||||||
[4,7,10]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
|
||||||
sum
|
|
||||||
----------
|
|
||||||
[4,7,10]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY[]::vector[]) v;
|
|
||||||
sum
|
|
||||||
-----
|
|
||||||
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
|
||||||
ERROR: different vector dimensions 2 and 1
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[3e38]'::vector, '[3e38]']) v;
|
|
||||||
ERROR: value out of range: overflow
|
|
||||||
|
|||||||
@@ -1,26 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_cosine_ops);
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
SELECT * FROM t ORDER BY val <=> '[3,3,3]';
|
|
||||||
val
|
|
||||||
---------
|
|
||||||
[1,1,1]
|
|
||||||
[1,2,3]
|
|
||||||
[1,2,4]
|
|
||||||
(3 rows)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
|
|
||||||
count
|
|
||||||
-------
|
|
||||||
3
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
|
|
||||||
count
|
|
||||||
-------
|
|
||||||
3
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_ip_ops);
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
SELECT * FROM t ORDER BY val <#> '[3,3,3]';
|
|
||||||
val
|
|
||||||
---------
|
|
||||||
[1,2,4]
|
|
||||||
[1,2,3]
|
|
||||||
[1,1,1]
|
|
||||||
[0,0,0]
|
|
||||||
(4 rows)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector)) t2;
|
|
||||||
count
|
|
||||||
-------
|
|
||||||
4
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,36 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
val
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
[1,2,4]
|
|
||||||
[1,1,1]
|
|
||||||
[0,0,0]
|
|
||||||
(4 rows)
|
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
|
||||||
val
|
|
||||||
---------
|
|
||||||
[0,0,0]
|
|
||||||
[1,1,1]
|
|
||||||
[1,2,3]
|
|
||||||
[1,2,4]
|
|
||||||
(4 rows)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM t;
|
|
||||||
count
|
|
||||||
-------
|
|
||||||
5
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
TRUNCATE t;
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
val
|
|
||||||
-----
|
|
||||||
(0 rows)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
CREATE TABLE t (val vector(3));
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 1);
|
|
||||||
ERROR: value 1 out of bounds for option "m"
|
|
||||||
DETAIL: Valid values are between "2" and "100".
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 101);
|
|
||||||
ERROR: value 101 out of bounds for option "m"
|
|
||||||
DETAIL: Valid values are between "2" and "100".
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 3);
|
|
||||||
ERROR: value 3 out of bounds for option "ef_construction"
|
|
||||||
DETAIL: Valid values are between "4" and "1000".
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 1001);
|
|
||||||
ERROR: value 1001 out of bounds for option "ef_construction"
|
|
||||||
DETAIL: Valid values are between "4" and "1000".
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31);
|
|
||||||
ERROR: ef_construction must be greater than or equal to 2 * m
|
|
||||||
SHOW hnsw.ef_search;
|
|
||||||
hnsw.ef_search
|
|
||||||
----------------
|
|
||||||
40
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SET hnsw.ef_search = 0;
|
|
||||||
ERROR: 0 is outside the valid range for parameter "hnsw.ef_search" (1 .. 1000)
|
|
||||||
SET hnsw.ef_search = 1001;
|
|
||||||
ERROR: 1001 is outside the valid range for parameter "hnsw.ef_search" (1 .. 1000)
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,13 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
CREATE UNLOGGED TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
val
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
[1,1,1]
|
|
||||||
[0,0,0]
|
|
||||||
(3 rows)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -4,22 +4,10 @@ SELECT '[1,2,3]'::vector;
|
|||||||
[1,2,3]
|
[1,2,3]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[-1,-2,-3]'::vector;
|
SELECT '[-1,2,3]'::vector;
|
||||||
vector
|
vector
|
||||||
------------
|
----------
|
||||||
[-1,-2,-3]
|
[-1,2,3]
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[1.,2.,3.]'::vector;
|
|
||||||
vector
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT ' [ 1, 2 , 3 ] '::vector;
|
|
||||||
vector
|
|
||||||
---------
|
|
||||||
[1,2,3]
|
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[1.23456]'::vector;
|
SELECT '[1.23456]'::vector;
|
||||||
@@ -29,7 +17,7 @@ SELECT '[1.23456]'::vector;
|
|||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[hello,1]'::vector;
|
SELECT '[hello,1]'::vector;
|
||||||
ERROR: invalid input syntax for type vector: "[hello,1]"
|
ERROR: invalid input syntax for type vector: "hello"
|
||||||
LINE 1: SELECT '[hello,1]'::vector;
|
LINE 1: SELECT '[hello,1]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[NaN,1]'::vector;
|
SELECT '[NaN,1]'::vector;
|
||||||
@@ -44,35 +32,13 @@ SELECT '[-Infinity,1]'::vector;
|
|||||||
ERROR: infinite value not allowed in vector
|
ERROR: infinite value not allowed in vector
|
||||||
LINE 1: SELECT '[-Infinity,1]'::vector;
|
LINE 1: SELECT '[-Infinity,1]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[1.5e38,-1.5e38]'::vector;
|
|
||||||
vector
|
|
||||||
--------------------
|
|
||||||
[1.5e+38,-1.5e+38]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[1.5e+38,-1.5e+38]'::vector;
|
|
||||||
vector
|
|
||||||
--------------------
|
|
||||||
[1.5e+38,-1.5e+38]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[1.5e-38,-1.5e-38]'::vector;
|
|
||||||
vector
|
|
||||||
--------------------
|
|
||||||
[1.5e-38,-1.5e-38]
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT '[4e38,1]'::vector;
|
|
||||||
ERROR: infinite value not allowed in vector
|
|
||||||
LINE 1: SELECT '[4e38,1]'::vector;
|
|
||||||
^
|
|
||||||
SELECT '[1,2,3'::vector;
|
SELECT '[1,2,3'::vector;
|
||||||
ERROR: malformed vector literal: "[1,2,3"
|
ERROR: malformed vector literal
|
||||||
LINE 1: SELECT '[1,2,3'::vector;
|
LINE 1: SELECT '[1,2,3'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Unexpected end of input.
|
DETAIL: Unexpected end of input.
|
||||||
SELECT '[1,2,3]9'::vector;
|
SELECT '[1,2,3]9'::vector;
|
||||||
ERROR: malformed vector literal: "[1,2,3]9"
|
ERROR: malformed vector literal
|
||||||
LINE 1: SELECT '[1,2,3]9'::vector;
|
LINE 1: SELECT '[1,2,3]9'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Junk after closing right brace.
|
DETAIL: Junk after closing right brace.
|
||||||
@@ -81,41 +47,14 @@ ERROR: malformed vector literal: "1,2,3"
|
|||||||
LINE 1: SELECT '1,2,3'::vector;
|
LINE 1: SELECT '1,2,3'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Vector contents must start with "[".
|
DETAIL: Vector contents must start with "[".
|
||||||
SELECT ''::vector;
|
|
||||||
ERROR: malformed vector literal: ""
|
|
||||||
LINE 1: SELECT ''::vector;
|
|
||||||
^
|
|
||||||
DETAIL: Vector contents must start with "[".
|
|
||||||
SELECT '['::vector;
|
|
||||||
ERROR: malformed vector literal: "["
|
|
||||||
LINE 1: SELECT '['::vector;
|
|
||||||
^
|
|
||||||
DETAIL: Unexpected end of input.
|
|
||||||
SELECT '[,'::vector;
|
|
||||||
ERROR: malformed vector literal: "[,"
|
|
||||||
LINE 1: SELECT '[,'::vector;
|
|
||||||
^
|
|
||||||
DETAIL: Unexpected end of input.
|
|
||||||
SELECT '[]'::vector;
|
SELECT '[]'::vector;
|
||||||
ERROR: vector must have at least 1 dimension
|
ERROR: vector must have at least 1 dimension
|
||||||
LINE 1: SELECT '[]'::vector;
|
LINE 1: SELECT '[]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[1,]'::vector;
|
SELECT '[1,]'::vector;
|
||||||
ERROR: invalid input syntax for type vector: "[1,]"
|
ERROR: invalid input syntax for type vector: "]"
|
||||||
LINE 1: SELECT '[1,]'::vector;
|
LINE 1: SELECT '[1,]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[1a]'::vector;
|
|
||||||
ERROR: invalid input syntax for type vector: "[1a]"
|
|
||||||
LINE 1: SELECT '[1a]'::vector;
|
|
||||||
^
|
|
||||||
SELECT '[1,,3]'::vector;
|
|
||||||
ERROR: malformed vector literal: "[1,,3]"
|
|
||||||
LINE 1: SELECT '[1,,3]'::vector;
|
|
||||||
^
|
|
||||||
SELECT '[1, ,3]'::vector;
|
|
||||||
ERROR: invalid input syntax for type vector: "[1, ,3]"
|
|
||||||
LINE 1: SELECT '[1, ,3]'::vector;
|
|
||||||
^
|
|
||||||
SELECT '[1,2,3]'::vector(2);
|
SELECT '[1,2,3]'::vector(2);
|
||||||
ERROR: expected 2 dimensions, not 3
|
ERROR: expected 2 dimensions, not 3
|
||||||
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
||||||
|
|||||||
@@ -11,16 +11,9 @@ SELECT * FROM t ORDER BY val <=> '[3,3,3]';
|
|||||||
[1,2,4]
|
[1,2,4]
|
||||||
(3 rows)
|
(3 rows)
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
|
SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector);
|
||||||
count
|
val
|
||||||
-------
|
-----
|
||||||
3
|
(0 rows)
|
||||||
(1 row)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
|
|
||||||
count
|
|
||||||
-------
|
|
||||||
3
|
|
||||||
(1 row)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -12,10 +12,9 @@ SELECT * FROM t ORDER BY val <#> '[3,3,3]';
|
|||||||
[0,0,0]
|
[0,0,0]
|
||||||
(4 rows)
|
(4 rows)
|
||||||
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector)) t2;
|
SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector);
|
||||||
count
|
val
|
||||||
-------
|
-----
|
||||||
4
|
(0 rows)
|
||||||
(1 row)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
SET enable_seqscan = off;
|
SET enable_seqscan = off;
|
||||||
CREATE TABLE t (val vector(3));
|
CREATE TABLE t (val vector(3));
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 1);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 1);
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
INSERT INTO t (val) VALUES ('[1,2,4]');
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
||||||
val
|
val
|
||||||
@@ -13,13 +13,9 @@ SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|||||||
(4 rows)
|
(4 rows)
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
||||||
val
|
val
|
||||||
---------
|
-----
|
||||||
[0,0,0]
|
(0 rows)
|
||||||
[1,1,1]
|
|
||||||
[1,2,3]
|
|
||||||
[1,2,4]
|
|
||||||
(4 rows)
|
|
||||||
|
|
||||||
SELECT COUNT(*) FROM t;
|
SELECT COUNT(*) FROM t;
|
||||||
count
|
count
|
||||||
@@ -27,13 +23,4 @@ SELECT COUNT(*) FROM t;
|
|||||||
5
|
5
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
TRUNCATE t;
|
|
||||||
NOTICE: ivfflat index created with little data
|
|
||||||
DETAIL: This will cause low recall.
|
|
||||||
HINT: Drop the index until the table has more data.
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
val
|
|
||||||
-----
|
|
||||||
(0 rows)
|
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -1,8 +1,9 @@
|
|||||||
|
SET enable_seqscan = off;
|
||||||
CREATE TABLE t (val vector(3));
|
CREATE TABLE t (val vector(3));
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 0);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 0);
|
||||||
ERROR: value 0 out of bounds for option "lists"
|
ERROR: value 0 out of bounds for option "lists"
|
||||||
DETAIL: Valid values are between "1" and "32768".
|
DETAIL: Valid values are between "1" and "32768".
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 32769);
|
||||||
ERROR: value 32769 out of bounds for option "lists"
|
ERROR: value 32769 out of bounds for option "lists"
|
||||||
DETAIL: Valid values are between "1" and "32768".
|
DETAIL: Valid values are between "1" and "32768".
|
||||||
SHOW ivfflat.probes;
|
SHOW ivfflat.probes;
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
SET enable_seqscan = off;
|
SET enable_seqscan = off;
|
||||||
CREATE UNLOGGED TABLE t (val vector(3));
|
CREATE UNLOGGED TABLE t (val vector(3));
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 1);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 1);
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
||||||
val
|
val
|
||||||
---------
|
---------
|
||||||
|
|||||||
@@ -8,10 +8,8 @@ SELECT '{NaN}'::real[]::vector;
|
|||||||
SELECT '{Infinity}'::real[]::vector;
|
SELECT '{Infinity}'::real[]::vector;
|
||||||
SELECT '{-Infinity}'::real[]::vector;
|
SELECT '{-Infinity}'::real[]::vector;
|
||||||
SELECT '{}'::real[]::vector;
|
SELECT '{}'::real[]::vector;
|
||||||
SELECT '{{1}}'::real[]::vector;
|
|
||||||
SELECT '[1,2,3]'::vector::real[];
|
SELECT '[1,2,3]'::vector::real[];
|
||||||
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
||||||
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
|
|
||||||
|
|
||||||
-- ensure no error
|
-- ensure no error
|
||||||
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
||||||
|
|||||||
@@ -1,82 +1,26 @@
|
|||||||
SELECT '[1,2,3]'::vector + '[4,5,6]';
|
SELECT '[1,2,3]'::vector + '[4,5,6]';
|
||||||
SELECT '[3e38]'::vector + '[3e38]';
|
|
||||||
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
||||||
SELECT '[-3e38]'::vector - '[3e38]';
|
|
||||||
SELECT '[1,2,3]'::vector * '[4,5,6]';
|
|
||||||
SELECT '[1e37]'::vector * '[1e37]';
|
|
||||||
SELECT '[1e-37]'::vector * '[1e-37]';
|
|
||||||
|
|
||||||
SELECT ('[1,2,3]'::vector)[0];
|
|
||||||
SELECT ('[1,2,3]'::vector)[1];
|
|
||||||
SELECT ('[1,2,3]'::vector)[2];
|
|
||||||
SELECT ('[1,2,3]'::vector)[3];
|
|
||||||
SELECT ('[1,2,3]'::vector)[4];
|
|
||||||
SELECT ('[1,2,3]'::vector)[1:1];
|
|
||||||
SELECT ('[1,2,3]'::vector)[1:2];
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:4];
|
|
||||||
SELECT ('[1,2,3]'::vector)[-2:2];
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:1];
|
|
||||||
SELECT ('[1,2,3]'::vector)[:];
|
|
||||||
SELECT ('[1,2,3]'::vector)[:2];
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:];
|
|
||||||
SELECT ('[1,2,3]'::vector)[:4];
|
|
||||||
SELECT ('[1,2,3]'::vector)[-2:];
|
|
||||||
SELECT ('[1,2,3]'::vector)[NULL];
|
|
||||||
SELECT ('[1,2,3]'::vector)[NULL:2];
|
|
||||||
SELECT ('[1,2,3]'::vector)[2:NULL];
|
|
||||||
SELECT ('[1,2,3]'::vector)[1][1];
|
|
||||||
|
|
||||||
SELECT '[1,2,3]'::vector = '[1,2,3]';
|
|
||||||
SELECT '[1,2,3]'::vector = '[1,2]';
|
|
||||||
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[1,2,3]');
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[0,0,0]');
|
|
||||||
SELECT vector_cmp('[0,0,0]', '[1,2,3]');
|
|
||||||
SELECT vector_cmp('[1,2]', '[1,2,3]');
|
|
||||||
SELECT vector_cmp('[1,2,3]', '[1,2]');
|
|
||||||
SELECT vector_cmp('[1,2]', '[2,3,4]');
|
|
||||||
SELECT vector_cmp('[2,3]', '[1,2,3]');
|
|
||||||
|
|
||||||
SELECT vector_dims('[1,2,3]');
|
SELECT vector_dims('[1,2,3]');
|
||||||
|
|
||||||
SELECT round(vector_norm('[1,1]')::numeric, 5);
|
SELECT round(vector_norm('[1,1]')::numeric, 5);
|
||||||
SELECT vector_norm('[3,4]');
|
SELECT vector_norm('[3,4]');
|
||||||
SELECT vector_norm('[0,1]');
|
SELECT vector_norm('[0,1]');
|
||||||
SELECT vector_norm('[3e37,4e37]')::real;
|
|
||||||
|
|
||||||
SELECT l2_distance('[0,0]', '[3,4]');
|
SELECT l2_distance('[0,0]', '[3,4]');
|
||||||
SELECT l2_distance('[0,0]', '[0,1]');
|
SELECT l2_distance('[0,0]', '[0,1]');
|
||||||
SELECT l2_distance('[1,2]', '[3]');
|
SELECT l2_distance('[1,2]', '[3]');
|
||||||
SELECT l2_distance('[3e38]', '[-3e38]');
|
|
||||||
|
|
||||||
SELECT inner_product('[1,2]', '[3,4]');
|
SELECT inner_product('[1,2]', '[3,4]');
|
||||||
SELECT inner_product('[1,2]', '[3]');
|
SELECT inner_product('[1,2]', '[3]');
|
||||||
SELECT inner_product('[3e38]', '[3e38]');
|
|
||||||
|
|
||||||
SELECT cosine_distance('[1,2]', '[2,4]');
|
SELECT cosine_distance('[1,2]', '[2,4]');
|
||||||
SELECT cosine_distance('[1,2]', '[0,0]');
|
SELECT cosine_distance('[1,2]', '[0,0]');
|
||||||
SELECT cosine_distance('[1,1]', '[1,1]');
|
SELECT cosine_distance('[1,1]', '[1,1]');
|
||||||
SELECT cosine_distance('[1,0]', '[0,2]');
|
|
||||||
SELECT cosine_distance('[1,1]', '[-1,-1]');
|
SELECT cosine_distance('[1,1]', '[-1,-1]');
|
||||||
SELECT cosine_distance('[1,2]', '[3]');
|
SELECT cosine_distance('[1,2]', '[3]');
|
||||||
SELECT cosine_distance('[1,1]', '[1.1,1.1]');
|
|
||||||
SELECT cosine_distance('[1,1]', '[-1.1,-1.1]');
|
|
||||||
SELECT cosine_distance('[3e38]', '[3e38]');
|
|
||||||
|
|
||||||
SELECT l1_distance('[0,0]', '[3,4]');
|
|
||||||
SELECT l1_distance('[0,0]', '[0,1]');
|
|
||||||
SELECT l1_distance('[1,2]', '[3]');
|
|
||||||
SELECT l1_distance('[3e38]', '[-3e38]');
|
|
||||||
|
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[3e38]'::vector, '[3e38]']) v;
|
|
||||||
SELECT vector_avg(array_agg(n)) FROM generate_series(1, 16002) n;
|
|
||||||
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY[]::vector[]) v;
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
|
||||||
SELECT sum(v) FROM unnest(ARRAY['[3e38]'::vector, '[3e38]']) v;
|
|
||||||
|
|||||||
@@ -1,13 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_cosine_ops);
|
|
||||||
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <=> '[3,3,3]';
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,12 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_ip_ops);
|
|
||||||
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <#> '[3,3,3]';
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector)) t2;
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
|
|
||||||
CREATE TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
|
|
||||||
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
|
||||||
SELECT COUNT(*) FROM t;
|
|
||||||
|
|
||||||
TRUNCATE t;
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,13 +0,0 @@
|
|||||||
CREATE TABLE t (val vector(3));
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 1);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 101);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 3);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (ef_construction = 1001);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops) WITH (m = 16, ef_construction = 31);
|
|
||||||
|
|
||||||
SHOW hnsw.ef_search;
|
|
||||||
|
|
||||||
SET hnsw.ef_search = 0;
|
|
||||||
SET hnsw.ef_search = 1001;
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
SET enable_seqscan = off;
|
|
||||||
|
|
||||||
CREATE UNLOGGED TABLE t (val vector(3));
|
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
|
||||||
CREATE INDEX ON t USING hnsw (val vector_l2_ops);
|
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
@@ -1,27 +1,15 @@
|
|||||||
SELECT '[1,2,3]'::vector;
|
SELECT '[1,2,3]'::vector;
|
||||||
SELECT '[-1,-2,-3]'::vector;
|
SELECT '[-1,2,3]'::vector;
|
||||||
SELECT '[1.,2.,3.]'::vector;
|
|
||||||
SELECT ' [ 1, 2 , 3 ] '::vector;
|
|
||||||
SELECT '[1.23456]'::vector;
|
SELECT '[1.23456]'::vector;
|
||||||
SELECT '[hello,1]'::vector;
|
SELECT '[hello,1]'::vector;
|
||||||
SELECT '[NaN,1]'::vector;
|
SELECT '[NaN,1]'::vector;
|
||||||
SELECT '[Infinity,1]'::vector;
|
SELECT '[Infinity,1]'::vector;
|
||||||
SELECT '[-Infinity,1]'::vector;
|
SELECT '[-Infinity,1]'::vector;
|
||||||
SELECT '[1.5e38,-1.5e38]'::vector;
|
|
||||||
SELECT '[1.5e+38,-1.5e+38]'::vector;
|
|
||||||
SELECT '[1.5e-38,-1.5e-38]'::vector;
|
|
||||||
SELECT '[4e38,1]'::vector;
|
|
||||||
SELECT '[1,2,3'::vector;
|
SELECT '[1,2,3'::vector;
|
||||||
SELECT '[1,2,3]9'::vector;
|
SELECT '[1,2,3]9'::vector;
|
||||||
SELECT '1,2,3'::vector;
|
SELECT '1,2,3'::vector;
|
||||||
SELECT ''::vector;
|
|
||||||
SELECT '['::vector;
|
|
||||||
SELECT '[,'::vector;
|
|
||||||
SELECT '[]'::vector;
|
SELECT '[]'::vector;
|
||||||
SELECT '[1,]'::vector;
|
SELECT '[1,]'::vector;
|
||||||
SELECT '[1a]'::vector;
|
|
||||||
SELECT '[1,,3]'::vector;
|
|
||||||
SELECT '[1, ,3]'::vector;
|
|
||||||
SELECT '[1,2,3]'::vector(2);
|
SELECT '[1,2,3]'::vector(2);
|
||||||
|
|
||||||
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ CREATE INDEX ON t USING ivfflat (val vector_cosine_ops) WITH (lists = 1);
|
|||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
INSERT INTO t (val) VALUES ('[1,2,4]');
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <=> '[3,3,3]';
|
SELECT * FROM t ORDER BY val <=> '[3,3,3]';
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> '[0,0,0]') t2;
|
SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector);
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <=> (SELECT NULL::vector)) t2;
|
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -7,6 +7,6 @@ CREATE INDEX ON t USING ivfflat (val vector_ip_ops) WITH (lists = 1);
|
|||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
INSERT INTO t (val) VALUES ('[1,2,4]');
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <#> '[3,3,3]';
|
SELECT * FROM t ORDER BY val <#> '[3,3,3]';
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector)) t2;
|
SELECT * FROM t ORDER BY val <#> (SELECT NULL::vector);
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ SET enable_seqscan = off;
|
|||||||
|
|
||||||
CREATE TABLE t (val vector(3));
|
CREATE TABLE t (val vector(3));
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 1);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 1);
|
||||||
|
|
||||||
INSERT INTO t (val) VALUES ('[1,2,4]');
|
INSERT INTO t (val) VALUES ('[1,2,4]');
|
||||||
|
|
||||||
@@ -10,7 +10,4 @@ SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|||||||
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
SELECT * FROM t ORDER BY val <-> (SELECT NULL::vector);
|
||||||
SELECT COUNT(*) FROM t;
|
SELECT COUNT(*) FROM t;
|
||||||
|
|
||||||
TRUNCATE t;
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
|
||||||
|
|
||||||
DROP TABLE t;
|
DROP TABLE t;
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
|
SET enable_seqscan = off;
|
||||||
|
|
||||||
CREATE TABLE t (val vector(3));
|
CREATE TABLE t (val vector(3));
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 0);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 0);
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 32769);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 32769);
|
||||||
|
|
||||||
SHOW ivfflat.probes;
|
SHOW ivfflat.probes;
|
||||||
|
|
||||||
|
|||||||
15
test/sql/ivfflat_parallel.sql
Normal file
15
test/sql/ivfflat_parallel.sql
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
-- SET force_parallel_mode = on;
|
||||||
|
SET parallel_setup_cost = 10;
|
||||||
|
SET parallel_tuple_cost = 0.001;
|
||||||
|
SET min_parallel_table_scan_size = 1;
|
||||||
|
SET min_parallel_index_scan_size = 1;
|
||||||
|
|
||||||
|
CREATE TABLE t (id integer, val vector(3));
|
||||||
|
INSERT INTO t (id, val) SELECT n, ARRAY[random(), random(), random()] FROM generate_series(1,1000000) n;
|
||||||
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 10);
|
||||||
|
SET ivfflat.probes = 2;
|
||||||
|
|
||||||
|
EXPLAIN SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
|
||||||
|
SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
|
||||||
|
|
||||||
|
DROP TABLE t;
|
||||||
@@ -2,7 +2,7 @@ SET enable_seqscan = off;
|
|||||||
|
|
||||||
CREATE UNLOGGED TABLE t (val vector(3));
|
CREATE UNLOGGED TABLE t (val vector(3));
|
||||||
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
INSERT INTO t (val) VALUES ('[0,0,0]'), ('[1,2,3]'), ('[1,1,1]'), (NULL);
|
||||||
CREATE INDEX ON t USING ivfflat (val vector_l2_ops) WITH (lists = 1);
|
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 1);
|
||||||
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
SELECT * FROM t ORDER BY val <-> '[3,3,3]';
|
||||||
|
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 31;
|
||||||
|
|
||||||
my $dim = 32;
|
my $dim = 32;
|
||||||
|
|
||||||
@@ -19,13 +19,14 @@ sub test_index_replay
|
|||||||
|
|
||||||
# Wait for replica to catch up
|
# Wait for replica to catch up
|
||||||
my $applname = $node_replica->name;
|
my $applname = $node_replica->name;
|
||||||
|
|
||||||
|
my $server_version_num = $node_primary->safe_psql("postgres", "SHOW server_version_num");
|
||||||
my $caughtup_query = "SELECT pg_current_wal_lsn() <= replay_lsn FROM pg_stat_replication WHERE application_name = '$applname';";
|
my $caughtup_query = "SELECT pg_current_wal_lsn() <= replay_lsn FROM pg_stat_replication WHERE application_name = '$applname';";
|
||||||
$node_primary->poll_query_until('postgres', $caughtup_query)
|
$node_primary->poll_query_until('postgres', $caughtup_query)
|
||||||
or die "Timed out while waiting for replica 1 to catch up";
|
or die "Timed out while waiting for replica 1 to catch up";
|
||||||
|
|
||||||
my @r = ();
|
my @r = ();
|
||||||
for (1 .. $dim)
|
for (1 .. $dim) {
|
||||||
{
|
|
||||||
push(@r, rand());
|
push(@r, rand());
|
||||||
}
|
}
|
||||||
my $sql = join(",", @r);
|
my $sql = join(",", @r);
|
||||||
@@ -51,13 +52,11 @@ my $array_sql = join(",", ('random()') x $dim);
|
|||||||
# Initialize primary node
|
# Initialize primary node
|
||||||
$node_primary = get_new_node('primary');
|
$node_primary = get_new_node('primary');
|
||||||
$node_primary->init(allows_streaming => 1);
|
$node_primary->init(allows_streaming => 1);
|
||||||
if ($dim > 32)
|
if ($dim > 32) {
|
||||||
{
|
|
||||||
# TODO use wal_keep_segments for Postgres < 13
|
# TODO use wal_keep_segments for Postgres < 13
|
||||||
$node_primary->append_conf('postgresql.conf', qq(wal_keep_size = 1GB));
|
$node_primary->append_conf('postgresql.conf', qq(wal_keep_size = 1GB));
|
||||||
}
|
}
|
||||||
if ($dim > 1500)
|
if ($dim > 1500) {
|
||||||
{
|
|
||||||
$node_primary->append_conf('postgresql.conf', qq(maintenance_work_mem = 128MB));
|
$node_primary->append_conf('postgresql.conf', qq(maintenance_work_mem = 128MB));
|
||||||
}
|
}
|
||||||
$node_primary->start;
|
$node_primary->start;
|
||||||
@@ -68,7 +67,8 @@ $node_primary->backup($backup_name);
|
|||||||
|
|
||||||
# Create streaming replica linking to primary
|
# Create streaming replica linking to primary
|
||||||
$node_replica = get_new_node('replica');
|
$node_replica = get_new_node('replica');
|
||||||
$node_replica->init_from_backup($node_primary, $backup_name, has_streaming => 1);
|
$node_replica->init_from_backup($node_primary, $backup_name,
|
||||||
|
has_streaming => 1);
|
||||||
$node_replica->start;
|
$node_replica->start;
|
||||||
|
|
||||||
# Create ivfflat index on primary
|
# Create ivfflat index on primary
|
||||||
@@ -77,7 +77,7 @@ $node_primary->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim));
|
|||||||
$node_primary->safe_psql("postgres",
|
$node_primary->safe_psql("postgres",
|
||||||
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series(1, 100000) i;"
|
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series(1, 100000) i;"
|
||||||
);
|
);
|
||||||
$node_primary->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v vector_l2_ops);");
|
$node_primary->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v);");
|
||||||
|
|
||||||
# Test that queries give same result
|
# Test that queries give same result
|
||||||
test_index_replay('initial');
|
test_index_replay('initial');
|
||||||
@@ -95,5 +95,3 @@ for my $i (1 .. 10)
|
|||||||
);
|
);
|
||||||
test_index_replay("insert $i");
|
test_index_replay("insert $i");
|
||||||
}
|
}
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -2,13 +2,12 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 1;
|
||||||
|
|
||||||
my $dim = 3;
|
my $dim = 3;
|
||||||
|
|
||||||
my @r = ();
|
my @r = ();
|
||||||
for (1 .. $dim)
|
for (1 .. $dim) {
|
||||||
{
|
|
||||||
my $v = int(rand(1000)) + 1;
|
my $v = int(rand(1000)) + 1;
|
||||||
push(@r, "i % $v");
|
push(@r, "i % $v");
|
||||||
}
|
}
|
||||||
@@ -25,7 +24,7 @@ $node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim));");
|
|||||||
$node->safe_psql("postgres",
|
$node->safe_psql("postgres",
|
||||||
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series(1, 100000) i;"
|
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series(1, 100000) i;"
|
||||||
);
|
);
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v vector_l2_ops);");
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v);");
|
||||||
|
|
||||||
# Get size
|
# Get size
|
||||||
my $size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
my $size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
||||||
@@ -40,5 +39,3 @@ $node->safe_psql("postgres",
|
|||||||
# Check size
|
# Check size
|
||||||
my $new_size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
my $new_size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
||||||
is($size, $new_size, "size does not change");
|
is($size, $new_size, "size does not change");
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,128 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $node;
|
|
||||||
my @queries = ();
|
|
||||||
my @expected;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
|
||||||
{
|
|
||||||
my ($probes, $min, $operator) = @_;
|
|
||||||
my $correct = 0;
|
|
||||||
my $total = 0;
|
|
||||||
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET ivfflat.probes = $probes;
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v $operator '$queries[0]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx on tst/);
|
|
||||||
|
|
||||||
for my $i (0 .. $#queries)
|
|
||||||
{
|
|
||||||
my $actual = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET ivfflat.probes = $probes;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
my @actual_ids = split("\n", $actual);
|
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
|
||||||
|
|
||||||
my @expected_ids = split("\n", $expected[$i]);
|
|
||||||
|
|
||||||
foreach (@expected_ids)
|
|
||||||
{
|
|
||||||
if (exists($actual_set{$_}))
|
|
||||||
{
|
|
||||||
$correct++;
|
|
||||||
}
|
|
||||||
$total++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $operator);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
$node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector(3));");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
# Generate queries
|
|
||||||
for (1 .. 20)
|
|
||||||
{
|
|
||||||
my $r1 = rand();
|
|
||||||
my $r2 = rand();
|
|
||||||
my $r3 = rand();
|
|
||||||
push(@queries, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Check each index type
|
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
|
||||||
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
|
||||||
|
|
||||||
for my $i (0 .. $#operators)
|
|
||||||
{
|
|
||||||
my $operator = $operators[$i];
|
|
||||||
my $opclass = $opclasses[$i];
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries)
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", "SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;");
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Build index serially
|
|
||||||
$node->safe_psql("postgres", qq(
|
|
||||||
SET max_parallel_maintenance_workers = 0;
|
|
||||||
CREATE INDEX idx ON tst USING ivfflat (v $opclass);
|
|
||||||
));
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
if ($operator ne "<#>")
|
|
||||||
{
|
|
||||||
# TODO Fix test (uniform random vectors all have similar inner product)
|
|
||||||
test_recall(1, 0.71, $operator);
|
|
||||||
test_recall(10, 0.95, $operator);
|
|
||||||
}
|
|
||||||
# Account for equal distances
|
|
||||||
test_recall(100, 0.9925, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
|
|
||||||
# Build index in parallel
|
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres", qq(
|
|
||||||
SET client_min_messages = DEBUG;
|
|
||||||
SET min_parallel_table_scan_size = 1;
|
|
||||||
CREATE INDEX idx ON tst USING ivfflat (v $opclass);
|
|
||||||
));
|
|
||||||
is($ret, 0, $stderr);
|
|
||||||
like($stderr, qr/using \d+ parallel workers/);
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
if ($operator ne "<#>")
|
|
||||||
{
|
|
||||||
# TODO Fix test (uniform random vectors all have similar inner product)
|
|
||||||
test_recall(1, 0.71, $operator);
|
|
||||||
test_recall(10, 0.95, $operator);
|
|
||||||
}
|
|
||||||
# Account for equal distances
|
|
||||||
test_recall(100, 0.9925, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
}
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
88
test/t/003_recall.pl
Normal file
88
test/t/003_recall.pl
Normal file
@@ -0,0 +1,88 @@
|
|||||||
|
use strict;
|
||||||
|
use warnings;
|
||||||
|
use PostgresNode;
|
||||||
|
use TestLib;
|
||||||
|
use Test::More tests => 9;
|
||||||
|
|
||||||
|
my $node;
|
||||||
|
my @queries = ();
|
||||||
|
my @expected;
|
||||||
|
my $limit = 20;
|
||||||
|
|
||||||
|
sub test_recall
|
||||||
|
{
|
||||||
|
my ($probes, $min, $operator) = @_;
|
||||||
|
my $correct = 0;
|
||||||
|
my $total = 0;
|
||||||
|
|
||||||
|
for my $i (0 .. $#queries) {
|
||||||
|
my $actual = $node->safe_psql("postgres", qq(
|
||||||
|
SET enable_seqscan = off;
|
||||||
|
SET ivfflat.probes = $probes;
|
||||||
|
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
||||||
|
));
|
||||||
|
my @actual_ids = split("\n", $actual);
|
||||||
|
my %actual_set = map { $_ => 1 } @actual_ids;
|
||||||
|
|
||||||
|
my @expected_ids = split("\n", $expected[$i]);
|
||||||
|
|
||||||
|
foreach (@expected_ids) {
|
||||||
|
if (exists($actual_set{$_})) {
|
||||||
|
$correct++;
|
||||||
|
}
|
||||||
|
$total++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
cmp_ok($correct / $total, ">=", $min, $operator);
|
||||||
|
}
|
||||||
|
|
||||||
|
# Initialize node
|
||||||
|
$node = get_new_node('node');
|
||||||
|
$node->init;
|
||||||
|
$node->start;
|
||||||
|
|
||||||
|
# Create table
|
||||||
|
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
||||||
|
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector(3));");
|
||||||
|
$node->safe_psql("postgres",
|
||||||
|
"INSERT INTO tst SELECT i, ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
||||||
|
);
|
||||||
|
|
||||||
|
# Generate queries
|
||||||
|
for (1..20) {
|
||||||
|
my $r1 = rand();
|
||||||
|
my $r2 = rand();
|
||||||
|
my $r3 = rand();
|
||||||
|
push(@queries, "[$r1,$r2,$r3]");
|
||||||
|
}
|
||||||
|
|
||||||
|
# Check each index type
|
||||||
|
my @operators = ("<->", "<#>", "<=>");
|
||||||
|
|
||||||
|
foreach (@operators) {
|
||||||
|
my $operator = $_;
|
||||||
|
|
||||||
|
# Get exact results
|
||||||
|
@expected = ();
|
||||||
|
foreach (@queries) {
|
||||||
|
my $res = $node->safe_psql("postgres", "SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;");
|
||||||
|
push(@expected, $res);
|
||||||
|
}
|
||||||
|
|
||||||
|
# Add index
|
||||||
|
my $opclass;
|
||||||
|
if ($operator == "<->") {
|
||||||
|
$opclass = "vector_l2_ops";
|
||||||
|
} elsif ($operator == "<#>") {
|
||||||
|
$opclass = "vector_ip_ops";
|
||||||
|
} else {
|
||||||
|
$opclass = "vector_cosine_ops";
|
||||||
|
}
|
||||||
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v $opclass);");
|
||||||
|
|
||||||
|
# Test approximate results
|
||||||
|
test_recall(1, 0.75, $operator);
|
||||||
|
test_recall(10, 0.95, $operator);
|
||||||
|
test_recall(100, 1.0, $operator);
|
||||||
|
}
|
||||||
@@ -2,7 +2,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 3;
|
||||||
|
|
||||||
# Initialize node
|
# Initialize node
|
||||||
my $node = get_new_node('node');
|
my $node = get_new_node('node');
|
||||||
@@ -20,7 +20,7 @@ sub test_centers
|
|||||||
{
|
{
|
||||||
my ($lists, $min) = @_;
|
my ($lists, $min) = @_;
|
||||||
|
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres", "CREATE INDEX ON tst USING ivfflat (v vector_l2_ops) WITH (lists = $lists);");
|
my ($ret, $stdout, $stderr) = $node->psql("postgres", "CREATE INDEX ON tst USING ivfflat (v) WITH (lists = $lists);");
|
||||||
is($ret, 0, $stderr);
|
is($ret, 0, $stderr);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -34,5 +34,3 @@ $node->safe_psql("postgres",
|
|||||||
|
|
||||||
# Test no error for duplicate centers
|
# Test no error for duplicate centers
|
||||||
test_centers(10);
|
test_centers(10);
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -2,7 +2,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 60;
|
||||||
|
|
||||||
# Initialize node
|
# Initialize node
|
||||||
my $node = get_new_node('node');
|
my $node = get_new_node('node');
|
||||||
@@ -18,21 +18,24 @@ $node->safe_psql("postgres",
|
|||||||
|
|
||||||
# Check each index type
|
# Check each index type
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
my @operators = ("<->", "<#>", "<=>");
|
||||||
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
foreach (@operators) {
|
||||||
|
my $operator = $_;
|
||||||
for my $i (0 .. $#operators)
|
|
||||||
{
|
|
||||||
my $operator = $operators[$i];
|
|
||||||
my $opclass = $opclasses[$i];
|
|
||||||
|
|
||||||
# Add index
|
# Add index
|
||||||
|
my $opclass;
|
||||||
|
if ($operator == "<->") {
|
||||||
|
$opclass = "vector_l2_ops";
|
||||||
|
} elsif ($operator == "<#>") {
|
||||||
|
$opclass = "vector_ip_ops";
|
||||||
|
} else {
|
||||||
|
$opclass = "vector_cosine_ops";
|
||||||
|
}
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v $opclass);");
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v $opclass);");
|
||||||
|
|
||||||
# Test 100% recall
|
# Test 100% recall
|
||||||
for (1 .. 20)
|
for (1..20) {
|
||||||
{
|
my $i = int(rand() * 100000);
|
||||||
my $id = int(rand() * 100000);
|
my $query = $node->safe_psql("postgres", "SELECT v FROM tst WHERE i = $i;");
|
||||||
my $query = $node->safe_psql("postgres", "SELECT v FROM tst WHERE i = $id;");
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
my $res = $node->safe_psql("postgres", qq(
|
||||||
SET enable_seqscan = off;
|
SET enable_seqscan = off;
|
||||||
SELECT v FROM tst ORDER BY v <-> '$query' LIMIT 1;
|
SELECT v FROM tst ORDER BY v <-> '$query' LIMIT 1;
|
||||||
@@ -40,5 +43,3 @@ for my $i (0 .. $#operators)
|
|||||||
is($res, $query);
|
is($res, $query);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -2,7 +2,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 3;
|
||||||
|
|
||||||
# Initialize node
|
# Initialize node
|
||||||
my $node = get_new_node('node');
|
my $node = get_new_node('node');
|
||||||
@@ -16,8 +16,8 @@ $node->safe_psql("postgres",
|
|||||||
"INSERT INTO tst SELECT ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
"INSERT INTO tst SELECT ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
||||||
);
|
);
|
||||||
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX lists50 ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 50);");
|
$node->safe_psql("postgres", "CREATE INDEX lists50 ON tst USING ivfflat (v) WITH (lists = 50);");
|
||||||
$node->safe_psql("postgres", "CREATE INDEX lists100 ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 100);");
|
$node->safe_psql("postgres", "CREATE INDEX lists100 ON tst USING ivfflat (v) WITH (lists = 100);");
|
||||||
|
|
||||||
# Test prefers more lists
|
# Test prefers more lists
|
||||||
my $res = $node->safe_psql("postgres", "EXPLAIN SELECT v FROM tst ORDER BY v <-> '[0.5,0.5,0.5]' LIMIT 10;");
|
my $res = $node->safe_psql("postgres", "EXPLAIN SELECT v FROM tst ORDER BY v <-> '[0.5,0.5,0.5]' LIMIT 10;");
|
||||||
@@ -26,8 +26,6 @@ unlike($res, qr/lists50/);
|
|||||||
|
|
||||||
# Test errors with too much memory
|
# Test errors with too much memory
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres",
|
my ($ret, $stdout, $stderr) = $node->psql("postgres",
|
||||||
"CREATE INDEX lists10000 ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 10000);"
|
"CREATE INDEX lists10000 ON tst USING ivfflat (v) WITH (lists = 10000);"
|
||||||
);
|
);
|
||||||
like($stderr, qr/memory required is/);
|
like($stderr, qr/memory required is/);
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -2,7 +2,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 7;
|
||||||
|
|
||||||
my $dim = 768;
|
my $dim = 768;
|
||||||
|
|
||||||
@@ -19,7 +19,7 @@ $node->safe_psql("postgres", "CREATE TABLE tst (v vector($dim));");
|
|||||||
$node->safe_psql("postgres",
|
$node->safe_psql("postgres",
|
||||||
"INSERT INTO tst SELECT ARRAY[$array_sql] FROM generate_series(1, 10000) i;"
|
"INSERT INTO tst SELECT ARRAY[$array_sql] FROM generate_series(1, 10000) i;"
|
||||||
);
|
);
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v vector_l2_ops);");
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v);");
|
||||||
|
|
||||||
$node->pgbench(
|
$node->pgbench(
|
||||||
"--no-vacuum --client=5 --transactions=100",
|
"--no-vacuum --client=5 --transactions=100",
|
||||||
@@ -28,7 +28,7 @@ $node->pgbench(
|
|||||||
[qr{^$}],
|
[qr{^$}],
|
||||||
"concurrent INSERTs",
|
"concurrent INSERTs",
|
||||||
{
|
{
|
||||||
"007_ivfflat_inserts" => "INSERT INTO tst SELECT ARRAY[$array_sql] FROM generate_series(1, 10) i;"
|
"007_inserts" => "INSERT INTO tst SELECT ARRAY[$array_sql] FROM generate_series(1, 10) i;"
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -53,5 +53,3 @@ $count = $node->safe_psql("postgres", qq(
|
|||||||
));
|
));
|
||||||
is($count, $expected);
|
is($count, $expected);
|
||||||
is(idx_scan(), 1);
|
is(idx_scan(), 1);
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,49 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (r1 real, r2 real, r3 real, v vector(3));");
|
|
||||||
$node->safe_psql("postgres", qq(
|
|
||||||
INSERT INTO tst SELECT r1, r2, r3, ARRAY[r1, r2, r3] FROM (
|
|
||||||
SELECT random() + 1.01 AS r1, random() + 2.01 AS r2, random() + 3.01 AS r3 FROM generate_series(1, 1000000) t
|
|
||||||
) i;
|
|
||||||
));
|
|
||||||
|
|
||||||
sub test_aggregate
|
|
||||||
{
|
|
||||||
my ($agg) = @_;
|
|
||||||
|
|
||||||
# Test value
|
|
||||||
my $res = $node->safe_psql("postgres", "SELECT $agg(v) FROM tst;");
|
|
||||||
like($res, qr/\[1\.5/);
|
|
||||||
like($res, qr/,2\.5/);
|
|
||||||
like($res, qr/,3\.5/);
|
|
||||||
|
|
||||||
# Test matches real for avg
|
|
||||||
# Cannot test sum since sum(real) varies between calls
|
|
||||||
if ($agg eq 'avg')
|
|
||||||
{
|
|
||||||
my $r1 = $node->safe_psql("postgres", "SELECT $agg(r1)::float4 FROM tst;");
|
|
||||||
my $r2 = $node->safe_psql("postgres", "SELECT $agg(r2)::float4 FROM tst;");
|
|
||||||
my $r3 = $node->safe_psql("postgres", "SELECT $agg(r3)::float4 FROM tst;");
|
|
||||||
is($res, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Test explain
|
|
||||||
my $explain = $node->safe_psql("postgres", "EXPLAIN SELECT $agg(v) FROM tst;");
|
|
||||||
like($explain, qr/Partial Aggregate/);
|
|
||||||
}
|
|
||||||
|
|
||||||
test_aggregate('avg');
|
|
||||||
test_aggregate('sum');
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
35
test/t/008_avg.pl
Normal file
35
test/t/008_avg.pl
Normal file
@@ -0,0 +1,35 @@
|
|||||||
|
use strict;
|
||||||
|
use warnings;
|
||||||
|
use PostgresNode;
|
||||||
|
use TestLib;
|
||||||
|
use Test::More tests => 5;
|
||||||
|
|
||||||
|
# Initialize node
|
||||||
|
my $node = get_new_node('node');
|
||||||
|
$node->init;
|
||||||
|
$node->start;
|
||||||
|
|
||||||
|
# Create table
|
||||||
|
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
||||||
|
$node->safe_psql("postgres", "CREATE TABLE tst (r1 real, r2 real, r3 real, v vector(3));");
|
||||||
|
$node->safe_psql("postgres", qq(
|
||||||
|
INSERT INTO tst SELECT r1, r2, r3, ARRAY[r1, r2, r3] FROM (
|
||||||
|
SELECT random() + 1.01 AS r1, random() + 2.01 AS r2, random() + 3.01 AS r3 FROM generate_series(1, 1000000) t
|
||||||
|
) i;
|
||||||
|
));
|
||||||
|
|
||||||
|
# Test avg
|
||||||
|
my $avg = $node->safe_psql("postgres", "SELECT AVG(v) FROM tst;");
|
||||||
|
like($avg, qr/\[1\.5/);
|
||||||
|
like($avg, qr/,2\.5/);
|
||||||
|
like($avg, qr/,3\.5/);
|
||||||
|
|
||||||
|
# Test matches real
|
||||||
|
my $r1 = $node->safe_psql("postgres", "SELECT AVG(r1)::float4 FROM tst;");
|
||||||
|
my $r2 = $node->safe_psql("postgres", "SELECT AVG(r2)::float4 FROM tst;");
|
||||||
|
my $r3 = $node->safe_psql("postgres", "SELECT AVG(r3)::float4 FROM tst;");
|
||||||
|
is($avg, "[$r1,$r2,$r3]");
|
||||||
|
|
||||||
|
# Test explain
|
||||||
|
my $explain = $node->safe_psql("postgres", "EXPLAIN SELECT AVG(v) FROM tst;");
|
||||||
|
like($explain, qr/Partial Aggregate/);
|
||||||
@@ -2,7 +2,7 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More;
|
use Test::More tests => 1;
|
||||||
|
|
||||||
my $dim = 1024;
|
my $dim = 1024;
|
||||||
|
|
||||||
@@ -30,5 +30,3 @@ my ($ret, $stdout, $stderr) = $node->psql("postgres",
|
|||||||
"INSERT INTO tst SELECT array_agg(n), array_agg(n), array_agg(n) FROM generate_series(1, $dim) n"
|
"INSERT INTO tst SELECT array_agg(n), array_agg(n), array_agg(n) FROM generate_series(1, $dim) n"
|
||||||
);
|
);
|
||||||
like($stderr, qr/row is too big/);
|
like($stderr, qr/row is too big/);
|
||||||
|
|
||||||
done_testing();
|
|
||||||
|
|||||||
@@ -1,99 +0,0 @@
|
|||||||
# Based on postgres/contrib/bloom/t/001_wal.pl
|
|
||||||
|
|
||||||
# Test generic xlog record work for hnsw index replication.
|
|
||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $dim = 32;
|
|
||||||
|
|
||||||
my $node_primary;
|
|
||||||
my $node_replica;
|
|
||||||
|
|
||||||
# Run few queries on both primary and replica and check their results match.
|
|
||||||
sub test_index_replay
|
|
||||||
{
|
|
||||||
my ($test_name) = @_;
|
|
||||||
|
|
||||||
# Wait for replica to catch up
|
|
||||||
my $applname = $node_replica->name;
|
|
||||||
my $caughtup_query = "SELECT pg_current_wal_lsn() <= replay_lsn FROM pg_stat_replication WHERE application_name = '$applname';";
|
|
||||||
$node_primary->poll_query_until('postgres', $caughtup_query)
|
|
||||||
or die "Timed out while waiting for replica 1 to catch up";
|
|
||||||
|
|
||||||
my @r = ();
|
|
||||||
for (1 .. $dim)
|
|
||||||
{
|
|
||||||
push(@r, rand());
|
|
||||||
}
|
|
||||||
my $sql = join(",", @r);
|
|
||||||
|
|
||||||
my $queries = qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT * FROM tst ORDER BY v <-> '[$sql]' LIMIT 10;
|
|
||||||
);
|
|
||||||
|
|
||||||
# Run test queries and compare their result
|
|
||||||
my $primary_result = $node_primary->safe_psql("postgres", $queries);
|
|
||||||
my $replica_result = $node_replica->safe_psql("postgres", $queries);
|
|
||||||
|
|
||||||
is($primary_result, $replica_result, "$test_name: query result matches");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Use ARRAY[random(), random(), random(), ...] over
|
|
||||||
# SELECT array_agg(random()) FROM generate_series(1, $dim)
|
|
||||||
# to generate different values for each row
|
|
||||||
my $array_sql = join(",", ('random()') x $dim);
|
|
||||||
|
|
||||||
# Initialize primary node
|
|
||||||
$node_primary = get_new_node('primary');
|
|
||||||
$node_primary->init(allows_streaming => 1);
|
|
||||||
if ($dim > 32)
|
|
||||||
{
|
|
||||||
# TODO use wal_keep_segments for Postgres < 13
|
|
||||||
$node_primary->append_conf('postgresql.conf', qq(wal_keep_size = 1GB));
|
|
||||||
}
|
|
||||||
if ($dim > 1500)
|
|
||||||
{
|
|
||||||
$node_primary->append_conf('postgresql.conf', qq(maintenance_work_mem = 128MB));
|
|
||||||
}
|
|
||||||
$node_primary->start;
|
|
||||||
my $backup_name = 'my_backup';
|
|
||||||
|
|
||||||
# Take backup
|
|
||||||
$node_primary->backup($backup_name);
|
|
||||||
|
|
||||||
# Create streaming replica linking to primary
|
|
||||||
$node_replica = get_new_node('replica');
|
|
||||||
$node_replica->init_from_backup($node_primary, $backup_name, has_streaming => 1);
|
|
||||||
$node_replica->start;
|
|
||||||
|
|
||||||
# Create hnsw index on primary
|
|
||||||
$node_primary->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node_primary->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim));");
|
|
||||||
$node_primary->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series(1, 1000) i;"
|
|
||||||
);
|
|
||||||
$node_primary->safe_psql("postgres", "CREATE INDEX ON tst USING hnsw (v vector_l2_ops);");
|
|
||||||
|
|
||||||
# Test that queries give same result
|
|
||||||
test_index_replay('initial');
|
|
||||||
|
|
||||||
# Run 10 cycles of table modification. Run test queries after each modification.
|
|
||||||
for my $i (1 .. 10)
|
|
||||||
{
|
|
||||||
$node_primary->safe_psql("postgres", "DELETE FROM tst WHERE i = $i;");
|
|
||||||
test_index_replay("delete $i");
|
|
||||||
$node_primary->safe_psql("postgres", "VACUUM tst;");
|
|
||||||
test_index_replay("vacuum $i");
|
|
||||||
my ($start, $end) = (1001 + ($i - 1) * 100, 1000 + $i * 100);
|
|
||||||
$node_primary->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i % 10, ARRAY[$array_sql] FROM generate_series($start, $end) i;"
|
|
||||||
);
|
|
||||||
test_index_replay("insert $i");
|
|
||||||
}
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $dim = 3;
|
|
||||||
|
|
||||||
my @r = ();
|
|
||||||
for (1 .. $dim)
|
|
||||||
{
|
|
||||||
my $v = int(rand(1000)) + 1;
|
|
||||||
push(@r, "i % $v");
|
|
||||||
}
|
|
||||||
my $array_sql = join(", ", @r);
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table and index
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim));");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[$array_sql] FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING hnsw (v vector_l2_ops);");
|
|
||||||
|
|
||||||
# Get size
|
|
||||||
my $size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
|
||||||
|
|
||||||
# Delete all, vacuum, and insert same data
|
|
||||||
$node->safe_psql("postgres", "DELETE FROM tst;");
|
|
||||||
$node->safe_psql("postgres", "VACUUM tst;");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[$array_sql] FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
# Check size
|
|
||||||
# May increase some due to different levels
|
|
||||||
my $new_size = $node->safe_psql("postgres", "SELECT pg_total_relation_size('tst_v_idx');");
|
|
||||||
cmp_ok($new_size, "<=", $size * 1.02, "size does not increase too much");
|
|
||||||
|
|
||||||
# Delete all but one
|
|
||||||
$node->safe_psql("postgres", "DELETE FROM tst WHERE i != 123;");
|
|
||||||
$node->safe_psql("postgres", "VACUUM tst;");
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v <-> '[0,0,0]' LIMIT 10;
|
|
||||||
));
|
|
||||||
is($res, 123);
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,128 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $node;
|
|
||||||
my @queries = ();
|
|
||||||
my @expected;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
|
||||||
{
|
|
||||||
my ($min, $operator) = @_;
|
|
||||||
my $correct = 0;
|
|
||||||
my $total = 0;
|
|
||||||
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v $operator '$queries[0]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan/);
|
|
||||||
|
|
||||||
for my $i (0 .. $#queries)
|
|
||||||
{
|
|
||||||
my $actual = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
my @actual_ids = split("\n", $actual);
|
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
|
||||||
|
|
||||||
my @expected_ids = split("\n", $expected[$i]);
|
|
||||||
|
|
||||||
foreach (@expected_ids)
|
|
||||||
{
|
|
||||||
if (exists($actual_set{$_}))
|
|
||||||
{
|
|
||||||
$correct++;
|
|
||||||
}
|
|
||||||
$total++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $operator);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
$node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector(3));");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[random(), random(), random()] FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
# Generate queries
|
|
||||||
for (1 .. 20)
|
|
||||||
{
|
|
||||||
my $r1 = rand();
|
|
||||||
my $r2 = rand();
|
|
||||||
my $r3 = rand();
|
|
||||||
push(@queries, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Check each index type
|
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
|
||||||
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
|
||||||
|
|
||||||
for my $i (0 .. $#operators)
|
|
||||||
{
|
|
||||||
my $operator = $operators[$i];
|
|
||||||
my $opclass = $opclasses[$i];
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries)
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", "SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;");
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Build index serially
|
|
||||||
$node->safe_psql("postgres", qq(
|
|
||||||
SET max_parallel_maintenance_workers = 0;
|
|
||||||
CREATE INDEX idx ON tst USING hnsw (v $opclass);
|
|
||||||
));
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
my $min = $operator eq "<#>" ? 0.80 : 0.99;
|
|
||||||
test_recall($min, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
|
|
||||||
# Build index in parallel in memory
|
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres", qq(
|
|
||||||
SET client_min_messages = DEBUG;
|
|
||||||
SET min_parallel_table_scan_size = 1;
|
|
||||||
CREATE INDEX idx ON tst USING hnsw (v $opclass);
|
|
||||||
));
|
|
||||||
is($ret, 0, $stderr);
|
|
||||||
like($stderr, qr/using \d+ parallel workers/);
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
test_recall($min, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
|
|
||||||
# Build index in parallel on disk
|
|
||||||
# Set parallel_workers on table to use workers with low maintenance_work_mem
|
|
||||||
($ret, $stdout, $stderr) = $node->psql("postgres", qq(
|
|
||||||
ALTER TABLE tst SET (parallel_workers = 2);
|
|
||||||
SET client_min_messages = DEBUG;
|
|
||||||
SET maintenance_work_mem = '4MB';
|
|
||||||
CREATE INDEX idx ON tst USING hnsw (v $opclass);
|
|
||||||
ALTER TABLE tst RESET (parallel_workers);
|
|
||||||
));
|
|
||||||
is($ret, 0, $stderr);
|
|
||||||
like($stderr, qr/using \d+ parallel workers/);
|
|
||||||
like($stderr, qr/hnsw graph no longer fits into maintenance_work_mem/);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
}
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,108 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $node;
|
|
||||||
my @queries = ();
|
|
||||||
my @expected;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
|
||||||
{
|
|
||||||
my ($min, $operator) = @_;
|
|
||||||
my $correct = 0;
|
|
||||||
my $total = 0;
|
|
||||||
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v $operator '$queries[0]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan/);
|
|
||||||
|
|
||||||
for my $i (0 .. $#queries)
|
|
||||||
{
|
|
||||||
my $actual = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
my @actual_ids = split("\n", $actual);
|
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
|
||||||
|
|
||||||
my @expected_ids = split("\n", $expected[$i]);
|
|
||||||
|
|
||||||
foreach (@expected_ids)
|
|
||||||
{
|
|
||||||
if (exists($actual_set{$_}))
|
|
||||||
{
|
|
||||||
$correct++;
|
|
||||||
}
|
|
||||||
$total++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $operator);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
$node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i serial, v vector(3));");
|
|
||||||
|
|
||||||
# Generate queries
|
|
||||||
for (1 .. 20)
|
|
||||||
{
|
|
||||||
my $r1 = rand();
|
|
||||||
my $r2 = rand();
|
|
||||||
my $r3 = rand();
|
|
||||||
push(@queries, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Check each index type
|
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
|
||||||
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
|
||||||
|
|
||||||
for my $i (0 .. $#operators)
|
|
||||||
{
|
|
||||||
my $operator = $operators[$i];
|
|
||||||
my $opclass = $opclasses[$i];
|
|
||||||
|
|
||||||
# Add index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING hnsw (v $opclass);");
|
|
||||||
|
|
||||||
# Use concurrent inserts
|
|
||||||
$node->pgbench(
|
|
||||||
"--no-vacuum --client=10 --transactions=1000",
|
|
||||||
0,
|
|
||||||
[qr{actually processed}],
|
|
||||||
[qr{^$}],
|
|
||||||
"concurrent INSERTs",
|
|
||||||
{
|
|
||||||
"013_hnsw_insert_recall_$opclass" => "INSERT INTO tst (v) VALUES (ARRAY[random(), random(), random()]);"
|
|
||||||
}
|
|
||||||
);
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries)
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_indexscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;
|
|
||||||
));
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
my $min = $operator eq "<#>" ? 0.80 : 0.99;
|
|
||||||
test_recall($min, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
$node->safe_psql("postgres", "TRUNCATE tst;");
|
|
||||||
}
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,74 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
# Ensures elements and neighbors on both same and different pages
|
|
||||||
my $dim = 1900;
|
|
||||||
|
|
||||||
my $array_sql = join(",", ('random()') x $dim);
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table and index
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (v vector($dim));");
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING hnsw (v vector_l2_ops);");
|
|
||||||
|
|
||||||
sub idx_scan
|
|
||||||
{
|
|
||||||
# Stats do not update instantaneously
|
|
||||||
# https://www.postgresql.org/docs/current/monitoring-stats.html#MONITORING-STATS-VIEWS
|
|
||||||
sleep(1);
|
|
||||||
$node->safe_psql("postgres", "SELECT idx_scan FROM pg_stat_user_indexes WHERE indexrelid = 'tst_v_idx'::regclass;");
|
|
||||||
}
|
|
||||||
|
|
||||||
for my $i (1 .. 20)
|
|
||||||
{
|
|
||||||
$node->pgbench(
|
|
||||||
"--no-vacuum --client=10 --transactions=1",
|
|
||||||
0,
|
|
||||||
[qr{actually processed}],
|
|
||||||
[qr{^$}],
|
|
||||||
"concurrent INSERTs",
|
|
||||||
{
|
|
||||||
"014_hnsw_inserts_$i" => "INSERT INTO tst VALUES (ARRAY[$array_sql]);"
|
|
||||||
}
|
|
||||||
);
|
|
||||||
|
|
||||||
my $count = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT COUNT(*) FROM (SELECT v FROM tst ORDER BY v <-> (SELECT v FROM tst LIMIT 1)) t;
|
|
||||||
));
|
|
||||||
is($count, 10);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "TRUNCATE tst;");
|
|
||||||
}
|
|
||||||
|
|
||||||
$node->pgbench(
|
|
||||||
"--no-vacuum --client=20 --transactions=5",
|
|
||||||
0,
|
|
||||||
[qr{actually processed}],
|
|
||||||
[qr{^$}],
|
|
||||||
"concurrent INSERTs",
|
|
||||||
{
|
|
||||||
"014_hnsw_inserts" => "INSERT INTO tst SELECT ARRAY[$array_sql] FROM generate_series(1, 10) i;"
|
|
||||||
}
|
|
||||||
);
|
|
||||||
|
|
||||||
my $count = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET hnsw.ef_search = 1000;
|
|
||||||
SELECT COUNT(*) FROM (SELECT v FROM tst ORDER BY v <-> (SELECT v FROM tst LIMIT 1)) t;
|
|
||||||
));
|
|
||||||
# Elements may lose all incoming connections with the HNSW algorithm
|
|
||||||
# Vacuuming can fix this if one of the elements neighbors is deleted
|
|
||||||
cmp_ok($count, ">=", 997);
|
|
||||||
|
|
||||||
is(idx_scan(), 21);
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,58 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (v vector(3));");
|
|
||||||
|
|
||||||
sub insert_vectors
|
|
||||||
{
|
|
||||||
for my $i (1 .. 20)
|
|
||||||
{
|
|
||||||
$node->safe_psql("postgres", "INSERT INTO tst VALUES ('[1,1,1]');");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
sub test_duplicates
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET hnsw.ef_search = 1;
|
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM tst ORDER BY v <-> '[1,1,1]') t;
|
|
||||||
));
|
|
||||||
is($res, 10);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Test duplicates with build
|
|
||||||
insert_vectors();
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING hnsw (v vector_l2_ops);");
|
|
||||||
test_duplicates();
|
|
||||||
|
|
||||||
# Reset
|
|
||||||
$node->safe_psql("postgres", "TRUNCATE tst;");
|
|
||||||
|
|
||||||
# Test duplicates with inserts
|
|
||||||
insert_vectors();
|
|
||||||
test_duplicates();
|
|
||||||
|
|
||||||
# Test fallback path for inserts
|
|
||||||
$node->pgbench(
|
|
||||||
"--no-vacuum --client=5 --transactions=100",
|
|
||||||
0,
|
|
||||||
[qr{actually processed}],
|
|
||||||
[qr{^$}],
|
|
||||||
"concurrent INSERTs",
|
|
||||||
{
|
|
||||||
"015_hnsw_duplicates" => "INSERT INTO tst VALUES ('[1,1,1]');"
|
|
||||||
}
|
|
||||||
);
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,97 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $node;
|
|
||||||
my @queries = ();
|
|
||||||
my @expected;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
|
||||||
{
|
|
||||||
my ($min, $ef_search, $test_name) = @_;
|
|
||||||
my $correct = 0;
|
|
||||||
my $total = 0;
|
|
||||||
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET hnsw.ef_search = $ef_search;
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v <-> '$queries[0]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan/);
|
|
||||||
|
|
||||||
for my $i (0 .. $#queries)
|
|
||||||
{
|
|
||||||
my $actual = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET hnsw.ef_search = $ef_search;
|
|
||||||
SELECT i FROM tst ORDER BY v <-> '$queries[$i]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
my @actual_ids = split("\n", $actual);
|
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
|
||||||
|
|
||||||
my @expected_ids = split("\n", $expected[$i]);
|
|
||||||
|
|
||||||
foreach (@expected_ids)
|
|
||||||
{
|
|
||||||
if (exists($actual_set{$_}))
|
|
||||||
{
|
|
||||||
$correct++;
|
|
||||||
}
|
|
||||||
$total++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $test_name);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
$node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector(3));");
|
|
||||||
$node->safe_psql("postgres", "ALTER TABLE tst SET (autovacuum_enabled = false);");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[random(), random(), random()] FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
# Add index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING hnsw (v vector_l2_ops) WITH (m = 4, ef_construction = 8);");
|
|
||||||
|
|
||||||
# Delete data
|
|
||||||
$node->safe_psql("postgres", "DELETE FROM tst WHERE i > 2500;");
|
|
||||||
|
|
||||||
# Generate queries
|
|
||||||
for (1 .. 20)
|
|
||||||
{
|
|
||||||
my $r1 = rand();
|
|
||||||
my $r2 = rand();
|
|
||||||
my $r3 = rand();
|
|
||||||
push(@queries, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries)
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_indexscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v <-> '$_' LIMIT $limit;
|
|
||||||
));
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
test_recall(0.20, $limit, "before vacuum");
|
|
||||||
test_recall(0.95, 100, "before vacuum");
|
|
||||||
|
|
||||||
# TODO Test concurrent inserts with vacuum
|
|
||||||
$node->safe_psql("postgres", "VACUUM tst;");
|
|
||||||
|
|
||||||
test_recall(0.95, $limit, "after vacuum");
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,117 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $node;
|
|
||||||
my @queries = ();
|
|
||||||
my @expected;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
|
||||||
{
|
|
||||||
my ($probes, $min, $operator) = @_;
|
|
||||||
my $correct = 0;
|
|
||||||
my $total = 0;
|
|
||||||
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET ivfflat.probes = $probes;
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v $operator '$queries[0]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx on tst/);
|
|
||||||
|
|
||||||
for my $i (0 .. $#queries)
|
|
||||||
{
|
|
||||||
my $actual = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SET ivfflat.probes = $probes;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
|
||||||
));
|
|
||||||
my @actual_ids = split("\n", $actual);
|
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
|
||||||
|
|
||||||
my @expected_ids = split("\n", $expected[$i]);
|
|
||||||
|
|
||||||
foreach (@expected_ids)
|
|
||||||
{
|
|
||||||
if (exists($actual_set{$_}))
|
|
||||||
{
|
|
||||||
$correct++;
|
|
||||||
}
|
|
||||||
$total++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $operator);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
$node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i serial, v vector(3));");
|
|
||||||
|
|
||||||
# Generate queries
|
|
||||||
for (1 .. 20)
|
|
||||||
{
|
|
||||||
my $r1 = rand();
|
|
||||||
my $r2 = rand();
|
|
||||||
my $r3 = rand();
|
|
||||||
push(@queries, "[$r1,$r2,$r3]");
|
|
||||||
}
|
|
||||||
|
|
||||||
# Check each index type
|
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
|
||||||
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
|
||||||
|
|
||||||
for my $i (0 .. $#operators)
|
|
||||||
{
|
|
||||||
my $operator = $operators[$i];
|
|
||||||
my $opclass = $opclasses[$i];
|
|
||||||
|
|
||||||
# Add index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING ivfflat (v $opclass);");
|
|
||||||
|
|
||||||
# Use concurrent inserts
|
|
||||||
$node->pgbench(
|
|
||||||
"--no-vacuum --client=10 --transactions=1000",
|
|
||||||
0,
|
|
||||||
[qr{actually processed}],
|
|
||||||
[qr{^$}],
|
|
||||||
"concurrent INSERTs",
|
|
||||||
{
|
|
||||||
"017_ivfflat_insert_recall_$opclass" => "INSERT INTO tst (v) SELECT ARRAY[random(), random(), random()] FROM generate_series(1, 10) i;"
|
|
||||||
}
|
|
||||||
);
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries)
|
|
||||||
{
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_indexscan = off;
|
|
||||||
SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;
|
|
||||||
));
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
if ($operator ne "<#>")
|
|
||||||
{
|
|
||||||
# TODO Fix test (uniform random vectors all have similar inner product)
|
|
||||||
test_recall(1, 0.71, $operator);
|
|
||||||
test_recall(10, 0.95, $operator);
|
|
||||||
}
|
|
||||||
# Account for equal distances
|
|
||||||
test_recall(100, 0.9925, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
$node->safe_psql("postgres", "TRUNCATE tst;");
|
|
||||||
}
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,114 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $dim = 3;
|
|
||||||
my $nc = 50;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
my $array_sql = join(",", ('random()') x $dim);
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table and index
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim), c int4, t text);");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[$array_sql], i % $nc, 'test ' || i FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING hnsw (v vector_l2_ops);");
|
|
||||||
$node->safe_psql("postgres", "ANALYZE tst;");
|
|
||||||
|
|
||||||
# Generate query
|
|
||||||
my @r = ();
|
|
||||||
for (1 .. $dim)
|
|
||||||
{
|
|
||||||
push(@r, rand());
|
|
||||||
}
|
|
||||||
my $query = "[" . join(",", @r) . "]";
|
|
||||||
my $c = int(rand() * $nc);
|
|
||||||
|
|
||||||
# Test attribute filtering
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c != $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed comparison
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c >= 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with many rows removed comparison
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c < 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed like
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE t LIKE '%%test%%' ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with many rows removed like
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE t LIKE '%%other%%' ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Seq Scan/);
|
|
||||||
|
|
||||||
# Test distance filtering
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test distance filtering greater than distance
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' > 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test distance filtering without order
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1;
|
|
||||||
));
|
|
||||||
like($explain, qr/Seq Scan/);
|
|
||||||
|
|
||||||
# Test distance filtering without limit
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1 ORDER BY v <-> '$query';
|
|
||||||
));
|
|
||||||
like($explain, qr/Seq Scan/);
|
|
||||||
|
|
||||||
# Test attribute index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX attribute_idx ON tst (c);");
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Use attribute index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test partial index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX partial_idx ON tst USING hnsw (v vector_l2_ops) WHERE (c = $c);");
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using partial_idx/);
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,116 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More;
|
|
||||||
|
|
||||||
my $dim = 3;
|
|
||||||
my $nc = 50;
|
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
my $array_sql = join(",", ('random()') x $dim);
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table and index
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4, v vector($dim), c int4, t text);");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[$array_sql], i % $nc, 'test ' || i FROM generate_series(1, 10000) i;"
|
|
||||||
);
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 100);");
|
|
||||||
$node->safe_psql("postgres", "ANALYZE tst;");
|
|
||||||
|
|
||||||
# Generate query
|
|
||||||
my @r = ();
|
|
||||||
for (1 .. $dim)
|
|
||||||
{
|
|
||||||
push(@r, rand());
|
|
||||||
}
|
|
||||||
my $query = "[" . join(",", @r) . "]";
|
|
||||||
my $c = int(rand() * $nc);
|
|
||||||
|
|
||||||
# Test attribute filtering
|
|
||||||
my $explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c != $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed comparison
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c >= 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with many rows removed comparison
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c < 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with few rows removed like
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE t LIKE '%%test%%' ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute filtering with many rows removed like
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE t LIKE '%%other%%' ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Seq Scan/);
|
|
||||||
|
|
||||||
# Test distance filtering
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test distance filtering greater than distance
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' > 1 ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test distance filtering without order
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1;
|
|
||||||
));
|
|
||||||
like($explain, qr/Seq Scan/);
|
|
||||||
|
|
||||||
# Test distance filtering without limit
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE v <-> '$query' < 1 ORDER BY v <-> '$query';
|
|
||||||
));
|
|
||||||
# TODO Do not use index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test attribute index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX attribute_idx ON tst (c);");
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Use attribute index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
# Test partial index
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX partial_idx ON tst USING ivfflat (v vector_l2_ops) WITH (lists = 5) WHERE (c = $c);");
|
|
||||||
$explain = $node->safe_psql("postgres", qq(
|
|
||||||
EXPLAIN ANALYZE SELECT i FROM tst WHERE c = $c ORDER BY v <-> '$query' LIMIT $limit;
|
|
||||||
));
|
|
||||||
# TODO Use partial index
|
|
||||||
like($explain, qr/Index Scan using idx/);
|
|
||||||
|
|
||||||
done_testing();
|
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
comment = 'vector data type and ivfflat and hnsw access methods'
|
comment = 'vector data type and ivfflat access method'
|
||||||
default_version = '0.6.1'
|
default_version = '0.4.1'
|
||||||
module_pathname = '$libdir/vector'
|
module_pathname = '$libdir/vector'
|
||||||
relocatable = true
|
relocatable = true
|
||||||
|
|||||||
Reference in New Issue
Block a user