mirror of
https://github.com/pgvector/pgvector.git
synced 2026-07-22 12:07:34 +08:00
Compare commits
169 Commits
parallel-i
...
v0.4.4
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1982121694 | ||
|
|
9c8c4483db | ||
|
|
426ae1f16e | ||
|
|
06c3e68bef | ||
|
|
e5a620e02c | ||
|
|
092bb80f58 | ||
|
|
b690cd4d5d | ||
|
|
4e0d11acfe | ||
|
|
aa63f80b69 | ||
|
|
b8a7355731 | ||
|
|
3332669489 | ||
|
|
08c70bb57f | ||
|
|
eff0de6a64 | ||
|
|
3cf7ce6543 | ||
|
|
3cb6440744 | ||
|
|
b6a0d2b12b | ||
|
|
78632e3301 | ||
|
|
f8c85905c3 | ||
|
|
0a98a953cd | ||
|
|
a577c2df80 | ||
|
|
7ee9e86b10 | ||
|
|
5fdf5573a0 | ||
|
|
2b939edfee | ||
|
|
987026a559 | ||
|
|
d158eefa60 | ||
|
|
a7bbb0772d | ||
|
|
6ad276aa54 | ||
|
|
c03ce7d62a | ||
|
|
629fa6f0cd | ||
|
|
a33e72d58e | ||
|
|
49e6a72d36 | ||
|
|
b158a5fa48 | ||
|
|
81cc04df61 | ||
|
|
d6ab4892fe | ||
|
|
cbaf470f2e | ||
|
|
8cb32cac76 | ||
|
|
4ce915cf16 | ||
|
|
41b766c24b | ||
|
|
d11fe7bbfb | ||
|
|
2c35074f3a | ||
|
|
b4c1c3ab63 | ||
|
|
edcbafca79 | ||
|
|
cbec1b3f48 | ||
|
|
f81d863dfd | ||
|
|
b8c7a4f4b6 | ||
|
|
7446cbde8f | ||
|
|
9f2359894f | ||
|
|
3c78130868 | ||
|
|
768cd5d5d5 | ||
|
|
7335a122db | ||
|
|
2115630fb0 | ||
|
|
836be51298 | ||
|
|
972d9d61cb | ||
|
|
8be2b6c244 | ||
|
|
0134debfb8 | ||
|
|
2f93781c3b | ||
|
|
4d910f30fd | ||
|
|
a20add331f | ||
|
|
f03381bc62 | ||
|
|
198390333e | ||
|
|
41c68bf692 | ||
|
|
13cf29088d | ||
|
|
1aea0dfcd8 | ||
|
|
b6430bae62 | ||
|
|
8294a0a562 | ||
|
|
b31c8062c3 | ||
|
|
73ff7c3c68 | ||
|
|
482a5f8b66 | ||
|
|
e971fdd4fd | ||
|
|
6330abb7df | ||
|
|
7938b476ea | ||
|
|
0ef0467a0f | ||
|
|
dee2c4feb1 | ||
|
|
29d9ec6f1e | ||
|
|
0200134397 | ||
|
|
ceddbac6bf | ||
|
|
0deb443458 | ||
|
|
d1fb0d8e27 | ||
|
|
9e5f7fd5ec | ||
|
|
7f744b02c8 | ||
|
|
a9c6af89e8 | ||
|
|
212af771bd | ||
|
|
b37d154b26 | ||
|
|
51bd223b4a | ||
|
|
491b6b18f9 | ||
|
|
4576a9f9a4 | ||
|
|
451ac59a03 | ||
|
|
6f94c5e897 | ||
|
|
e9c88d6f25 | ||
|
|
a912d1af9a | ||
|
|
fa401b7883 | ||
|
|
e97ef5fbac | ||
|
|
dfe487145f | ||
|
|
c2f331908f | ||
|
|
7911a3b395 | ||
|
|
0d46281c02 | ||
|
|
59071dc78d | ||
|
|
5b3878b7fe | ||
|
|
18e7319a40 | ||
|
|
69672cd84d | ||
|
|
300adba2f1 | ||
|
|
e362279199 | ||
|
|
53301021f6 | ||
|
|
8f589f6d09 | ||
|
|
dcf206128a | ||
|
|
3244d40e8a | ||
|
|
7d8dbcaa3c | ||
|
|
7f575f55fb | ||
|
|
94e7487d5f | ||
|
|
74a3cd597f | ||
|
|
db8ed738b8 | ||
|
|
54c550420b | ||
|
|
cc539a0a27 | ||
|
|
d885e2bcfa | ||
|
|
a445355a48 | ||
|
|
d5b17a3624 | ||
|
|
6383078029 | ||
|
|
5146c7cc57 | ||
|
|
ac63f9858b | ||
|
|
76a4166857 | ||
|
|
31fb6963a3 | ||
|
|
18c06cb9b1 | ||
|
|
f858796c64 | ||
|
|
f32f695844 | ||
|
|
1b013a94f7 | ||
|
|
00148dfa1f | ||
|
|
67fc791d95 | ||
|
|
8bf360ed84 | ||
|
|
f79d28347b | ||
|
|
20cf63de0a | ||
|
|
587cbcf15b | ||
|
|
c09edb5b8f | ||
|
|
c63501cca4 | ||
|
|
58f0c922d2 | ||
|
|
36e73d2818 | ||
|
|
dd92d0ece3 | ||
|
|
6ede7681a5 | ||
|
|
8733729149 | ||
|
|
03a5789132 | ||
|
|
e5b612a856 | ||
|
|
9a7d3532f5 | ||
|
|
91315dfeff | ||
|
|
6e3101d527 | ||
|
|
96ae1a6a72 | ||
|
|
55aeba8bd6 | ||
|
|
aebe1bae02 | ||
|
|
14355b9312 | ||
|
|
b5c66d0416 | ||
|
|
f534d9878a | ||
|
|
f3df137db6 | ||
|
|
489cdb5068 | ||
|
|
161f48793e | ||
|
|
f0f7ffca41 | ||
|
|
138d9be616 | ||
|
|
fb98e73255 | ||
|
|
4754cac40c | ||
|
|
8432efb7d8 | ||
|
|
c38410259c | ||
|
|
609d9fbf0a | ||
|
|
7946424639 | ||
|
|
d51310dfa0 | ||
|
|
30f2893aeb | ||
|
|
5d0f88529e | ||
|
|
1d020abdd1 | ||
|
|
d0fdd42652 | ||
|
|
8473468925 | ||
|
|
9c01524466 | ||
|
|
121baa411e | ||
|
|
50005d7326 |
28
.github/workflows/build.yml
vendored
28
.github/workflows/build.yml
vendored
@@ -8,6 +8,8 @@ jobs:
|
|||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
|
- postgres: 16
|
||||||
|
os: ubuntu-22.04
|
||||||
- postgres: 15
|
- postgres: 15
|
||||||
os: ubuntu-22.04
|
os: ubuntu-22.04
|
||||||
- postgres: 14
|
- postgres: 14
|
||||||
@@ -17,7 +19,7 @@ jobs:
|
|||||||
- postgres: 12
|
- postgres: 12
|
||||||
os: ubuntu-20.04
|
os: ubuntu-20.04
|
||||||
- postgres: 11
|
- postgres: 11
|
||||||
os: ubuntu-18.04
|
os: ubuntu-20.04
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v3
|
||||||
- uses: ankane/setup-postgres@v1
|
- uses: ankane/setup-postgres@v1
|
||||||
@@ -25,6 +27,8 @@ jobs:
|
|||||||
postgres-version: ${{ matrix.postgres }}
|
postgres-version: ${{ matrix.postgres }}
|
||||||
dev-files: true
|
dev-files: true
|
||||||
- run: make
|
- run: make
|
||||||
|
env:
|
||||||
|
PG_CFLAGS: -Wall -Wextra -Werror -Wno-unused-parameter -Wno-sign-compare
|
||||||
- run: |
|
- run: |
|
||||||
export PG_CONFIG=`which pg_config`
|
export PG_CONFIG=`which pg_config`
|
||||||
sudo --preserve-env=PG_CONFIG make install
|
sudo --preserve-env=PG_CONFIG make install
|
||||||
@@ -44,6 +48,8 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
postgres-version: 14
|
postgres-version: 14
|
||||||
- run: make
|
- run: make
|
||||||
|
env:
|
||||||
|
PG_CFLAGS: -Wall -Wextra -Werror -Wno-unused-parameter
|
||||||
- run: make install
|
- run: make install
|
||||||
- run: make installcheck
|
- run: make installcheck
|
||||||
- if: ${{ failure() }}
|
- if: ${{ failure() }}
|
||||||
@@ -54,6 +60,7 @@ jobs:
|
|||||||
wget -q https://github.com/postgres/postgres/archive/refs/tags/REL_14_5.tar.gz
|
wget -q https://github.com/postgres/postgres/archive/refs/tags/REL_14_5.tar.gz
|
||||||
tar xf REL_14_5.tar.gz
|
tar xf REL_14_5.tar.gz
|
||||||
- run: make prove_installcheck PROVE_FLAGS="-I ./postgres-REL_14_5/src/test/perl" PERL5LIB="/Users/runner/perl5/lib/perl5"
|
- run: make prove_installcheck PROVE_FLAGS="-I ./postgres-REL_14_5/src/test/perl" PERL5LIB="/Users/runner/perl5/lib/perl5"
|
||||||
|
- run: make clean && /usr/local/opt/llvm@15/bin/scan-build --status-bugs make
|
||||||
windows:
|
windows:
|
||||||
runs-on: windows-latest
|
runs-on: windows-latest
|
||||||
if: ${{ !startsWith(github.ref_name, 'mac') }}
|
if: ${{ !startsWith(github.ref_name, 'mac') }}
|
||||||
@@ -70,3 +77,22 @@ jobs:
|
|||||||
nmake /NOLOGO /F Makefile.win clean && ^
|
nmake /NOLOGO /F Makefile.win clean && ^
|
||||||
nmake /NOLOGO /F Makefile.win uninstall
|
nmake /NOLOGO /F Makefile.win uninstall
|
||||||
shell: cmd
|
shell: cmd
|
||||||
|
i386:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
container:
|
||||||
|
image: debian:11
|
||||||
|
options: --platform linux/386
|
||||||
|
steps:
|
||||||
|
- run: apt-get update && apt-get install -y build-essential git libipc-run-perl postgresql-13 postgresql-server-dev-13 sudo
|
||||||
|
- run: service postgresql start
|
||||||
|
- run: |
|
||||||
|
git clone https://github.com/${{ github.repository }}.git pgvector
|
||||||
|
cd pgvector
|
||||||
|
git checkout ${{ github.ref }}
|
||||||
|
make
|
||||||
|
make install
|
||||||
|
chown -R postgres .
|
||||||
|
sudo -u postgres make installcheck
|
||||||
|
sudo -u postgres make prove_installcheck
|
||||||
|
env:
|
||||||
|
PG_CFLAGS: -Wall -Wextra -Werror -Wno-unused-parameter -Wno-sign-compare
|
||||||
|
|||||||
23
CHANGELOG.md
23
CHANGELOG.md
@@ -1,3 +1,26 @@
|
|||||||
|
## 0.4.4 (2023-06-12)
|
||||||
|
|
||||||
|
- Improved error message for malformed vector literal
|
||||||
|
- Fixed segmentation fault with text input
|
||||||
|
- Fixed consecutive delimiters with text input
|
||||||
|
|
||||||
|
## 0.4.3 (2023-06-10)
|
||||||
|
|
||||||
|
- Improved cost estimation
|
||||||
|
- Improved support for spaces with text input
|
||||||
|
- Fixed infinite and NaN values with binary input
|
||||||
|
- Fixed infinite values with vector addition and subtraction
|
||||||
|
- Fixed infinite values with list centers
|
||||||
|
- Fixed compilation error when `float8` is pass by reference
|
||||||
|
- Fixed compilation error on PowerPC
|
||||||
|
- Fixed segmentation fault with index creation on i386
|
||||||
|
|
||||||
|
## 0.4.2 (2023-05-13)
|
||||||
|
|
||||||
|
- Added notice when index created with little data
|
||||||
|
- Fixed dimensions check for some direct function calls
|
||||||
|
- Fixed installation error with Postgres 12.0-12.2
|
||||||
|
|
||||||
## 0.4.1 (2023-03-21)
|
## 0.4.1 (2023-03-21)
|
||||||
|
|
||||||
- Improved performance of cosine distance
|
- Improved performance of cosine distance
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
FROM postgres:15
|
ARG PG_MAJOR=15
|
||||||
|
FROM postgres:$PG_MAJOR
|
||||||
|
ARG PG_MAJOR
|
||||||
|
|
||||||
COPY . /tmp/pgvector
|
COPY . /tmp/pgvector
|
||||||
|
|
||||||
RUN apt-get update && \
|
RUN apt-get update && \
|
||||||
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-15 && \
|
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-$PG_MAJOR && \
|
||||||
cd /tmp/pgvector && \
|
cd /tmp/pgvector && \
|
||||||
make clean && \
|
make clean && \
|
||||||
make OPTFLAGS="" && \
|
make OPTFLAGS="" && \
|
||||||
@@ -11,6 +13,6 @@ RUN apt-get update && \
|
|||||||
mkdir /usr/share/doc/pgvector && \
|
mkdir /usr/share/doc/pgvector && \
|
||||||
cp LICENSE README.md /usr/share/doc/pgvector && \
|
cp LICENSE README.md /usr/share/doc/pgvector && \
|
||||||
rm -r /tmp/pgvector && \
|
rm -r /tmp/pgvector && \
|
||||||
apt-get remove -y build-essential postgresql-server-dev-15 && \
|
apt-get remove -y build-essential postgresql-server-dev-$PG_MAJOR && \
|
||||||
apt-get autoremove -y && \
|
apt-get autoremove -y && \
|
||||||
rm -rf /var/lib/apt/lists/*
|
rm -rf /var/lib/apt/lists/*
|
||||||
|
|||||||
2
LICENSE
2
LICENSE
@@ -1,4 +1,4 @@
|
|||||||
Portions Copyright (c) 1996-2022, PostgreSQL Global Development Group
|
Portions Copyright (c) 1996-2023, PostgreSQL Global Development Group
|
||||||
|
|
||||||
Portions Copyright (c) 1994, The Regents of the University of California
|
Portions Copyright (c) 1994, The Regents of the University of California
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
"name": "vector",
|
"name": "vector",
|
||||||
"abstract": "Open-source vector similarity search for Postgres",
|
"abstract": "Open-source vector similarity search for Postgres",
|
||||||
"description": "Supports L2 distance, inner product, and cosine distance",
|
"description": "Supports L2 distance, inner product, and cosine distance",
|
||||||
"version": "0.4.1",
|
"version": "0.4.4",
|
||||||
"maintainer": [
|
"maintainer": [
|
||||||
"Andrew Kane <andrew@ankane.org>"
|
"Andrew Kane <andrew@ankane.org>"
|
||||||
],
|
],
|
||||||
@@ -20,7 +20,7 @@
|
|||||||
"vector": {
|
"vector": {
|
||||||
"file": "sql/vector.sql",
|
"file": "sql/vector.sql",
|
||||||
"docfile": "README.md",
|
"docfile": "README.md",
|
||||||
"version": "0.4.1",
|
"version": "0.4.4",
|
||||||
"abstract": "Open-source vector similarity search for Postgres"
|
"abstract": "Open-source vector similarity search for Postgres"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|||||||
14
Makefile
14
Makefile
@@ -1,5 +1,5 @@
|
|||||||
EXTENSION = vector
|
EXTENSION = vector
|
||||||
EXTVERSION = 0.4.1
|
EXTVERSION = 0.4.4
|
||||||
|
|
||||||
MODULE_big = vector
|
MODULE_big = vector
|
||||||
DATA = $(wildcard sql/*--*.sql)
|
DATA = $(wildcard sql/*--*.sql)
|
||||||
@@ -14,10 +14,16 @@ OPTFLAGS = -march=native
|
|||||||
# Mac ARM doesn't support -march=native
|
# Mac ARM doesn't support -march=native
|
||||||
ifeq ($(shell uname -s), Darwin)
|
ifeq ($(shell uname -s), Darwin)
|
||||||
ifeq ($(shell uname -p), arm)
|
ifeq ($(shell uname -p), arm)
|
||||||
|
# no difference with -march=armv8.5-a
|
||||||
OPTFLAGS =
|
OPTFLAGS =
|
||||||
endif
|
endif
|
||||||
endif
|
endif
|
||||||
|
|
||||||
|
# PowerPC doesn't support -march=native
|
||||||
|
ifneq ($(filter ppc64%, $(shell uname -m)), )
|
||||||
|
OPTFLAGS =
|
||||||
|
endif
|
||||||
|
|
||||||
# For auto-vectorization:
|
# For auto-vectorization:
|
||||||
# - GCC (needs -ftree-vectorize OR -O3) - https://gcc.gnu.org/projects/tree-ssa/vectorization.html
|
# - GCC (needs -ftree-vectorize OR -O3) - https://gcc.gnu.org/projects/tree-ssa/vectorization.html
|
||||||
# - Clang (could use pragma instead) - https://llvm.org/docs/Vectorizers.html
|
# - Clang (could use pragma instead) - https://llvm.org/docs/Vectorizers.html
|
||||||
@@ -62,3 +68,9 @@ dist:
|
|||||||
|
|
||||||
docker:
|
docker:
|
||||||
docker build --pull --no-cache --platform linux/amd64 -t ankane/pgvector:latest .
|
docker build --pull --no-cache --platform linux/amd64 -t ankane/pgvector:latest .
|
||||||
|
|
||||||
|
.PHONY: docker-release
|
||||||
|
|
||||||
|
docker-release:
|
||||||
|
docker buildx build --push --pull --no-cache --platform linux/amd64,linux/arm64 -t ankane/pgvector:latest .
|
||||||
|
docker buildx build --push --platform linux/amd64,linux/arm64 -t ankane/pgvector:v$(EXTVERSION) .
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
EXTENSION = vector
|
EXTENSION = vector
|
||||||
EXTVERSION = 0.4.1
|
EXTVERSION = 0.4.4
|
||||||
|
|
||||||
OBJS = src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\vector.obj
|
OBJS = src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\vector.obj
|
||||||
|
|
||||||
|
|||||||
311
README.md
311
README.md
@@ -2,13 +2,13 @@
|
|||||||
|
|
||||||
Open-source vector similarity search for Postgres
|
Open-source vector similarity search for Postgres
|
||||||
|
|
||||||
```sql
|
Supports
|
||||||
CREATE TABLE items (embedding vector(3));
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops);
|
|
||||||
SELECT * FROM items ORDER BY embedding <-> '[1,2,3]' LIMIT 5;
|
|
||||||
```
|
|
||||||
|
|
||||||
Supports L2 distance, inner product, and cosine distance
|
- exact and approximate nearest neighbor search
|
||||||
|
- L2 distance, inner product, and cosine distance
|
||||||
|
- any [language](#languages) with a Postgres client
|
||||||
|
|
||||||
|
Plus [ACID](https://en.wikipedia.org/wiki/ACID) compliance, point-in-time recovery, JOINs, and all of the other [great features](https://www.postgresql.org/about/) of Postgres
|
||||||
|
|
||||||
[](https://github.com/pgvector/pgvector/actions)
|
[](https://github.com/pgvector/pgvector/actions)
|
||||||
|
|
||||||
@@ -17,53 +17,101 @@ Supports L2 distance, inner product, and cosine distance
|
|||||||
Compile and install the extension (supports Postgres 11+)
|
Compile and install the extension (supports Postgres 11+)
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
|
cd /tmp
|
||||||
|
git clone --branch v0.4.4 https://github.com/pgvector/pgvector.git
|
||||||
cd pgvector
|
cd pgvector
|
||||||
make
|
make
|
||||||
make install # may need sudo
|
make install # may need sudo
|
||||||
```
|
```
|
||||||
|
|
||||||
Then load it in databases where you want to use it
|
See the [installation notes](#installation-notes) if you run into issues
|
||||||
|
|
||||||
```sql
|
You can also install it with [Docker](#docker), [Homebrew](#homebrew), [PGXN](#pgxn), [APT](#apt), [Yum](#yum), or [conda-forge](#conda-forge), and it comes preinstalled with [Postgres.app](#postgresapp) and many [hosted providers](#hosted-postgres)
|
||||||
CREATE EXTENSION vector;
|
|
||||||
```
|
|
||||||
|
|
||||||
You can also install it with [Docker](#docker), [Homebrew](#homebrew), [PGXN](#pgxn), or [conda-forge](#conda-forge)
|
|
||||||
|
|
||||||
## Getting Started
|
## Getting Started
|
||||||
|
|
||||||
|
Enable the extension (do this once in each database where you want to use it)
|
||||||
|
|
||||||
|
```tsql
|
||||||
|
CREATE EXTENSION vector;
|
||||||
|
```
|
||||||
|
|
||||||
Create a vector column with 3 dimensions
|
Create a vector column with 3 dimensions
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE items (embedding vector(3));
|
CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3));
|
||||||
```
|
```
|
||||||
|
|
||||||
Insert values
|
Insert vectors
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
INSERT INTO items VALUES ('[1,2,3]'), ('[4,5,6]');
|
INSERT INTO items (embedding) VALUES ('[1,2,3]'), ('[4,5,6]');
|
||||||
```
|
```
|
||||||
|
|
||||||
Get the nearest neighbor by L2 distance
|
Get the nearest neighbors by L2 distance
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 1;
|
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
```
|
```
|
||||||
|
|
||||||
Also supports inner product (`<#>`) and cosine distance (`<=>`)
|
Also supports inner product (`<#>`) and cosine distance (`<=>`)
|
||||||
|
|
||||||
Note: `<#>` returns the negative inner product since Postgres only supports `ASC` order index scans on operators
|
Note: `<#>` returns the negative inner product since Postgres only supports `ASC` order index scans on operators
|
||||||
|
|
||||||
## Querying
|
## Storing
|
||||||
|
|
||||||
Use a `SELECT` clause to get the distance
|
Create a new table with a vector column
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT embedding <-> '[3,1,2]' AS distance FROM items;
|
CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3));
|
||||||
```
|
```
|
||||||
|
|
||||||
Use a `WHERE` clause to get rows within a certain distance
|
Or add a vector column to an existing table
|
||||||
|
|
||||||
|
```sql
|
||||||
|
ALTER TABLE items ADD COLUMN embedding vector(3);
|
||||||
|
```
|
||||||
|
|
||||||
|
Insert vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
INSERT INTO items (embedding) VALUES ('[1,2,3]'), ('[4,5,6]');
|
||||||
|
```
|
||||||
|
|
||||||
|
Upsert vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
INSERT INTO items (id, embedding) VALUES (1, '[1,2,3]'), (2, '[4,5,6]')
|
||||||
|
ON CONFLICT (id) DO UPDATE SET embedding = EXCLUDED.embedding;
|
||||||
|
```
|
||||||
|
|
||||||
|
Update vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
UPDATE items SET embedding = '[1,2,3]' WHERE id = 1;
|
||||||
|
```
|
||||||
|
|
||||||
|
Delete vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
DELETE FROM items WHERE id = 1;
|
||||||
|
```
|
||||||
|
|
||||||
|
## Querying
|
||||||
|
|
||||||
|
Get the nearest neighbors to a vector
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
Get the nearest neighbors to a row
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT * FROM items WHERE id != 1 ORDER BY embedding <-> (SELECT embedding FROM items WHERE id = 1) LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
Get rows within a certain distance
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
||||||
@@ -71,55 +119,80 @@ SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
|||||||
|
|
||||||
Note: Combine with `ORDER BY` and `LIMIT` to use an index
|
Note: Combine with `ORDER BY` and `LIMIT` to use an index
|
||||||
|
|
||||||
Get the average of vectors
|
#### Distances
|
||||||
|
|
||||||
|
Get the distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT embedding <-> '[3,1,2]' AS distance FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
For inner product, multiply by -1 (since `<#>` returns the negative inner product)
|
||||||
|
|
||||||
|
```tsql
|
||||||
|
SELECT (embedding <#> '[3,1,2]') * -1 AS inner_product FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
For cosine similarity, use 1 - cosine distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT 1 - (embedding <=> '[3,1,2]') AS cosine_similarity FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Aggregates
|
||||||
|
|
||||||
|
Average vectors
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT AVG(embedding) FROM items;
|
SELECT AVG(embedding) FROM items;
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Average groups of vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT category_id, AVG(embedding) FROM items GROUP BY category_id;
|
||||||
|
```
|
||||||
|
|
||||||
## Indexing
|
## Indexing
|
||||||
|
|
||||||
Speed up queries with an approximate index. Add an index for each distance function you want to use.
|
By default, pgvector performs exact nearest neighbor search, which provides perfect recall.
|
||||||
|
|
||||||
|
You can add an index to use approximate nearest neighbor search, which trades some recall for performance. Unlike typical indexes, you will see different results for queries after adding an approximate index.
|
||||||
|
|
||||||
|
Three keys to achieving good recall are:
|
||||||
|
|
||||||
|
1. Create the index *after* the table has some data
|
||||||
|
2. Choose an appropriate number of lists - a good place to start is `rows / 1000` for up to 1M rows and `sqrt(rows)` for over 1M rows
|
||||||
|
3. When querying, specify an appropriate number of [probes](#query-options) (higher is better for recall, lower is better for speed) - a good place to start is `sqrt(lists)`
|
||||||
|
|
||||||
|
Add an index for each distance function you want to use.
|
||||||
|
|
||||||
L2 distance
|
L2 distance
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Inner product
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_ip_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Cosine distance
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_cosine_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Indexes should be created after the table has some data for optimal clustering. Also, unlike typical indexes which only affect performance, you may see different results for queries after adding an approximate index. Vectors with up to 2,000 dimensions can be indexed.
|
|
||||||
|
|
||||||
### Index Options
|
|
||||||
|
|
||||||
Specify the number of inverted lists (100 by default)
|
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100);
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100);
|
||||||
```
|
```
|
||||||
|
|
||||||
A lower value provides better recall at the cost of speed. A good place to start is:
|
Inner product
|
||||||
|
|
||||||
- `rows / 1000` for up to 1M rows
|
```sql
|
||||||
- `sqrt(rows)` for over 1M rows
|
CREATE INDEX ON items USING ivfflat (embedding vector_ip_ops) WITH (lists = 100);
|
||||||
|
```
|
||||||
|
|
||||||
|
Cosine distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
CREATE INDEX ON items USING ivfflat (embedding vector_cosine_ops) WITH (lists = 100);
|
||||||
|
```
|
||||||
|
|
||||||
|
Vectors with up to 2,000 dimensions can be indexed.
|
||||||
|
|
||||||
### Query Options
|
### Query Options
|
||||||
|
|
||||||
Specify the number of probes (1 by default)
|
Specify the number of probes (1 by default)
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SET ivfflat.probes = 1;
|
SET ivfflat.probes = 10;
|
||||||
```
|
```
|
||||||
|
|
||||||
A higher value provides better recall at the cost of speed, and it can be set to the number of lists for exact nearest neighbor search (at which point the planner won’t use the index)
|
A higher value provides better recall at the cost of speed, and it can be set to the number of lists for exact nearest neighbor search (at which point the planner won’t use the index)
|
||||||
@@ -128,7 +201,7 @@ Use `SET LOCAL` inside a transaction to set it for a single query
|
|||||||
|
|
||||||
```sql
|
```sql
|
||||||
BEGIN;
|
BEGIN;
|
||||||
SET LOCAL ivfflat.probes = 1;
|
SET LOCAL ivfflat.probes = 10;
|
||||||
SELECT ...
|
SELECT ...
|
||||||
COMMIT;
|
COMMIT;
|
||||||
```
|
```
|
||||||
@@ -150,46 +223,72 @@ The phases are:
|
|||||||
|
|
||||||
Note: `tuples_done` and `tuples_total` are only populated during the `loading tuples` phase
|
Note: `tuples_done` and `tuples_total` are only populated during the `loading tuples` phase
|
||||||
|
|
||||||
### Partial Indexes
|
### Filtering
|
||||||
|
|
||||||
Consider [partial indexes](https://www.postgresql.org/docs/current/indexes-partial.html) for queries with a `WHERE` clause
|
There are a few ways to index nearest neighbor queries with a `WHERE` clause
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT * FROM items WHERE category_id = 123 ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
SELECT * FROM items WHERE category_id = 123 ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
```
|
```
|
||||||
|
|
||||||
can be indexed with:
|
Create an index on one [or more](https://www.postgresql.org/docs/current/indexes-multicolumn.html) of the `WHERE` columns for exact search
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WHERE (category_id = 123);
|
CREATE INDEX ON items (category_id);
|
||||||
```
|
```
|
||||||
|
|
||||||
To index many different values of `category_id`, consider [partitioning](https://www.postgresql.org/docs/current/ddl-partitioning.html) on `category_id`.
|
Or a [partial index](https://www.postgresql.org/docs/current/indexes-partial.html) on the vector column for approximate search
|
||||||
|
|
||||||
|
```sql
|
||||||
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100)
|
||||||
|
WHERE (category_id = 123);
|
||||||
|
```
|
||||||
|
|
||||||
|
Use [partitioning](https://www.postgresql.org/docs/current/ddl-partitioning.html) for approximate search on many different values of the `WHERE` columns
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE items (embedding vector(3), category_id int) PARTITION BY LIST(category_id);
|
CREATE TABLE items (embedding vector(3), category_id int) PARTITION BY LIST(category_id);
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Hybrid Search
|
||||||
|
|
||||||
|
Use together with Postgres [full-text search](https://www.postgresql.org/docs/current/textsearch-intro.html) for hybrid search ([Python example](https://github.com/pgvector/pgvector-python/blob/master/examples/hybrid_search.py)).
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT id, content FROM items, to_tsquery('hello & search') query
|
||||||
|
WHERE textsearch @@ query ORDER BY ts_rank_cd(textsearch, query) DESC LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
## Performance
|
## Performance
|
||||||
|
|
||||||
|
Use `EXPLAIN ANALYZE` to debug performance.
|
||||||
|
|
||||||
|
```sql
|
||||||
|
EXPLAIN ANALYZE SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
### Exact Search
|
||||||
|
|
||||||
To speed up queries without an index, increase `max_parallel_workers_per_gather`.
|
To speed up queries without an index, increase `max_parallel_workers_per_gather`.
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SET max_parallel_workers_per_gather = 4;
|
SET max_parallel_workers_per_gather = 4;
|
||||||
```
|
```
|
||||||
|
|
||||||
|
If vectors are normalized to length 1 (like [OpenAI embeddings](https://platform.openai.com/docs/guides/embeddings/which-distance-function-should-i-use)), use inner product for best performance.
|
||||||
|
|
||||||
|
```tsql
|
||||||
|
SELECT * FROM items ORDER BY embedding <#> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
### Approximate Search
|
||||||
|
|
||||||
To speed up queries with an index, increase the number of inverted lists (at the expense of recall).
|
To speed up queries with an index, increase the number of inverted lists (at the expense of recall).
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 1000);
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 1000);
|
||||||
```
|
```
|
||||||
|
|
||||||
Use `EXPLAIN ANALYZE` to debug performance.
|
|
||||||
|
|
||||||
```sql
|
|
||||||
EXPLAIN ANALYZE SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 1;
|
|
||||||
```
|
|
||||||
|
|
||||||
## Languages
|
## Languages
|
||||||
|
|
||||||
Use pgvector from any language with a Postgres client. You can even generate and store vectors in one language and query them in another.
|
Use pgvector from any language with a Postgres client. You can even generate and store vectors in one language and query them in another.
|
||||||
@@ -198,8 +297,10 @@ Language | Libraries / Examples
|
|||||||
--- | ---
|
--- | ---
|
||||||
C++ | [pgvector-cpp](https://github.com/pgvector/pgvector-cpp)
|
C++ | [pgvector-cpp](https://github.com/pgvector/pgvector-cpp)
|
||||||
C# | [pgvector-dotnet](https://github.com/pgvector/pgvector-dotnet)
|
C# | [pgvector-dotnet](https://github.com/pgvector/pgvector-dotnet)
|
||||||
|
Crystal | [pgvector-crystal](https://github.com/pgvector/pgvector-crystal)
|
||||||
Elixir | [pgvector-elixir](https://github.com/pgvector/pgvector-elixir)
|
Elixir | [pgvector-elixir](https://github.com/pgvector/pgvector-elixir)
|
||||||
Go | [pgvector-go](https://github.com/pgvector/pgvector-go)
|
Go | [pgvector-go](https://github.com/pgvector/pgvector-go)
|
||||||
|
Haskell | [pgvector-haskell](https://github.com/pgvector/pgvector-haskell)
|
||||||
Java, Scala | [pgvector-java](https://github.com/pgvector/pgvector-java)
|
Java, Scala | [pgvector-java](https://github.com/pgvector/pgvector-java)
|
||||||
Julia | [pgvector-julia](https://github.com/pgvector/pgvector-julia)
|
Julia | [pgvector-julia](https://github.com/pgvector/pgvector-julia)
|
||||||
Lua | [pgvector-lua](https://github.com/pgvector/pgvector-lua)
|
Lua | [pgvector-lua](https://github.com/pgvector/pgvector-lua)
|
||||||
@@ -210,6 +311,7 @@ Python | [pgvector-python](https://github.com/pgvector/pgvector-python)
|
|||||||
R | [pgvector-r](https://github.com/pgvector/pgvector-r)
|
R | [pgvector-r](https://github.com/pgvector/pgvector-r)
|
||||||
Ruby | [pgvector-ruby](https://github.com/pgvector/pgvector-ruby), [Neighbor](https://github.com/ankane/neighbor)
|
Ruby | [pgvector-ruby](https://github.com/pgvector/pgvector-ruby), [Neighbor](https://github.com/ankane/neighbor)
|
||||||
Rust | [pgvector-rust](https://github.com/pgvector/pgvector-rust)
|
Rust | [pgvector-rust](https://github.com/pgvector/pgvector-rust)
|
||||||
|
Swift | [pgvector-swift](https://github.com/pgvector/pgvector-swift)
|
||||||
|
|
||||||
## Frequently Asked Questions
|
## Frequently Asked Questions
|
||||||
|
|
||||||
@@ -223,10 +325,11 @@ Yes, pgvector uses the write-ahead log (WAL), which allows for replication and p
|
|||||||
|
|
||||||
#### What if I want to index vectors with more than 2,000 dimensions?
|
#### What if I want to index vectors with more than 2,000 dimensions?
|
||||||
|
|
||||||
Two things you can try are:
|
You’ll need to use [dimensionality reduction](https://en.wikipedia.org/wiki/Dimensionality_reduction) at the moment.
|
||||||
|
|
||||||
1. use dimensionality reduction
|
#### Why am I seeing less results after adding an index?
|
||||||
2. compile Postgres with a larger block size (`./configure --with-blocksize=32`) and edit the limit in `src/ivfflat.h`
|
|
||||||
|
The index was likely created with too little data for the number of lists. Drop the index until the table has more data.
|
||||||
|
|
||||||
## Reference
|
## Reference
|
||||||
|
|
||||||
@@ -260,6 +363,46 @@ Function | Description
|
|||||||
--- | ---
|
--- | ---
|
||||||
avg(vector) → vector | arithmetic mean
|
avg(vector) → vector | arithmetic mean
|
||||||
|
|
||||||
|
## Installation Notes
|
||||||
|
|
||||||
|
### Postgres Location
|
||||||
|
|
||||||
|
If your machine has multiple Postgres installations, specify the path to [pg_config](https://www.postgresql.org/docs/current/app-pgconfig.html) with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
export PG_CONFIG=/Applications/Postgres.app/Contents/Versions/latest/bin/pg_config
|
||||||
|
```
|
||||||
|
|
||||||
|
Then re-run the installation instructions (run `make clean` before `make` if needed). If `sudo` is needed for `make install`, use:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo --preserve-env=PG_CONFIG make install
|
||||||
|
```
|
||||||
|
|
||||||
|
### Missing Header
|
||||||
|
|
||||||
|
If compilation fails with `fatal error: postgres.h: No such file or directory`, make sure Postgres development files are installed on the server.
|
||||||
|
|
||||||
|
For Ubuntu and Debian, use:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo apt install postgresql-server-dev-15
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Replace `15` with your Postgres server version
|
||||||
|
|
||||||
|
### Windows
|
||||||
|
|
||||||
|
Support for Windows is currently experimental. Use `nmake` to build:
|
||||||
|
|
||||||
|
```cmd
|
||||||
|
set "PGROOT=C:\Program Files\PostgreSQL\15"
|
||||||
|
git clone --branch v0.4.4 https://github.com/pgvector/pgvector.git
|
||||||
|
cd pgvector
|
||||||
|
nmake /F Makefile.win
|
||||||
|
nmake /F Makefile.win install
|
||||||
|
```
|
||||||
|
|
||||||
## Additional Installation Methods
|
## Additional Installation Methods
|
||||||
|
|
||||||
### Docker
|
### Docker
|
||||||
@@ -275,9 +418,9 @@ This adds pgvector to the [Postgres image](https://hub.docker.com/_/postgres) (r
|
|||||||
You can also build the image manually:
|
You can also build the image manually:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
|
git clone --branch v0.4.4 https://github.com/pgvector/pgvector.git
|
||||||
cd pgvector
|
cd pgvector
|
||||||
docker build -t pgvector .
|
docker build --build-arg PG_MAJOR=15 -t myuser/pgvector .
|
||||||
```
|
```
|
||||||
|
|
||||||
### Homebrew
|
### Homebrew
|
||||||
@@ -288,6 +431,8 @@ With Homebrew Postgres, you can use:
|
|||||||
brew install pgvector
|
brew install pgvector
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Note: This only adds it to the `postgresql@14` formula
|
||||||
|
|
||||||
### PGXN
|
### PGXN
|
||||||
|
|
||||||
Install from the [PostgreSQL Extension Network](https://pgxn.org/dist/vector) with:
|
Install from the [PostgreSQL Extension Network](https://pgxn.org/dist/vector) with:
|
||||||
@@ -296,6 +441,28 @@ Install from the [PostgreSQL Extension Network](https://pgxn.org/dist/vector) wi
|
|||||||
pgxn install vector
|
pgxn install vector
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### APT
|
||||||
|
|
||||||
|
Debian and Ubuntu packages are available from the [PostgreSQL APT Repository](https://wiki.postgresql.org/wiki/Apt). Follow the [setup instructions](https://wiki.postgresql.org/wiki/Apt#Quickstart) and run:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo apt install postgresql-15-pgvector
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Replace `15` with your Postgres server version
|
||||||
|
|
||||||
|
### Yum
|
||||||
|
|
||||||
|
RPM packages are available from the [PostgreSQL Yum Repository](https://yum.postgresql.org/). Follow the [setup instructions](https://www.postgresql.org/download/linux/redhat/) for your distribution and run:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo yum install pgvector_15
|
||||||
|
# or
|
||||||
|
sudo dnf install pgvector_15
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Replace `15` with your Postgres server version
|
||||||
|
|
||||||
### conda-forge
|
### conda-forge
|
||||||
|
|
||||||
With Conda Postgres, install from [conda-forge](https://anaconda.org/conda-forge/pgvector) with:
|
With Conda Postgres, install from [conda-forge](https://anaconda.org/conda-forge/pgvector) with:
|
||||||
@@ -306,17 +473,19 @@ conda install -c conda-forge pgvector
|
|||||||
|
|
||||||
This method is [community-maintained](https://github.com/conda-forge/pgvector-feedstock) by [@mmcauliffe](https://github.com/mmcauliffe)
|
This method is [community-maintained](https://github.com/conda-forge/pgvector-feedstock) by [@mmcauliffe](https://github.com/mmcauliffe)
|
||||||
|
|
||||||
|
### Postgres.app
|
||||||
|
|
||||||
|
Download the [latest release](https://postgresapp.com/downloads.html) with Postgres 15+.
|
||||||
|
|
||||||
## Hosted Postgres
|
## Hosted Postgres
|
||||||
|
|
||||||
pgvector is available on [these providers](https://github.com/pgvector/pgvector/issues/54).
|
pgvector is available on [these providers](https://github.com/pgvector/pgvector/issues/54).
|
||||||
|
|
||||||
To request a new extension on other providers:
|
To request a new extension on other providers:
|
||||||
|
|
||||||
- Amazon RDS - follow the instructions on [this page](https://aws.amazon.com/rds/postgresql/faqs/)
|
|
||||||
- Google Cloud SQL - vote or comment on [this page](https://issuetracker.google.com/issues/265172065)
|
- Google Cloud SQL - vote or comment on [this page](https://issuetracker.google.com/issues/265172065)
|
||||||
- Azure Database - vote or comment on [this page](https://feedback.azure.com/d365community/idea/7b423322-6189-ed11-a81b-000d3ae49307)
|
- DigitalOcean Managed Databases - vote or comment on [this page](https://ideas.digitalocean.com/managed-database/p/pgvector-extension-for-postgresql)
|
||||||
- DigitalOcean Managed Databases - vote or comment on [this page](https://ideas.digitalocean.com/app-framework-services/p/pgvector-extension-for-postgresql)
|
- Heroku Postgres - vote or comment on [this page](https://github.com/heroku/roadmap/issues/156)
|
||||||
- Render - vote or comment on [this page](https://feedback.render.com/features/p/add-pgvector-extension-to-postgresql)
|
|
||||||
|
|
||||||
## Upgrading
|
## Upgrading
|
||||||
|
|
||||||
|
|||||||
2
sql/vector--0.4.1--0.4.2.sql
Normal file
2
sql/vector--0.4.1--0.4.2.sql
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
||||||
|
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.2'" to load this file. \quit
|
||||||
2
sql/vector--0.4.2--0.4.3.sql
Normal file
2
sql/vector--0.4.2--0.4.3.sql
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
||||||
|
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.3'" to load this file. \quit
|
||||||
2
sql/vector--0.4.3--0.4.4.sql
Normal file
2
sql/vector--0.4.3--0.4.4.sql
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
|
||||||
|
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.4'" to load this file. \quit
|
||||||
@@ -147,7 +147,7 @@ AddTupleToSort(Relation index, ItemPointer tid, Datum *values, IvfflatBuildState
|
|||||||
{
|
{
|
||||||
double distance;
|
double distance;
|
||||||
double minDistance = DBL_MAX;
|
double minDistance = DBL_MAX;
|
||||||
int closestCenter = -1;
|
int closestCenter = 0;
|
||||||
VectorArray centers = buildstate->centers;
|
VectorArray centers = buildstate->centers;
|
||||||
TupleTableSlot *slot = buildstate->slot;
|
TupleTableSlot *slot = buildstate->slot;
|
||||||
int i;
|
int i;
|
||||||
@@ -431,8 +431,18 @@ ComputeCenters(IvfflatBuildState * buildstate)
|
|||||||
/* TODO Ensure within maintenance_work_mem */
|
/* TODO Ensure within maintenance_work_mem */
|
||||||
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
||||||
if (buildstate->heap != NULL)
|
if (buildstate->heap != NULL)
|
||||||
|
{
|
||||||
SampleRows(buildstate);
|
SampleRows(buildstate);
|
||||||
|
|
||||||
|
if (buildstate->samples->length < buildstate->lists)
|
||||||
|
{
|
||||||
|
ereport(NOTICE,
|
||||||
|
(errmsg("ivfflat index created with little data"),
|
||||||
|
errdetail("This will cause low recall."),
|
||||||
|
errhint("Drop the index until the table has more data.")));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* Calculate centers */
|
/* Calculate centers */
|
||||||
IvfflatBench("k-means", IvfflatKmeans(buildstate->index, buildstate->samples, buildstate->centers));
|
IvfflatBench("k-means", IvfflatKmeans(buildstate->index, buildstate->samples, buildstate->centers));
|
||||||
|
|
||||||
@@ -558,6 +568,21 @@ PrintKmeansMetrics(IvfflatBuildState * buildstate)
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Scan table for tuples to index
|
||||||
|
*/
|
||||||
|
static void
|
||||||
|
ScanTable(IvfflatBuildState * buildstate)
|
||||||
|
{
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, true, BuildCallback, (void *) buildstate, NULL);
|
||||||
|
#else
|
||||||
|
buildstate->reltuples = IndexBuildHeapScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, BuildCallback, (void *) buildstate, NULL);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Create entry pages
|
* Create entry pages
|
||||||
*/
|
*/
|
||||||
@@ -565,7 +590,7 @@ static void
|
|||||||
CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
||||||
{
|
{
|
||||||
AttrNumber attNums[] = {1};
|
AttrNumber attNums[] = {1};
|
||||||
Oid sortOperators[] = {Float8LessOperator};
|
Oid sortOperators[] = {Int4LessOperator};
|
||||||
Oid sortCollations[] = {InvalidOid};
|
Oid sortCollations[] = {InvalidOid};
|
||||||
bool nullsFirstFlags[] = {false};
|
bool nullsFirstFlags[] = {false};
|
||||||
|
|
||||||
@@ -575,25 +600,17 @@ CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
|||||||
|
|
||||||
/* Add tuples to sort */
|
/* Add tuples to sort */
|
||||||
if (buildstate->heap != NULL)
|
if (buildstate->heap != NULL)
|
||||||
{
|
IvfflatBench("assign tuples", ScanTable(buildstate));
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
true, true, BuildCallback, (void *) buildstate, NULL);
|
|
||||||
#else
|
|
||||||
buildstate->reltuples = IndexBuildHeapScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
true, BuildCallback, (void *) buildstate, NULL);
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Sort */
|
/* Sort */
|
||||||
tuplesort_performsort(buildstate->sortstate);
|
IvfflatBench("sort tuples", tuplesort_performsort(buildstate->sortstate));
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
PrintKmeansMetrics(buildstate);
|
PrintKmeansMetrics(buildstate);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* Insert */
|
/* Insert */
|
||||||
InsertTuples(buildstate->index, buildstate, forkNum);
|
IvfflatBench("load tuples", InsertTuples(buildstate->index, buildstate, forkNum));
|
||||||
tuplesort_end(buildstate->sortstate);
|
tuplesort_end(buildstate->sortstate);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -611,7 +628,7 @@ BuildIndex(Relation heap, Relation index, IndexInfo *indexInfo,
|
|||||||
/* Create pages */
|
/* Create pages */
|
||||||
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
||||||
CreateListPages(index, buildstate->centers, buildstate->dimensions, buildstate->lists, forkNum, &buildstate->listInfo);
|
CreateListPages(index, buildstate->centers, buildstate->dimensions, buildstate->lists, forkNum, &buildstate->listInfo);
|
||||||
IvfflatBench("CreateEntryPages", CreateEntryPages(buildstate, forkNum));
|
CreateEntryPages(buildstate, forkNum);
|
||||||
|
|
||||||
FreeBuildState(buildstate);
|
FreeBuildState(buildstate);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "utils/guc.h"
|
#include "utils/guc.h"
|
||||||
#include "utils/selfuncs.h"
|
#include "utils/selfuncs.h"
|
||||||
|
#include "utils/spccache.h"
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
#include "commands/progress.h"
|
#include "commands/progress.h"
|
||||||
@@ -63,13 +64,13 @@ ivfflatbuildphasename(int64 phasenum)
|
|||||||
static void
|
static void
|
||||||
ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
||||||
Cost *indexStartupCost, Cost *indexTotalCost,
|
Cost *indexStartupCost, Cost *indexTotalCost,
|
||||||
Selectivity *indexSelectivity, double *indexCorrelation
|
Selectivity *indexSelectivity, double *indexCorrelation,
|
||||||
,double *indexPages
|
double *indexPages)
|
||||||
)
|
|
||||||
{
|
{
|
||||||
GenericCosts costs;
|
GenericCosts costs;
|
||||||
int lists;
|
int lists;
|
||||||
double ratio;
|
double ratio;
|
||||||
|
double spc_seq_page_cost;
|
||||||
Relation indexRel;
|
Relation indexRel;
|
||||||
#if PG_VERSION_NUM < 120000
|
#if PG_VERSION_NUM < 120000
|
||||||
List *qinfos;
|
List *qinfos;
|
||||||
@@ -88,6 +89,22 @@ ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
|||||||
|
|
||||||
MemSet(&costs, 0, sizeof(costs));
|
MemSet(&costs, 0, sizeof(costs));
|
||||||
|
|
||||||
|
indexRel = index_open(path->indexinfo->indexoid, NoLock);
|
||||||
|
lists = IvfflatGetLists(indexRel);
|
||||||
|
index_close(indexRel, NoLock);
|
||||||
|
|
||||||
|
/* Get the ratio of lists that we need to visit */
|
||||||
|
ratio = ((double) ivfflat_probes) / lists;
|
||||||
|
if (ratio > 1.0)
|
||||||
|
ratio = 1.0;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* This gives us the subset of tuples to visit. This value is passed into
|
||||||
|
* the generic cost estimator to determine the number of pages to visit
|
||||||
|
* during the index scan.
|
||||||
|
*/
|
||||||
|
costs.numIndexTuples = path->indexinfo->tuples * ratio;
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
genericcostestimate(root, path, loop_count, &costs);
|
genericcostestimate(root, path, loop_count, &costs);
|
||||||
#else
|
#else
|
||||||
@@ -95,17 +112,31 @@ ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
|||||||
genericcostestimate(root, path, loop_count, qinfos, &costs);
|
genericcostestimate(root, path, loop_count, qinfos, &costs);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
indexRel = index_open(path->indexinfo->indexoid, NoLock);
|
get_tablespace_page_costs(path->indexinfo->reltablespace, NULL, &spc_seq_page_cost);
|
||||||
lists = IvfflatGetLists(indexRel);
|
|
||||||
index_close(indexRel, NoLock);
|
|
||||||
|
|
||||||
ratio = ((double) ivfflat_probes) / lists;
|
/* Adjust cost if needed since TOAST not included in seq scan cost */
|
||||||
if (ratio > 1)
|
if (costs.numIndexPages > path->indexinfo->rel->pages && ratio < 0.5)
|
||||||
ratio = 1;
|
{
|
||||||
|
/* Change all page cost from random to sequential */
|
||||||
|
costs.indexTotalCost -= costs.numIndexPages * (costs.spc_random_page_cost - spc_seq_page_cost);
|
||||||
|
|
||||||
costs.indexTotalCost *= ratio;
|
/* Remove cost of extra pages */
|
||||||
|
costs.indexTotalCost -= (costs.numIndexPages - path->indexinfo->rel->pages) * spc_seq_page_cost;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/* Change some page cost from random to sequential */
|
||||||
|
costs.indexTotalCost -= 0.5 * costs.numIndexPages * (costs.spc_random_page_cost - spc_seq_page_cost);
|
||||||
|
}
|
||||||
|
|
||||||
/* Startup cost and total cost are same */
|
/*
|
||||||
|
* If the list selectivity is lower than what is returned from the generic
|
||||||
|
* cost estimator, use that.
|
||||||
|
*/
|
||||||
|
if (ratio < costs.indexSelectivity)
|
||||||
|
costs.indexSelectivity = ratio;
|
||||||
|
|
||||||
|
/* Use total cost since most work happens before first tuple is returned */
|
||||||
*indexStartupCost = costs.indexTotalCost;
|
*indexStartupCost = costs.indexTotalCost;
|
||||||
*indexTotalCost = costs.indexTotalCost;
|
*indexTotalCost = costs.indexTotalCost;
|
||||||
*indexSelectivity = costs.indexSelectivity;
|
*indexSelectivity = costs.indexSelectivity;
|
||||||
|
|||||||
@@ -23,6 +23,10 @@ FindInsertPage(Relation rel, Datum *values, BlockNumber *insertPage, ListInfo *
|
|||||||
OffsetNumber offno;
|
OffsetNumber offno;
|
||||||
OffsetNumber maxoffno;
|
OffsetNumber maxoffno;
|
||||||
|
|
||||||
|
/* Avoid compiler warning */
|
||||||
|
listInfo->blkno = nextblkno;
|
||||||
|
listInfo->offno = FirstOffsetNumber;
|
||||||
|
|
||||||
procinfo = index_getprocinfo(rel, 1, IVFFLAT_DISTANCE_PROC);
|
procinfo = index_getprocinfo(rel, 1, IVFFLAT_DISTANCE_PROC);
|
||||||
collation = rel->rd_indcollation[0];
|
collation = rel->rd_indcollation[0];
|
||||||
|
|
||||||
@@ -39,7 +43,7 @@ FindInsertPage(Relation rel, Datum *values, BlockNumber *insertPage, ListInfo *
|
|||||||
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
list = (IvfflatList) PageGetItem(cpage, PageGetItemId(cpage, offno));
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, values[0], PointerGetDatum(&list->center)));
|
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, values[0], PointerGetDatum(&list->center)));
|
||||||
|
|
||||||
if (distance < minDistance)
|
if (distance < minDistance || !BlockNumberIsValid(*insertPage))
|
||||||
{
|
{
|
||||||
*insertPage = list->insertPage;
|
*insertPage = list->insertPage;
|
||||||
listInfo->blkno = nextblkno;
|
listInfo->blkno = nextblkno;
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
#include <math.h>
|
||||||
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
@@ -211,7 +212,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
/* Check memory requirements */
|
/* Check memory requirements */
|
||||||
/* Add one to error message to ceil */
|
/* Add one to error message to ceil */
|
||||||
if (totalSize / 1024 > maintenance_work_mem)
|
if (totalSize > (Size) maintenance_work_mem * 1024L)
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
||||||
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
||||||
@@ -251,7 +252,7 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
minDistance = DBL_MAX;
|
minDistance = DBL_MAX;
|
||||||
closestCenter = -1;
|
closestCenter = 0;
|
||||||
|
|
||||||
/* Find closest center */
|
/* Find closest center */
|
||||||
for (k = 0; k < numCenters; k++)
|
for (k = 0; k < numCenters; k++)
|
||||||
@@ -398,6 +399,14 @@ ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
|
|
||||||
if (centerCounts[j] > 0)
|
if (centerCounts[j] > 0)
|
||||||
{
|
{
|
||||||
|
/* Double avoids overflow, but requires more memory */
|
||||||
|
/* TODO Update bounds */
|
||||||
|
for (k = 0; k < dimensions; k++)
|
||||||
|
{
|
||||||
|
if (isinf(vec->x[k]))
|
||||||
|
vec->x[k] = vec->x[k] > 0 ? FLT_MAX : -FLT_MAX;
|
||||||
|
}
|
||||||
|
|
||||||
for (k = 0; k < dimensions; k++)
|
for (k = 0; k < dimensions; k++)
|
||||||
vec->x[k] /= centerCounts[j];
|
vec->x[k] /= centerCounts[j];
|
||||||
}
|
}
|
||||||
@@ -461,12 +470,29 @@ CheckCenters(Relation index, VectorArray centers)
|
|||||||
{
|
{
|
||||||
FmgrInfo *normprocinfo;
|
FmgrInfo *normprocinfo;
|
||||||
Oid collation;
|
Oid collation;
|
||||||
|
Vector *vec;
|
||||||
int i;
|
int i;
|
||||||
|
int j;
|
||||||
double norm;
|
double norm;
|
||||||
|
|
||||||
if (centers->length != centers->maxlen)
|
if (centers->length != centers->maxlen)
|
||||||
elog(ERROR, "Not enough centers. Please report a bug.");
|
elog(ERROR, "Not enough centers. Please report a bug.");
|
||||||
|
|
||||||
|
/* Ensure no NaN or infinite values */
|
||||||
|
for (i = 0; i < centers->length; i++)
|
||||||
|
{
|
||||||
|
vec = VectorArrayGet(centers, i);
|
||||||
|
|
||||||
|
for (j = 0; j < vec->dim; j++)
|
||||||
|
{
|
||||||
|
if (isnan(vec->x[j]))
|
||||||
|
elog(ERROR, "NaN detected. Please report a bug.");
|
||||||
|
|
||||||
|
if (isinf(vec->x[j]))
|
||||||
|
elog(ERROR, "Infinite value detected. Please report a bug.");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* Ensure no duplicate centers */
|
/* Ensure no duplicate centers */
|
||||||
/* Fine to sort in-place */
|
/* Fine to sort in-place */
|
||||||
qsort(centers->items, centers->length, VECTOR_SIZE(centers->dim), CompareVectors);
|
qsort(centers->items, centers->length, VECTOR_SIZE(centers->dim), CompareVectors);
|
||||||
|
|||||||
@@ -111,6 +111,7 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
Datum datum;
|
Datum datum;
|
||||||
bool isnull;
|
bool isnull;
|
||||||
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
||||||
|
double tuples = 0;
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
||||||
@@ -159,6 +160,8 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
ExecStoreVirtualTuple(slot);
|
ExecStoreVirtualTuple(slot);
|
||||||
|
|
||||||
tuplesort_puttupleslot(so->sortstate, slot);
|
tuplesort_puttupleslot(so->sortstate, slot);
|
||||||
|
|
||||||
|
tuples++;
|
||||||
}
|
}
|
||||||
|
|
||||||
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
||||||
@@ -167,6 +170,15 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
FreeAccessStrategy(bas);
|
||||||
|
|
||||||
|
/* TODO Scan more lists */
|
||||||
|
if (tuples < 100)
|
||||||
|
ereport(DEBUG1,
|
||||||
|
(errmsg("index scan found few tuples"),
|
||||||
|
errdetail("Index may have been created with little data."),
|
||||||
|
errhint("Recreate the index and possibly decrease lists.")));
|
||||||
|
|
||||||
tuplesort_performsort(so->sortstate);
|
tuplesort_performsort(so->sortstate);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -132,6 +132,8 @@ ivfflatbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
FreeAccessStrategy(bas);
|
||||||
|
|
||||||
return stats;
|
return stats;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
176
src/vector.c
176
src/vector.c
@@ -42,7 +42,7 @@ CheckDims(Vector * a, Vector * b)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Ensure expected dimension
|
* Ensure expected dimensions
|
||||||
*/
|
*/
|
||||||
static inline void
|
static inline void
|
||||||
CheckExpectedDim(int32 typmod, int dim)
|
CheckExpectedDim(int32 typmod, int dim)
|
||||||
@@ -53,7 +53,9 @@ CheckExpectedDim(int32 typmod, int dim)
|
|||||||
errmsg("expected %d dimensions, not %d", typmod, dim)));
|
errmsg("expected %d dimensions, not %d", typmod, dim)));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Ensure valid dimensions
|
||||||
|
*/
|
||||||
static inline void
|
static inline void
|
||||||
CheckDim(int dim)
|
CheckDim(int dim)
|
||||||
{
|
{
|
||||||
@@ -79,13 +81,28 @@ CheckElement(float value)
|
|||||||
(errcode(ERRCODE_DATA_EXCEPTION),
|
(errcode(ERRCODE_DATA_EXCEPTION),
|
||||||
errmsg("NaN not allowed in vector")));
|
errmsg("NaN not allowed in vector")));
|
||||||
|
|
||||||
|
|
||||||
if (isinf(value))
|
if (isinf(value))
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_DATA_EXCEPTION),
|
(errcode(ERRCODE_DATA_EXCEPTION),
|
||||||
errmsg("infinite value not allowed in vector")));
|
errmsg("infinite value not allowed in vector")));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Check for whitespace, since array_isspace() is static
|
||||||
|
*/
|
||||||
|
static inline bool
|
||||||
|
vector_isspace(char ch)
|
||||||
|
{
|
||||||
|
if (ch == ' ' ||
|
||||||
|
ch == '\t' ||
|
||||||
|
ch == '\n' ||
|
||||||
|
ch == '\r' ||
|
||||||
|
ch == '\v' ||
|
||||||
|
ch == '\f')
|
||||||
|
return true;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Check state array
|
* Check state array
|
||||||
*/
|
*/
|
||||||
@@ -100,7 +117,7 @@ CheckStateArray(ArrayType *statearray, const char *caller)
|
|||||||
return (float8 *) ARR_DATA_PTR(statearray);
|
return (float8 *) ARR_DATA_PTR(statearray);
|
||||||
}
|
}
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 120000
|
#if PG_VERSION_NUM < 120003
|
||||||
static pg_noinline void
|
static pg_noinline void
|
||||||
float_overflow_error(void)
|
float_overflow_error(void)
|
||||||
{
|
{
|
||||||
@@ -110,30 +127,6 @@ float_overflow_error(void)
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/*
|
|
||||||
* Print vector - useful for debugging
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
PrintVector(char *msg, Vector * vector)
|
|
||||||
{
|
|
||||||
StringInfoData buf;
|
|
||||||
int dim = vector->dim;
|
|
||||||
int i;
|
|
||||||
|
|
||||||
initStringInfo(&buf);
|
|
||||||
|
|
||||||
appendStringInfoChar(&buf, '[');
|
|
||||||
for (i = 0; i < dim; i++)
|
|
||||||
{
|
|
||||||
if (i > 0)
|
|
||||||
appendStringInfoString(&buf, ",");
|
|
||||||
appendStringInfoString(&buf, float8out_internal(vector->x[i]));
|
|
||||||
}
|
|
||||||
appendStringInfoChar(&buf, ']');
|
|
||||||
|
|
||||||
elog(INFO, "%s = %s", msg, buf.data);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Convert textual representation to internal representation
|
* Convert textual representation to internal representation
|
||||||
*/
|
*/
|
||||||
@@ -149,11 +142,15 @@ vector_in(PG_FUNCTION_ARGS)
|
|||||||
char *pt;
|
char *pt;
|
||||||
char *stringEnd;
|
char *stringEnd;
|
||||||
Vector *result;
|
Vector *result;
|
||||||
|
char *lit = pstrdup(str);
|
||||||
|
|
||||||
|
while (vector_isspace(*str))
|
||||||
|
str++;
|
||||||
|
|
||||||
if (*str != '[')
|
if (*str != '[')
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
errmsg("malformed vector literal: \"%s\"", str),
|
errmsg("malformed vector literal: \"%s\"", lit),
|
||||||
errdetail("Vector contents must start with \"[\".")));
|
errdetail("Vector contents must start with \"[\".")));
|
||||||
|
|
||||||
str++;
|
str++;
|
||||||
@@ -167,6 +164,15 @@ vector_in(PG_FUNCTION_ARGS)
|
|||||||
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
||||||
errmsg("vector cannot have more than %d dimensions", VECTOR_MAX_DIM)));
|
errmsg("vector cannot have more than %d dimensions", VECTOR_MAX_DIM)));
|
||||||
|
|
||||||
|
while (vector_isspace(*pt))
|
||||||
|
pt++;
|
||||||
|
|
||||||
|
/* Check for empty string like float4in */
|
||||||
|
if (*pt == '\0')
|
||||||
|
ereport(ERROR,
|
||||||
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
|
errmsg("invalid input syntax for type vector: \"%s\"", lit)));
|
||||||
|
|
||||||
/* Use strtof like float4in to avoid a double-rounding problem */
|
/* Use strtof like float4in to avoid a double-rounding problem */
|
||||||
x[dim] = strtof(pt, &stringEnd);
|
x[dim] = strtof(pt, &stringEnd);
|
||||||
CheckElement(x[dim]);
|
CheckElement(x[dim]);
|
||||||
@@ -175,33 +181,53 @@ vector_in(PG_FUNCTION_ARGS)
|
|||||||
if (stringEnd == pt)
|
if (stringEnd == pt)
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
errmsg("invalid input syntax for type vector: \"%s\"", pt)));
|
errmsg("invalid input syntax for type vector: \"%s\"", lit)));
|
||||||
|
|
||||||
|
while (vector_isspace(*stringEnd))
|
||||||
|
stringEnd++;
|
||||||
|
|
||||||
if (*stringEnd != '\0' && *stringEnd != ']')
|
if (*stringEnd != '\0' && *stringEnd != ']')
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
errmsg("invalid input syntax for type vector: \"%s\"", pt)));
|
errmsg("invalid input syntax for type vector: \"%s\"", lit)));
|
||||||
|
|
||||||
pt = strtok(NULL, ",");
|
pt = strtok(NULL, ",");
|
||||||
}
|
}
|
||||||
|
|
||||||
if (*stringEnd != ']')
|
if (stringEnd == NULL || *stringEnd != ']')
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
errmsg("malformed vector literal"),
|
errmsg("malformed vector literal: \"%s\"", lit),
|
||||||
errdetail("Unexpected end of input.")));
|
errdetail("Unexpected end of input.")));
|
||||||
|
|
||||||
if (stringEnd[1] != '\0')
|
stringEnd++;
|
||||||
|
|
||||||
|
/* Only whitespace is allowed after the closing brace */
|
||||||
|
while (vector_isspace(*stringEnd))
|
||||||
|
stringEnd++;
|
||||||
|
|
||||||
|
if (*stringEnd != '\0')
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
errmsg("malformed vector literal"),
|
errmsg("malformed vector literal: \"%s\"", lit),
|
||||||
errdetail("Junk after closing right brace.")));
|
errdetail("Junk after closing right brace.")));
|
||||||
|
|
||||||
|
/* Ensure no consecutive delimiters since strtok skips */
|
||||||
|
for (pt = lit + 1; *pt != '\0'; pt++)
|
||||||
|
{
|
||||||
|
if (pt[-1] == ',' && *pt == ',')
|
||||||
|
ereport(ERROR,
|
||||||
|
(errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
|
||||||
|
errmsg("malformed vector literal: \"%s\"", lit)));
|
||||||
|
}
|
||||||
|
|
||||||
if (dim < 1)
|
if (dim < 1)
|
||||||
ereport(ERROR,
|
ereport(ERROR,
|
||||||
(errcode(ERRCODE_DATA_EXCEPTION),
|
(errcode(ERRCODE_DATA_EXCEPTION),
|
||||||
errmsg("vector must have at least 1 dimension")));
|
errmsg("vector must have at least 1 dimension")));
|
||||||
|
|
||||||
|
pfree(lit);
|
||||||
|
|
||||||
CheckExpectedDim(typmod, dim);
|
CheckExpectedDim(typmod, dim);
|
||||||
|
|
||||||
result = InitVector(dim);
|
result = InitVector(dim);
|
||||||
@@ -272,6 +298,18 @@ vector_out(PG_FUNCTION_ARGS)
|
|||||||
PG_RETURN_CSTRING(buf);
|
PG_RETURN_CSTRING(buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Print vector - useful for debugging
|
||||||
|
*/
|
||||||
|
void
|
||||||
|
PrintVector(char *msg, Vector * vector)
|
||||||
|
{
|
||||||
|
char *out = DatumGetPointer(DirectFunctionCall1(vector_out, PointerGetDatum(vector)));
|
||||||
|
|
||||||
|
elog(INFO, "%s = %s", msg, out);
|
||||||
|
pfree(out);
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Convert type modifier
|
* Convert type modifier
|
||||||
*/
|
*/
|
||||||
@@ -330,7 +368,10 @@ vector_recv(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
result = InitVector(dim);
|
result = InitVector(dim);
|
||||||
for (i = 0; i < dim; i++)
|
for (i = 0; i < dim; i++)
|
||||||
|
{
|
||||||
result->x[i] = pq_getmsgfloat4(buf);
|
result->x[i] = pq_getmsgfloat4(buf);
|
||||||
|
CheckElement(result->x[i]);
|
||||||
|
}
|
||||||
|
|
||||||
PG_RETURN_POINTER(result);
|
PG_RETURN_POINTER(result);
|
||||||
}
|
}
|
||||||
@@ -396,10 +437,8 @@ array_to_vector(PG_FUNCTION_ARGS)
|
|||||||
get_typlenbyvalalign(ARR_ELEMTYPE(array), &typlen, &typbyval, &typalign);
|
get_typlenbyvalalign(ARR_ELEMTYPE(array), &typlen, &typbyval, &typalign);
|
||||||
deconstruct_array(array, ARR_ELEMTYPE(array), typlen, typbyval, typalign, &elemsp, &nullsp, &nelemsp);
|
deconstruct_array(array, ARR_ELEMTYPE(array), typlen, typbyval, typalign, &elemsp, &nullsp, &nelemsp);
|
||||||
|
|
||||||
if (typmod == -1)
|
CheckDim(nelemsp);
|
||||||
CheckDim(nelemsp);
|
CheckExpectedDim(typmod, nelemsp);
|
||||||
else
|
|
||||||
CheckExpectedDim(typmod, nelemsp);
|
|
||||||
|
|
||||||
result = InitVector(nelemsp);
|
result = InitVector(nelemsp);
|
||||||
for (i = 0; i < nelemsp; i++)
|
for (i = 0; i < nelemsp; i++)
|
||||||
@@ -409,6 +448,7 @@ array_to_vector(PG_FUNCTION_ARGS)
|
|||||||
(errcode(ERRCODE_NULL_VALUE_NOT_ALLOWED),
|
(errcode(ERRCODE_NULL_VALUE_NOT_ALLOWED),
|
||||||
errmsg("array must not containing NULLs")));
|
errmsg("array must not containing NULLs")));
|
||||||
|
|
||||||
|
/* TODO Move outside loop in 0.5.0 */
|
||||||
if (ARR_ELEMTYPE(array) == INT4OID)
|
if (ARR_ELEMTYPE(array) == INT4OID)
|
||||||
result->x[i] = DatumGetInt32(elemsp[i]);
|
result->x[i] = DatumGetInt32(elemsp[i]);
|
||||||
else if (ARR_ELEMTYPE(array) == FLOAT8OID)
|
else if (ARR_ELEMTYPE(array) == FLOAT8OID)
|
||||||
@@ -436,17 +476,19 @@ Datum
|
|||||||
vector_to_float4(PG_FUNCTION_ARGS)
|
vector_to_float4(PG_FUNCTION_ARGS)
|
||||||
{
|
{
|
||||||
Vector *vec = PG_GETARG_VECTOR_P(0);
|
Vector *vec = PG_GETARG_VECTOR_P(0);
|
||||||
Datum *d;
|
Datum *datums;
|
||||||
ArrayType *result;
|
ArrayType *result;
|
||||||
int i;
|
int i;
|
||||||
|
|
||||||
d = (Datum *) palloc(sizeof(Datum) * vec->dim);
|
datums = (Datum *) palloc(sizeof(Datum) * vec->dim);
|
||||||
|
|
||||||
for (i = 0; i < vec->dim; i++)
|
for (i = 0; i < vec->dim; i++)
|
||||||
d[i] = Float4GetDatum(vec->x[i]);
|
datums[i] = Float4GetDatum(vec->x[i]);
|
||||||
|
|
||||||
/* Use TYPALIGN_INT for float4 */
|
/* Use TYPALIGN_INT for float4 */
|
||||||
result = construct_array(d, vec->dim, FLOAT4OID, sizeof(float4), true, TYPALIGN_INT);
|
result = construct_array(datums, vec->dim, FLOAT4OID, sizeof(float4), true, TYPALIGN_INT);
|
||||||
|
|
||||||
|
pfree(datums);
|
||||||
|
|
||||||
PG_RETURN_POINTER(result);
|
PG_RETURN_POINTER(result);
|
||||||
}
|
}
|
||||||
@@ -467,6 +509,7 @@ l2_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
diff = ax[i] - bx[i];
|
diff = ax[i] - bx[i];
|
||||||
@@ -493,6 +536,7 @@ vector_l2_squared_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
diff = ax[i] - bx[i];
|
diff = ax[i] - bx[i];
|
||||||
@@ -517,6 +561,7 @@ inner_product(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
|
|
||||||
@@ -538,6 +583,7 @@ vector_negative_inner_product(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
|
|
||||||
@@ -561,6 +607,7 @@ cosine_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
@@ -587,6 +634,7 @@ vector_spherical_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += a->x[i] * b->x[i];
|
distance += a->x[i] * b->x[i];
|
||||||
|
|
||||||
@@ -622,6 +670,7 @@ vector_norm(PG_FUNCTION_ARGS)
|
|||||||
float *ax = a->x;
|
float *ax = a->x;
|
||||||
double norm = 0.0;
|
double norm = 0.0;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
norm += ax[i] * ax[i];
|
norm += ax[i] * ax[i];
|
||||||
|
|
||||||
@@ -646,9 +695,18 @@ vector_add(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
result = InitVector(a->dim);
|
result = InitVector(a->dim);
|
||||||
rx = result->x;
|
rx = result->x;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0, imax = a->dim; i < imax; i++)
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
rx[i] = ax[i] + bx[i];
|
rx[i] = ax[i] + bx[i];
|
||||||
|
|
||||||
|
/* Check for overflow */
|
||||||
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
|
{
|
||||||
|
if (isinf(rx[i]))
|
||||||
|
float_overflow_error();
|
||||||
|
}
|
||||||
|
|
||||||
PG_RETURN_POINTER(result);
|
PG_RETURN_POINTER(result);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -670,9 +728,18 @@ vector_sub(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
result = InitVector(a->dim);
|
result = InitVector(a->dim);
|
||||||
rx = result->x;
|
rx = result->x;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0, imax = a->dim; i < imax; i++)
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
rx[i] = ax[i] - bx[i];
|
rx[i] = ax[i] - bx[i];
|
||||||
|
|
||||||
|
/* Check for overflow */
|
||||||
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
|
{
|
||||||
|
if (isinf(rx[i]))
|
||||||
|
float_overflow_error();
|
||||||
|
}
|
||||||
|
|
||||||
PG_RETURN_POINTER(result);
|
PG_RETURN_POINTER(result);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -818,12 +885,12 @@ vector_accum(PG_FUNCTION_ARGS)
|
|||||||
n = statevalues[0] + 1.0;
|
n = statevalues[0] + 1.0;
|
||||||
|
|
||||||
statedatums = CreateStateDatums(dim);
|
statedatums = CreateStateDatums(dim);
|
||||||
statedatums[0] = Float8GetDatumFast(n);
|
statedatums[0] = Float8GetDatum(n);
|
||||||
|
|
||||||
if (newarr)
|
if (newarr)
|
||||||
{
|
{
|
||||||
for (int i = 0; i < dim; i++)
|
for (int i = 0; i < dim; i++)
|
||||||
statedatums[i + 1] = Float8GetDatumFast((double) x[i]);
|
statedatums[i + 1] = Float8GetDatum((double) x[i]);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -831,10 +898,11 @@ vector_accum(PG_FUNCTION_ARGS)
|
|||||||
{
|
{
|
||||||
double v = statevalues[i + 1] + x[i];
|
double v = statevalues[i + 1] + x[i];
|
||||||
|
|
||||||
|
/* Check for overflow */
|
||||||
if (isinf(v))
|
if (isinf(v))
|
||||||
float_overflow_error();
|
float_overflow_error();
|
||||||
|
|
||||||
statedatums[i + 1] = Float8GetDatumFast(v);
|
statedatums[i + 1] = Float8GetDatum(v);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -879,7 +947,7 @@ vector_combine(PG_FUNCTION_ARGS)
|
|||||||
dim = STATE_DIMS(statearray2);
|
dim = STATE_DIMS(statearray2);
|
||||||
statedatums = CreateStateDatums(dim);
|
statedatums = CreateStateDatums(dim);
|
||||||
for (int i = 1; i <= dim; i++)
|
for (int i = 1; i <= dim; i++)
|
||||||
statedatums[i] = Float8GetDatumFast(statevalues2[i]);
|
statedatums[i] = Float8GetDatum(statevalues2[i]);
|
||||||
}
|
}
|
||||||
else if (n2 == 0.0)
|
else if (n2 == 0.0)
|
||||||
{
|
{
|
||||||
@@ -887,7 +955,7 @@ vector_combine(PG_FUNCTION_ARGS)
|
|||||||
dim = STATE_DIMS(statearray1);
|
dim = STATE_DIMS(statearray1);
|
||||||
statedatums = CreateStateDatums(dim);
|
statedatums = CreateStateDatums(dim);
|
||||||
for (int i = 1; i <= dim; i++)
|
for (int i = 1; i <= dim; i++)
|
||||||
statedatums[i] = Float8GetDatumFast(statevalues1[i]);
|
statedatums[i] = Float8GetDatum(statevalues1[i]);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -899,14 +967,15 @@ vector_combine(PG_FUNCTION_ARGS)
|
|||||||
{
|
{
|
||||||
double v = statevalues1[i] + statevalues2[i];
|
double v = statevalues1[i] + statevalues2[i];
|
||||||
|
|
||||||
|
/* Check for overflow */
|
||||||
if (isinf(v))
|
if (isinf(v))
|
||||||
float_overflow_error();
|
float_overflow_error();
|
||||||
|
|
||||||
statedatums[i] = Float8GetDatumFast(v);
|
statedatums[i] = Float8GetDatum(v);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
statedatums[0] = Float8GetDatumFast(n);
|
statedatums[0] = Float8GetDatum(n);
|
||||||
|
|
||||||
result = construct_array(statedatums, dim + 1,
|
result = construct_array(statedatums, dim + 1,
|
||||||
FLOAT8OID,
|
FLOAT8OID,
|
||||||
@@ -929,7 +998,6 @@ vector_avg(PG_FUNCTION_ARGS)
|
|||||||
float8 n;
|
float8 n;
|
||||||
uint16 dim;
|
uint16 dim;
|
||||||
Vector *result;
|
Vector *result;
|
||||||
float v;
|
|
||||||
|
|
||||||
/* Check array before using */
|
/* Check array before using */
|
||||||
statevalues = CheckStateArray(statearray, "vector_avg");
|
statevalues = CheckStateArray(statearray, "vector_avg");
|
||||||
@@ -941,12 +1009,12 @@ vector_avg(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
/* Create vector */
|
/* Create vector */
|
||||||
dim = STATE_DIMS(statearray);
|
dim = STATE_DIMS(statearray);
|
||||||
|
CheckDim(dim);
|
||||||
result = InitVector(dim);
|
result = InitVector(dim);
|
||||||
for (int i = 0; i < dim; i++)
|
for (int i = 0; i < dim; i++)
|
||||||
{
|
{
|
||||||
v = statevalues[i + 1] / n;
|
result->x[i] = statevalues[i + 1] / n;
|
||||||
CheckElement(v);
|
CheckElement(result->x[i]);
|
||||||
result->x[i] = v;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
PG_RETURN_POINTER(result);
|
PG_RETURN_POINTER(result);
|
||||||
|
|||||||
@@ -46,6 +46,8 @@ SELECT '[1,2,3]'::vector::real[];
|
|||||||
|
|
||||||
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
||||||
ERROR: vector cannot have more than 16000 dimensions
|
ERROR: vector cannot have more than 16000 dimensions
|
||||||
|
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
|
||||||
|
ERROR: vector cannot have more than 16000 dimensions
|
||||||
-- ensure no error
|
-- ensure no error
|
||||||
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
||||||
?column?
|
?column?
|
||||||
|
|||||||
@@ -4,12 +4,16 @@ SELECT '[1,2,3]'::vector + '[4,5,6]';
|
|||||||
[5,7,9]
|
[5,7,9]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[3e38]'::vector + '[3e38]';
|
||||||
|
ERROR: value out of range: overflow
|
||||||
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
||||||
?column?
|
?column?
|
||||||
------------
|
------------
|
||||||
[-3,-3,-3]
|
[-3,-3,-3]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[-3e38]'::vector - '[3e38]';
|
||||||
|
ERROR: value out of range: overflow
|
||||||
SELECT vector_dims('[1,2,3]');
|
SELECT vector_dims('[1,2,3]');
|
||||||
vector_dims
|
vector_dims
|
||||||
-------------
|
-------------
|
||||||
@@ -102,3 +106,5 @@ SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
|||||||
|
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
||||||
ERROR: expected 2 dimensions, not 1
|
ERROR: expected 2 dimensions, not 1
|
||||||
|
SELECT vector_avg(array_agg(n)) FROM generate_series(1, 16002) n;
|
||||||
|
ERROR: vector cannot have more than 16000 dimensions
|
||||||
|
|||||||
@@ -4,10 +4,22 @@ SELECT '[1,2,3]'::vector;
|
|||||||
[1,2,3]
|
[1,2,3]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[-1,2,3]'::vector;
|
SELECT '[-1,-2,-3]'::vector;
|
||||||
vector
|
vector
|
||||||
----------
|
------------
|
||||||
[-1,2,3]
|
[-1,-2,-3]
|
||||||
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[1.,2.,3.]'::vector;
|
||||||
|
vector
|
||||||
|
---------
|
||||||
|
[1,2,3]
|
||||||
|
(1 row)
|
||||||
|
|
||||||
|
SELECT ' [ 1, 2 , 3 ] '::vector;
|
||||||
|
vector
|
||||||
|
---------
|
||||||
|
[1,2,3]
|
||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[1.23456]'::vector;
|
SELECT '[1.23456]'::vector;
|
||||||
@@ -17,7 +29,7 @@ SELECT '[1.23456]'::vector;
|
|||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
SELECT '[hello,1]'::vector;
|
SELECT '[hello,1]'::vector;
|
||||||
ERROR: invalid input syntax for type vector: "hello"
|
ERROR: invalid input syntax for type vector: "[hello,1]"
|
||||||
LINE 1: SELECT '[hello,1]'::vector;
|
LINE 1: SELECT '[hello,1]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[NaN,1]'::vector;
|
SELECT '[NaN,1]'::vector;
|
||||||
@@ -32,13 +44,35 @@ SELECT '[-Infinity,1]'::vector;
|
|||||||
ERROR: infinite value not allowed in vector
|
ERROR: infinite value not allowed in vector
|
||||||
LINE 1: SELECT '[-Infinity,1]'::vector;
|
LINE 1: SELECT '[-Infinity,1]'::vector;
|
||||||
^
|
^
|
||||||
|
SELECT '[1.5e38,-1.5e38]'::vector;
|
||||||
|
vector
|
||||||
|
--------------------
|
||||||
|
[1.5e+38,-1.5e+38]
|
||||||
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[1.5e+38,-1.5e+38]'::vector;
|
||||||
|
vector
|
||||||
|
--------------------
|
||||||
|
[1.5e+38,-1.5e+38]
|
||||||
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[1.5e-38,-1.5e-38]'::vector;
|
||||||
|
vector
|
||||||
|
--------------------
|
||||||
|
[1.5e-38,-1.5e-38]
|
||||||
|
(1 row)
|
||||||
|
|
||||||
|
SELECT '[4e38,1]'::vector;
|
||||||
|
ERROR: infinite value not allowed in vector
|
||||||
|
LINE 1: SELECT '[4e38,1]'::vector;
|
||||||
|
^
|
||||||
SELECT '[1,2,3'::vector;
|
SELECT '[1,2,3'::vector;
|
||||||
ERROR: malformed vector literal
|
ERROR: malformed vector literal: "[1,2,3"
|
||||||
LINE 1: SELECT '[1,2,3'::vector;
|
LINE 1: SELECT '[1,2,3'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Unexpected end of input.
|
DETAIL: Unexpected end of input.
|
||||||
SELECT '[1,2,3]9'::vector;
|
SELECT '[1,2,3]9'::vector;
|
||||||
ERROR: malformed vector literal
|
ERROR: malformed vector literal: "[1,2,3]9"
|
||||||
LINE 1: SELECT '[1,2,3]9'::vector;
|
LINE 1: SELECT '[1,2,3]9'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Junk after closing right brace.
|
DETAIL: Junk after closing right brace.
|
||||||
@@ -47,14 +81,36 @@ ERROR: malformed vector literal: "1,2,3"
|
|||||||
LINE 1: SELECT '1,2,3'::vector;
|
LINE 1: SELECT '1,2,3'::vector;
|
||||||
^
|
^
|
||||||
DETAIL: Vector contents must start with "[".
|
DETAIL: Vector contents must start with "[".
|
||||||
|
SELECT '['::vector;
|
||||||
|
ERROR: malformed vector literal: "["
|
||||||
|
LINE 1: SELECT '['::vector;
|
||||||
|
^
|
||||||
|
DETAIL: Unexpected end of input.
|
||||||
|
SELECT '[,'::vector;
|
||||||
|
ERROR: malformed vector literal: "[,"
|
||||||
|
LINE 1: SELECT '[,'::vector;
|
||||||
|
^
|
||||||
|
DETAIL: Unexpected end of input.
|
||||||
SELECT '[]'::vector;
|
SELECT '[]'::vector;
|
||||||
ERROR: vector must have at least 1 dimension
|
ERROR: vector must have at least 1 dimension
|
||||||
LINE 1: SELECT '[]'::vector;
|
LINE 1: SELECT '[]'::vector;
|
||||||
^
|
^
|
||||||
SELECT '[1,]'::vector;
|
SELECT '[1,]'::vector;
|
||||||
ERROR: invalid input syntax for type vector: "]"
|
ERROR: invalid input syntax for type vector: "[1,]"
|
||||||
LINE 1: SELECT '[1,]'::vector;
|
LINE 1: SELECT '[1,]'::vector;
|
||||||
^
|
^
|
||||||
|
SELECT '[1a]'::vector;
|
||||||
|
ERROR: invalid input syntax for type vector: "[1a]"
|
||||||
|
LINE 1: SELECT '[1a]'::vector;
|
||||||
|
^
|
||||||
|
SELECT '[1,,3]'::vector;
|
||||||
|
ERROR: malformed vector literal: "[1,,3]"
|
||||||
|
LINE 1: SELECT '[1,,3]'::vector;
|
||||||
|
^
|
||||||
|
SELECT '[1, ,3]'::vector;
|
||||||
|
ERROR: invalid input syntax for type vector: "[1, ,3]"
|
||||||
|
LINE 1: SELECT '[1, ,3]'::vector;
|
||||||
|
^
|
||||||
SELECT '[1,2,3]'::vector(2);
|
SELECT '[1,2,3]'::vector(2);
|
||||||
ERROR: expected 2 dimensions, not 3
|
ERROR: expected 2 dimensions, not 3
|
||||||
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ SELECT '{-Infinity}'::real[]::vector;
|
|||||||
SELECT '{}'::real[]::vector;
|
SELECT '{}'::real[]::vector;
|
||||||
SELECT '[1,2,3]'::vector::real[];
|
SELECT '[1,2,3]'::vector::real[];
|
||||||
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
SELECT array_agg(n)::vector FROM generate_series(1, 16001) n;
|
||||||
|
SELECT array_to_vector(array_agg(n), 16001, false) FROM generate_series(1, 16001) n;
|
||||||
|
|
||||||
-- ensure no error
|
-- ensure no error
|
||||||
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
SELECT ARRAY[1,2,3] = ARRAY[1,2,3];
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
SELECT '[1,2,3]'::vector + '[4,5,6]';
|
SELECT '[1,2,3]'::vector + '[4,5,6]';
|
||||||
|
SELECT '[3e38]'::vector + '[3e38]';
|
||||||
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
SELECT '[1,2,3]'::vector - '[4,5,6]';
|
||||||
|
SELECT '[-3e38]'::vector - '[3e38]';
|
||||||
|
|
||||||
SELECT vector_dims('[1,2,3]');
|
SELECT vector_dims('[1,2,3]');
|
||||||
|
|
||||||
@@ -24,3 +26,4 @@ SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;
|
|||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]', NULL]) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
SELECT avg(v) FROM unnest(ARRAY[]::vector[]) v;
|
||||||
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
SELECT avg(v) FROM unnest(ARRAY['[1,2]'::vector, '[3]']) v;
|
||||||
|
SELECT vector_avg(array_agg(n)) FROM generate_series(1, 16002) n;
|
||||||
|
|||||||
@@ -1,15 +1,26 @@
|
|||||||
SELECT '[1,2,3]'::vector;
|
SELECT '[1,2,3]'::vector;
|
||||||
SELECT '[-1,2,3]'::vector;
|
SELECT '[-1,-2,-3]'::vector;
|
||||||
|
SELECT '[1.,2.,3.]'::vector;
|
||||||
|
SELECT ' [ 1, 2 , 3 ] '::vector;
|
||||||
SELECT '[1.23456]'::vector;
|
SELECT '[1.23456]'::vector;
|
||||||
SELECT '[hello,1]'::vector;
|
SELECT '[hello,1]'::vector;
|
||||||
SELECT '[NaN,1]'::vector;
|
SELECT '[NaN,1]'::vector;
|
||||||
SELECT '[Infinity,1]'::vector;
|
SELECT '[Infinity,1]'::vector;
|
||||||
SELECT '[-Infinity,1]'::vector;
|
SELECT '[-Infinity,1]'::vector;
|
||||||
|
SELECT '[1.5e38,-1.5e38]'::vector;
|
||||||
|
SELECT '[1.5e+38,-1.5e+38]'::vector;
|
||||||
|
SELECT '[1.5e-38,-1.5e-38]'::vector;
|
||||||
|
SELECT '[4e38,1]'::vector;
|
||||||
SELECT '[1,2,3'::vector;
|
SELECT '[1,2,3'::vector;
|
||||||
SELECT '[1,2,3]9'::vector;
|
SELECT '[1,2,3]9'::vector;
|
||||||
SELECT '1,2,3'::vector;
|
SELECT '1,2,3'::vector;
|
||||||
|
SELECT '['::vector;
|
||||||
|
SELECT '[,'::vector;
|
||||||
SELECT '[]'::vector;
|
SELECT '[]'::vector;
|
||||||
SELECT '[1,]'::vector;
|
SELECT '[1,]'::vector;
|
||||||
|
SELECT '[1a]'::vector;
|
||||||
|
SELECT '[1,,3]'::vector;
|
||||||
|
SELECT '[1, ,3]'::vector;
|
||||||
SELECT '[1,2,3]'::vector(2);
|
SELECT '[1,2,3]'::vector(2);
|
||||||
|
|
||||||
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
SELECT unnest('{"[1,2,3]", "[4,5,6]"}'::vector[]);
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
comment = 'vector data type and ivfflat access method'
|
comment = 'vector data type and ivfflat access method'
|
||||||
default_version = '0.4.1'
|
default_version = '0.4.4'
|
||||||
module_pathname = '$libdir/vector'
|
module_pathname = '$libdir/vector'
|
||||||
relocatable = true
|
relocatable = true
|
||||||
|
|||||||
Reference in New Issue
Block a user