Compare commits

...

58 Commits

Author SHA1 Message Date
Andrew Kane
54870b9bdc Debug [skip ci] 2023-03-25 16:04:48 -07:00
Andrew Kane
0950968c11 Started support for parallel index scan [skip ci] 2023-03-25 15:51:59 -07:00
Andrew Kane
ec12d79cbc Added link to pgvector-perl [skip ci] 2023-03-23 21:42:43 -07:00
Andrew Kane
d3eb56df07 Improved tests 2023-03-22 16:35:56 -07:00
Andrew Kane
13f7aa50c3 Added Render link [skip ci] 2023-03-22 16:29:06 -07:00
Andrew Kane
81e9e72fbc Updated guidance on lists [skip ci] 2023-03-22 14:00:47 -07:00
Andrew Kane
8d95510302 Updated readme [skip ci] 2023-03-22 13:32:25 -07:00
Andrew Kane
53bb2ed0cd Updated Homebrew instructions [skip ci] 2023-03-21 13:20:57 -07:00
Andrew Kane
fac8b9c8d9 Version bump to 0.4.1 [skip ci] 2023-03-21 12:39:59 -07:00
Andrew Kane
4e68f4f800 Updated comment [skip ci] 2023-03-21 11:55:26 -07:00
Andrew Kane
b69ac51ad7 Added comment [skip ci] 2023-03-21 11:43:30 -07:00
Andrew Kane
29c7af5d18 Improved test 2023-03-21 11:32:16 -07:00
Andrew Kane
bc9e2a37ec Improved performance of cosine distance 2023-03-21 11:25:25 -07:00
Andrew Kane
18eb04e278 Updated readme [skip ci] 2023-03-12 14:09:36 -07:00
Andrew Kane
42da2b334b Restored previous behavior and added comment 2023-03-12 13:57:40 -07:00
Andrew Kane
dbfc6a35d9 Improved vacuumcleanup stats 2023-03-12 13:34:26 -07:00
Andrew Kane
bbae64b784 Removed unneeded query from test 2023-03-12 13:26:23 -07:00
Andrew Kane
a6e8f36415 Updated readme [skip ci] 2023-03-12 13:15:47 -07:00
Andrew Kane
44985abbc6 Updated comment [skip ci] 2023-03-12 13:15:31 -07:00
Andrew Kane
9897ba7f64 Fixed CI 2023-03-12 13:13:04 -07:00
Andrew Kane
bb75ce2cf2 Fixed index scan count 2023-03-12 12:24:20 -07:00
Andrew Kane
385b2437a9 Updated readme [skip ci] 2023-03-05 18:18:03 -08:00
Andrew Kane
6c1536ee78 Moved section [skip ci] 2023-03-05 18:16:43 -08:00
Andrew Kane
4d8eae1c3e Updated readme [skip ci] 2023-03-05 18:13:32 -08:00
Andrew Kane
68378066a8 Added example of AVG [skip ci] 2023-03-05 17:54:02 -08:00
Andrew Kane
4f8620fe9e Added EXPLAIN ANALYZE to readme [skip ci] 2023-03-05 17:48:01 -08:00
Andrew Kane
a1ab0a453c Moved section [skip ci] 2023-03-05 17:44:53 -08:00
Andrew Kane
f1983ce672 Added section on querying - #64 [skip ci] 2023-03-05 17:34:49 -08:00
Andrew Kane
0a3615b82f Updated readme [skip ci] 2023-03-05 14:17:39 -08:00
Andrew Kane
e8a67dcf1c Added link to pgvector-lua [skip ci] 2023-03-05 13:44:52 -08:00
Andrew Kane
e83121eee4 Added link to pgvector-dotnet [skip ci] 2023-03-05 11:45:40 -08:00
Andrew Kane
77154f8cbb Updated readme [skip ci] 2023-03-03 16:33:44 -08:00
Andrew Kane
051c42a05a Updated readme [skip ci] 2023-03-03 16:28:23 -08:00
Andrew Kane
9077d85407 Reordered libraries [skip ci] 2023-03-03 16:18:55 -08:00
Andrew Kane
d74c82ff97 Updated pgvector-java link [skip ci] 2023-03-03 16:08:01 -08:00
Andrew Kane
cb6b7a3893 Added link to pgvector-scala [skip ci] 2023-03-03 11:50:03 -08:00
Andrew Kane
cb7d787d7c Added link to pgvector-julia [skip ci] 2023-03-02 22:16:23 -08:00
Andrew Kane
fc09ee200d Updated readme [skip ci] 2023-03-02 19:33:21 -08:00
Andrew Kane
eedcd64e08 Updated readme [skip ci] 2023-03-02 19:27:09 -08:00
Andrew Kane
c1671a4982 Updated readme [skip ci] 2023-03-02 19:22:00 -08:00
Andrew Kane
010055b288 Updated readme [skip ci] 2023-03-02 16:57:01 -08:00
Andrew Kane
8b759b695b Added link to pgvector-r [skip ci] 2023-03-02 16:48:14 -08:00
Andrew Kane
e1e53860d4 Added log to .gitignore [skip ci] 2023-02-26 09:57:31 -08:00
Andrew Kane
a85250c7bc Added platform to Docker task [skip ci] 2023-02-25 16:11:28 -08:00
Andrew Kane
4984411906 Added dylib to .gitignore [skip ci] 2023-02-25 15:28:00 -08:00
Andrew Kane
e4b0d41d30 Fixed warning 2023-02-23 14:17:31 -08:00
Andrew Kane
1b16a28906 Added test for casting from numeric[] 2023-02-23 14:15:50 -08:00
Andrew Kane
0b3dc0887f Fixed compilation with Postgres 16 - fixes #61 2023-02-23 14:08:27 -08:00
Andrew Kane
bb71b22e2e Updated readme [skip ci] 2023-02-22 19:42:40 -08:00
Andrew Kane
3b9df1c145 Moved to thread [skip ci] 2023-01-28 13:56:44 -08:00
Andrew Kane
11f66e6010 Added Supabase to readme [skip ci] 2023-01-28 11:13:47 -08:00
Andrew Kane
9b13db5c5c Updated link [skip ci] 2023-01-18 04:31:42 -08:00
Andrew Kane
ca9d4ed82a Updated readme [skip ci] 2023-01-13 10:30:34 -08:00
Andrew Kane
b1ffe2d4d3 Updated readme [skip ci] 2023-01-13 00:41:13 -08:00
Andrew Kane
246f8814bc Updated readme [skip ci] 2023-01-13 00:40:37 -08:00
Andrew Kane
a5d8bb2688 Added link to feedstock [skip ci] 2023-01-12 02:15:18 -08:00
Andrew Kane
95e496ba00 Added conda-forge to readme - closes #47 [skip ci] 2023-01-12 02:02:18 -08:00
Andrew Kane
6bfc2cb9a1 Use table for libraries [skip ci] 2023-01-11 14:34:33 -08:00
19 changed files with 238 additions and 71 deletions

2
.gitignore vendored
View File

@@ -1,4 +1,5 @@
/dist/
/log/
/results/
/tmp_check/
/sql/vector--?.?.?.sql
@@ -7,6 +8,7 @@ regression.*
*.so
*.bc
*.dll
*.dylib
*.obj
*.lib
*.exp

View File

@@ -1,3 +1,8 @@
## 0.4.1 (2023-03-21)
- Improved performance of cosine distance
- Fixed index scan count
## 0.4.0 (2023-01-11)
If upgrading with Postgres < 13, see [this note](https://github.com/pgvector/pgvector#040).

View File

@@ -2,7 +2,7 @@
"name": "vector",
"abstract": "Open-source vector similarity search for Postgres",
"description": "Supports L2 distance, inner product, and cosine distance",
"version": "0.4.0",
"version": "0.4.1",
"maintainer": [
"Andrew Kane <andrew@ankane.org>"
],
@@ -20,7 +20,7 @@
"vector": {
"file": "sql/vector.sql",
"docfile": "README.md",
"version": "0.4.0",
"version": "0.4.1",
"abstract": "Open-source vector similarity search for Postgres"
}
},

View File

@@ -1,5 +1,5 @@
EXTENSION = vector
EXTVERSION = 0.4.0
EXTVERSION = 0.4.1
MODULE_big = vector
DATA = $(wildcard sql/*--*.sql)
@@ -61,4 +61,4 @@ dist:
.PHONY: docker
docker:
docker build --pull --no-cache -t ankane/pgvector:latest .
docker build --pull --no-cache --platform linux/amd64 -t ankane/pgvector:latest .

View File

@@ -1,5 +1,5 @@
EXTENSION = vector
EXTVERSION = 0.4.0
EXTVERSION = 0.4.1
OBJS = src\ivfbuild.obj src\ivfflat.obj src\ivfinsert.obj src\ivfkmeans.obj src\ivfscan.obj src\ivfutils.obj src\ivfvacuum.obj src\vector.obj

143
README.md
View File

@@ -17,7 +17,7 @@ Supports L2 distance, inner product, and cosine distance
Compile and install the extension (supports Postgres 11+)
```sh
git clone --branch v0.4.0 https://github.com/pgvector/pgvector.git
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
cd pgvector
make
make install # may need sudo
@@ -29,7 +29,7 @@ Then load it in databases where you want to use it
CREATE EXTENSION vector;
```
You can also install it with [Docker](#docker), [Homebrew](#homebrew), or [PGXN](#pgxn)
You can also install it with [Docker](#docker), [Homebrew](#homebrew), [PGXN](#pgxn), or [conda-forge](#conda-forge)
## Getting Started
@@ -55,6 +55,28 @@ Also supports inner product (`<#>`) and cosine distance (`<=>`)
Note: `<#>` returns the negative inner product since Postgres only supports `ASC` order index scans on operators
## Querying
Use a `SELECT` clause to get the distance
```sql
SELECT embedding <-> '[3,1,2]' AS distance FROM items;
```
Use a `WHERE` clause to get rows within a certain distance
```sql
SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
```
Note: Combine with `ORDER BY` and `LIMIT` to use an index
Get the average of vectors
```sql
SELECT AVG(embedding) FROM items;
```
## Indexing
Speed up queries with an approximate index. Add an index for each distance function you want to use.
@@ -87,7 +109,10 @@ Specify the number of inverted lists (100 by default)
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100);
```
A [good place to start](https://github.com/facebookresearch/faiss/issues/112) is `4 * sqrt(rows)`
A lower value provides better recall at the cost of speed. A good place to start is:
- `rows / 1000` for up to 1M rows
- `sqrt(rows)` for over 1M rows
### Query Options
@@ -97,7 +122,7 @@ Specify the number of probes (1 by default)
SET ivfflat.probes = 1;
```
A higher value improves recall at the cost of speed.
A higher value provides better recall at the cost of speed, and it can be set to the number of lists for exact nearest neighbor search (at which point the planner wont use the index)
Use `SET LOCAL` inside a transaction to set it for a single query
@@ -159,6 +184,50 @@ To speed up queries with an index, increase the number of inverted lists (at the
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 1000);
```
Use `EXPLAIN ANALYZE` to debug performance.
```sql
EXPLAIN ANALYZE SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 1;
```
## Languages
Use pgvector from any language with a Postgres client. You can even generate and store vectors in one language and query them in another.
Language | Libraries / Examples
--- | ---
C++ | [pgvector-cpp](https://github.com/pgvector/pgvector-cpp)
C# | [pgvector-dotnet](https://github.com/pgvector/pgvector-dotnet)
Elixir | [pgvector-elixir](https://github.com/pgvector/pgvector-elixir)
Go | [pgvector-go](https://github.com/pgvector/pgvector-go)
Java, Scala | [pgvector-java](https://github.com/pgvector/pgvector-java)
Julia | [pgvector-julia](https://github.com/pgvector/pgvector-julia)
Lua | [pgvector-lua](https://github.com/pgvector/pgvector-lua)
Node.js | [pgvector-node](https://github.com/pgvector/pgvector-node)
Perl | [pgvector-perl](https://github.com/pgvector/pgvector-perl)
PHP | [pgvector-php](https://github.com/pgvector/pgvector-php)
Python | [pgvector-python](https://github.com/pgvector/pgvector-python)
R | [pgvector-r](https://github.com/pgvector/pgvector-r)
Ruby | [pgvector-ruby](https://github.com/pgvector/pgvector-ruby), [Neighbor](https://github.com/ankane/neighbor)
Rust | [pgvector-rust](https://github.com/pgvector/pgvector-rust)
## Frequently Asked Questions
#### How many vectors can be stored in a single table?
A non-partitioned table has a limit of 32 TB by default in Postgres. A partitioned table can have thousands of partitions of that size.
#### Is replication supported?
Yes, pgvector uses the write-ahead log (WAL), which allows for replication and point-in-time recovery.
#### What if I want to index vectors with more than 2,000 dimensions?
Two things you can try are:
1. use dimensionality reduction
2. compile Postgres with a larger block size (`./configure --with-blocksize=32`) and edit the limit in `src/ivfflat.h`
## Reference
### Vector Type
@@ -187,40 +256,9 @@ vector_norm(vector) → double precision | Euclidean norm
### Aggregate Functions
Function | Description | Partial Mode
--- | --- | ---
avg(vector) → vector | arithmetic mean | Yes
## Libraries
Libraries that use pgvector:
- [pgvector-python](https://github.com/pgvector/pgvector-python) (Python)
- [Neighbor](https://github.com/ankane/neighbor) (Ruby)
- [pgvector-ruby](https://github.com/pgvector/pgvector-ruby) (Ruby)
- [pgvector-node](https://github.com/pgvector/pgvector-node) (Node.js)
- [pgvector-go](https://github.com/pgvector/pgvector-go) (Go)
- [pgvector-php](https://github.com/pgvector/pgvector-php) (PHP)
- [pgvector-rust](https://github.com/pgvector/pgvector-rust) (Rust)
- [pgvector-cpp](https://github.com/pgvector/pgvector-cpp) (C++)
- [pgvector-elixir](https://github.com/pgvector/pgvector-elixir) (Elixir)
## Frequently Asked Questions
#### How many vectors can be stored in a single table?
A non-partitioned table has a limit of 32 TB by default in Postgres. A partitioned table can have thousands of partitions of that size.
#### Is replication supported?
Yes, pgvector uses the write-ahead log (WAL), which allows for replication and point-in-time recovery.
#### What if I want to index vectors with more than 2,000 dimensions?
Two things you can try are:
1. use dimensionality reduction
2. compile Postgres with a larger block size (`./configure --with-blocksize=32`) and edit the limit in `src/ivfflat.h`
Function | Description
--- | ---
avg(vector) → vector | arithmetic mean
## Additional Installation Methods
@@ -232,12 +270,12 @@ Get the [Docker image](https://hub.docker.com/r/ankane/pgvector) with:
docker pull ankane/pgvector
```
This adds pgvector to the [Postgres image](https://hub.docker.com/_/postgres).
This adds pgvector to the [Postgres image](https://hub.docker.com/_/postgres) (run it the same way).
You can also build the image manually
You can also build the image manually:
```sh
git clone --branch v0.4.0 https://github.com/pgvector/pgvector.git
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
cd pgvector
docker build -t pgvector .
```
@@ -247,7 +285,7 @@ docker build -t pgvector .
With Homebrew Postgres, you can use:
```sh
brew install pgvector/brew/pgvector
brew install pgvector
```
### PGXN
@@ -258,14 +296,27 @@ Install from the [PostgreSQL Extension Network](https://pgxn.org/dist/vector) wi
pgxn install vector
```
### conda-forge
With Conda Postgres, install from [conda-forge](https://anaconda.org/conda-forge/pgvector) with:
```sh
conda install -c conda-forge pgvector
```
This method is [community-maintained](https://github.com/conda-forge/pgvector-feedstock) by [@mmcauliffe](https://github.com/mmcauliffe)
## Hosted Postgres
Some Postgres providers only support specific extensions. To request a new extension:
pgvector is available on [these providers](https://github.com/pgvector/pgvector/issues/54).
To request a new extension on other providers:
- Amazon RDS - follow the instructions on [this page](https://aws.amazon.com/rds/postgresql/faqs/)
- Google Cloud SQL - follow the instructions on [this page](https://cloud.google.com/sql/docs/postgres/extensions#requesting-support-for-a-new-extension)
- Google Cloud SQL - vote or comment on [this page](https://issuetracker.google.com/issues/265172065)
- Azure Database - vote or comment on [this page](https://feedback.azure.com/d365community/idea/7b423322-6189-ed11-a81b-000d3ae49307)
- DigitalOcean Managed Databases - vote or comment on [this page](https://ideas.digitalocean.com/app-framework-services/p/pgvector-extension-for-postgresql)
- Azure Database for PostgreSQL - vote or comment on [this page](https://feedback.azure.com/d365community/idea/7b423322-6189-ed11-a81b-000d3ae49307)
- Render - vote or comment on [this page](https://feedback.render.com/features/p/add-pgvector-extension-to-postgresql)
## Upgrading
@@ -279,7 +330,7 @@ ALTER EXTENSION vector UPDATE;
### 0.4.0
For Postgres < 13, remove this line from `sql/vector--0.3.2--0.4.0.sql`:
If upgrading with Postgres < 13, remove this line from `sql/vector--0.3.2--0.4.0.sql`:
```sql
ALTER TYPE vector SET (STORAGE = extended);

View File

@@ -0,0 +1,2 @@
-- complain if script is sourced in psql, rather than via CREATE EXTENSION
\echo Use "ALTER EXTENSION vector UPDATE TO '0.4.1'" to load this file. \quit

View File

@@ -103,7 +103,13 @@ ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
if (ratio > 1)
ratio = 1;
// cost estimates for parallel workers applied outside of amcostestimate
elog(INFO, "parallel_workers = %d, parallel aware = %d", path->path.parallel_workers, path->path.parallel_aware);
costs.indexTotalCost *= ratio;
costs.numIndexPages *= ratio;
elog(INFO, "ivfflatcostestimate = %f", costs.indexTotalCost);
/* Startup cost and total cost are same */
*indexStartupCost = costs.indexTotalCost;
@@ -151,6 +157,25 @@ ivfflatvalidate(Oid opclassoid)
return true;
}
static Size
ivfflatestimateparallelscan()
{
elog(INFO, "ivfflatestimateparallelscan");
return 0;
}
static void
ivfflatinitparallelscan(void *target)
{
elog(INFO, "ivfflatinitparallelscan");
}
static void
ivfflatparallelrescan(IndexScanDesc scan)
{
elog(INFO, "ivfflatparallelrescan");
}
/*
* Define index handler
*
@@ -178,7 +203,7 @@ ivfflathandler(PG_FUNCTION_ARGS)
amroutine->amstorage = false;
amroutine->amclusterable = false;
amroutine->ampredlocks = false;
amroutine->amcanparallel = false;
amroutine->amcanparallel = true;
amroutine->amcaninclude = false;
#if PG_VERSION_NUM >= 130000
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
@@ -212,9 +237,9 @@ ivfflathandler(PG_FUNCTION_ARGS)
amroutine->amrestrpos = NULL;
/* Interface functions to support parallel index scans */
amroutine->amestimateparallelscan = NULL;
amroutine->aminitparallelscan = NULL;
amroutine->amparallelrescan = NULL;
amroutine->amestimateparallelscan = ivfflatestimateparallelscan;
amroutine->aminitparallelscan = ivfflatinitparallelscan;
amroutine->amparallelrescan = ivfflatparallelrescan;
PG_RETURN_POINTER(amroutine);
}

View File

@@ -5,6 +5,7 @@
#include "access/relscan.h"
#include "ivfflat.h"
#include "miscadmin.h"
#include "pgstat.h"
#include "storage/bufmgr.h"
#include "catalog/pg_operator_d.h"
@@ -267,6 +268,9 @@ ivfflatgettuple(IndexScanDesc scan, ScanDirection dir)
{
Datum value;
/* Count index scan for stats */
pgstat_count_index_scan(scan->indexRelation);
/* Safety check */
if (scan->orderByData == NULL)
elog(ERROR, "cannot scan ivfflat index without order");

View File

@@ -143,6 +143,11 @@ ivfflatvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats)
{
Relation rel = info->index;
if (info->analyze_only)
return stats;
/* stats is NULL if ambulkdelete not called */
/* OK to return NULL if index not changed */
if (stats == NULL)
return NULL;

View File

@@ -416,7 +416,7 @@ array_to_vector(PG_FUNCTION_ARGS)
else if (ARR_ELEMTYPE(array) == FLOAT4OID)
result->x[i] = DatumGetFloat4(elemsp[i]);
else if (ARR_ELEMTYPE(array) == NUMERICOID)
result->x[i] = DatumGetFloat4(DirectFunctionCall1(numeric_float4, NumericGetDatum(elemsp[i])));
result->x[i] = DatumGetFloat4(DirectFunctionCall1(numeric_float4, elemsp[i]));
else
ereport(ERROR,
(errcode(ERRCODE_DATA_EXCEPTION),
@@ -568,7 +568,8 @@ cosine_distance(PG_FUNCTION_ARGS)
normb += bx[i] * bx[i];
}
PG_RETURN_FLOAT8(1 - (distance / (sqrt(norma) * sqrt(normb))));
/* Use sqrt(a * b) over sqrt(a) * sqrt(b) */
PG_RETURN_FLOAT8(1 - (distance / sqrt(norma * normb)));
}
/*
@@ -822,7 +823,7 @@ vector_accum(PG_FUNCTION_ARGS)
if (newarr)
{
for (int i = 0; i < dim; i++)
statedatums[i + 1] = Float8GetDatumFast(x[i]);
statedatums[i + 1] = Float8GetDatumFast((double) x[i]);
}
else
{

View File

@@ -3,6 +3,10 @@
#include "postgres.h"
#if PG_VERSION_NUM >= 160000
#include "varatt.h"
#endif
#define VECTOR_MAX_DIM 16000
#define VECTOR_SIZE(_dim) (offsetof(Vector, x) + sizeof(float)*(_dim))

View File

@@ -22,6 +22,12 @@ SELECT ARRAY[1,2,3]::float8[]::vector;
[1,2,3]
(1 row)
SELECT ARRAY[1,2,3]::numeric[]::vector;
array
---------
[1,2,3]
(1 row)
SELECT '{NULL}'::real[]::vector;
ERROR: array must not containing NULLs
SELECT '{NaN}'::real[]::vector;

View File

@@ -22,10 +22,28 @@ SELECT round(vector_norm('[1,1]')::numeric, 5);
1.41421
(1 row)
SELECT round(l2_distance('[1,2]', '[0,0]')::numeric, 5);
round
---------
2.23607
SELECT vector_norm('[3,4]');
vector_norm
-------------
5
(1 row)
SELECT vector_norm('[0,1]');
vector_norm
-------------
1
(1 row)
SELECT l2_distance('[0,0]', '[3,4]');
l2_distance
-------------
5
(1 row)
SELECT l2_distance('[0,0]', '[0,1]');
l2_distance
-------------
1
(1 row)
SELECT l2_distance('[1,2]', '[3]');
@@ -38,10 +56,10 @@ SELECT inner_product('[1,2]', '[3,4]');
SELECT inner_product('[1,2]', '[3]');
ERROR: different vector dimensions 2 and 1
SELECT round(cosine_distance('[1,2]', '[2,4]')::numeric, 5);
round
---------
0.00000
SELECT cosine_distance('[1,2]', '[2,4]');
cosine_distance
-----------------
0
(1 row)
SELECT cosine_distance('[1,2]', '[0,0]');
@@ -50,6 +68,18 @@ SELECT cosine_distance('[1,2]', '[0,0]');
NaN
(1 row)
SELECT cosine_distance('[1,1]', '[1,1]');
cosine_distance
-----------------
0
(1 row)
SELECT cosine_distance('[1,1]', '[-1,-1]');
cosine_distance
-----------------
2
(1 row)
SELECT cosine_distance('[1,2]', '[3]');
ERROR: different vector dimensions 2 and 1
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;

View File

@@ -2,6 +2,7 @@ SELECT ARRAY[1,2,3]::vector;
SELECT ARRAY[1.0,2.0,3.0]::vector;
SELECT ARRAY[1,2,3]::float4[]::vector;
SELECT ARRAY[1,2,3]::float8[]::vector;
SELECT ARRAY[1,2,3]::numeric[]::vector;
SELECT '{NULL}'::real[]::vector;
SELECT '{NaN}'::real[]::vector;
SELECT '{Infinity}'::real[]::vector;

View File

@@ -2,16 +2,22 @@ SELECT '[1,2,3]'::vector + '[4,5,6]';
SELECT '[1,2,3]'::vector - '[4,5,6]';
SELECT vector_dims('[1,2,3]');
SELECT round(vector_norm('[1,1]')::numeric, 5);
SELECT round(l2_distance('[1,2]', '[0,0]')::numeric, 5);
SELECT round(vector_norm('[1,1]')::numeric, 5);
SELECT vector_norm('[3,4]');
SELECT vector_norm('[0,1]');
SELECT l2_distance('[0,0]', '[3,4]');
SELECT l2_distance('[0,0]', '[0,1]');
SELECT l2_distance('[1,2]', '[3]');
SELECT inner_product('[1,2]', '[3,4]');
SELECT inner_product('[1,2]', '[3]');
SELECT round(cosine_distance('[1,2]', '[2,4]')::numeric, 5);
SELECT cosine_distance('[1,2]', '[2,4]');
SELECT cosine_distance('[1,2]', '[0,0]');
SELECT cosine_distance('[1,1]', '[1,1]');
SELECT cosine_distance('[1,1]', '[-1,-1]');
SELECT cosine_distance('[1,2]', '[3]');
SELECT avg(v) FROM unnest(ARRAY['[1,2,3]'::vector, '[3,5,7]']) v;

View File

@@ -0,0 +1,15 @@
-- SET force_parallel_mode = on;
SET parallel_setup_cost = 10;
SET parallel_tuple_cost = 0.001;
SET min_parallel_table_scan_size = 1;
SET min_parallel_index_scan_size = 1;
CREATE TABLE t (id integer, val vector(3));
INSERT INTO t (id, val) SELECT n, ARRAY[random(), random(), random()] FROM generate_series(1,1000000) n;
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 10);
SET ivfflat.probes = 2;
EXPLAIN SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
DROP TABLE t;

View File

@@ -2,7 +2,7 @@ use strict;
use warnings;
use PostgresNode;
use TestLib;
use Test::More tests => 5;
use Test::More tests => 7;
my $dim = 768;
@@ -32,10 +32,19 @@ $node->pgbench(
}
);
sub idx_scan
{
# Stats do not update instantaneously
# https://www.postgresql.org/docs/current/monitoring-stats.html#MONITORING-STATS-VIEWS
sleep(1);
$node->safe_psql("postgres", "SELECT idx_scan FROM pg_stat_user_indexes WHERE indexrelid = 'tst_v_idx'::regclass;");
}
my $expected = 10000 + 5 * 100 * 10;
my $count = $node->safe_psql("postgres", "SELECT COUNT(*) FROM tst;");
is($count, $expected);
is(idx_scan(), 0);
$count = $node->safe_psql("postgres", qq(
SET enable_seqscan = off;
@@ -43,3 +52,4 @@ $count = $node->safe_psql("postgres", qq(
SELECT COUNT(*) FROM (SELECT v FROM tst ORDER BY v <-> (SELECT v FROM tst LIMIT 1)) t;
));
is($count, $expected);
is(idx_scan(), 1);

View File

@@ -1,4 +1,4 @@
comment = 'vector data type and ivfflat access method'
default_version = '0.4.0'
default_version = '0.4.1'
module_pathname = '$libdir/vector'
relocatable = true