mirror of
https://github.com/pgvector/pgvector.git
synced 2026-07-22 12:07:34 +08:00
Compare commits
43 Commits
parallel-i
...
bound2
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5f66019f49 | ||
|
|
8bf360ed84 | ||
|
|
f79d28347b | ||
|
|
20cf63de0a | ||
|
|
587cbcf15b | ||
|
|
c09edb5b8f | ||
|
|
c63501cca4 | ||
|
|
58f0c922d2 | ||
|
|
36e73d2818 | ||
|
|
dd92d0ece3 | ||
|
|
6ede7681a5 | ||
|
|
8733729149 | ||
|
|
03a5789132 | ||
|
|
e5b612a856 | ||
|
|
9a7d3532f5 | ||
|
|
91315dfeff | ||
|
|
6e3101d527 | ||
|
|
96ae1a6a72 | ||
|
|
55aeba8bd6 | ||
|
|
aebe1bae02 | ||
|
|
14355b9312 | ||
|
|
b5c66d0416 | ||
|
|
f534d9878a | ||
|
|
f3df137db6 | ||
|
|
489cdb5068 | ||
|
|
161f48793e | ||
|
|
f0f7ffca41 | ||
|
|
138d9be616 | ||
|
|
fb98e73255 | ||
|
|
4754cac40c | ||
|
|
8432efb7d8 | ||
|
|
c38410259c | ||
|
|
609d9fbf0a | ||
|
|
7946424639 | ||
|
|
d51310dfa0 | ||
|
|
30f2893aeb | ||
|
|
5d0f88529e | ||
|
|
1d020abdd1 | ||
|
|
d0fdd42652 | ||
|
|
8473468925 | ||
|
|
9c01524466 | ||
|
|
121baa411e | ||
|
|
50005d7326 |
2
.github/workflows/build.yml
vendored
2
.github/workflows/build.yml
vendored
@@ -17,7 +17,7 @@ jobs:
|
|||||||
- postgres: 12
|
- postgres: 12
|
||||||
os: ubuntu-20.04
|
os: ubuntu-20.04
|
||||||
- postgres: 11
|
- postgres: 11
|
||||||
os: ubuntu-18.04
|
os: ubuntu-20.04
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v3
|
||||||
- uses: ankane/setup-postgres@v1
|
- uses: ankane/setup-postgres@v1
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
FROM postgres:15
|
ARG PG_MAJOR=15
|
||||||
|
FROM postgres:$PG_MAJOR
|
||||||
|
ARG PG_MAJOR
|
||||||
|
|
||||||
COPY . /tmp/pgvector
|
COPY . /tmp/pgvector
|
||||||
|
|
||||||
RUN apt-get update && \
|
RUN apt-get update && \
|
||||||
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-15 && \
|
apt-get install -y --no-install-recommends build-essential postgresql-server-dev-$PG_MAJOR && \
|
||||||
cd /tmp/pgvector && \
|
cd /tmp/pgvector && \
|
||||||
make clean && \
|
make clean && \
|
||||||
make OPTFLAGS="" && \
|
make OPTFLAGS="" && \
|
||||||
@@ -11,6 +13,6 @@ RUN apt-get update && \
|
|||||||
mkdir /usr/share/doc/pgvector && \
|
mkdir /usr/share/doc/pgvector && \
|
||||||
cp LICENSE README.md /usr/share/doc/pgvector && \
|
cp LICENSE README.md /usr/share/doc/pgvector && \
|
||||||
rm -r /tmp/pgvector && \
|
rm -r /tmp/pgvector && \
|
||||||
apt-get remove -y build-essential postgresql-server-dev-15 && \
|
apt-get remove -y build-essential postgresql-server-dev-$PG_MAJOR && \
|
||||||
apt-get autoremove -y && \
|
apt-get autoremove -y && \
|
||||||
rm -rf /var/lib/apt/lists/*
|
rm -rf /var/lib/apt/lists/*
|
||||||
|
|||||||
1
Makefile
1
Makefile
@@ -14,6 +14,7 @@ OPTFLAGS = -march=native
|
|||||||
# Mac ARM doesn't support -march=native
|
# Mac ARM doesn't support -march=native
|
||||||
ifeq ($(shell uname -s), Darwin)
|
ifeq ($(shell uname -s), Darwin)
|
||||||
ifeq ($(shell uname -p), arm)
|
ifeq ($(shell uname -p), arm)
|
||||||
|
# no difference with -march=armv8.5-a
|
||||||
OPTFLAGS =
|
OPTFLAGS =
|
||||||
endif
|
endif
|
||||||
endif
|
endif
|
||||||
|
|||||||
229
README.md
229
README.md
@@ -2,13 +2,7 @@
|
|||||||
|
|
||||||
Open-source vector similarity search for Postgres
|
Open-source vector similarity search for Postgres
|
||||||
|
|
||||||
```sql
|
Supports exact and approximate nearest neighbor search for L2 distance, inner product, and cosine distance
|
||||||
CREATE TABLE items (embedding vector(3));
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops);
|
|
||||||
SELECT * FROM items ORDER BY embedding <-> '[1,2,3]' LIMIT 5;
|
|
||||||
```
|
|
||||||
|
|
||||||
Supports L2 distance, inner product, and cosine distance
|
|
||||||
|
|
||||||
[](https://github.com/pgvector/pgvector/actions)
|
[](https://github.com/pgvector/pgvector/actions)
|
||||||
|
|
||||||
@@ -17,6 +11,7 @@ Supports L2 distance, inner product, and cosine distance
|
|||||||
Compile and install the extension (supports Postgres 11+)
|
Compile and install the extension (supports Postgres 11+)
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
|
cd /tmp
|
||||||
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
|
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
|
||||||
cd pgvector
|
cd pgvector
|
||||||
make
|
make
|
||||||
@@ -29,41 +24,88 @@ Then load it in databases where you want to use it
|
|||||||
CREATE EXTENSION vector;
|
CREATE EXTENSION vector;
|
||||||
```
|
```
|
||||||
|
|
||||||
You can also install it with [Docker](#docker), [Homebrew](#homebrew), [PGXN](#pgxn), or [conda-forge](#conda-forge)
|
See the [installation notes](#installation-notes) if you run into issues
|
||||||
|
|
||||||
|
You can also install it with [Docker](#docker), [Homebrew](#homebrew), [PGXN](#pgxn), [Yum](#yum), or [conda-forge](#conda-forge)
|
||||||
|
|
||||||
## Getting Started
|
## Getting Started
|
||||||
|
|
||||||
Create a vector column with 3 dimensions
|
Create a vector column with 3 dimensions
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE items (embedding vector(3));
|
CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3));
|
||||||
```
|
```
|
||||||
|
|
||||||
Insert values
|
Insert values
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
INSERT INTO items VALUES ('[1,2,3]'), ('[4,5,6]');
|
INSERT INTO items (embedding) VALUES ('[1,2,3]'), ('[4,5,6]');
|
||||||
```
|
```
|
||||||
|
|
||||||
Get the nearest neighbor by L2 distance
|
Get the nearest neighbors by L2 distance
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 1;
|
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
```
|
```
|
||||||
|
|
||||||
Also supports inner product (`<#>`) and cosine distance (`<=>`)
|
Also supports inner product (`<#>`) and cosine distance (`<=>`)
|
||||||
|
|
||||||
Note: `<#>` returns the negative inner product since Postgres only supports `ASC` order index scans on operators
|
Note: `<#>` returns the negative inner product since Postgres only supports `ASC` order index scans on operators
|
||||||
|
|
||||||
## Querying
|
## Storing
|
||||||
|
|
||||||
Use a `SELECT` clause to get the distance
|
Create a new table with a vector column
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT embedding <-> '[3,1,2]' AS distance FROM items;
|
CREATE TABLE items (id bigserial PRIMARY KEY, embedding vector(3));
|
||||||
```
|
```
|
||||||
|
|
||||||
Use a `WHERE` clause to get rows within a certain distance
|
Or add a vector column to an existing table
|
||||||
|
|
||||||
|
```sql
|
||||||
|
ALTER TABLE items ADD COLUMN embedding vector(3);
|
||||||
|
```
|
||||||
|
|
||||||
|
Insert vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
INSERT INTO items (embedding) VALUES ('[1,2,3]'), ('[4,5,6]');
|
||||||
|
```
|
||||||
|
|
||||||
|
Upsert vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
INSERT INTO items (id, embedding) VALUES (1, '[1,2,3]'), (2, '[4,5,6]')
|
||||||
|
ON CONFLICT (id) DO UPDATE SET embedding = EXCLUDED.embedding;
|
||||||
|
```
|
||||||
|
|
||||||
|
Update vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
UPDATE items SET embedding = '[1,2,3]' WHERE id = 1;
|
||||||
|
```
|
||||||
|
|
||||||
|
Delete vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
DELETE FROM items WHERE id = 1;
|
||||||
|
```
|
||||||
|
|
||||||
|
## Querying
|
||||||
|
|
||||||
|
Get the nearest neighbors to a vector
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
Get the nearest neighbors to a row
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT * FROM items WHERE id != 1 ORDER BY embedding <-> (SELECT embedding FROM items WHERE id = 1) LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
Get rows within a certain distance
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
||||||
@@ -71,48 +113,77 @@ SELECT * FROM items WHERE embedding <-> '[3,1,2]' < 5;
|
|||||||
|
|
||||||
Note: Combine with `ORDER BY` and `LIMIT` to use an index
|
Note: Combine with `ORDER BY` and `LIMIT` to use an index
|
||||||
|
|
||||||
Get the average of vectors
|
#### Distances
|
||||||
|
|
||||||
|
Get the distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT embedding <-> '[3,1,2]' AS distance FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
For inner product, multiply by -1 (since `<#>` returns the negative inner product)
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT -1 * (embedding <#> '[3,1,2]') AS inner_product FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
For cosine similarity, use 1 - cosine distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT 1 - (embedding <=> '[3,1,2]') AS cosine_similarity FROM items;
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Averaging
|
||||||
|
|
||||||
|
Average vectors
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SELECT AVG(embedding) FROM items;
|
SELECT AVG(embedding) FROM items;
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Average groups of vectors
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT category_id, AVG(embedding) FROM items GROUP BY category_id;
|
||||||
|
```
|
||||||
|
|
||||||
## Indexing
|
## Indexing
|
||||||
|
|
||||||
Speed up queries with an approximate index. Add an index for each distance function you want to use.
|
By default, pgvector performs exact nearest neighbor search, which provides perfect recall.
|
||||||
|
|
||||||
|
You can add an index to use approximate nearest neighbor search, which trades some recall for performance. Unlike typical indexes, you will see different results for queries after adding an approximate index.
|
||||||
|
|
||||||
|
Two keys to achieving good recall are:
|
||||||
|
|
||||||
|
1. Create the index *after* the table has some data
|
||||||
|
2. Choose an appropriate number of lists (lower is better for recall, higher is better for speed)
|
||||||
|
|
||||||
|
A good place to start is:
|
||||||
|
|
||||||
|
- `rows / 1000` for up to 1M rows
|
||||||
|
- `sqrt(rows)` for over 1M rows
|
||||||
|
|
||||||
|
Add an index for each distance function you want to use.
|
||||||
|
|
||||||
L2 distance
|
L2 distance
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Inner product
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_ip_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Cosine distance
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_cosine_ops);
|
|
||||||
```
|
|
||||||
|
|
||||||
Indexes should be created after the table has some data for optimal clustering. Also, unlike typical indexes which only affect performance, you may see different results for queries after adding an approximate index. Vectors with up to 2,000 dimensions can be indexed.
|
|
||||||
|
|
||||||
### Index Options
|
|
||||||
|
|
||||||
Specify the number of inverted lists (100 by default)
|
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100);
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100);
|
||||||
```
|
```
|
||||||
|
|
||||||
A lower value provides better recall at the cost of speed. A good place to start is:
|
Inner product
|
||||||
|
|
||||||
- `rows / 1000` for up to 1M rows
|
```sql
|
||||||
- `sqrt(rows)` for over 1M rows
|
CREATE INDEX ON items USING ivfflat (embedding vector_ip_ops) WITH (lists = 100);
|
||||||
|
```
|
||||||
|
|
||||||
|
Cosine distance
|
||||||
|
|
||||||
|
```sql
|
||||||
|
CREATE INDEX ON items USING ivfflat (embedding vector_cosine_ops) WITH (lists = 100);
|
||||||
|
```
|
||||||
|
|
||||||
|
Vectors with up to 2,000 dimensions can be indexed.
|
||||||
|
|
||||||
### Query Options
|
### Query Options
|
||||||
|
|
||||||
@@ -161,7 +232,7 @@ SELECT * FROM items WHERE category_id = 123 ORDER BY embedding <-> '[3,1,2]' LIM
|
|||||||
can be indexed with:
|
can be indexed with:
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WHERE (category_id = 123);
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 100) WHERE (category_id = 123);
|
||||||
```
|
```
|
||||||
|
|
||||||
To index many different values of `category_id`, consider [partitioning](https://www.postgresql.org/docs/current/ddl-partitioning.html) on `category_id`.
|
To index many different values of `category_id`, consider [partitioning](https://www.postgresql.org/docs/current/ddl-partitioning.html) on `category_id`.
|
||||||
@@ -172,24 +243,34 @@ CREATE TABLE items (embedding vector(3), category_id int) PARTITION BY LIST(cate
|
|||||||
|
|
||||||
## Performance
|
## Performance
|
||||||
|
|
||||||
|
Use `EXPLAIN ANALYZE` to debug performance.
|
||||||
|
|
||||||
|
```sql
|
||||||
|
EXPLAIN ANALYZE SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
### Exact Search
|
||||||
|
|
||||||
To speed up queries without an index, increase `max_parallel_workers_per_gather`.
|
To speed up queries without an index, increase `max_parallel_workers_per_gather`.
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
SET max_parallel_workers_per_gather = 4;
|
SET max_parallel_workers_per_gather = 4;
|
||||||
```
|
```
|
||||||
|
|
||||||
|
If vectors are normalized to length 1 (like [OpenAI embeddings](https://platform.openai.com/docs/guides/embeddings/which-distance-function-should-i-use)), use inner product for best performance.
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT * FROM items ORDER BY embedding <#> '[3,1,2]' LIMIT 5;
|
||||||
|
```
|
||||||
|
|
||||||
|
### Approximate Search
|
||||||
|
|
||||||
To speed up queries with an index, increase the number of inverted lists (at the expense of recall).
|
To speed up queries with an index, increase the number of inverted lists (at the expense of recall).
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 1000);
|
CREATE INDEX ON items USING ivfflat (embedding vector_l2_ops) WITH (lists = 1000);
|
||||||
```
|
```
|
||||||
|
|
||||||
Use `EXPLAIN ANALYZE` to debug performance.
|
|
||||||
|
|
||||||
```sql
|
|
||||||
EXPLAIN ANALYZE SELECT * FROM items ORDER BY embedding <-> '[3,1,2]' LIMIT 1;
|
|
||||||
```
|
|
||||||
|
|
||||||
## Languages
|
## Languages
|
||||||
|
|
||||||
Use pgvector from any language with a Postgres client. You can even generate and store vectors in one language and query them in another.
|
Use pgvector from any language with a Postgres client. You can even generate and store vectors in one language and query them in another.
|
||||||
@@ -260,6 +341,42 @@ Function | Description
|
|||||||
--- | ---
|
--- | ---
|
||||||
avg(vector) → vector | arithmetic mean
|
avg(vector) → vector | arithmetic mean
|
||||||
|
|
||||||
|
## Installation Notes
|
||||||
|
|
||||||
|
### Postgres Location
|
||||||
|
|
||||||
|
If your machine has multiple Postgres installations, specify the path to [pg_config](https://www.postgresql.org/docs/current/app-pgconfig.html) with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
export PG_CONFIG=/Applications/Postgres.app/Contents/Versions/latest/bin/pg_config
|
||||||
|
```
|
||||||
|
|
||||||
|
Then re-run the installation instructions (run `make clean` before `make` if needed)
|
||||||
|
|
||||||
|
### Missing Header
|
||||||
|
|
||||||
|
If compilation fails with `fatal error: postgres.h: No such file or directory`, make sure Postgres development files are installed on the server.
|
||||||
|
|
||||||
|
For Ubuntu and Debian, use:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo apt-get install postgresql-server-dev-15
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Replace `15` with your Postgres server version
|
||||||
|
|
||||||
|
### Windows
|
||||||
|
|
||||||
|
Support for Windows is currently experimental. Use `nmake` to build:
|
||||||
|
|
||||||
|
```cmd
|
||||||
|
set "PGROOT=C:\Program Files\PostgreSQL\15"
|
||||||
|
git clone --branch v0.4.1 https://github.com/pgvector/pgvector.git
|
||||||
|
cd pgvector
|
||||||
|
nmake /F Makefile.win
|
||||||
|
nmake /F Makefile.win install
|
||||||
|
```
|
||||||
|
|
||||||
## Additional Installation Methods
|
## Additional Installation Methods
|
||||||
|
|
||||||
### Docker
|
### Docker
|
||||||
@@ -296,6 +413,18 @@ Install from the [PostgreSQL Extension Network](https://pgxn.org/dist/vector) wi
|
|||||||
pgxn install vector
|
pgxn install vector
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Yum
|
||||||
|
|
||||||
|
RPM packages are available from the [PostgreSQL Yum Repository](https://yum.postgresql.org/). Follow the [setup instructions](https://www.postgresql.org/download/linux/redhat/) for your distribution and run:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo yum install pgvector_15
|
||||||
|
# or
|
||||||
|
sudo dnf install pgvector_15
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Replace `15` with your Postgres server version
|
||||||
|
|
||||||
### conda-forge
|
### conda-forge
|
||||||
|
|
||||||
With Conda Postgres, install from [conda-forge](https://anaconda.org/conda-forge/pgvector) with:
|
With Conda Postgres, install from [conda-forge](https://anaconda.org/conda-forge/pgvector) with:
|
||||||
|
|||||||
@@ -13,6 +13,7 @@
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
int ivfflat_probes;
|
int ivfflat_probes;
|
||||||
|
int ivfflat_bound;
|
||||||
static relopt_kind ivfflat_relopt_kind;
|
static relopt_kind ivfflat_relopt_kind;
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -32,6 +33,10 @@ _PG_init(void)
|
|||||||
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
||||||
"Valid range is 1..lists.", &ivfflat_probes,
|
"Valid range is 1..lists.", &ivfflat_probes,
|
||||||
1, 1, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
1, 1, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
||||||
|
|
||||||
|
DefineCustomIntVariable("ivfflat.bound", "Sets the max results from index (experimental)",
|
||||||
|
NULL, &ivfflat_bound,
|
||||||
|
0, 0, INT_MAX, PGC_USERSET, 0, NULL, NULL, NULL);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -103,13 +108,7 @@ ivfflatcostestimate(PlannerInfo *root, IndexPath *path, double loop_count,
|
|||||||
if (ratio > 1)
|
if (ratio > 1)
|
||||||
ratio = 1;
|
ratio = 1;
|
||||||
|
|
||||||
// cost estimates for parallel workers applied outside of amcostestimate
|
|
||||||
elog(INFO, "parallel_workers = %d, parallel aware = %d", path->path.parallel_workers, path->path.parallel_aware);
|
|
||||||
|
|
||||||
costs.indexTotalCost *= ratio;
|
costs.indexTotalCost *= ratio;
|
||||||
costs.numIndexPages *= ratio;
|
|
||||||
|
|
||||||
elog(INFO, "ivfflatcostestimate = %f", costs.indexTotalCost);
|
|
||||||
|
|
||||||
/* Startup cost and total cost are same */
|
/* Startup cost and total cost are same */
|
||||||
*indexStartupCost = costs.indexTotalCost;
|
*indexStartupCost = costs.indexTotalCost;
|
||||||
@@ -157,25 +156,6 @@ ivfflatvalidate(Oid opclassoid)
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
static Size
|
|
||||||
ivfflatestimateparallelscan()
|
|
||||||
{
|
|
||||||
elog(INFO, "ivfflatestimateparallelscan");
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
static void
|
|
||||||
ivfflatinitparallelscan(void *target)
|
|
||||||
{
|
|
||||||
elog(INFO, "ivfflatinitparallelscan");
|
|
||||||
}
|
|
||||||
|
|
||||||
static void
|
|
||||||
ivfflatparallelrescan(IndexScanDesc scan)
|
|
||||||
{
|
|
||||||
elog(INFO, "ivfflatparallelrescan");
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Define index handler
|
* Define index handler
|
||||||
*
|
*
|
||||||
@@ -203,7 +183,7 @@ ivfflathandler(PG_FUNCTION_ARGS)
|
|||||||
amroutine->amstorage = false;
|
amroutine->amstorage = false;
|
||||||
amroutine->amclusterable = false;
|
amroutine->amclusterable = false;
|
||||||
amroutine->ampredlocks = false;
|
amroutine->ampredlocks = false;
|
||||||
amroutine->amcanparallel = true;
|
amroutine->amcanparallel = false;
|
||||||
amroutine->amcaninclude = false;
|
amroutine->amcaninclude = false;
|
||||||
#if PG_VERSION_NUM >= 130000
|
#if PG_VERSION_NUM >= 130000
|
||||||
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
|
amroutine->amusemaintenanceworkmem = false; /* not used during VACUUM */
|
||||||
@@ -237,9 +217,9 @@ ivfflathandler(PG_FUNCTION_ARGS)
|
|||||||
amroutine->amrestrpos = NULL;
|
amroutine->amrestrpos = NULL;
|
||||||
|
|
||||||
/* Interface functions to support parallel index scans */
|
/* Interface functions to support parallel index scans */
|
||||||
amroutine->amestimateparallelscan = ivfflatestimateparallelscan;
|
amroutine->amestimateparallelscan = NULL;
|
||||||
amroutine->aminitparallelscan = ivfflatinitparallelscan;
|
amroutine->aminitparallelscan = NULL;
|
||||||
amroutine->amparallelrescan = ivfflatparallelrescan;
|
amroutine->amparallelrescan = NULL;
|
||||||
|
|
||||||
PG_RETURN_POINTER(amroutine);
|
PG_RETURN_POINTER(amroutine);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -78,6 +78,7 @@
|
|||||||
|
|
||||||
/* Variables */
|
/* Variables */
|
||||||
extern int ivfflat_probes;
|
extern int ivfflat_probes;
|
||||||
|
extern int ivfflat_bound;
|
||||||
|
|
||||||
/* Exported functions */
|
/* Exported functions */
|
||||||
PGDLLEXPORT void _PG_init(void);
|
PGDLLEXPORT void _PG_init(void);
|
||||||
|
|||||||
@@ -111,6 +111,7 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
Datum datum;
|
Datum datum;
|
||||||
bool isnull;
|
bool isnull;
|
||||||
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
||||||
|
double tuples = 0;
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
TupleTableSlot *slot = MakeSingleTupleTableSlot(so->tupdesc, &TTSOpsVirtual);
|
||||||
@@ -125,6 +126,10 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
*/
|
*/
|
||||||
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
||||||
|
|
||||||
|
/* Set the max number of results */
|
||||||
|
if (ivfflat_bound > 0)
|
||||||
|
tuplesort_set_bound(so->sortstate, ivfflat_bound);
|
||||||
|
|
||||||
/* Search closest probes lists */
|
/* Search closest probes lists */
|
||||||
while (!pairingheap_is_empty(so->listQueue))
|
while (!pairingheap_is_empty(so->listQueue))
|
||||||
{
|
{
|
||||||
@@ -159,6 +164,8 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
ExecStoreVirtualTuple(slot);
|
ExecStoreVirtualTuple(slot);
|
||||||
|
|
||||||
tuplesort_puttupleslot(so->sortstate, slot);
|
tuplesort_puttupleslot(so->sortstate, slot);
|
||||||
|
|
||||||
|
tuples++;
|
||||||
}
|
}
|
||||||
|
|
||||||
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
searchPage = IvfflatPageGetOpaque(page)->nextblkno;
|
||||||
@@ -167,6 +174,13 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* TODO Scan more lists */
|
||||||
|
if (tuples < 100)
|
||||||
|
ereport(DEBUG1,
|
||||||
|
(errmsg("index scan found few tuples"),
|
||||||
|
errdetail("index may have been created without data or lists is too high"),
|
||||||
|
errhint("recreate the index and possibly decrease lists")));
|
||||||
|
|
||||||
tuplesort_performsort(so->sortstate);
|
tuplesort_performsort(so->sortstate);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
11
src/vector.c
11
src/vector.c
@@ -467,6 +467,7 @@ l2_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
diff = ax[i] - bx[i];
|
diff = ax[i] - bx[i];
|
||||||
@@ -493,6 +494,7 @@ vector_l2_squared_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
diff = ax[i] - bx[i];
|
diff = ax[i] - bx[i];
|
||||||
@@ -517,6 +519,7 @@ inner_product(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
|
|
||||||
@@ -538,6 +541,7 @@ vector_negative_inner_product(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
|
|
||||||
@@ -561,6 +565,7 @@ cosine_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
{
|
{
|
||||||
distance += ax[i] * bx[i];
|
distance += ax[i] * bx[i];
|
||||||
@@ -587,6 +592,7 @@ vector_spherical_distance(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
CheckDims(a, b);
|
CheckDims(a, b);
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
distance += a->x[i] * b->x[i];
|
distance += a->x[i] * b->x[i];
|
||||||
|
|
||||||
@@ -622,6 +628,7 @@ vector_norm(PG_FUNCTION_ARGS)
|
|||||||
float *ax = a->x;
|
float *ax = a->x;
|
||||||
double norm = 0.0;
|
double norm = 0.0;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0; i < a->dim; i++)
|
for (int i = 0; i < a->dim; i++)
|
||||||
norm += ax[i] * ax[i];
|
norm += ax[i] * ax[i];
|
||||||
|
|
||||||
@@ -646,6 +653,8 @@ vector_add(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
result = InitVector(a->dim);
|
result = InitVector(a->dim);
|
||||||
rx = result->x;
|
rx = result->x;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0, imax = a->dim; i < imax; i++)
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
rx[i] = ax[i] + bx[i];
|
rx[i] = ax[i] + bx[i];
|
||||||
|
|
||||||
@@ -670,6 +679,8 @@ vector_sub(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
result = InitVector(a->dim);
|
result = InitVector(a->dim);
|
||||||
rx = result->x;
|
rx = result->x;
|
||||||
|
|
||||||
|
/* Auto-vectorized */
|
||||||
for (int i = 0, imax = a->dim; i < imax; i++)
|
for (int i = 0, imax = a->dim; i < imax; i++)
|
||||||
rx[i] = ax[i] - bx[i];
|
rx[i] = ax[i] - bx[i];
|
||||||
|
|
||||||
|
|||||||
@@ -1,15 +0,0 @@
|
|||||||
-- SET force_parallel_mode = on;
|
|
||||||
SET parallel_setup_cost = 10;
|
|
||||||
SET parallel_tuple_cost = 0.001;
|
|
||||||
SET min_parallel_table_scan_size = 1;
|
|
||||||
SET min_parallel_index_scan_size = 1;
|
|
||||||
|
|
||||||
CREATE TABLE t (id integer, val vector(3));
|
|
||||||
INSERT INTO t (id, val) SELECT n, ARRAY[random(), random(), random()] FROM generate_series(1,1000000) n;
|
|
||||||
CREATE INDEX ON t USING ivfflat (val) WITH (lists = 10);
|
|
||||||
SET ivfflat.probes = 2;
|
|
||||||
|
|
||||||
EXPLAIN SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
|
|
||||||
SELECT * FROM t ORDER BY val <-> '[0.5,0.5,0.5]' LIMIT 5;
|
|
||||||
|
|
||||||
DROP TABLE t;
|
|
||||||
Reference in New Issue
Block a user