mirror of
https://github.com/pgvector/pgvector.git
synced 2026-07-23 04:20:56 +08:00
Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7c6694e0ef |
@@ -1,8 +1,3 @@
|
|||||||
## 0.2.6 (unreleased)
|
|
||||||
|
|
||||||
- Switched to mini-batch k-means
|
|
||||||
- Improved performance of index creation for Postgres < 12
|
|
||||||
|
|
||||||
## 0.2.5 (2022-02-11)
|
## 0.2.5 (2022-02-11)
|
||||||
|
|
||||||
- Reduced memory usage during index creation
|
- Reduced memory usage during index creation
|
||||||
|
|||||||
@@ -119,9 +119,10 @@ SELECT phase, tuples_done, tuples_total FROM pg_stat_progress_create_index;
|
|||||||
The phases are:
|
The phases are:
|
||||||
|
|
||||||
1. `initializing`
|
1. `initializing`
|
||||||
2. `performing k-means`
|
2. `sampling table`
|
||||||
3. `sorting tuples`
|
3. `performing k-means`
|
||||||
4. `loading tuples`
|
4. `sorting tuples`
|
||||||
|
5. `loading tuples`
|
||||||
|
|
||||||
Note: `tuples_done` and `tuples_total` are only populated during the `loading tuples` phase
|
Note: `tuples_done` and `tuples_total` are only populated during the `loading tuples` phase
|
||||||
|
|
||||||
@@ -263,7 +264,7 @@ Thanks to:
|
|||||||
|
|
||||||
- [PASE: PostgreSQL Ultra-High-Dimensional Approximate Nearest Neighbor Search Extension](https://dl.acm.org/doi/pdf/10.1145/3318464.3386131)
|
- [PASE: PostgreSQL Ultra-High-Dimensional Approximate Nearest Neighbor Search Extension](https://dl.acm.org/doi/pdf/10.1145/3318464.3386131)
|
||||||
- [Faiss: A Library for Efficient Similarity Search and Clustering of Dense Vectors](https://github.com/facebookresearch/faiss)
|
- [Faiss: A Library for Efficient Similarity Search and Clustering of Dense Vectors](https://github.com/facebookresearch/faiss)
|
||||||
- [Web-Scale k-means Clustering](https://www.eecs.tufts.edu/~dsculley/papers/fastkmeans.pdf)
|
- [Using the Triangle Inequality to Accelerate k-means](https://www.aaai.org/Papers/ICML/2003/ICML03-022.pdf)
|
||||||
- [k-means++: The Advantage of Careful Seeding](https://theory.stanford.edu/~sergei/papers/kMeansPP-soda.pdf)
|
- [k-means++: The Advantage of Careful Seeding](https://theory.stanford.edu/~sergei/papers/kMeansPP-soda.pdf)
|
||||||
- [Concept Decompositions for Large Sparse Text Data using Clustering](https://www.cs.utexas.edu/users/inderjit/public_papers/concept_mlj.pdf)
|
- [Concept Decompositions for Large Sparse Text Data using Clustering](https://www.cs.utexas.edu/users/inderjit/public_papers/concept_mlj.pdf)
|
||||||
|
|
||||||
|
|||||||
196
src/ivfbuild.c
196
src/ivfbuild.c
@@ -42,6 +42,87 @@
|
|||||||
#define UpdateProgress(index, val) ((void)val)
|
#define UpdateProgress(index, val) ((void)val)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Callback for sampling
|
||||||
|
*/
|
||||||
|
static void
|
||||||
|
SampleCallback(Relation index, CALLBACK_ITEM_POINTER, Datum *values,
|
||||||
|
bool *isnull, bool tupleIsAlive, void *state)
|
||||||
|
{
|
||||||
|
IvfflatBuildState *buildstate = (IvfflatBuildState *) state;
|
||||||
|
VectorArray samples = buildstate->samples;
|
||||||
|
int targsamples = samples->maxlen;
|
||||||
|
Datum value = values[0];
|
||||||
|
|
||||||
|
/* Skip nulls */
|
||||||
|
if (isnull[0])
|
||||||
|
return;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Normalize with KMEANS_NORM_PROC since spherical distance function
|
||||||
|
* expects unit vectors
|
||||||
|
*/
|
||||||
|
if (buildstate->kmeansnormprocinfo != NULL)
|
||||||
|
{
|
||||||
|
if (!IvfflatNormValue(buildstate->kmeansnormprocinfo, buildstate->collation, &value, buildstate->normvec))
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (samples->length < targsamples)
|
||||||
|
{
|
||||||
|
VectorArraySet(samples, samples->length, DatumGetVector(value));
|
||||||
|
samples->length++;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
if (buildstate->rowstoskip < 0)
|
||||||
|
buildstate->rowstoskip = reservoir_get_next_S(&buildstate->rstate, samples->length, targsamples);
|
||||||
|
|
||||||
|
if (buildstate->rowstoskip <= 0)
|
||||||
|
{
|
||||||
|
int k = (int) (targsamples * sampler_random_fract(buildstate->rstate.randstate));
|
||||||
|
|
||||||
|
Assert(k >= 0 && k < targsamples);
|
||||||
|
VectorArraySet(samples, k, DatumGetVector(value));
|
||||||
|
}
|
||||||
|
|
||||||
|
buildstate->rowstoskip -= 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Sample rows with same logic as ANALYZE
|
||||||
|
*/
|
||||||
|
static void
|
||||||
|
SampleRows(IvfflatBuildState * buildstate)
|
||||||
|
{
|
||||||
|
int targsamples = buildstate->samples->maxlen;
|
||||||
|
BlockNumber totalblocks = RelationGetNumberOfBlocks(buildstate->heap);
|
||||||
|
|
||||||
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_SAMPLE);
|
||||||
|
|
||||||
|
buildstate->rowstoskip = -1;
|
||||||
|
|
||||||
|
BlockSampler_Init(&buildstate->bs, totalblocks, targsamples, random());
|
||||||
|
|
||||||
|
reservoir_init_selection_state(&buildstate->rstate, targsamples);
|
||||||
|
while (BlockSampler_HasMore(&buildstate->bs))
|
||||||
|
{
|
||||||
|
BlockNumber targblock = BlockSampler_Next(&buildstate->bs);
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM >= 120000
|
||||||
|
table_index_build_range_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
false, true, true, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
||||||
|
#elif PG_VERSION_NUM >= 110000
|
||||||
|
IndexBuildHeapRangeScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, true, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
||||||
|
#else
|
||||||
|
IndexBuildHeapRangeScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
|
true, true, targblock, 1, SampleCallback, (void *) buildstate);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Callback for table_index_build_scan
|
* Callback for table_index_build_scan
|
||||||
*/
|
*/
|
||||||
@@ -86,18 +167,18 @@ BuildCallback(Relation index, CALLBACK_ITEM_POINTER, Datum *values,
|
|||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
buildstate->inertia += minDistance;
|
buildstate->inertia += minDistance;
|
||||||
buildstate->listSums[closestCenter] += minDistance;
|
|
||||||
buildstate->listCounts[closestCenter]++;
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* Create a virtual tuple */
|
/* Create a virtual tuple */
|
||||||
ExecClearTuple(slot);
|
ExecClearTuple(slot);
|
||||||
slot->tts_values[0] = Int32GetDatum(closestCenter);
|
slot->tts_values[0] = Int32GetDatum(closestCenter);
|
||||||
slot->tts_isnull[0] = false;
|
slot->tts_isnull[0] = false;
|
||||||
slot->tts_values[1] = PointerGetDatum(tid);
|
slot->tts_values[1] = Int32GetDatum(ItemPointerGetBlockNumberNoCheck(tid));
|
||||||
slot->tts_isnull[1] = false;
|
slot->tts_isnull[1] = false;
|
||||||
slot->tts_values[2] = value;
|
slot->tts_values[2] = Int32GetDatum(ItemPointerGetOffsetNumberNoCheck(tid));
|
||||||
slot->tts_isnull[2] = false;
|
slot->tts_isnull[2] = false;
|
||||||
|
slot->tts_values[3] = value;
|
||||||
|
slot->tts_isnull[3] = false;
|
||||||
ExecStoreVirtualTuple(slot);
|
ExecStoreVirtualTuple(slot);
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -119,6 +200,8 @@ GetNextTuple(Tuplesortstate *sortstate, TupleDesc tupdesc, TupleTableSlot *slot,
|
|||||||
{
|
{
|
||||||
Datum value;
|
Datum value;
|
||||||
bool isnull;
|
bool isnull;
|
||||||
|
int tupblk;
|
||||||
|
int tupoff;
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 100000
|
#if PG_VERSION_NUM >= 100000
|
||||||
if (tuplesort_gettupleslot(sortstate, true, false, slot, NULL))
|
if (tuplesort_gettupleslot(sortstate, true, false, slot, NULL))
|
||||||
@@ -127,11 +210,13 @@ GetNextTuple(Tuplesortstate *sortstate, TupleDesc tupdesc, TupleTableSlot *slot,
|
|||||||
#endif
|
#endif
|
||||||
{
|
{
|
||||||
*list = DatumGetInt32(slot_getattr(slot, 1, &isnull));
|
*list = DatumGetInt32(slot_getattr(slot, 1, &isnull));
|
||||||
value = slot_getattr(slot, 3, &isnull);
|
tupblk = DatumGetInt32(slot_getattr(slot, 2, &isnull));
|
||||||
|
tupoff = DatumGetInt32(slot_getattr(slot, 3, &isnull));
|
||||||
|
value = slot_getattr(slot, 4, &isnull);
|
||||||
|
|
||||||
/* Form the index tuple */
|
/* Form the index tuple */
|
||||||
*itup = index_form_tuple(tupdesc, &value, &isnull);
|
*itup = index_form_tuple(tupdesc, &value, &isnull);
|
||||||
(*itup)->t_tid = *((ItemPointer) DatumGetPointer(slot_getattr(slot, 2, &isnull)));
|
ItemPointerSet(&(*itup)->t_tid, tupblk, tupoff);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
*list = -1;
|
*list = -1;
|
||||||
@@ -240,16 +325,17 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
|
|||||||
|
|
||||||
/* Create tuple description for sorting */
|
/* Create tuple description for sorting */
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
buildstate->tupdesc = CreateTemplateTupleDesc(3);
|
buildstate->tupdesc = CreateTemplateTupleDesc(4);
|
||||||
#else
|
#else
|
||||||
buildstate->tupdesc = CreateTemplateTupleDesc(3, false);
|
buildstate->tupdesc = CreateTemplateTupleDesc(4, false);
|
||||||
#endif
|
#endif
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 1, "list", INT4OID, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 1, "list", INT4OID, -1, 0);
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 2, "tid", TIDOID, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 2, "blkno", INT4OID, -1, 0);
|
||||||
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 3, "offset", INT4OID, -1, 0);
|
||||||
#if PG_VERSION_NUM >= 110000
|
#if PG_VERSION_NUM >= 110000
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 3, "vector", RelationGetDescr(index)->attrs[0].atttypid, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 4, "vector", RelationGetDescr(index)->attrs[0].atttypid, -1, 0);
|
||||||
#else
|
#else
|
||||||
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 3, "vector", RelationGetDescr(index)->attrs[0]->atttypid, -1, 0);
|
TupleDescInitEntry(buildstate->tupdesc, (AttrNumber) 4, "vector", RelationGetDescr(index)->attrs[0]->atttypid, -1, 0);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
@@ -266,8 +352,6 @@ InitBuildState(IvfflatBuildState * buildstate, Relation heap, Relation index, In
|
|||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
buildstate->inertia = 0;
|
buildstate->inertia = 0;
|
||||||
buildstate->listSums = palloc0(sizeof(double) * buildstate->lists);
|
|
||||||
buildstate->listCounts = palloc0(sizeof(int) * buildstate->lists);
|
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -280,11 +364,38 @@ FreeBuildState(IvfflatBuildState * buildstate)
|
|||||||
pfree(buildstate->centers);
|
pfree(buildstate->centers);
|
||||||
pfree(buildstate->listInfo);
|
pfree(buildstate->listInfo);
|
||||||
pfree(buildstate->normvec);
|
pfree(buildstate->normvec);
|
||||||
|
}
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
/*
|
||||||
pfree(buildstate->listSums);
|
* Compute centers
|
||||||
pfree(buildstate->listCounts);
|
*/
|
||||||
#endif
|
static void
|
||||||
|
ComputeCenters(IvfflatBuildState * buildstate)
|
||||||
|
{
|
||||||
|
int numSamples;
|
||||||
|
|
||||||
|
/* Target 50 samples per list, with at least 10000 samples */
|
||||||
|
/* The number of samples has a large effect on index build time */
|
||||||
|
numSamples = buildstate->lists * 50;
|
||||||
|
if (numSamples < 10000)
|
||||||
|
numSamples = 10000;
|
||||||
|
|
||||||
|
/* Skip samples for unlogged table */
|
||||||
|
if (buildstate->heap == NULL)
|
||||||
|
numSamples = 1;
|
||||||
|
|
||||||
|
/* Sample rows */
|
||||||
|
/* TODO Ensure within maintenance_work_mem */
|
||||||
|
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
||||||
|
if (buildstate->heap != NULL)
|
||||||
|
SampleRows(buildstate);
|
||||||
|
|
||||||
|
/* Calculate centers */
|
||||||
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS);
|
||||||
|
IvfflatBench("k-means", IvfflatKmeans(buildstate->index, buildstate->samples, buildstate->centers));
|
||||||
|
|
||||||
|
/* Free samples before we allocate more memory */
|
||||||
|
pfree(buildstate->samples);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -360,51 +471,6 @@ CreateListPages(Relation index, VectorArray centers, int dimensions,
|
|||||||
pfree(list);
|
pfree(list);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
|
||||||
* Print k-means metrics
|
|
||||||
*/
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
|
||||||
static void
|
|
||||||
PrintKmeansMetrics(IvfflatBuildState * buildstate)
|
|
||||||
{
|
|
||||||
elog(INFO, "inertia: %.3e", buildstate->inertia);
|
|
||||||
|
|
||||||
/* Calculate Davies-Bouldin index */
|
|
||||||
if (buildstate->lists > 1)
|
|
||||||
{
|
|
||||||
double db = 0.0;
|
|
||||||
|
|
||||||
/* Calculate average distance */
|
|
||||||
for (int i = 0; i < buildstate->lists; i++)
|
|
||||||
{
|
|
||||||
if (buildstate->listCounts[i] > 0)
|
|
||||||
buildstate->listSums[i] /= buildstate->listCounts[i];
|
|
||||||
}
|
|
||||||
|
|
||||||
for (int i = 0; i < buildstate->lists; i++)
|
|
||||||
{
|
|
||||||
double max = 0.0;
|
|
||||||
double distance;
|
|
||||||
|
|
||||||
for (int j = 0; j < buildstate->lists; j++)
|
|
||||||
{
|
|
||||||
if (j == i)
|
|
||||||
continue;
|
|
||||||
|
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(buildstate->procinfo, buildstate->collation, PointerGetDatum(VectorArrayGet(buildstate->centers, i)), PointerGetDatum(VectorArrayGet(buildstate->centers, j))));
|
|
||||||
distance = (buildstate->listSums[i] + buildstate->listSums[j]) / distance;
|
|
||||||
|
|
||||||
if (distance > max)
|
|
||||||
max = distance;
|
|
||||||
}
|
|
||||||
db += max;
|
|
||||||
}
|
|
||||||
db /= buildstate->lists;
|
|
||||||
elog(INFO, "davies-bouldin: %.3f", db);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Create entry pages
|
* Create entry pages
|
||||||
*/
|
*/
|
||||||
@@ -443,7 +509,7 @@ CreateEntryPages(IvfflatBuildState * buildstate, ForkNumber forkNum)
|
|||||||
tuplesort_performsort(buildstate->sortstate);
|
tuplesort_performsort(buildstate->sortstate);
|
||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
PrintKmeansMetrics(buildstate);
|
elog(INFO, "inertia: %.3e", buildstate->inertia);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* Insert */
|
/* Insert */
|
||||||
@@ -460,9 +526,7 @@ BuildIndex(Relation heap, Relation index, IndexInfo *indexInfo,
|
|||||||
{
|
{
|
||||||
InitBuildState(buildstate, heap, index, indexInfo);
|
InitBuildState(buildstate, heap, index, indexInfo);
|
||||||
|
|
||||||
/* Perform k-means clustering */
|
ComputeCenters(buildstate);
|
||||||
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_IVFFLAT_PHASE_KMEANS);
|
|
||||||
IvfflatBench("k-means", IvfflatKmeans(buildstate));
|
|
||||||
|
|
||||||
/* Create pages */
|
/* Create pages */
|
||||||
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
CreateMetaPage(index, buildstate->dimensions, buildstate->lists, forkNum);
|
||||||
|
|||||||
@@ -13,6 +13,7 @@
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
int ivfflat_probes;
|
int ivfflat_probes;
|
||||||
|
int ivfflat_bound;
|
||||||
static relopt_kind ivfflat_relopt_kind;
|
static relopt_kind ivfflat_relopt_kind;
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -32,6 +33,10 @@ _PG_init(void)
|
|||||||
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
DefineCustomIntVariable("ivfflat.probes", "Sets the number of probes",
|
||||||
"Valid range is 1..lists.", &ivfflat_probes,
|
"Valid range is 1..lists.", &ivfflat_probes,
|
||||||
1, 1, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
1, 1, IVFFLAT_MAX_LISTS, PGC_USERSET, 0, NULL, NULL, NULL);
|
||||||
|
|
||||||
|
DefineCustomIntVariable("ivfflat.bound", "Sets the max results from index (experimental)",
|
||||||
|
NULL, &ivfflat_bound,
|
||||||
|
0, 0, INT_MAX, PGC_USERSET, 0, NULL, NULL, NULL);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -45,6 +50,8 @@ ivfflatbuildphasename(int64 phasenum)
|
|||||||
{
|
{
|
||||||
case PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE:
|
case PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE:
|
||||||
return "initializing";
|
return "initializing";
|
||||||
|
case PROGRESS_IVFFLAT_PHASE_SAMPLE:
|
||||||
|
return "sampling table";
|
||||||
case PROGRESS_IVFFLAT_PHASE_KMEANS:
|
case PROGRESS_IVFFLAT_PHASE_KMEANS:
|
||||||
return "performing k-means";
|
return "performing k-means";
|
||||||
case PROGRESS_IVFFLAT_PHASE_SORT:
|
case PROGRESS_IVFFLAT_PHASE_SORT:
|
||||||
|
|||||||
@@ -37,9 +37,10 @@
|
|||||||
|
|
||||||
/* Build phases */
|
/* Build phases */
|
||||||
/* PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE is 1 */
|
/* PROGRESS_CREATEIDX_SUBPHASE_INITIALIZE is 1 */
|
||||||
#define PROGRESS_IVFFLAT_PHASE_KMEANS 2
|
#define PROGRESS_IVFFLAT_PHASE_SAMPLE 2
|
||||||
#define PROGRESS_IVFFLAT_PHASE_SORT 3
|
#define PROGRESS_IVFFLAT_PHASE_KMEANS 3
|
||||||
#define PROGRESS_IVFFLAT_PHASE_LOAD 4
|
#define PROGRESS_IVFFLAT_PHASE_SORT 4
|
||||||
|
#define PROGRESS_IVFFLAT_PHASE_LOAD 5
|
||||||
|
|
||||||
#define IVFFLAT_LIST_SIZE(_dim) (offsetof(IvfflatListData, center) + VECTOR_SIZE(_dim))
|
#define IVFFLAT_LIST_SIZE(_dim) (offsetof(IvfflatListData, center) + VECTOR_SIZE(_dim))
|
||||||
|
|
||||||
@@ -61,8 +62,14 @@
|
|||||||
#define IvfflatBench(name, code) (code)
|
#define IvfflatBench(name, code) (code)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#if PG_VERSION_NUM < 100000
|
||||||
|
#define ItemPointerGetBlockNumberNoCheck ItemPointerGetBlockNumber
|
||||||
|
#define ItemPointerGetOffsetNumberNoCheck ItemPointerGetOffsetNumber
|
||||||
|
#endif
|
||||||
|
|
||||||
/* Variables */
|
/* Variables */
|
||||||
extern int ivfflat_probes;
|
extern int ivfflat_probes;
|
||||||
|
extern int ivfflat_bound;
|
||||||
|
|
||||||
typedef struct VectorArrayData
|
typedef struct VectorArrayData
|
||||||
{
|
{
|
||||||
@@ -116,8 +123,6 @@ typedef struct IvfflatBuildState
|
|||||||
|
|
||||||
#ifdef IVFFLAT_KMEANS_DEBUG
|
#ifdef IVFFLAT_KMEANS_DEBUG
|
||||||
double inertia;
|
double inertia;
|
||||||
double *listSums;
|
|
||||||
int *listCounts;
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* Sampling */
|
/* Sampling */
|
||||||
@@ -161,7 +166,6 @@ typedef IvfflatListData * IvfflatList;
|
|||||||
|
|
||||||
typedef struct IvfflatScanList
|
typedef struct IvfflatScanList
|
||||||
{
|
{
|
||||||
pairingheap_node ph_node;
|
|
||||||
BlockNumber startPage;
|
BlockNumber startPage;
|
||||||
double distance;
|
double distance;
|
||||||
} IvfflatScanList;
|
} IvfflatScanList;
|
||||||
@@ -183,8 +187,6 @@ typedef struct IvfflatScanOpaqueData
|
|||||||
FmgrInfo *normprocinfo;
|
FmgrInfo *normprocinfo;
|
||||||
Oid collation;
|
Oid collation;
|
||||||
|
|
||||||
/* Lists */
|
|
||||||
pairingheap *listQueue;
|
|
||||||
IvfflatScanList lists[FLEXIBLE_ARRAY_MEMBER]; /* must come last */
|
IvfflatScanList lists[FLEXIBLE_ARRAY_MEMBER]; /* must come last */
|
||||||
} IvfflatScanOpaqueData;
|
} IvfflatScanOpaqueData;
|
||||||
|
|
||||||
@@ -199,7 +201,7 @@ typedef IvfflatScanOpaqueData * IvfflatScanOpaque;
|
|||||||
void _PG_init(void);
|
void _PG_init(void);
|
||||||
VectorArray VectorArrayInit(int maxlen, int dimensions);
|
VectorArray VectorArrayInit(int maxlen, int dimensions);
|
||||||
void PrintVectorArray(char *msg, VectorArray arr);
|
void PrintVectorArray(char *msg, VectorArray arr);
|
||||||
void IvfflatKmeans(IvfflatBuildState * buildstate);
|
void IvfflatKmeans(Relation index, VectorArray samples, VectorArray centers);
|
||||||
FmgrInfo *IvfflatOptionalProcInfo(Relation rel, uint16 procnum);
|
FmgrInfo *IvfflatOptionalProcInfo(Relation rel, uint16 procnum);
|
||||||
bool IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result);
|
bool IvfflatNormValue(FmgrInfo *procinfo, Oid collation, Datum *value, Vector * result);
|
||||||
int IvfflatGetLists(Relation index);
|
int IvfflatGetLists(Relation index);
|
||||||
|
|||||||
466
src/ivfkmeans.c
466
src/ivfkmeans.c
@@ -2,20 +2,8 @@
|
|||||||
|
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
|
|
||||||
#include "catalog/index.h"
|
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "storage/bufmgr.h"
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
#include "access/tableam.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 130000
|
|
||||||
#define CALLBACK_ITEM_POINTER ItemPointer tid
|
|
||||||
#else
|
|
||||||
#define CALLBACK_ITEM_POINTER HeapTuple hup
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Initialize with kmeans++
|
* Initialize with kmeans++
|
||||||
@@ -23,7 +11,7 @@
|
|||||||
* https://theory.stanford.edu/~sergei/papers/kMeansPP-soda.pdf
|
* https://theory.stanford.edu/~sergei/papers/kMeansPP-soda.pdf
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
InitCenters(Relation index, VectorArray samples, VectorArray centers)
|
InitCenters(Relation index, VectorArray samples, VectorArray centers, float *lowerBound)
|
||||||
{
|
{
|
||||||
FmgrInfo *procinfo;
|
FmgrInfo *procinfo;
|
||||||
Oid collation;
|
Oid collation;
|
||||||
@@ -47,7 +35,7 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
for (j = 0; j < numSamples; j++)
|
for (j = 0; j < numSamples; j++)
|
||||||
weight[j] = DBL_MAX;
|
weight[j] = DBL_MAX;
|
||||||
|
|
||||||
for (i = 0; i < numCenters - 1; i++)
|
for (i = 0; i < numCenters; i++)
|
||||||
{
|
{
|
||||||
CHECK_FOR_INTERRUPTS();
|
CHECK_FOR_INTERRUPTS();
|
||||||
|
|
||||||
@@ -61,6 +49,9 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
/* TODO Use triangle inequality to reduce distance calculations */
|
/* TODO Use triangle inequality to reduce distance calculations */
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, i))));
|
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, i))));
|
||||||
|
|
||||||
|
/* Set lower bound */
|
||||||
|
lowerBound[j * numCenters + i] = distance;
|
||||||
|
|
||||||
/* Use distance squared for weighted probability distribution */
|
/* Use distance squared for weighted probability distribution */
|
||||||
distance *= distance;
|
distance *= distance;
|
||||||
|
|
||||||
@@ -70,6 +61,10 @@ InitCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
sum += weight[j];
|
sum += weight[j];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Only compute lower bound on last iteration */
|
||||||
|
if (i + 1 == numCenters)
|
||||||
|
break;
|
||||||
|
|
||||||
/* Choose new center using weighted probability distribution. */
|
/* Choose new center using weighted probability distribution. */
|
||||||
choice = sum * (((double) random()) / MAX_RANDOM_VALUE);
|
choice = sum * (((double) random()) / MAX_RANDOM_VALUE);
|
||||||
for (j = 0; j < numSamples - 1; j++)
|
for (j = 0; j < numSamples - 1; j++)
|
||||||
@@ -161,202 +156,299 @@ QuickCenters(Relation index, VectorArray samples, VectorArray centers)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Callback for sampling
|
* Use Elkan for performance. This requires distance function to satisfy triangle inequality.
|
||||||
*/
|
|
||||||
static void
|
|
||||||
SampleCallback(Relation index, CALLBACK_ITEM_POINTER, Datum *values,
|
|
||||||
bool *isnull, bool tupleIsAlive, void *state)
|
|
||||||
{
|
|
||||||
IvfflatBuildState *buildstate = (IvfflatBuildState *) state;
|
|
||||||
VectorArray samples = buildstate->samples;
|
|
||||||
int targsamples = samples->maxlen;
|
|
||||||
Datum value = values[0];
|
|
||||||
|
|
||||||
/* Skip nulls */
|
|
||||||
if (isnull[0])
|
|
||||||
return;
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Normalize with KMEANS_NORM_PROC since spherical distance function
|
|
||||||
* expects unit vectors
|
|
||||||
*/
|
|
||||||
if (buildstate->kmeansnormprocinfo != NULL)
|
|
||||||
{
|
|
||||||
if (!IvfflatNormValue(buildstate->kmeansnormprocinfo, buildstate->collation, &value, buildstate->normvec))
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (samples->length < targsamples)
|
|
||||||
{
|
|
||||||
VectorArraySet(samples, samples->length, DatumGetVector(value));
|
|
||||||
samples->length++;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (buildstate->rowstoskip < 0)
|
|
||||||
buildstate->rowstoskip = reservoir_get_next_S(&buildstate->rstate, samples->length, targsamples);
|
|
||||||
|
|
||||||
if (buildstate->rowstoskip <= 0)
|
|
||||||
{
|
|
||||||
int k = (int) (targsamples * sampler_random_fract(buildstate->rstate.randstate));
|
|
||||||
|
|
||||||
Assert(k >= 0 && k < targsamples);
|
|
||||||
VectorArraySet(samples, k, DatumGetVector(value));
|
|
||||||
}
|
|
||||||
|
|
||||||
buildstate->rowstoskip -= 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Sample rows with same logic as ANALYZE
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
SampleRows(IvfflatBuildState * buildstate)
|
|
||||||
{
|
|
||||||
int targsamples = buildstate->samples->maxlen;
|
|
||||||
BlockNumber totalblocks = RelationGetNumberOfBlocks(buildstate->heap);
|
|
||||||
|
|
||||||
buildstate->rowstoskip = -1;
|
|
||||||
buildstate->samples->length = 0;
|
|
||||||
|
|
||||||
BlockSampler_Init(&buildstate->bs, totalblocks, targsamples, random());
|
|
||||||
|
|
||||||
reservoir_init_selection_state(&buildstate->rstate, targsamples);
|
|
||||||
while (BlockSampler_HasMore(&buildstate->bs))
|
|
||||||
{
|
|
||||||
BlockNumber targblock = BlockSampler_Next(&buildstate->bs);
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
table_index_build_range_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
false, true, false, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
|
||||||
#elif PG_VERSION_NUM >= 110000
|
|
||||||
IndexBuildHeapRangeScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
false, true, targblock, 1, SampleCallback, (void *) buildstate, NULL);
|
|
||||||
#else
|
|
||||||
IndexBuildHeapRangeScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
|
||||||
false, true, targblock, 1, SampleCallback, (void *) buildstate);
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Use mini-batch k-means
|
|
||||||
*
|
*
|
||||||
* We use L2 distance for L2 (not L2 squared like index scan)
|
* We use L2 distance for L2 (not L2 squared like index scan)
|
||||||
* and angular distance for inner product and cosine distance
|
* and angular distance for inner product and cosine distance
|
||||||
*
|
*
|
||||||
* https://www.eecs.tufts.edu/~dsculley/papers/fastkmeans.pdf
|
* https://www.aaai.org/Papers/ICML/2003/ICML03-022.pdf
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
MiniBatchKmeans(IvfflatBuildState * buildstate)
|
ElkanKmeans(Relation index, VectorArray samples, VectorArray centers)
|
||||||
{
|
{
|
||||||
VectorArray centers = buildstate->centers;
|
FmgrInfo *procinfo;
|
||||||
VectorArray m = buildstate->samples;
|
FmgrInfo *normprocinfo;
|
||||||
int b = m->maxlen;
|
Oid collation;
|
||||||
int t = 20;
|
Vector *vec;
|
||||||
double distance;
|
Vector *newCenter;
|
||||||
double minDistance;
|
int iteration;
|
||||||
int closestCenter;
|
|
||||||
int i;
|
|
||||||
int j;
|
int j;
|
||||||
int k;
|
int k;
|
||||||
Vector *c;
|
int dimensions = centers->dim;
|
||||||
Vector *x;
|
int numCenters = centers->maxlen;
|
||||||
int *v;
|
int numSamples = samples->length;
|
||||||
int *d;
|
VectorArray newCenters;
|
||||||
double eta;
|
int *centerCounts;
|
||||||
|
int *closestCenters;
|
||||||
|
float *lowerBound;
|
||||||
|
float *upperBound;
|
||||||
|
float *s;
|
||||||
|
float *halfcdist;
|
||||||
|
float *newcdist;
|
||||||
|
int changes;
|
||||||
|
double minDistance;
|
||||||
|
int closestCenter;
|
||||||
|
double distance;
|
||||||
|
bool rj;
|
||||||
|
bool rjreset;
|
||||||
|
double dxcx;
|
||||||
|
double dxc;
|
||||||
|
|
||||||
|
/* Calculate allocation sizes */
|
||||||
|
Size samplesSize = VECTOR_ARRAY_SIZE(samples->maxlen, samples->dim);
|
||||||
|
Size centersSize = VECTOR_ARRAY_SIZE(centers->maxlen, centers->dim);
|
||||||
|
Size newCentersSize = VECTOR_ARRAY_SIZE(numCenters, dimensions);
|
||||||
|
Size centerCountsSize = sizeof(int) * numCenters;
|
||||||
|
Size closestCentersSize = sizeof(int) * numSamples;
|
||||||
|
Size lowerBoundSize = sizeof(float) * numSamples * numCenters;
|
||||||
|
Size upperBoundSize = sizeof(float) * numSamples;
|
||||||
|
Size sSize = sizeof(float) * numCenters;
|
||||||
|
Size halfcdistSize = sizeof(float) * numCenters * numCenters;
|
||||||
|
Size newcdistSize = sizeof(float) * numCenters;
|
||||||
|
|
||||||
|
/* Calculate total size */
|
||||||
|
Size totalSize = samplesSize + centersSize + newCentersSize + centerCountsSize + closestCentersSize + lowerBoundSize + upperBoundSize + sSize + halfcdistSize + newcdistSize;
|
||||||
|
|
||||||
|
/* Check memory requirements */
|
||||||
|
/* Add one to error message to ceil */
|
||||||
|
if (totalSize / 1024 > maintenance_work_mem)
|
||||||
|
ereport(ERROR,
|
||||||
|
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
||||||
|
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
||||||
|
totalSize / (1024 * 1024) + 1, maintenance_work_mem / 1024)));
|
||||||
|
|
||||||
/* Set support functions */
|
/* Set support functions */
|
||||||
FmgrInfo *procinfo = index_getprocinfo(buildstate->index, 1, IVFFLAT_KMEANS_DISTANCE_PROC);
|
procinfo = index_getprocinfo(index, 1, IVFFLAT_KMEANS_DISTANCE_PROC);
|
||||||
FmgrInfo *normprocinfo = buildstate->kmeansnormprocinfo;
|
normprocinfo = IvfflatOptionalProcInfo(index, IVFFLAT_KMEANS_NORM_PROC);
|
||||||
Oid collation = buildstate->index->rd_indcollation[0];
|
collation = index->rd_indcollation[0];
|
||||||
|
|
||||||
|
/* Allocate space */
|
||||||
|
/* Use float instead of double to save memory */
|
||||||
|
centerCounts = palloc(centerCountsSize);
|
||||||
|
closestCenters = palloc(closestCentersSize);
|
||||||
|
lowerBound = palloc_extended(lowerBoundSize, MCXT_ALLOC_HUGE);
|
||||||
|
upperBound = palloc(upperBoundSize);
|
||||||
|
s = palloc(sSize);
|
||||||
|
halfcdist = palloc(halfcdistSize);
|
||||||
|
newcdist = palloc(newcdistSize);
|
||||||
|
|
||||||
|
newCenters = VectorArrayInit(numCenters, dimensions);
|
||||||
|
for (j = 0; j < numCenters; j++)
|
||||||
|
{
|
||||||
|
vec = VectorArrayGet(newCenters, j);
|
||||||
|
SET_VARSIZE(vec, VECTOR_SIZE(dimensions));
|
||||||
|
vec->dim = dimensions;
|
||||||
|
}
|
||||||
|
|
||||||
/* Pick initial centers */
|
/* Pick initial centers */
|
||||||
InitCenters(buildstate->index, buildstate->samples, buildstate->centers);
|
InitCenters(index, samples, centers, lowerBound);
|
||||||
|
|
||||||
v = palloc(sizeof(int) * centers->maxlen);
|
/* Assign each x to its closest initial center c(x) = argmin d(x,c) */
|
||||||
d = palloc(sizeof(int) * b);
|
for (j = 0; j < numSamples; j++)
|
||||||
|
{
|
||||||
|
minDistance = DBL_MAX;
|
||||||
|
closestCenter = -1;
|
||||||
|
|
||||||
for (int i = 0; i < centers->length; i++)
|
vec = VectorArrayGet(samples, j);
|
||||||
v[i] = 0;
|
|
||||||
|
|
||||||
for (i = 0; i < t; i++)
|
/* Find closest center */
|
||||||
|
for (k = 0; k < numCenters; k++)
|
||||||
|
{
|
||||||
|
/* TODO Use Lemma 1 in k-means++ initialization */
|
||||||
|
distance = lowerBound[j * numCenters + k];
|
||||||
|
|
||||||
|
if (distance < minDistance)
|
||||||
|
{
|
||||||
|
minDistance = distance;
|
||||||
|
closestCenter = k;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
upperBound[j] = minDistance;
|
||||||
|
closestCenters[j] = closestCenter;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Give 500 iterations to converge */
|
||||||
|
for (iteration = 0; iteration < 500; iteration++)
|
||||||
{
|
{
|
||||||
/* Can take a while, so ensure we can interrupt */
|
/* Can take a while, so ensure we can interrupt */
|
||||||
CHECK_FOR_INTERRUPTS();
|
CHECK_FOR_INTERRUPTS();
|
||||||
|
|
||||||
/* Get b examples picked randomly from X */
|
changes = 0;
|
||||||
SampleRows(buildstate);
|
|
||||||
|
|
||||||
/* Cache nearest center to x */
|
/* Step 1: For all centers, compute distance */
|
||||||
for (j = 0; j < m->length; j++)
|
for (j = 0; j < numCenters; j++)
|
||||||
|
{
|
||||||
|
vec = VectorArrayGet(centers, j);
|
||||||
|
|
||||||
|
for (k = j + 1; k < numCenters; k++)
|
||||||
|
{
|
||||||
|
distance = 0.5 * DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
||||||
|
halfcdist[j * numCenters + k] = distance;
|
||||||
|
halfcdist[k * numCenters + j] = distance;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* For all centers c, compute s(c) */
|
||||||
|
for (j = 0; j < numCenters; j++)
|
||||||
{
|
{
|
||||||
/* compute closest */
|
|
||||||
minDistance = DBL_MAX;
|
minDistance = DBL_MAX;
|
||||||
closestCenter = -1;
|
|
||||||
|
|
||||||
x = VectorArrayGet(m, j);
|
for (k = 0; k < numCenters; k++)
|
||||||
|
|
||||||
/* Find closest center */
|
|
||||||
for (k = 0; k < centers->length; k++)
|
|
||||||
{
|
{
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(x), PointerGetDatum(VectorArrayGet(centers, k))));
|
if (j == k)
|
||||||
|
continue;
|
||||||
|
|
||||||
|
distance = halfcdist[j * numCenters + k];
|
||||||
if (distance < minDistance)
|
if (distance < minDistance)
|
||||||
{
|
|
||||||
minDistance = distance;
|
minDistance = distance;
|
||||||
closestCenter = k;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
d[j] = closestCenter;
|
s[j] = minDistance;
|
||||||
}
|
}
|
||||||
|
|
||||||
for (j = 0; j < m->length; j++)
|
rjreset = iteration != 0;
|
||||||
|
|
||||||
|
for (j = 0; j < numSamples; j++)
|
||||||
{
|
{
|
||||||
x = VectorArrayGet(m, j);
|
/* Step 2: Identify all points x such that u(x) <= s(c(x)) */
|
||||||
|
if (upperBound[j] <= s[closestCenters[j]])
|
||||||
|
continue;
|
||||||
|
|
||||||
/* Get cached center for this x */
|
rj = rjreset;
|
||||||
c = VectorArrayGet(centers, d[j]);
|
|
||||||
|
|
||||||
/* Update per-center counts */
|
for (k = 0; k < numCenters; k++)
|
||||||
v[d[j]]++;
|
|
||||||
|
|
||||||
/* Get per-center learning rate */
|
|
||||||
eta = 1.0 / v[d[j]];
|
|
||||||
|
|
||||||
/* Take gradient step */
|
|
||||||
for (k = 0; k < c->dim; k++)
|
|
||||||
c->x[k] = (1 - eta) * c->x[k] + eta * x->x[k];
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Check for empty centers (likely duplicates) */
|
|
||||||
if (i == 0)
|
|
||||||
{
|
|
||||||
for (j = 0; j < centers->length; j++)
|
|
||||||
{
|
{
|
||||||
if (v[j] == 0)
|
/* Step 3: For all remaining points x and centers c */
|
||||||
{
|
if (k == closestCenters[j])
|
||||||
c = VectorArrayGet(centers, j);
|
continue;
|
||||||
|
|
||||||
|
if (upperBound[j] <= lowerBound[j * numCenters + k])
|
||||||
|
continue;
|
||||||
|
|
||||||
|
if (upperBound[j] <= halfcdist[closestCenters[j] * numCenters + k])
|
||||||
|
continue;
|
||||||
|
|
||||||
|
vec = VectorArrayGet(samples, j);
|
||||||
|
|
||||||
|
/* Step 3a */
|
||||||
|
if (rj)
|
||||||
|
{
|
||||||
|
dxcx = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, closestCenters[j]))));
|
||||||
|
|
||||||
|
/* d(x,c(x)) computed, which is a form of d(x,c) */
|
||||||
|
lowerBound[j * numCenters + closestCenters[j]] = dxcx;
|
||||||
|
upperBound[j] = dxcx;
|
||||||
|
|
||||||
|
rj = false;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
dxcx = upperBound[j];
|
||||||
|
|
||||||
|
/* Step 3b */
|
||||||
|
if (dxcx > lowerBound[j * numCenters + k] || dxcx > halfcdist[closestCenters[j] * numCenters + k])
|
||||||
|
{
|
||||||
|
dxc = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(vec), PointerGetDatum(VectorArrayGet(centers, k))));
|
||||||
|
|
||||||
|
/* d(x,c) calculated */
|
||||||
|
lowerBound[j * numCenters + k] = dxc;
|
||||||
|
|
||||||
|
if (dxc < dxcx)
|
||||||
|
{
|
||||||
|
closestCenters[j] = k;
|
||||||
|
|
||||||
|
/* c(x) changed */
|
||||||
|
upperBound[j] = dxc;
|
||||||
|
|
||||||
|
changes++;
|
||||||
|
}
|
||||||
|
|
||||||
/* TODO Handle empty centers properly */
|
|
||||||
for (k = 0; k < c->dim; k++)
|
|
||||||
c->x[k] = ((double) random()) / MAX_RANDOM_VALUE;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Normalize if needed */
|
/* Step 4: For each center c, let m(c) be mean of all points assigned */
|
||||||
if (normprocinfo != NULL)
|
for (j = 0; j < numCenters; j++)
|
||||||
{
|
{
|
||||||
for (j = 0; j < centers->length; j++)
|
vec = VectorArrayGet(newCenters, j);
|
||||||
ApplyNorm(normprocinfo, collation, VectorArrayGet(centers, j));
|
for (k = 0; k < dimensions; k++)
|
||||||
|
vec->x[k] = 0.0;
|
||||||
|
|
||||||
|
centerCounts[j] = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
for (j = 0; j < numSamples; j++)
|
||||||
|
{
|
||||||
|
vec = VectorArrayGet(samples, j);
|
||||||
|
closestCenter = closestCenters[j];
|
||||||
|
|
||||||
|
/* Increment sum and count of closest center */
|
||||||
|
newCenter = VectorArrayGet(newCenters, closestCenter);
|
||||||
|
for (k = 0; k < dimensions; k++)
|
||||||
|
newCenter->x[k] += vec->x[k];
|
||||||
|
|
||||||
|
centerCounts[closestCenter] += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (j = 0; j < numCenters; j++)
|
||||||
|
{
|
||||||
|
vec = VectorArrayGet(newCenters, j);
|
||||||
|
|
||||||
|
if (centerCounts[j] > 0)
|
||||||
|
{
|
||||||
|
for (k = 0; k < dimensions; k++)
|
||||||
|
vec->x[k] /= centerCounts[j];
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/* TODO Handle empty centers properly */
|
||||||
|
for (k = 0; k < dimensions; k++)
|
||||||
|
vec->x[k] = ((double) random()) / MAX_RANDOM_VALUE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Normalize if needed */
|
||||||
|
if (normprocinfo != NULL)
|
||||||
|
ApplyNorm(normprocinfo, collation, vec);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Step 5 */
|
||||||
|
for (j = 0; j < numCenters; j++)
|
||||||
|
newcdist[j] = DatumGetFloat8(FunctionCall2Coll(procinfo, collation, PointerGetDatum(VectorArrayGet(centers, j)), PointerGetDatum(VectorArrayGet(newCenters, j))));
|
||||||
|
|
||||||
|
for (j = 0; j < numSamples; j++)
|
||||||
|
{
|
||||||
|
for (k = 0; k < numCenters; k++)
|
||||||
|
{
|
||||||
|
distance = lowerBound[j * numCenters + k] - newcdist[k];
|
||||||
|
|
||||||
|
if (distance < 0)
|
||||||
|
distance = 0;
|
||||||
|
|
||||||
|
lowerBound[j * numCenters + k] = distance;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Step 6 */
|
||||||
|
/* We reset r(x) before Step 3 in the next iteration */
|
||||||
|
for (j = 0; j < numSamples; j++)
|
||||||
|
upperBound[j] += newcdist[closestCenters[j]];
|
||||||
|
|
||||||
|
/* Step 7 */
|
||||||
|
for (j = 0; j < numCenters; j++)
|
||||||
|
memcpy(VectorArrayGet(centers, j), VectorArrayGet(newCenters, j), VECTOR_SIZE(dimensions));
|
||||||
|
|
||||||
|
if (changes == 0 && iteration != 0)
|
||||||
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
pfree(v);
|
pfree(newCenters);
|
||||||
pfree(d);
|
pfree(centerCounts);
|
||||||
|
pfree(closestCenters);
|
||||||
|
pfree(lowerBound);
|
||||||
|
pfree(upperBound);
|
||||||
|
pfree(s);
|
||||||
|
pfree(halfcdist);
|
||||||
|
pfree(newcdist);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -399,48 +491,16 @@ CheckCenters(Relation index, VectorArray centers)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Perform k-means clustering
|
* Perform naive k-means centering
|
||||||
* We use spherical k-means for inner product and cosine
|
* We use spherical k-means for inner product and cosine
|
||||||
*/
|
*/
|
||||||
void
|
void
|
||||||
IvfflatKmeans(IvfflatBuildState * buildstate)
|
IvfflatKmeans(Relation index, VectorArray samples, VectorArray centers)
|
||||||
{
|
{
|
||||||
int numSamples;
|
if (samples->length <= centers->maxlen)
|
||||||
Size totalSize;
|
QuickCenters(index, samples, centers);
|
||||||
|
|
||||||
/* Target 10 samples per list, with at least 10000 samples */
|
|
||||||
/* The number of samples has a large effect on index build time */
|
|
||||||
numSamples = buildstate->lists * 10;
|
|
||||||
if (numSamples < 10000)
|
|
||||||
numSamples = 10000;
|
|
||||||
|
|
||||||
/* Skip samples for unlogged table */
|
|
||||||
if (buildstate->heap == NULL)
|
|
||||||
numSamples = 1;
|
|
||||||
|
|
||||||
/* Calculate total size */
|
|
||||||
totalSize = VECTOR_ARRAY_SIZE(numSamples, buildstate->dimensions);
|
|
||||||
|
|
||||||
/* Check memory requirements */
|
|
||||||
/* Add one to error message to ceil */
|
|
||||||
if (totalSize / 1024 > maintenance_work_mem)
|
|
||||||
ereport(ERROR,
|
|
||||||
(errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
|
|
||||||
errmsg("memory required is %zu MB, maintenance_work_mem is %d MB",
|
|
||||||
totalSize / (1024 * 1024) + 1, maintenance_work_mem / 1024)));
|
|
||||||
|
|
||||||
/* Sample rows */
|
|
||||||
buildstate->samples = VectorArrayInit(numSamples, buildstate->dimensions);
|
|
||||||
if (buildstate->heap != NULL)
|
|
||||||
SampleRows(buildstate);
|
|
||||||
|
|
||||||
if (buildstate->samples->length <= buildstate->centers->maxlen)
|
|
||||||
QuickCenters(buildstate->index, buildstate->samples, buildstate->centers);
|
|
||||||
else
|
else
|
||||||
MiniBatchKmeans(buildstate);
|
ElkanKmeans(index, samples, centers);
|
||||||
|
|
||||||
CheckCenters(buildstate->index, buildstate->centers);
|
CheckCenters(index, centers);
|
||||||
|
|
||||||
/* Free samples before we allocate more memory */
|
|
||||||
pfree(buildstate->samples);
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,7 +1,5 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include <float.h>
|
|
||||||
|
|
||||||
#include "access/relscan.h"
|
#include "access/relscan.h"
|
||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
@@ -19,12 +17,14 @@
|
|||||||
* Compare list distances
|
* Compare list distances
|
||||||
*/
|
*/
|
||||||
static int
|
static int
|
||||||
CompareLists(const pairingheap_node *a, const pairingheap_node *b, void *arg)
|
CompareLists(const void *a, const void *b)
|
||||||
{
|
{
|
||||||
if (((const IvfflatScanList *) a)->distance > ((const IvfflatScanList *) b)->distance)
|
double diff = (((IvfflatScanList *) a)->distance - ((IvfflatScanList *) b)->distance);
|
||||||
|
|
||||||
|
if (diff > 0)
|
||||||
return 1;
|
return 1;
|
||||||
|
|
||||||
if (((const IvfflatScanList *) a)->distance < ((const IvfflatScanList *) b)->distance)
|
if (diff < 0)
|
||||||
return -1;
|
return -1;
|
||||||
|
|
||||||
return 0;
|
return 0;
|
||||||
@@ -45,8 +45,6 @@ GetScanLists(IndexScanDesc scan, Datum value)
|
|||||||
int listCount = 0;
|
int listCount = 0;
|
||||||
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
IvfflatScanOpaque so = (IvfflatScanOpaque) scan->opaque;
|
||||||
double distance;
|
double distance;
|
||||||
IvfflatScanList *scanlist;
|
|
||||||
double maxDistance = DBL_MAX;
|
|
||||||
|
|
||||||
/* Search all list pages */
|
/* Search all list pages */
|
||||||
while (BlockNumberIsValid(nextblkno))
|
while (BlockNumberIsValid(nextblkno))
|
||||||
@@ -64,39 +62,22 @@ GetScanLists(IndexScanDesc scan, Datum value)
|
|||||||
/* Use procinfo from the index instead of scan key for performance */
|
/* Use procinfo from the index instead of scan key for performance */
|
||||||
distance = DatumGetFloat8(FunctionCall2Coll(so->procinfo, so->collation, PointerGetDatum(&list->center), value));
|
distance = DatumGetFloat8(FunctionCall2Coll(so->procinfo, so->collation, PointerGetDatum(&list->center), value));
|
||||||
|
|
||||||
if (listCount < so->probes)
|
so->lists[listCount].startPage = list->startPage;
|
||||||
{
|
so->lists[listCount].distance = distance;
|
||||||
scanlist = &so->lists[listCount];
|
listCount++;
|
||||||
scanlist->startPage = list->startPage;
|
|
||||||
scanlist->distance = distance;
|
|
||||||
listCount++;
|
|
||||||
|
|
||||||
/* Add to heap */
|
|
||||||
pairingheap_add(so->listQueue, &scanlist->ph_node);
|
|
||||||
|
|
||||||
/* Calculate max distance */
|
|
||||||
if (listCount == so->probes)
|
|
||||||
maxDistance = ((IvfflatScanList *) pairingheap_first(so->listQueue))->distance;
|
|
||||||
}
|
|
||||||
else if (distance < maxDistance)
|
|
||||||
{
|
|
||||||
/* Remove */
|
|
||||||
scanlist = (IvfflatScanList *) pairingheap_remove_first(so->listQueue);
|
|
||||||
|
|
||||||
/* Reuse */
|
|
||||||
scanlist->startPage = list->startPage;
|
|
||||||
scanlist->distance = distance;
|
|
||||||
pairingheap_add(so->listQueue, &scanlist->ph_node);
|
|
||||||
|
|
||||||
/* Update max distance */
|
|
||||||
maxDistance = ((IvfflatScanList *) pairingheap_first(so->listQueue))->distance;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
nextblkno = IvfflatPageGetOpaque(cpage)->nextblkno;
|
nextblkno = IvfflatPageGetOpaque(cpage)->nextblkno;
|
||||||
|
|
||||||
UnlockReleaseBuffer(cbuf);
|
UnlockReleaseBuffer(cbuf);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Sort by distance */
|
||||||
|
/* TODO Use heap for performance */
|
||||||
|
qsort(so->lists, listCount, sizeof(IvfflatScanList), CompareLists);
|
||||||
|
|
||||||
|
if (so->probes > listCount)
|
||||||
|
so->probes = listCount;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -114,6 +95,7 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
OffsetNumber maxoffno;
|
OffsetNumber maxoffno;
|
||||||
Datum datum;
|
Datum datum;
|
||||||
bool isnull;
|
bool isnull;
|
||||||
|
int i;
|
||||||
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
TupleDesc tupdesc = RelationGetDescr(scan->indexRelation);
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
@@ -129,10 +111,14 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
*/
|
*/
|
||||||
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
BufferAccessStrategy bas = GetAccessStrategy(BAS_BULKREAD);
|
||||||
|
|
||||||
|
/* Set the max number of results */
|
||||||
|
if (ivfflat_bound > 0)
|
||||||
|
tuplesort_set_bound(so->sortstate, ivfflat_bound);
|
||||||
|
|
||||||
/* Search closest probes lists */
|
/* Search closest probes lists */
|
||||||
while (!pairingheap_is_empty(so->listQueue))
|
for (i = 0; i < so->probes; i++)
|
||||||
{
|
{
|
||||||
searchPage = ((IvfflatScanList *) pairingheap_remove_first(so->listQueue))->startPage;
|
searchPage = so->lists[i].startPage;
|
||||||
|
|
||||||
/* Search all entry pages for list */
|
/* Search all entry pages for list */
|
||||||
while (BlockNumberIsValid(searchPage))
|
while (BlockNumberIsValid(searchPage))
|
||||||
@@ -156,10 +142,12 @@ GetScanItems(IndexScanDesc scan, Datum value)
|
|||||||
ExecClearTuple(slot);
|
ExecClearTuple(slot);
|
||||||
slot->tts_values[0] = FunctionCall2Coll(so->procinfo, so->collation, datum, value);
|
slot->tts_values[0] = FunctionCall2Coll(so->procinfo, so->collation, datum, value);
|
||||||
slot->tts_isnull[0] = false;
|
slot->tts_isnull[0] = false;
|
||||||
slot->tts_values[1] = PointerGetDatum(&itup->t_tid);
|
slot->tts_values[1] = Int32GetDatum((int) ItemPointerGetBlockNumberNoCheck(&itup->t_tid));
|
||||||
slot->tts_isnull[1] = false;
|
slot->tts_isnull[1] = false;
|
||||||
slot->tts_values[2] = Int32GetDatum((int) searchPage);
|
slot->tts_values[2] = Int32GetDatum((int) ItemPointerGetOffsetNumberNoCheck(&itup->t_tid));
|
||||||
slot->tts_isnull[2] = false;
|
slot->tts_isnull[2] = false;
|
||||||
|
slot->tts_values[3] = Int32GetDatum((int) searchPage);
|
||||||
|
slot->tts_isnull[3] = false;
|
||||||
ExecStoreVirtualTuple(slot);
|
ExecStoreVirtualTuple(slot);
|
||||||
|
|
||||||
tuplesort_puttupleslot(so->sortstate, slot);
|
tuplesort_puttupleslot(so->sortstate, slot);
|
||||||
@@ -187,18 +175,13 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
Oid sortOperators[] = {Float8LessOperator};
|
Oid sortOperators[] = {Float8LessOperator};
|
||||||
Oid sortCollations[] = {InvalidOid};
|
Oid sortCollations[] = {InvalidOid};
|
||||||
bool nullsFirstFlags[] = {false};
|
bool nullsFirstFlags[] = {false};
|
||||||
int probes = ivfflat_probes;
|
|
||||||
|
|
||||||
scan = RelationGetIndexScan(index, nkeys, norderbys);
|
scan = RelationGetIndexScan(index, nkeys, norderbys);
|
||||||
lists = IvfflatGetLists(scan->indexRelation);
|
lists = IvfflatGetLists(scan->indexRelation);
|
||||||
|
|
||||||
if (probes > lists)
|
so = (IvfflatScanOpaque) palloc(offsetof(IvfflatScanOpaqueData, lists) + lists * sizeof(IvfflatScanList));
|
||||||
probes = lists;
|
|
||||||
|
|
||||||
so = (IvfflatScanOpaque) palloc(offsetof(IvfflatScanOpaqueData, lists) + probes * sizeof(IvfflatScanList));
|
|
||||||
so->buf = InvalidBuffer;
|
so->buf = InvalidBuffer;
|
||||||
so->first = true;
|
so->first = true;
|
||||||
so->probes = probes;
|
|
||||||
|
|
||||||
/* Set support functions */
|
/* Set support functions */
|
||||||
so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
|
so->procinfo = index_getprocinfo(index, 1, IVFFLAT_DISTANCE_PROC);
|
||||||
@@ -207,13 +190,14 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
|
|
||||||
/* Create tuple description for sorting */
|
/* Create tuple description for sorting */
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
so->tupdesc = CreateTemplateTupleDesc(3);
|
so->tupdesc = CreateTemplateTupleDesc(4);
|
||||||
#else
|
#else
|
||||||
so->tupdesc = CreateTemplateTupleDesc(3, false);
|
so->tupdesc = CreateTemplateTupleDesc(4, false);
|
||||||
#endif
|
#endif
|
||||||
TupleDescInitEntry(so->tupdesc, (AttrNumber) 1, "distance", FLOAT8OID, -1, 0);
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 1, "distance", FLOAT8OID, -1, 0);
|
||||||
TupleDescInitEntry(so->tupdesc, (AttrNumber) 2, "tid", TIDOID, -1, 0);
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 2, "blkno", INT4OID, -1, 0);
|
||||||
TupleDescInitEntry(so->tupdesc, (AttrNumber) 3, "indexblkno", INT4OID, -1, 0);
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 3, "offset", INT4OID, -1, 0);
|
||||||
|
TupleDescInitEntry(so->tupdesc, (AttrNumber) 4, "indexblkno", INT4OID, -1, 0);
|
||||||
|
|
||||||
/* Prep sort */
|
/* Prep sort */
|
||||||
#if PG_VERSION_NUM >= 110000
|
#if PG_VERSION_NUM >= 110000
|
||||||
@@ -228,8 +212,6 @@ ivfflatbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
so->slot = MakeSingleTupleTableSlot(so->tupdesc);
|
so->slot = MakeSingleTupleTableSlot(so->tupdesc);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
so->listQueue = pairingheap_allocate(CompareLists, scan);
|
|
||||||
|
|
||||||
scan->opaque = so;
|
scan->opaque = so;
|
||||||
|
|
||||||
return scan;
|
return scan;
|
||||||
@@ -249,7 +231,7 @@ ivfflatrescan(IndexScanDesc scan, ScanKey keys, int nkeys, ScanKey orderbys, int
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
so->first = true;
|
so->first = true;
|
||||||
pairingheap_reset(so->listQueue);
|
so->probes = ivfflat_probes;
|
||||||
|
|
||||||
if (keys && scan->numberOfKeys > 0)
|
if (keys && scan->numberOfKeys > 0)
|
||||||
memmove(scan->keyData, keys, scan->numberOfKeys * sizeof(ScanKeyData));
|
memmove(scan->keyData, keys, scan->numberOfKeys * sizeof(ScanKeyData));
|
||||||
@@ -308,13 +290,14 @@ ivfflatgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
if (tuplesort_gettupleslot(so->sortstate, true, so->slot, NULL))
|
if (tuplesort_gettupleslot(so->sortstate, true, so->slot, NULL))
|
||||||
#endif
|
#endif
|
||||||
{
|
{
|
||||||
ItemPointer tid = (ItemPointer) DatumGetPointer(slot_getattr(so->slot, 2, &so->isnull));
|
BlockNumber blkno = DatumGetInt32(slot_getattr(so->slot, 2, &so->isnull));
|
||||||
BlockNumber indexblkno = DatumGetInt32(slot_getattr(so->slot, 3, &so->isnull));
|
OffsetNumber offset = DatumGetInt32(slot_getattr(so->slot, 3, &so->isnull));
|
||||||
|
BlockNumber indexblkno = DatumGetInt32(slot_getattr(so->slot, 4, &so->isnull));
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
scan->xs_heaptid = *tid;
|
ItemPointerSet(&scan->xs_heaptid, blkno, offset);
|
||||||
#else
|
#else
|
||||||
scan->xs_ctup.t_self = *tid;
|
ItemPointerSet(&scan->xs_ctup.t_self, blkno, offset);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
if (BufferIsValid(so->buf))
|
if (BufferIsValid(so->buf))
|
||||||
@@ -347,7 +330,6 @@ ivfflatendscan(IndexScanDesc scan)
|
|||||||
if (BufferIsValid(so->buf))
|
if (BufferIsValid(so->buf))
|
||||||
ReleaseBuffer(so->buf);
|
ReleaseBuffer(so->buf);
|
||||||
|
|
||||||
pairingheap_free(so->listQueue);
|
|
||||||
tuplesort_end(so->sortstate);
|
tuplesort_end(so->sortstate);
|
||||||
|
|
||||||
pfree(so);
|
pfree(so);
|
||||||
|
|||||||
@@ -2,16 +2,15 @@ use strict;
|
|||||||
use warnings;
|
use warnings;
|
||||||
use PostgresNode;
|
use PostgresNode;
|
||||||
use TestLib;
|
use TestLib;
|
||||||
use Test::More tests => 9;
|
use Test::More tests => 2;
|
||||||
|
|
||||||
my $node;
|
my $node;
|
||||||
my @queries = ();
|
my @queries = ();
|
||||||
my @expected;
|
my @expected = ();
|
||||||
my $limit = 20;
|
|
||||||
|
|
||||||
sub test_recall
|
sub test_recall
|
||||||
{
|
{
|
||||||
my ($probes, $min, $operator) = @_;
|
my ($probes, $min) = @_;
|
||||||
my $correct = 0;
|
my $correct = 0;
|
||||||
my $total = 0;
|
my $total = 0;
|
||||||
|
|
||||||
@@ -19,7 +18,7 @@ sub test_recall
|
|||||||
my $actual = $node->safe_psql("postgres", qq(
|
my $actual = $node->safe_psql("postgres", qq(
|
||||||
SET enable_seqscan = off;
|
SET enable_seqscan = off;
|
||||||
SET ivfflat.probes = $probes;
|
SET ivfflat.probes = $probes;
|
||||||
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
SELECT i FROM tst ORDER BY v <-> '$queries[$i]' LIMIT 10;
|
||||||
));
|
));
|
||||||
my @actual_ids = split("\n", $actual);
|
my @actual_ids = split("\n", $actual);
|
||||||
my %actual_set = map { $_ => 1 } @actual_ids;
|
my %actual_set = map { $_ => 1 } @actual_ids;
|
||||||
@@ -34,7 +33,7 @@ sub test_recall
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
cmp_ok($correct / $total, ">=", $min, $operator);
|
cmp_ok($correct / $total, ">=", $min);
|
||||||
}
|
}
|
||||||
|
|
||||||
# Initialize node
|
# Initialize node
|
||||||
@@ -57,32 +56,17 @@ for (1..20) {
|
|||||||
push(@queries, "[$r1,$r2,$r3]");
|
push(@queries, "[$r1,$r2,$r3]");
|
||||||
}
|
}
|
||||||
|
|
||||||
# Check each index type
|
# Get exact results
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
foreach (@queries) {
|
||||||
|
my $res = $node->safe_psql("postgres", "SELECT i FROM tst ORDER BY v <-> '$_' LIMIT 10;");
|
||||||
foreach (@operators) {
|
push(@expected, $res);
|
||||||
my $operator = $_;
|
|
||||||
|
|
||||||
# Get exact results
|
|
||||||
@expected = ();
|
|
||||||
foreach (@queries) {
|
|
||||||
my $res = $node->safe_psql("postgres", "SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;");
|
|
||||||
push(@expected, $res);
|
|
||||||
}
|
|
||||||
|
|
||||||
# Add index
|
|
||||||
my $opclass;
|
|
||||||
if ($operator == "<->") {
|
|
||||||
$opclass = "vector_l2_ops";
|
|
||||||
} elsif ($operator == "<#>") {
|
|
||||||
$opclass = "vector_ip_ops";
|
|
||||||
} else {
|
|
||||||
$opclass = "vector_cosine_ops";
|
|
||||||
}
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v $opclass);");
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
test_recall(1, 0.75, $operator);
|
|
||||||
test_recall(10, 0.95, $operator);
|
|
||||||
test_recall(100, 1.0, $operator);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Add index
|
||||||
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v);");
|
||||||
|
|
||||||
|
# Test approximate results
|
||||||
|
test_recall(1, 0.8);
|
||||||
|
|
||||||
|
# Test probes
|
||||||
|
test_recall(100, 1.0);
|
||||||
|
|||||||
@@ -1,45 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More tests => 60;
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (i int4 primary key, v vector(3));");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT i, ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
# Check each index type
|
|
||||||
my @operators = ("<->", "<#>", "<=>");
|
|
||||||
foreach (@operators) {
|
|
||||||
my $operator = $_;
|
|
||||||
|
|
||||||
# Add index
|
|
||||||
my $opclass;
|
|
||||||
if ($operator == "<->") {
|
|
||||||
$opclass = "vector_l2_ops";
|
|
||||||
} elsif ($operator == "<#>") {
|
|
||||||
$opclass = "vector_ip_ops";
|
|
||||||
} else {
|
|
||||||
$opclass = "vector_cosine_ops";
|
|
||||||
}
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX ON tst USING ivfflat (v $opclass);");
|
|
||||||
|
|
||||||
# Test 100% recall
|
|
||||||
for (1..20) {
|
|
||||||
my $i = int(rand() * 100000);
|
|
||||||
my $query = $node->safe_psql("postgres", "SELECT v FROM tst WHERE i = $i;");
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
|
||||||
SET enable_seqscan = off;
|
|
||||||
SELECT v FROM tst ORDER BY v <-> '$query' LIMIT 1;
|
|
||||||
));
|
|
||||||
is($res, $query);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,31 +0,0 @@
|
|||||||
use strict;
|
|
||||||
use warnings;
|
|
||||||
use PostgresNode;
|
|
||||||
use TestLib;
|
|
||||||
use Test::More tests => 3;
|
|
||||||
|
|
||||||
# Initialize node
|
|
||||||
my $node = get_new_node('node');
|
|
||||||
$node->init;
|
|
||||||
$node->start;
|
|
||||||
|
|
||||||
# Create table
|
|
||||||
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
|
||||||
$node->safe_psql("postgres", "CREATE TABLE tst (v vector(3));");
|
|
||||||
$node->safe_psql("postgres",
|
|
||||||
"INSERT INTO tst SELECT ARRAY[random(), random(), random()] FROM generate_series(1, 100000) i;"
|
|
||||||
);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX lists50 ON tst USING ivfflat (v) WITH (lists = 50);");
|
|
||||||
$node->safe_psql("postgres", "CREATE INDEX lists100 ON tst USING ivfflat (v) WITH (lists = 100);");
|
|
||||||
|
|
||||||
# Test prefers more lists
|
|
||||||
my $res = $node->safe_psql("postgres", "EXPLAIN SELECT v FROM tst ORDER BY v <-> '[0.5,0.5,0.5]' LIMIT 10;");
|
|
||||||
like($res, qr/lists100/);
|
|
||||||
unlike($res, qr/lists50/);
|
|
||||||
|
|
||||||
# Test errors with too much memory
|
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres",
|
|
||||||
"CREATE INDEX lists10000 ON tst USING ivfflat (v) WITH (lists = 10000);"
|
|
||||||
);
|
|
||||||
like($stderr, qr/memory required is/);
|
|
||||||
Reference in New Issue
Block a user