mirror of
https://github.com/pgvector/pgvector.git
synced 2026-07-24 21:02:40 +08:00
Compare commits
8 Commits
hnsw-paral
...
kill-prior
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
68c67be35a | ||
|
|
3f49343a79 | ||
|
|
d6ac7b93bb | ||
|
|
9ebec1529b | ||
|
|
77ff4c18f0 | ||
|
|
88dabaa41c | ||
|
|
1809ffa52b | ||
|
|
024f283ee8 |
55
src/hnsw.h
55
src/hnsw.h
@@ -4,7 +4,6 @@
|
|||||||
#include "postgres.h"
|
#include "postgres.h"
|
||||||
|
|
||||||
#include "access/generic_xlog.h"
|
#include "access/generic_xlog.h"
|
||||||
#include "access/parallel.h"
|
|
||||||
#include "access/reloptions.h"
|
#include "access/reloptions.h"
|
||||||
#include "nodes/execnodes.h"
|
#include "nodes/execnodes.h"
|
||||||
#include "port.h" /* for random() */
|
#include "port.h" /* for random() */
|
||||||
@@ -15,10 +14,6 @@
|
|||||||
#error "Requires PostgreSQL 11+"
|
#error "Requires PostgreSQL 11+"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 120000
|
|
||||||
#include "access/relscan.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define HNSW_MAX_DIM 2000
|
#define HNSW_MAX_DIM 2000
|
||||||
|
|
||||||
/* Support functions */
|
/* Support functions */
|
||||||
@@ -137,49 +132,6 @@ typedef struct HnswOptions
|
|||||||
int efConstruction; /* size of dynamic candidate list */
|
int efConstruction; /* size of dynamic candidate list */
|
||||||
} HnswOptions;
|
} HnswOptions;
|
||||||
|
|
||||||
typedef struct HnswSpool
|
|
||||||
{
|
|
||||||
Relation heap;
|
|
||||||
Relation index;
|
|
||||||
} HnswSpool;
|
|
||||||
|
|
||||||
typedef struct HnswShared
|
|
||||||
{
|
|
||||||
/* Immutable state */
|
|
||||||
Oid heaprelid;
|
|
||||||
Oid indexrelid;
|
|
||||||
bool isconcurrent;
|
|
||||||
int scantuplesortstates;
|
|
||||||
|
|
||||||
/* Worker progress */
|
|
||||||
ConditionVariable workersdonecv;
|
|
||||||
|
|
||||||
/* Mutex for mutable state */
|
|
||||||
slock_t mutex;
|
|
||||||
|
|
||||||
/* Mutable state */
|
|
||||||
int nparticipantsdone;
|
|
||||||
double reltuples;
|
|
||||||
double indtuples;
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM < 120000
|
|
||||||
ParallelHeapScanDescData heapdesc; /* must come last */
|
|
||||||
#endif
|
|
||||||
} HnswShared;
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
#define ParallelTableScanFromHnswShared(shared) \
|
|
||||||
(ParallelTableScanDesc) ((char *) (shared) + BUFFERALIGN(sizeof(HnswShared)))
|
|
||||||
#endif
|
|
||||||
|
|
||||||
typedef struct HnswLeader
|
|
||||||
{
|
|
||||||
ParallelContext *pcxt;
|
|
||||||
int nparticipanttuplesorts;
|
|
||||||
HnswShared *hnswshared;
|
|
||||||
Snapshot snapshot;
|
|
||||||
} HnswLeader;
|
|
||||||
|
|
||||||
typedef struct HnswBuildState
|
typedef struct HnswBuildState
|
||||||
{
|
{
|
||||||
/* Info */
|
/* Info */
|
||||||
@@ -213,9 +165,6 @@ typedef struct HnswBuildState
|
|||||||
|
|
||||||
/* Memory */
|
/* Memory */
|
||||||
MemoryContext tmpCtx;
|
MemoryContext tmpCtx;
|
||||||
|
|
||||||
/* Parallel builds */
|
|
||||||
HnswLeader *hnswleader;
|
|
||||||
} HnswBuildState;
|
} HnswBuildState;
|
||||||
|
|
||||||
typedef struct HnswMetaPageData
|
typedef struct HnswMetaPageData
|
||||||
@@ -270,6 +219,9 @@ typedef struct HnswScanOpaqueData
|
|||||||
{
|
{
|
||||||
bool first;
|
bool first;
|
||||||
Buffer buf;
|
Buffer buf;
|
||||||
|
ItemPointerData heaptid;
|
||||||
|
OffsetNumber offno;
|
||||||
|
int removedCount;
|
||||||
List *w;
|
List *w;
|
||||||
MemoryContext tmpCtx;
|
MemoryContext tmpCtx;
|
||||||
|
|
||||||
@@ -336,7 +288,6 @@ void HnswLoadElement(HnswElement element, float *distance, Datum *q, Relation i
|
|||||||
void HnswSetElementTuple(HnswElementTuple etup, HnswElement element);
|
void HnswSetElementTuple(HnswElementTuple etup, HnswElement element);
|
||||||
void HnswUpdateConnection(HnswElement element, HnswCandidate * hc, int m, int lc, int *updateIdx, Relation index, FmgrInfo *procinfo, Oid collation);
|
void HnswUpdateConnection(HnswElement element, HnswCandidate * hc, int m, int lc, int *updateIdx, Relation index, FmgrInfo *procinfo, Oid collation);
|
||||||
void HnswLoadNeighbors(HnswElement element, Relation index);
|
void HnswLoadNeighbors(HnswElement element, Relation index);
|
||||||
PGDLLEXPORT void HnswParallelBuildMain(dsm_segment *seg, shm_toc *toc);
|
|
||||||
|
|
||||||
/* Index access methods */
|
/* Index access methods */
|
||||||
IndexBuildResult *hnswbuild(Relation heap, Relation index, IndexInfo *indexInfo);
|
IndexBuildResult *hnswbuild(Relation heap, Relation index, IndexInfo *indexInfo);
|
||||||
|
|||||||
383
src/hnswbuild.c
383
src/hnswbuild.c
@@ -2,15 +2,12 @@
|
|||||||
|
|
||||||
#include <math.h>
|
#include <math.h>
|
||||||
|
|
||||||
#include "access/parallel.h"
|
|
||||||
#include "access/xact.h"
|
|
||||||
#include "catalog/index.h"
|
#include "catalog/index.h"
|
||||||
#include "hnsw.h"
|
#include "hnsw.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "lib/pairingheap.h"
|
#include "lib/pairingheap.h"
|
||||||
#include "nodes/pg_list.h"
|
#include "nodes/pg_list.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
#include "tcop/tcopprot.h"
|
|
||||||
#include "utils/memutils.h"
|
#include "utils/memutils.h"
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
#if PG_VERSION_NUM >= 140000
|
||||||
@@ -38,23 +35,6 @@
|
|||||||
#define UpdateProgress(index, val) ((void)val)
|
#define UpdateProgress(index, val) ((void)val)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
|
||||||
#include "utils/backend_status.h"
|
|
||||||
#include "utils/wait_event.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
#include "access/table.h"
|
|
||||||
#include "optimizer/optimizer.h"
|
|
||||||
#else
|
|
||||||
#include "access/heapam.h"
|
|
||||||
#include "optimizer/planner.h"
|
|
||||||
#include "pgstat.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define PARALLEL_KEY_HNSW_SHARED UINT64CONST(0xA000000000000001)
|
|
||||||
#define PARALLEL_KEY_QUERY_TEXT UINT64CONST(0xA000000000000002)
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Create the metapage
|
* Create the metapage
|
||||||
*/
|
*/
|
||||||
@@ -371,7 +351,6 @@ BuildCallback(Relation index, CALLBACK_ITEM_POINTER, Datum *values,
|
|||||||
|
|
||||||
oldCtx = MemoryContextSwitchTo(buildstate->tmpCtx);
|
oldCtx = MemoryContextSwitchTo(buildstate->tmpCtx);
|
||||||
|
|
||||||
/* TODO Fix progress for parallel builds */
|
|
||||||
if (HnswInsertTuple(buildstate->index, values, isnull, tid, buildstate->heap))
|
if (HnswInsertTuple(buildstate->index, values, isnull, tid, buildstate->heap))
|
||||||
UpdateProgress(PROGRESS_CREATEIDX_TUPLES_DONE, ++buildstate->indtuples);
|
UpdateProgress(PROGRESS_CREATEIDX_TUPLES_DONE, ++buildstate->indtuples);
|
||||||
|
|
||||||
@@ -469,8 +448,6 @@ InitBuildState(HnswBuildState * buildstate, Relation heap, Relation index, Index
|
|||||||
buildstate->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
buildstate->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
||||||
"Hnsw build temporary context",
|
"Hnsw build temporary context",
|
||||||
ALLOCSET_DEFAULT_SIZES);
|
ALLOCSET_DEFAULT_SIZES);
|
||||||
|
|
||||||
buildstate->hnswleader = NULL;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -483,373 +460,21 @@ FreeBuildState(HnswBuildState * buildstate)
|
|||||||
MemoryContextDelete(buildstate->tmpCtx);
|
MemoryContextDelete(buildstate->tmpCtx);
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
|
||||||
* Within leader, wait for end of heap scan
|
|
||||||
*/
|
|
||||||
static double
|
|
||||||
ParallelHeapScan(HnswBuildState * buildstate)
|
|
||||||
{
|
|
||||||
HnswShared *hnswshared = buildstate->hnswleader->hnswshared;
|
|
||||||
int nparticipanttuplesorts;
|
|
||||||
double reltuples;
|
|
||||||
|
|
||||||
nparticipanttuplesorts = buildstate->hnswleader->nparticipanttuplesorts;
|
|
||||||
for (;;)
|
|
||||||
{
|
|
||||||
SpinLockAcquire(&hnswshared->mutex);
|
|
||||||
if (hnswshared->nparticipantsdone == nparticipanttuplesorts)
|
|
||||||
{
|
|
||||||
buildstate->indtuples = hnswshared->indtuples;
|
|
||||||
reltuples = hnswshared->reltuples;
|
|
||||||
SpinLockRelease(&hnswshared->mutex);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
SpinLockRelease(&hnswshared->mutex);
|
|
||||||
|
|
||||||
ConditionVariableSleep(&hnswshared->workersdonecv,
|
|
||||||
WAIT_EVENT_PARALLEL_CREATE_INDEX_SCAN);
|
|
||||||
}
|
|
||||||
|
|
||||||
ConditionVariableCancelSleep();
|
|
||||||
|
|
||||||
return reltuples;
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Perform a worker's portion of a parallel insert
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswParallelScanAndInsert(HnswSpool * hnswspool, HnswShared * hnswshared, bool progress)
|
|
||||||
{
|
|
||||||
HnswBuildState buildstate;
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
TableScanDesc scan;
|
|
||||||
#else
|
|
||||||
HeapScanDesc scan;
|
|
||||||
#endif
|
|
||||||
double reltuples;
|
|
||||||
IndexInfo *indexInfo;
|
|
||||||
|
|
||||||
/* Join parallel scan */
|
|
||||||
indexInfo = BuildIndexInfo(hnswspool->index);
|
|
||||||
indexInfo->ii_Concurrent = hnswshared->isconcurrent;
|
|
||||||
InitBuildState(&buildstate, hnswspool->heap, hnswspool->index, indexInfo, MAIN_FORKNUM);
|
|
||||||
/* TODO Support in-memory builds */
|
|
||||||
buildstate.maxInMemoryElements = 0;
|
|
||||||
buildstate.flushed = true;
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
scan = table_beginscan_parallel(hnswspool->heap,
|
|
||||||
ParallelTableScanFromHnswShared(hnswshared));
|
|
||||||
reltuples = table_index_build_scan(hnswspool->heap, hnswspool->index, indexInfo,
|
|
||||||
true, progress, BuildCallback,
|
|
||||||
(void *) &buildstate, scan);
|
|
||||||
#else
|
|
||||||
scan = heap_beginscan_parallel(hnswspool->heap, &hnswshared->heapdesc);
|
|
||||||
reltuples = IndexBuildHeapScan(hnswspool->heap, hnswspool->index, indexInfo,
|
|
||||||
true, BuildCallback,
|
|
||||||
(void *) &buildstate, scan);
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Record statistics */
|
|
||||||
SpinLockAcquire(&hnswshared->mutex);
|
|
||||||
hnswshared->nparticipantsdone++;
|
|
||||||
hnswshared->reltuples += reltuples;
|
|
||||||
hnswshared->indtuples += buildstate.indtuples;
|
|
||||||
SpinLockRelease(&hnswshared->mutex);
|
|
||||||
|
|
||||||
/* Log statistics */
|
|
||||||
if (progress)
|
|
||||||
ereport(DEBUG1, (errmsg("leader processed " INT64_FORMAT " tuples", (int64) reltuples)));
|
|
||||||
else
|
|
||||||
ereport(DEBUG1, (errmsg("worker processed " INT64_FORMAT " tuples", (int64) reltuples)));
|
|
||||||
|
|
||||||
/* Notify leader */
|
|
||||||
ConditionVariableSignal(&hnswshared->workersdonecv);
|
|
||||||
|
|
||||||
FreeBuildState(&buildstate);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Perform work within a launched parallel process
|
|
||||||
*/
|
|
||||||
void
|
|
||||||
HnswParallelBuildMain(dsm_segment *seg, shm_toc *toc)
|
|
||||||
{
|
|
||||||
char *sharedquery;
|
|
||||||
HnswSpool *hnswspool;
|
|
||||||
HnswShared *hnswshared;
|
|
||||||
Relation heapRel;
|
|
||||||
Relation indexRel;
|
|
||||||
LOCKMODE heapLockmode;
|
|
||||||
LOCKMODE indexLockmode;
|
|
||||||
|
|
||||||
/* Set debug_query_string for individual workers first */
|
|
||||||
sharedquery = shm_toc_lookup(toc, PARALLEL_KEY_QUERY_TEXT, true);
|
|
||||||
debug_query_string = sharedquery;
|
|
||||||
|
|
||||||
/* Report the query string from leader */
|
|
||||||
pgstat_report_activity(STATE_RUNNING, debug_query_string);
|
|
||||||
|
|
||||||
/* Look up shared state */
|
|
||||||
hnswshared = shm_toc_lookup(toc, PARALLEL_KEY_HNSW_SHARED, false);
|
|
||||||
|
|
||||||
/* Open relations using lock modes known to be obtained by index.c */
|
|
||||||
if (!hnswshared->isconcurrent)
|
|
||||||
{
|
|
||||||
heapLockmode = ShareLock;
|
|
||||||
indexLockmode = AccessExclusiveLock;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
heapLockmode = ShareUpdateExclusiveLock;
|
|
||||||
indexLockmode = RowExclusiveLock;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Open relations within worker */
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
heapRel = table_open(hnswshared->heaprelid, heapLockmode);
|
|
||||||
#else
|
|
||||||
heapRel = heap_open(hnswshared->heaprelid, heapLockmode);
|
|
||||||
#endif
|
|
||||||
indexRel = index_open(hnswshared->indexrelid, indexLockmode);
|
|
||||||
|
|
||||||
/* Initialize worker's own spool */
|
|
||||||
hnswspool = (HnswSpool *) palloc0(sizeof(HnswSpool));
|
|
||||||
hnswspool->heap = heapRel;
|
|
||||||
hnswspool->index = indexRel;
|
|
||||||
|
|
||||||
/* Perform inserts */
|
|
||||||
HnswParallelScanAndInsert(hnswspool, hnswshared, false);
|
|
||||||
|
|
||||||
/* Close relations within worker */
|
|
||||||
index_close(indexRel, indexLockmode);
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
table_close(heapRel, heapLockmode);
|
|
||||||
#else
|
|
||||||
heap_close(heapRel, heapLockmode);
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* End parallel build
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswEndParallel(HnswLeader * hnswleader)
|
|
||||||
{
|
|
||||||
/* Shutdown worker processes */
|
|
||||||
WaitForParallelWorkersToFinish(hnswleader->pcxt);
|
|
||||||
|
|
||||||
/* Free last reference to MVCC snapshot, if one was used */
|
|
||||||
if (IsMVCCSnapshot(hnswleader->snapshot))
|
|
||||||
UnregisterSnapshot(hnswleader->snapshot);
|
|
||||||
DestroyParallelContext(hnswleader->pcxt);
|
|
||||||
ExitParallelMode();
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Return size of shared memory required for parallel index build
|
|
||||||
*/
|
|
||||||
static Size
|
|
||||||
ParallelEstimateShared(Relation heap, Snapshot snapshot)
|
|
||||||
{
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
return add_size(BUFFERALIGN(sizeof(HnswShared)), table_parallelscan_estimate(heap, snapshot));
|
|
||||||
#else
|
|
||||||
if (!IsMVCCSnapshot(snapshot))
|
|
||||||
{
|
|
||||||
Assert(snapshot == SnapshotAny);
|
|
||||||
return sizeof(HnswShared);
|
|
||||||
}
|
|
||||||
|
|
||||||
return add_size(offsetof(HnswShared, heapdesc) +
|
|
||||||
offsetof(ParallelHeapScanDescData, phs_snapshot_data),
|
|
||||||
EstimateSnapshotSpace(snapshot));
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Within leader, participate as a parallel worker
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswLeaderParticipateAsWorker(HnswBuildState * buildstate)
|
|
||||||
{
|
|
||||||
HnswLeader *hnswleader = buildstate->hnswleader;
|
|
||||||
HnswSpool *leaderworker;
|
|
||||||
|
|
||||||
/* Allocate memory and initialize private spool */
|
|
||||||
leaderworker = (HnswSpool *) palloc0(sizeof(HnswSpool));
|
|
||||||
leaderworker->heap = buildstate->heap;
|
|
||||||
leaderworker->index = buildstate->index;
|
|
||||||
|
|
||||||
/* Perform work common to all participants */
|
|
||||||
HnswParallelScanAndInsert(leaderworker, hnswleader->hnswshared, true);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Begin parallel build
|
|
||||||
*/
|
|
||||||
static void
|
|
||||||
HnswBeginParallel(HnswBuildState * buildstate, bool isconcurrent, int request)
|
|
||||||
{
|
|
||||||
ParallelContext *pcxt;
|
|
||||||
int scantuplesortstates;
|
|
||||||
Snapshot snapshot;
|
|
||||||
Size esthnswshared;
|
|
||||||
HnswShared *hnswshared;
|
|
||||||
HnswLeader *hnswleader = (HnswLeader *) palloc0(sizeof(HnswLeader));
|
|
||||||
bool leaderparticipates = true;
|
|
||||||
int querylen;
|
|
||||||
|
|
||||||
#ifdef DISABLE_LEADER_PARTICIPATION
|
|
||||||
leaderparticipates = false;
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Enter parallel mode and create context */
|
|
||||||
EnterParallelMode();
|
|
||||||
Assert(request > 0);
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
pcxt = CreateParallelContext("vector", "HnswParallelBuildMain", request);
|
|
||||||
#else
|
|
||||||
pcxt = CreateParallelContext("vector", "HnswParallelBuildMain", request, true);
|
|
||||||
#endif
|
|
||||||
|
|
||||||
scantuplesortstates = leaderparticipates ? request + 1 : request;
|
|
||||||
|
|
||||||
/* Get snapshot for table scan */
|
|
||||||
if (!isconcurrent)
|
|
||||||
snapshot = SnapshotAny;
|
|
||||||
else
|
|
||||||
snapshot = RegisterSnapshot(GetTransactionSnapshot());
|
|
||||||
|
|
||||||
/* Estimate size of workspaces */
|
|
||||||
esthnswshared = ParallelEstimateShared(buildstate->heap, snapshot);
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, esthnswshared);
|
|
||||||
shm_toc_estimate_keys(&pcxt->estimator, 1);
|
|
||||||
|
|
||||||
/* Finally, estimate PARALLEL_KEY_QUERY_TEXT space */
|
|
||||||
if (debug_query_string)
|
|
||||||
{
|
|
||||||
querylen = strlen(debug_query_string);
|
|
||||||
shm_toc_estimate_chunk(&pcxt->estimator, querylen + 1);
|
|
||||||
shm_toc_estimate_keys(&pcxt->estimator, 1);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
querylen = 0; /* keep compiler quiet */
|
|
||||||
|
|
||||||
/* Everyone's had a chance to ask for space, so now create the DSM */
|
|
||||||
InitializeParallelDSM(pcxt);
|
|
||||||
|
|
||||||
/* If no DSM segment was available, back out (do serial build) */
|
|
||||||
if (pcxt->seg == NULL)
|
|
||||||
{
|
|
||||||
if (IsMVCCSnapshot(snapshot))
|
|
||||||
UnregisterSnapshot(snapshot);
|
|
||||||
DestroyParallelContext(pcxt);
|
|
||||||
ExitParallelMode();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Store shared build state, for which we reserved space */
|
|
||||||
hnswshared = (HnswShared *) shm_toc_allocate(pcxt->toc, esthnswshared);
|
|
||||||
/* Initialize immutable state */
|
|
||||||
hnswshared->heaprelid = RelationGetRelid(buildstate->heap);
|
|
||||||
hnswshared->indexrelid = RelationGetRelid(buildstate->index);
|
|
||||||
hnswshared->isconcurrent = isconcurrent;
|
|
||||||
hnswshared->scantuplesortstates = scantuplesortstates;
|
|
||||||
ConditionVariableInit(&hnswshared->workersdonecv);
|
|
||||||
SpinLockInit(&hnswshared->mutex);
|
|
||||||
/* Initialize mutable state */
|
|
||||||
hnswshared->nparticipantsdone = 0;
|
|
||||||
hnswshared->reltuples = 0;
|
|
||||||
hnswshared->indtuples = 0;
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
|
||||||
table_parallelscan_initialize(buildstate->heap,
|
|
||||||
ParallelTableScanFromHnswShared(hnswshared),
|
|
||||||
snapshot);
|
|
||||||
#else
|
|
||||||
heap_parallelscan_initialize(&hnswshared->heapdesc, buildstate->heap, snapshot);
|
|
||||||
#endif
|
|
||||||
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_HNSW_SHARED, hnswshared);
|
|
||||||
|
|
||||||
/* Store query string for workers */
|
|
||||||
if (debug_query_string)
|
|
||||||
{
|
|
||||||
char *sharedquery;
|
|
||||||
|
|
||||||
sharedquery = (char *) shm_toc_allocate(pcxt->toc, querylen + 1);
|
|
||||||
memcpy(sharedquery, debug_query_string, querylen + 1);
|
|
||||||
shm_toc_insert(pcxt->toc, PARALLEL_KEY_QUERY_TEXT, sharedquery);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Launch workers, saving status for leader/caller */
|
|
||||||
LaunchParallelWorkers(pcxt);
|
|
||||||
hnswleader->pcxt = pcxt;
|
|
||||||
hnswleader->nparticipanttuplesorts = pcxt->nworkers_launched;
|
|
||||||
if (leaderparticipates)
|
|
||||||
hnswleader->nparticipanttuplesorts++;
|
|
||||||
hnswleader->hnswshared = hnswshared;
|
|
||||||
hnswleader->snapshot = snapshot;
|
|
||||||
|
|
||||||
/* If no workers were successfully launched, back out (do serial build) */
|
|
||||||
if (pcxt->nworkers_launched == 0)
|
|
||||||
{
|
|
||||||
HnswEndParallel(hnswleader);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Log participants */
|
|
||||||
ereport(DEBUG1, (errmsg("using %d parallel workers", pcxt->nworkers_launched)));
|
|
||||||
|
|
||||||
/* Save leader state now that it's clear build will be parallel */
|
|
||||||
buildstate->hnswleader = hnswleader;
|
|
||||||
|
|
||||||
/* Join heap scan ourselves */
|
|
||||||
if (leaderparticipates)
|
|
||||||
HnswLeaderParticipateAsWorker(buildstate);
|
|
||||||
|
|
||||||
/* Wait for all launched workers */
|
|
||||||
WaitForParallelWorkersToAttach(pcxt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Build graph
|
* Build graph
|
||||||
*/
|
*/
|
||||||
static void
|
static void
|
||||||
BuildGraph(HnswBuildState * buildstate, ForkNumber forkNum)
|
BuildGraph(HnswBuildState * buildstate, ForkNumber forkNum)
|
||||||
{
|
{
|
||||||
int parallel_workers = 0;
|
|
||||||
|
|
||||||
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_HNSW_PHASE_LOAD);
|
UpdateProgress(PROGRESS_CREATEIDX_SUBPHASE, PROGRESS_HNSW_PHASE_LOAD);
|
||||||
|
|
||||||
/* Calculate parallel workers */
|
|
||||||
parallel_workers = plan_create_index_workers(RelationGetRelid(buildstate->heap), RelationGetRelid(buildstate->index));
|
|
||||||
|
|
||||||
/* Attempt to launch parallel worker scan when required */
|
|
||||||
if (parallel_workers > 0)
|
|
||||||
{
|
|
||||||
/* TODO Support in-memory builds */
|
|
||||||
FlushPages(buildstate);
|
|
||||||
HnswBeginParallel(buildstate, buildstate->indexInfo->ii_Concurrent, parallel_workers);
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Add tuples to sort */
|
|
||||||
if (buildstate->hnswleader)
|
|
||||||
buildstate->reltuples = ParallelHeapScan(buildstate);
|
|
||||||
else
|
|
||||||
{
|
|
||||||
#if PG_VERSION_NUM >= 120000
|
#if PG_VERSION_NUM >= 120000
|
||||||
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
buildstate->reltuples = table_index_build_scan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
true, true, BuildCallback, (void *) buildstate, NULL);
|
true, true, BuildCallback, (void *) buildstate, NULL);
|
||||||
#else
|
#else
|
||||||
buildstate->reltuples = IndexBuildHeapScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
buildstate->reltuples = IndexBuildHeapScan(buildstate->heap, buildstate->index, buildstate->indexInfo,
|
||||||
true, BuildCallback, (void *) buildstate, NULL);
|
true, BuildCallback, (void *) buildstate, NULL);
|
||||||
#endif
|
#endif
|
||||||
}
|
|
||||||
|
|
||||||
/* End parallel build */
|
|
||||||
if (buildstate->hnswleader)
|
|
||||||
HnswEndParallel(buildstate->hnswleader);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
|
|||||||
@@ -58,6 +58,75 @@ GetDimensions(Relation index)
|
|||||||
return dimensions;
|
return dimensions;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Remove deleted heap TID
|
||||||
|
*/
|
||||||
|
static void
|
||||||
|
RemoveHeapTid(IndexScanDesc scan)
|
||||||
|
{
|
||||||
|
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
||||||
|
Relation index = scan->indexRelation;
|
||||||
|
Buffer buf = so->buf;
|
||||||
|
Page page;
|
||||||
|
GenericXLogState *state;
|
||||||
|
ItemId itemid;
|
||||||
|
HnswElementTuple etup;
|
||||||
|
Size etupSize;
|
||||||
|
int idx = -1;
|
||||||
|
|
||||||
|
/* Safety check */
|
||||||
|
if (!BufferIsValid(buf) || !OffsetNumberIsValid(so->offno) || !ItemPointerIsValid(&so->heaptid))
|
||||||
|
return;
|
||||||
|
|
||||||
|
/* Use WAL rather than hint */
|
||||||
|
LockBuffer(buf, BUFFER_LOCK_EXCLUSIVE);
|
||||||
|
state = GenericXLogStart(index);
|
||||||
|
page = GenericXLogRegisterBuffer(state, buf, 0);
|
||||||
|
itemid = PageGetItemId(page, so->offno);
|
||||||
|
etup = (HnswElementTuple) PageGetItem(page, itemid);
|
||||||
|
etupSize = ItemIdGetLength(itemid);
|
||||||
|
|
||||||
|
Assert(HnswIsElementTuple(etup));
|
||||||
|
|
||||||
|
/* Find index */
|
||||||
|
for (int i = 0; i < HNSW_HEAPTIDS; i++)
|
||||||
|
{
|
||||||
|
if (!ItemPointerIsValid(&etup->heaptids[i]))
|
||||||
|
break;
|
||||||
|
|
||||||
|
if (ItemPointerEquals(&etup->heaptids[i], &so->heaptid))
|
||||||
|
{
|
||||||
|
idx = i;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (idx == -1)
|
||||||
|
GenericXLogAbort(state);
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/* Move pointers forward */
|
||||||
|
for (int i = idx; i < HNSW_HEAPTIDS; i++)
|
||||||
|
{
|
||||||
|
if (i + 1 == HNSW_HEAPTIDS || !ItemPointerIsValid(&etup->heaptids[i + 1]))
|
||||||
|
ItemPointerSetInvalid(&etup->heaptids[i]);
|
||||||
|
else
|
||||||
|
ItemPointerCopy(&etup->heaptids[i + 1], &etup->heaptids[i]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Overwrite tuple */
|
||||||
|
if (!PageIndexTupleOverwrite(page, so->offno, (Item) etup, etupSize))
|
||||||
|
elog(ERROR, "failed to add index item to \"%s\"", RelationGetRelationName(index));
|
||||||
|
|
||||||
|
/* Commit */
|
||||||
|
MarkBufferDirty(buf);
|
||||||
|
GenericXLogFinish(state);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Unlock buffer */
|
||||||
|
LockBuffer(buf, BUFFER_LOCK_UNLOCK);
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Prepare for an index scan
|
* Prepare for an index scan
|
||||||
*/
|
*/
|
||||||
@@ -71,6 +140,9 @@ hnswbeginscan(Relation index, int nkeys, int norderbys)
|
|||||||
|
|
||||||
so = (HnswScanOpaque) palloc(sizeof(HnswScanOpaqueData));
|
so = (HnswScanOpaque) palloc(sizeof(HnswScanOpaqueData));
|
||||||
so->buf = InvalidBuffer;
|
so->buf = InvalidBuffer;
|
||||||
|
ItemPointerSetInvalid(&so->heaptid);
|
||||||
|
so->offno = InvalidOffsetNumber;
|
||||||
|
so->removedCount = 0;
|
||||||
so->first = true;
|
so->first = true;
|
||||||
so->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
so->tmpCtx = AllocSetContextCreate(CurrentMemoryContext,
|
||||||
"Hnsw scan temporary context",
|
"Hnsw scan temporary context",
|
||||||
@@ -95,6 +167,7 @@ hnswrescan(IndexScanDesc scan, ScanKey keys, int nkeys, ScanKey orderbys, int no
|
|||||||
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
HnswScanOpaque so = (HnswScanOpaque) scan->opaque;
|
||||||
|
|
||||||
so->first = true;
|
so->first = true;
|
||||||
|
ItemPointerSetInvalid(&so->heaptid);
|
||||||
MemoryContextReset(so->tmpCtx);
|
MemoryContextReset(so->tmpCtx);
|
||||||
|
|
||||||
if (keys && scan->numberOfKeys > 0)
|
if (keys && scan->numberOfKeys > 0)
|
||||||
@@ -158,12 +231,25 @@ hnswgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
|
|
||||||
so->first = false;
|
so->first = false;
|
||||||
}
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* Remove dead tuples. kill_prior_tuple will only be true if not in
|
||||||
|
* recovery. Limit the number removed per scan for performance.
|
||||||
|
*/
|
||||||
|
if (scan->kill_prior_tuple && so->removedCount < 3)
|
||||||
|
{
|
||||||
|
RemoveHeapTid(scan);
|
||||||
|
so->removedCount++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
while (list_length(so->w) > 0)
|
while (list_length(so->w) > 0)
|
||||||
{
|
{
|
||||||
HnswCandidate *hc = llast(so->w);
|
HnswCandidate *hc = llast(so->w);
|
||||||
ItemPointer tid;
|
ItemPointer tid;
|
||||||
BlockNumber indexblkno;
|
BlockNumber indexblkno;
|
||||||
|
OffsetNumber indexoffno;
|
||||||
|
|
||||||
/* Move to next element if no valid heap tids */
|
/* Move to next element if no valid heap tids */
|
||||||
if (list_length(hc->element->heaptids) == 0)
|
if (list_length(hc->element->heaptids) == 0)
|
||||||
@@ -174,6 +260,7 @@ hnswgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
|
|
||||||
tid = llast(hc->element->heaptids);
|
tid = llast(hc->element->heaptids);
|
||||||
indexblkno = hc->element->blkno;
|
indexblkno = hc->element->blkno;
|
||||||
|
indexoffno = hc->element->offno;
|
||||||
|
|
||||||
hc->element->heaptids = list_delete_last(hc->element->heaptids);
|
hc->element->heaptids = list_delete_last(hc->element->heaptids);
|
||||||
|
|
||||||
@@ -185,6 +272,10 @@ hnswgettuple(IndexScanDesc scan, ScanDirection dir)
|
|||||||
scan->xs_ctup.t_self = *tid;
|
scan->xs_ctup.t_self = *tid;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/* Keep track of info needed to remove dead tuples */
|
||||||
|
so->heaptid = *tid;
|
||||||
|
so->offno = indexoffno;
|
||||||
|
|
||||||
/* Unpin buffer */
|
/* Unpin buffer */
|
||||||
if (BufferIsValid(so->buf))
|
if (BufferIsValid(so->buf))
|
||||||
ReleaseBuffer(so->buf);
|
ReleaseBuffer(so->buf);
|
||||||
|
|||||||
@@ -551,6 +551,8 @@ HnswSearchLayer(Datum q, List *ep, int ef, int lc, Relation index, FmgrInfo *pro
|
|||||||
pairingheap *C = pairingheap_allocate(CompareNearestCandidates, NULL);
|
pairingheap *C = pairingheap_allocate(CompareNearestCandidates, NULL);
|
||||||
pairingheap *W = pairingheap_allocate(CompareFurthestCandidates, NULL);
|
pairingheap *W = pairingheap_allocate(CompareFurthestCandidates, NULL);
|
||||||
int wlen = 0;
|
int wlen = 0;
|
||||||
|
uint64 dead = 0;
|
||||||
|
uint64 maxAdditional = skipElement == NULL ? ef : PG_UINT64_MAX;
|
||||||
HASHCTL hash_ctl;
|
HASHCTL hash_ctl;
|
||||||
HTAB *v;
|
HTAB *v;
|
||||||
|
|
||||||
@@ -579,13 +581,14 @@ HnswSearchLayer(Datum q, List *ep, int ef, int lc, Relation index, FmgrInfo *pro
|
|||||||
pairingheap_add(C, &(CreatePairingHeapNode(hc)->ph_node));
|
pairingheap_add(C, &(CreatePairingHeapNode(hc)->ph_node));
|
||||||
pairingheap_add(W, &(CreatePairingHeapNode(hc)->ph_node));
|
pairingheap_add(W, &(CreatePairingHeapNode(hc)->ph_node));
|
||||||
|
|
||||||
/*
|
/* Do not count certain number of dead elements towards ef */
|
||||||
* Do not count elements being deleted towards ef when vacuuming. It
|
if (list_length(hc->element->heaptids) == 0)
|
||||||
* would be ideal to do this for inserts as well, but this could
|
{
|
||||||
* affect insert performance.
|
if ((++dead) <= maxAdditional)
|
||||||
*/
|
continue;
|
||||||
if (skipElement == NULL || list_length(hc->element->heaptids) != 0)
|
}
|
||||||
wlen++;
|
|
||||||
|
wlen++;
|
||||||
}
|
}
|
||||||
|
|
||||||
while (!pairingheap_is_empty(C))
|
while (!pairingheap_is_empty(C))
|
||||||
@@ -638,19 +641,18 @@ HnswSearchLayer(Datum q, List *ep, int ef, int lc, Relation index, FmgrInfo *pro
|
|||||||
pairingheap_add(C, &(CreatePairingHeapNode(ec)->ph_node));
|
pairingheap_add(C, &(CreatePairingHeapNode(ec)->ph_node));
|
||||||
pairingheap_add(W, &(CreatePairingHeapNode(ec)->ph_node));
|
pairingheap_add(W, &(CreatePairingHeapNode(ec)->ph_node));
|
||||||
|
|
||||||
/*
|
/* Do not count certain number of dead elements towards ef */
|
||||||
* Do not count elements being deleted towards ef when
|
if (list_length(e->element->heaptids) == 0)
|
||||||
* vacuuming. It would be ideal to do this for inserts as
|
|
||||||
* well, but this could affect insert performance.
|
|
||||||
*/
|
|
||||||
if (skipElement == NULL || list_length(e->element->heaptids) != 0)
|
|
||||||
{
|
{
|
||||||
wlen++;
|
if ((++dead) <= maxAdditional)
|
||||||
|
continue;
|
||||||
/* No need to decrement wlen */
|
|
||||||
if (wlen > ef)
|
|
||||||
pairingheap_remove_first(W);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
wlen++;
|
||||||
|
|
||||||
|
/* No need to decrement wlen */
|
||||||
|
if (wlen > ef)
|
||||||
|
pairingheap_remove_first(W);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,8 +10,8 @@
|
|||||||
#include "ivfflat.h"
|
#include "ivfflat.h"
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "storage/bufmgr.h"
|
#include "storage/bufmgr.h"
|
||||||
#include "utils/memutils.h"
|
|
||||||
#include "tcop/tcopprot.h"
|
#include "tcop/tcopprot.h"
|
||||||
|
#include "utils/memutils.h"
|
||||||
|
|
||||||
#if PG_VERSION_NUM >= 140000
|
#if PG_VERSION_NUM >= 140000
|
||||||
#include "utils/backend_progress.h"
|
#include "utils/backend_progress.h"
|
||||||
|
|||||||
@@ -94,7 +94,7 @@ for my $i (0 .. $#operators)
|
|||||||
# Test approximate results
|
# Test approximate results
|
||||||
if ($operator ne "<#>")
|
if ($operator ne "<#>")
|
||||||
{
|
{
|
||||||
# TODO fix test
|
# TODO Fix test (uniform random vectors all have similar inner product)
|
||||||
test_recall(1, 0.71, $operator);
|
test_recall(1, 0.71, $operator);
|
||||||
test_recall(10, 0.95, $operator);
|
test_recall(10, 0.95, $operator);
|
||||||
}
|
}
|
||||||
@@ -115,7 +115,7 @@ for my $i (0 .. $#operators)
|
|||||||
# Test approximate results
|
# Test approximate results
|
||||||
if ($operator ne "<#>")
|
if ($operator ne "<#>")
|
||||||
{
|
{
|
||||||
# TODO fix test
|
# TODO Fix test (uniform random vectors all have similar inner product)
|
||||||
test_recall(1, 0.71, $operator);
|
test_recall(1, 0.71, $operator);
|
||||||
test_recall(10, 0.95, $operator);
|
test_recall(10, 0.95, $operator);
|
||||||
}
|
}
|
||||||
@@ -38,8 +38,9 @@ sub test_index_replay
|
|||||||
);
|
);
|
||||||
|
|
||||||
# Run test queries and compare their result
|
# Run test queries and compare their result
|
||||||
my $primary_result = $node_primary->safe_psql("postgres", $queries);
|
# Query replica first since index scan on primary can generate WAL removing tuples
|
||||||
my $replica_result = $node_replica->safe_psql("postgres", $queries);
|
my $replica_result = $node_replica->safe_psql("postgres", $queries);
|
||||||
|
my $primary_result = $node_primary->safe_psql("postgres", $queries);
|
||||||
|
|
||||||
is($primary_result, $replica_result, "$test_name: query result matches");
|
is($primary_result, $replica_result, "$test_name: query result matches");
|
||||||
return;
|
return;
|
||||||
|
|||||||
@@ -83,31 +83,11 @@ for my $i (0 .. $#operators)
|
|||||||
push(@expected, $res);
|
push(@expected, $res);
|
||||||
}
|
}
|
||||||
|
|
||||||
# Build index serially
|
# Add index
|
||||||
$node->safe_psql("postgres", qq(
|
$node->safe_psql("postgres", "CREATE INDEX ON tst USING hnsw (v $opclass);");
|
||||||
SET max_parallel_maintenance_workers = 0;
|
|
||||||
CREATE INDEX idx ON tst USING hnsw (v $opclass);
|
|
||||||
));
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
my $min = $operator eq "<#>" ? 0.80 : 0.99;
|
my $min = $operator eq "<#>" ? 0.80 : 0.99;
|
||||||
test_recall($min, $operator);
|
test_recall($min, $operator);
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
|
|
||||||
# Build index in parallel
|
|
||||||
my ($ret, $stdout, $stderr) = $node->psql("postgres", qq(
|
|
||||||
SET client_min_messages = DEBUG;
|
|
||||||
SET min_parallel_table_scan_size = 1;
|
|
||||||
CREATE INDEX idx ON tst USING hnsw (v $opclass);
|
|
||||||
));
|
|
||||||
is($ret, 0, $stderr);
|
|
||||||
like($stderr, qr/using \d+ parallel workers/);
|
|
||||||
|
|
||||||
# Test approximate results
|
|
||||||
test_recall($min, $operator);
|
|
||||||
|
|
||||||
$node->safe_psql("postgres", "DROP INDEX idx;");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
done_testing();
|
done_testing();
|
||||||
|
|||||||
@@ -23,25 +23,27 @@ sub insert_vectors
|
|||||||
|
|
||||||
sub test_duplicates
|
sub test_duplicates
|
||||||
{
|
{
|
||||||
|
my ($exp) = @_;
|
||||||
|
|
||||||
my $res = $node->safe_psql("postgres", qq(
|
my $res = $node->safe_psql("postgres", qq(
|
||||||
SET enable_seqscan = off;
|
SET enable_seqscan = off;
|
||||||
SET hnsw.ef_search = 1;
|
SET hnsw.ef_search = 1;
|
||||||
SELECT COUNT(*) FROM (SELECT * FROM tst ORDER BY v <-> '[1,1,1]') t;
|
SELECT COUNT(*) FROM (SELECT * FROM tst ORDER BY v <-> '[1,1,1]') t;
|
||||||
));
|
));
|
||||||
is($res, 10);
|
is($res, $exp);
|
||||||
}
|
}
|
||||||
|
|
||||||
# Test duplicates with build
|
# Test duplicates with build
|
||||||
insert_vectors();
|
insert_vectors();
|
||||||
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING hnsw (v vector_l2_ops);");
|
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING hnsw (v vector_l2_ops);");
|
||||||
test_duplicates();
|
test_duplicates(10);
|
||||||
|
|
||||||
# Reset
|
# Reset
|
||||||
$node->safe_psql("postgres", "TRUNCATE tst;");
|
$node->safe_psql("postgres", "TRUNCATE tst;");
|
||||||
|
|
||||||
# Test duplicates with inserts
|
# Test duplicates with inserts
|
||||||
insert_vectors();
|
insert_vectors();
|
||||||
test_duplicates();
|
test_duplicates(10);
|
||||||
|
|
||||||
# Test fallback path for inserts
|
# Test fallback path for inserts
|
||||||
$node->pgbench(
|
$node->pgbench(
|
||||||
@@ -55,4 +57,15 @@ $node->pgbench(
|
|||||||
}
|
}
|
||||||
);
|
);
|
||||||
|
|
||||||
|
# Reset
|
||||||
|
$node->safe_psql("postgres", "TRUNCATE tst;");
|
||||||
|
|
||||||
|
# Test deletes with index scan
|
||||||
|
$node->safe_psql("postgres", "INSERT INTO tst SELECT '[1,1,1]' FROM generate_series(1, 10) i;");
|
||||||
|
$node->safe_psql("postgres", "DELETE FROM tst WHERE ctid IN (SELECT ctid FROM tst ORDER BY random() LIMIT 5);");
|
||||||
|
for (1 .. 3)
|
||||||
|
{
|
||||||
|
test_duplicates(5);
|
||||||
|
}
|
||||||
|
|
||||||
done_testing();
|
done_testing();
|
||||||
|
|||||||
@@ -89,7 +89,7 @@ foreach (@queries)
|
|||||||
test_recall(0.20, $limit, "before vacuum");
|
test_recall(0.20, $limit, "before vacuum");
|
||||||
test_recall(0.95, 100, "before vacuum");
|
test_recall(0.95, 100, "before vacuum");
|
||||||
|
|
||||||
# TODO test concurrent inserts with vacuum
|
# TODO Test concurrent inserts with vacuum
|
||||||
$node->safe_psql("postgres", "VACUUM tst;");
|
$node->safe_psql("postgres", "VACUUM tst;");
|
||||||
|
|
||||||
test_recall(0.95, $limit, "after vacuum");
|
test_recall(0.95, $limit, "after vacuum");
|
||||||
|
|||||||
117
test/t/017_ivfflat_insert_recall.pl
Normal file
117
test/t/017_ivfflat_insert_recall.pl
Normal file
@@ -0,0 +1,117 @@
|
|||||||
|
use strict;
|
||||||
|
use warnings;
|
||||||
|
use PostgresNode;
|
||||||
|
use TestLib;
|
||||||
|
use Test::More;
|
||||||
|
|
||||||
|
my $node;
|
||||||
|
my @queries = ();
|
||||||
|
my @expected;
|
||||||
|
my $limit = 20;
|
||||||
|
|
||||||
|
sub test_recall
|
||||||
|
{
|
||||||
|
my ($probes, $min, $operator) = @_;
|
||||||
|
my $correct = 0;
|
||||||
|
my $total = 0;
|
||||||
|
|
||||||
|
my $explain = $node->safe_psql("postgres", qq(
|
||||||
|
SET enable_seqscan = off;
|
||||||
|
SET ivfflat.probes = $probes;
|
||||||
|
EXPLAIN ANALYZE SELECT i FROM tst ORDER BY v $operator '$queries[0]' LIMIT $limit;
|
||||||
|
));
|
||||||
|
like($explain, qr/Index Scan using idx on tst/);
|
||||||
|
|
||||||
|
for my $i (0 .. $#queries)
|
||||||
|
{
|
||||||
|
my $actual = $node->safe_psql("postgres", qq(
|
||||||
|
SET enable_seqscan = off;
|
||||||
|
SET ivfflat.probes = $probes;
|
||||||
|
SELECT i FROM tst ORDER BY v $operator '$queries[$i]' LIMIT $limit;
|
||||||
|
));
|
||||||
|
my @actual_ids = split("\n", $actual);
|
||||||
|
my %actual_set = map { $_ => 1 } @actual_ids;
|
||||||
|
|
||||||
|
my @expected_ids = split("\n", $expected[$i]);
|
||||||
|
|
||||||
|
foreach (@expected_ids)
|
||||||
|
{
|
||||||
|
if (exists($actual_set{$_}))
|
||||||
|
{
|
||||||
|
$correct++;
|
||||||
|
}
|
||||||
|
$total++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
cmp_ok($correct / $total, ">=", $min, $operator);
|
||||||
|
}
|
||||||
|
|
||||||
|
# Initialize node
|
||||||
|
$node = get_new_node('node');
|
||||||
|
$node->init;
|
||||||
|
$node->start;
|
||||||
|
|
||||||
|
# Create table
|
||||||
|
$node->safe_psql("postgres", "CREATE EXTENSION vector;");
|
||||||
|
$node->safe_psql("postgres", "CREATE TABLE tst (i serial, v vector(3));");
|
||||||
|
|
||||||
|
# Generate queries
|
||||||
|
for (1 .. 20)
|
||||||
|
{
|
||||||
|
my $r1 = rand();
|
||||||
|
my $r2 = rand();
|
||||||
|
my $r3 = rand();
|
||||||
|
push(@queries, "[$r1,$r2,$r3]");
|
||||||
|
}
|
||||||
|
|
||||||
|
# Check each index type
|
||||||
|
my @operators = ("<->", "<#>", "<=>");
|
||||||
|
my @opclasses = ("vector_l2_ops", "vector_ip_ops", "vector_cosine_ops");
|
||||||
|
|
||||||
|
for my $i (0 .. $#operators)
|
||||||
|
{
|
||||||
|
my $operator = $operators[$i];
|
||||||
|
my $opclass = $opclasses[$i];
|
||||||
|
|
||||||
|
# Add index
|
||||||
|
$node->safe_psql("postgres", "CREATE INDEX idx ON tst USING ivfflat (v $opclass);");
|
||||||
|
|
||||||
|
# Use concurrent inserts
|
||||||
|
$node->pgbench(
|
||||||
|
"--no-vacuum --client=10 --transactions=1000",
|
||||||
|
0,
|
||||||
|
[qr{actually processed}],
|
||||||
|
[qr{^$}],
|
||||||
|
"concurrent INSERTs",
|
||||||
|
{
|
||||||
|
"017_ivfflat_insert_recall_$opclass" => "INSERT INTO tst (v) SELECT ARRAY[random(), random(), random()] FROM generate_series(1, 10) i;"
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
# Get exact results
|
||||||
|
@expected = ();
|
||||||
|
foreach (@queries)
|
||||||
|
{
|
||||||
|
my $res = $node->safe_psql("postgres", qq(
|
||||||
|
SET enable_indexscan = off;
|
||||||
|
SELECT i FROM tst ORDER BY v $operator '$_' LIMIT $limit;
|
||||||
|
));
|
||||||
|
push(@expected, $res);
|
||||||
|
}
|
||||||
|
|
||||||
|
# Test approximate results
|
||||||
|
if ($operator ne "<#>")
|
||||||
|
{
|
||||||
|
# TODO Fix test (uniform random vectors all have similar inner product)
|
||||||
|
test_recall(1, 0.71, $operator);
|
||||||
|
test_recall(10, 0.95, $operator);
|
||||||
|
}
|
||||||
|
# Account for equal distances
|
||||||
|
test_recall(100, 0.9925, $operator);
|
||||||
|
|
||||||
|
$node->safe_psql("postgres", "DROP INDEX idx;");
|
||||||
|
$node->safe_psql("postgres", "TRUNCATE tst;");
|
||||||
|
}
|
||||||
|
|
||||||
|
done_testing();
|
||||||
Reference in New Issue
Block a user