diff --git a/collection.go b/collection.go index d62eb29..230745c 100644 --- a/collection.go +++ b/collection.go @@ -584,9 +584,22 @@ func openCollectionDenseArtifact( return core.OpenScalarQuantizedIVFIndex(ctx, path, kind, reformer) case IndexTypeVamana: if spec.quantize == QuantizeTypeUndefined { - return core.OpenVamanaIndex(ctx, path) + return core.OpenVamanaIndexWithMmap(ctx, path, useMmap) } - return core.OpenScalarQuantizedVamanaIndex(ctx, path, kind, reformer) + if field.DataType == DataTypeVectorFP32 { + reader, keys, err := collectionEncodedDenseReader(ctx, field, documents) + if err != nil { + return nil, err + } + if reader != nil { + originals := make(map[uint64][]byte, len(keys)) + for position, key := range keys { + originals[key] = reader.rows[position] + } + return core.OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx, path, kind, reformer, originals, useMmap) + } + } + return core.OpenScalarQuantizedVamanaIndexWithMmap(ctx, path, kind, reformer, useMmap) case IndexTypeDiskANN: if spec.quantize == QuantizeTypeUndefined { candidateCount, candidateErr := collectionDenseCandidateCount(ctx, field, documents) @@ -664,7 +677,7 @@ func (c *Collection) segmentDocumentsLocked(ctx context.Context) ([]collectionSe if err != nil { return nil, err } - if spec.indexType == IndexTypeHNSW || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { + if spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { borrowedFields[field.Name] = struct{}{} } } @@ -1027,7 +1040,7 @@ func buildCollectionIndexes( } if field.DataType.IsDenseVector() { var exact collectionDenseIndex - useLazyExact := spec.indexType == IndexTypeHNSW || + useLazyExact := spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) || (spec.indexType == IndexTypeDiskANN && spec.quantize == QuantizeTypeUndefined) if field.DataType == DataTypeVectorFP32 && useLazyExact { @@ -1054,8 +1067,8 @@ func buildCollectionIndexes( var flat collectionDenseIndex if spec.quantize == QuantizeTypeUndefined || spec.indexType == IndexTypeHNSWRaBitQ || spec.indexType == IndexTypeIVFRaBitQ { flat = exact - } else if spec.indexType != IndexTypeHNSW { - // Quantized HNSW supplies a shared Flat view after opening the graph. + } else if spec.indexType != IndexTypeHNSW && spec.indexType != IndexTypeVamana { + // Quantized graphs supply a shared Flat view after opening. flat, err = buildCollectionDenseFlat(ctx, schema.Name, field, documents, spec) if err != nil { return fail(err) @@ -1081,9 +1094,11 @@ func buildCollectionIndexes( } indexes.denseNative[field.Name] = native if flat == nil { - quantized, ok := native.(*core.ScalarQuantizedHNSWIndex) + quantized, ok := native.(interface { + FlatIndex() *core.ScalarQuantizedFlatIndex + }) if !ok { - return fail(fmt.Errorf("quantized HNSW field %q has an incompatible native index", field.Name)) + return fail(fmt.Errorf("quantized graph field %q has an incompatible native index", field.Name)) } indexes.denseFlat[field.Name] = quantized.FlatIndex() } @@ -2529,7 +2544,7 @@ func buildCollectionDenseVamana( spec collectionVectorIndex, workers int, ) (collectionVamanaIndex, error) { - candidates, err := collectionDenseCandidates(ctx, field, documents) + count, err := collectionDenseCandidateCount(ctx, field, documents) if err != nil { return nil, err } @@ -2542,6 +2557,24 @@ func buildCollectionDenseVamana( options.MaxOcclusionSize = core.DefaultVamanaMaxOcclusionSize } options.SaturateGraph = spec.vamana.SaturateGraph + if field.DataType == DataTypeVectorFP32 { + candidates, err := collectionDenseBorrowedCandidates(ctx, field, documents) + if err != nil { + return nil, err + } + if spec.quantize == QuantizeTypeUndefined { + return core.BuildVamanaWithBorrowedVectors(ctx, int(field.Dimension), options, candidates, workers) + } + kind, err := toCoreQuantization(spec.quantize) + if err != nil { + return nil, err + } + reformer, err := collectionReformer(schemaName, field, spec) + if err != nil { + return nil, err + } + return core.BuildScalarQuantizedVamanaWithBorrowedVectors(ctx, int(field.Dimension), options, candidates, workers, kind, reformer) + } var builder *core.VamanaBuilder if field.DataType == DataTypeVectorFP16 && spec.quantize == QuantizeTypeUndefined { builder, err = core.NewVamanaBuilderFP16(int(field.Dimension), options) @@ -2551,17 +2584,27 @@ func buildCollectionDenseVamana( if err != nil { return nil, err } - for _, candidate := range candidates { - if err := builder.Add(ctx, candidate.Key, candidate.Vector); err != nil { + if err := builder.Reserve(count); err != nil { + return nil, err + } + for _, document := range documents { + if err := ctx.Err(); err != nil { + return nil, err + } + value, found := document.Fields[field.Name] + if !found || value == nil { + continue + } + vector, err := denseValueToFloat32Borrowed(value) + if err != nil { + return nil, fmt.Errorf("document %d field %q: %w", document.DocID, field.Name, err) + } + if err := builder.Add(ctx, document.DocID, vector); err != nil { return nil, err } - } - base, err := builder.BuildInterleavedWithWorkers(ctx, workers) - if err != nil { - return nil, err } if spec.quantize == QuantizeTypeUndefined { - return base, nil + return builder.BuildInterleavedWithWorkers(ctx, workers) } kind, err := toCoreQuantization(spec.quantize) if err != nil { @@ -2571,7 +2614,7 @@ func buildCollectionDenseVamana( if err != nil { return nil, err } - return core.NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) + return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) } func buildCollectionDenseDiskANN( diff --git a/collection_memory_test.go b/collection_memory_test.go index f8e6aef..e6fa6e1 100644 --- a/collection_memory_test.go +++ b/collection_memory_test.go @@ -26,12 +26,30 @@ import ( ) func TestHNSWRuntimeSharesFlatAndDefersExact(t *testing.T) { + testGraphRuntimeSharesFlatAndDefersExact(t, IndexTypeHNSW) +} + +func TestVamanaRuntimeSharesFlatAndDefersExact(t *testing.T) { + testGraphRuntimeSharesFlatAndDefersExact(t, IndexTypeVamana) +} + +func testGraphRuntimeSharesFlatAndDefersExact(t *testing.T, indexType IndexType) { ctx := context.Background() for _, quantize := range []QuantizeType{QuantizeTypeUndefined, QuantizeTypeFP16, QuantizeTypeInt8, QuantizeTypeInt4} { t.Run(fmt.Sprint(quantize), func(t *testing.T) { - params := NewHNSWIndexParams(MetricTypeL2) - params.M, params.EFConstruction, params.Quantize = 4, 16, quantize - params.Quantizer.EnableRotate = quantize == QuantizeTypeInt4 || quantize == QuantizeTypeInt8 + var params IndexParams + rotate := quantize == QuantizeTypeInt4 || quantize == QuantizeTypeInt8 + if indexType == IndexTypeVamana { + value := NewVamanaIndexParams(MetricTypeL2) + value.MaxDegree, value.SearchListSize, value.Quantize = 4, 16, quantize + value.Quantizer.EnableRotate = rotate + params = value + } else { + value := NewHNSWIndexParams(MetricTypeL2) + value.M, value.EFConstruction, value.Quantize = 4, 16, quantize + value.Quantizer.EnableRotate = rotate + params = value + } field := FieldSchema{Name: "embedding", DataType: DataTypeVectorFP32, Dimension: 4, Nullable: true, Index: params} schema := NewCollectionSchema("shared_hnsw", field) documents := annDenseDocuments(48) @@ -100,7 +118,7 @@ func TestHNSWRuntimeSharesFlatAndDefersExact(t *testing.T) { require.NoError(t, indexes.denseNative[field.Name].(interface { Save(context.Context, string) error }).Save(ctx, path)) - artifacts = map[string]string{collectionIndexArtifactKey(field.Name, collectionVectorArtifactKind(IndexTypeHNSW)): path} + artifacts = map[string]string{collectionIndexArtifactKey(field.Name, collectionVectorArtifactKind(indexType)): path} } require.NoError(t, indexes.Close()) } @@ -108,6 +126,16 @@ func TestHNSWRuntimeSharesFlatAndDefersExact(t *testing.T) { } } +func TestImmutableVamanaDocumentsSharedAcrossQuerySnapshots(t *testing.T) { + for _, kind := range []QuantizeType{QuantizeTypeUndefined, QuantizeTypeInt4} { + t.Run(fmt.Sprint(kind), func(t *testing.T) { + params := NewVamanaIndexParams(MetricTypeL2) + params.MaxDegree, params.SearchListSize, params.Quantize = 4, 16, kind + testImmutableDocumentsSharedAcrossQuerySnapshots(t, params) + }) + } +} + func TestImmutableHNSWDocumentsSharedAcrossQuerySnapshots(t *testing.T) { params := NewHNSWIndexParams(MetricTypeL2) params.M, params.EFConstruction, params.Quantize = 4, 16, QuantizeTypeInt4 diff --git a/docs/README.md b/docs/README.md index aa1af75..830a7a8 100644 --- a/docs/README.md +++ b/docs/README.md @@ -34,9 +34,14 @@ two-pass construction, contiguous-memory mode, ID maps, and refinement are disabled. INT4/INT8 enable rotation, verified through the native parameter getter. The Vamana-specific CSV columns record these settings and participate in grouping. -These runs use xvec commit `6a8b120d4284bf16853b2b4ad18465c599e72d82` (merged -PR #91), zvec-go `v0.7.0+rotate`, and the unchanged native zvec library built from -`8321c1314a559fd5f909e92498f43e5194bf9b99`. They use Go 1.27.1 with +The xvec rows were rerun on 2026-09-28 using commit +`5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997` plus the Vamana shared-vector memory patch +`586bbd53868e` (the suffix in `backend_version` identifies this uncommitted patch). +The zvec rows retain the previous measurements using zvec-go `v0.7.0+rotate` +and the native library built from `8321c1314a559fd5f909e92498f43e5194bf9b99`. +[Raw reports and provenance](benchmark-runs/vamana-borrowed-20260928/) retain +commands, binary/dataset/source checksums, process resource measurements, and +the previous CSV. All rows use Go 1.27.1 with `CGO_ENABLED=0`, `GOMAXPROCS=8`, `GOMEMLIMIT=24GiB`, and CPU affinity 0–7 on e2-standard-8. Each run uses a fresh collection, all 100,000 vectors, 1,000 serial queries, K=100, batch size 100, optimize concurrency 8, diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/README.md b/docs/benchmark-runs/vamana-borrowed-20260928/README.md new file mode 100644 index 0000000..9ed9abe --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/README.md @@ -0,0 +1,28 @@ +# Vamana shared-vector memory rerun, 2026-09-28 + +These artifacts support the four updated xvec rows in +[`../../benchmark-vamana.csv`](../../benchmark-vamana.csv). +The zvec rows are unchanged historical measurements. `previous.csv` contains +the runtime-memory rerun; the original historical CSV is in +`../vamana-memory-20260928/previous.csv`. + +- `xvec-*.json`: original benchmark reports and separate `*.resources.json` + files from `wait4` (KiB RSS, seconds for wall/user/system time). +- `xvec-*.log`: benchmark output. +- `metadata.json`: build version, command lines, environment settings, binary, + source-patch and dataset hashes, and successful exit statuses. +- `source.patch`: the exact production-code patch applied to the recorded base, + including the new borrowed-vector implementation. +- `previous.csv`: measurements before this rerun. +- `run.py`: the sequential runner and resource measurement implementation. +- `update_csv.py`: validation and mapping of raw metrics into CSV fields. + +The scripts record this machine's paths; adjust `root` and `repo` when +reproducing elsewhere. Use the source patch with the recorded base commit, +build with `CGO_ENABLED=0`, and download the three files from +`https://assets.zilliz.com/benchmark/cohere_small_100k/` into `dataset/` before +running. Use fresh collection paths. Downloads and compilation occur outside +the measured child processes. `run.py` includes untracked production files +in the patch checksum as well as the tracked-file diff. + +See [the analysis](../../benchmark-vamana-memory.md) for results and limitations. diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/metadata.json b/docs/benchmark-runs/vamana-borrowed-20260928/metadata.json new file mode 100644 index 0000000..168e531 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/metadata.json @@ -0,0 +1,221 @@ +{ + "machine": "e2-standard-8", + "cpu": "AMD EPYC 7B12", + "cpu_affinity": "0-7", + "gomaxprocs": 8, + "gomemlimit": "24GiB", + "cgo_enabled": 0, + "source_files": [ + "internal/db/collection.go", + "internal/core/algorithm/vamana_borrowed_vectors.go", + "internal/ailego/container/heap.go", + "collection.go", + "internal/core/algorithm/vamana_algorithm.go", + "internal/core/algorithm/vamana_quantized_searcher.go" + ], + "base_commit": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997", + "source_patch_sha256": "586bbd53868ea88db925c39cc3f51a91d8d14b4c63940b5432211b369f8357f5", + "backend_version": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e", + "binary_sha256": "16609c1faedfb95be7c41c37dff0bfb508a95c3096b1fddd9d72e0d3f9b91624", + "dataset_sha256": { + "neighbors.parquet": "ee07cdb43a7919bc1ad0525b3bf646036b43be5b7a0aee3ddcef06c21a42594a", + "shuffle_train.parquet": "9590e29f947e21cea6caae8a40e5a8f656937cbecd47859492bfae3fd9806540", + "test.parquet": "252a25003060713a268cd2bf5c5f8fab6159f13772924859487907bef64f1391" + }, + "runs": [ + { + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 1613092, + "peak_rss_mib": 1575.28515625, + "wall_seconds": 132.75145816099985, + "user_cpu_seconds": 814.904934, + "system_cpu_seconds": 9.082705 + }, + { + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 1544248, + "peak_rss_mib": 1508.0546875, + "wall_seconds": 124.68303719800042, + "user_cpu_seconds": 775.998832, + "system_cpu_seconds": 5.110706 + }, + { + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 1521700, + "peak_rss_mib": 1486.03515625, + "wall_seconds": 127.39010814100038, + "user_cpu_seconds": 785.333968, + "system_cpu_seconds": 5.479874 + }, + { + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2003564, + "peak_rss_mib": 1956.60546875, + "wall_seconds": 123.51656048799987, + "user_cpu_seconds": 773.204417, + "system_cpu_seconds": 8.120851 + } + ] +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/previous.csv b/docs/benchmark-runs/vamana-borrowed-20260928/previous.csv new file mode 100644 index 0000000..bf91ed3 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/previous.csv @@ -0,0 +1,9 @@ +machine,backend,backend_version,case,index_type,quantize_type,rotate,use_refiner,enable_mmap,vamana_max_degree,vamana_build_list,vamana_query_list,vamana_alpha,vamana_max_occlusion_size,vamana_saturate_graph,vamana_two_pass_build,vamana_use_contiguous_memory,vamana_use_id_map,k,batch_size,max_docs_per_segment,optimize_concurrency,query_concurrency,concurrency_duration_sec,serial_cooldown_sec,payload_profile,gomaxprocs,gomemlimit,cpu_affinity,go_version,inserted_count,insert_duration_sec,optimize_duration_sec,load_duration_sec,insert_rows_per_sec,serial_queries,serial_qps,recall_at_k_pct,serial_latency_avg_ms,serial_latency_p95_ms,serial_latency_p99_ms,concurrent_queries,concurrent_qps,concurrent_latency_avg_ms,concurrent_latency_p95_ms,concurrent_latency_p99_ms,peak_rss_kib,peak_rss_mib,wall_seconds,user_cpu_seconds,system_cpu_seconds +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-runtime.47525295f56e,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.935516,67.633822,74.57668,14418.536992,1000,228.357499,87.022,4.101898,5.313916,5.992039,46307,1542.966685,5.183016,6.933552,8.076526,2295552,2241.75,118.855395,713.076651,5.329071 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.726085,31.247228,37.006987,17463.938926,1000,388.831267,81.039,2.355339,3.555305,3.973072,79847,2661.173599,3.004251,4.884632,6.706344,580228,566.628906,73.133591,448.838842,10.488299 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-runtime.47525295f56e,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.592064,64.453186,71.052368,15169.756147,1000,228.649883,98.722,4.098633,5.387367,6.119528,44668,1488.734159,5.371761,7.204765,8.442134,2301332,2247.394531,114.822674,688.447641,5.169885 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.096447,28.252815,34.383313,16402.995921,1000,551.288631,98.355,1.622462,2.329426,2.686843,94285,3142.44868,2.5443,3.962252,5.503418,659776,644.3125,69.623703,420.824085,12.125019 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-runtime.47525295f56e,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.589116,63.622921,69.219881,17891.91568,1000,162.311627,99.347,5.882406,7.740576,8.300752,30785,1025.816249,7.795802,10.396236,11.781472,2310924,2256.761719,114.759851,686.68149,5.027617 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.294007,77.133349,82.463938,18889.284977,1000,382.938654,99.198,2.399853,3.193495,3.536057,68113,2270.108375,3.522075,5.363077,7.251954,802112,783.3125,118.577794,784.075142,17.826853 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-runtime.47525295f56e,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.152613,63.988753,68.150493,24081.221226,1000,140.196906,99.383,6.657831,8.646388,9.559788,25369,845.523271,9.456018,14.404441,19.155238,2199088,2147.546875,113.348038,686.835882,5.929749 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.376264,57.116215,61.51392,22850.542618,1000,357.309009,99.392,2.593317,3.435763,3.753755,55841,1860.602216,4.297414,6.194429,8.125008,788484,770.003906,97.844471,638.635022,14.392509 diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/run.py b/docs/benchmark-runs/vamana-borrowed-20260928/run.py new file mode 100644 index 0000000..5bc3d17 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/run.py @@ -0,0 +1,54 @@ +import hashlib, json, os, pathlib, subprocess, time +def file_hash(path): + with path.open('rb') as f: return hashlib.file_digest(f, 'sha256').hexdigest() + +root = pathlib.Path(__file__).parent +repo = pathlib.Path('/home/zhenghaoz/xvec') +source_files = ['internal/db/collection.go', 'internal/core/algorithm/vamana_borrowed_vectors.go', 'internal/ailego/container/heap.go', 'collection.go','internal/core/algorithm/vamana_algorithm.go','internal/core/algorithm/vamana_quantized_searcher.go'] +patch = subprocess.check_output(['git','diff','HEAD','--',*source_files], cwd=repo) +for name in source_files: + tracked = subprocess.run(['git','ls-files','--error-unmatch',name],cwd=repo,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL).returncode == 0 + if not tracked: + delta = subprocess.run(['git','diff','--no-index','--','/dev/null',name],cwd=repo,stdout=subprocess.PIPE,check=False) + assert delta.returncode == 1 + patch += delta.stdout +(root/'source.patch').write_bytes(patch) +revision = subprocess.check_output(['git','rev-parse','HEAD'],cwd=repo,text=True).strip() +patch_hash = hashlib.sha256(patch).hexdigest() +metadata = { + 'machine':'e2-standard-8', 'cpu':'AMD EPYC 7B12', 'cpu_affinity':'0-7', + 'gomaxprocs':8, 'gomemlimit':'24GiB', 'cgo_enabled':0, + 'source_files': source_files, 'base_commit': revision, 'source_patch_sha256': patch_hash, + 'backend_version': revision+'+vamana-borrowed.'+patch_hash[:12], + 'binary_sha256':file_hash(root/'vector-db-bench'), + 'dataset_sha256':{p.name:file_hash(p) for p in sorted((root/'dataset').glob('*.parquet'))}, + 'runs':[] +} +(root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') +for precision in ['int4','int8','fp16','fp32']: + name = 'xvec-'+precision + args = ['taskset','-c','0-7',str(root/'vector-db-bench'),'xvec', + '--path',str(root/(name+'.collection')),'--case-type','Performance768D100K', + '--dataset-dir',str(root/'dataset'),'--skip-download','--index-type','vamana', + '--ef-search','200','--k','100','--batch-size','100','--max-docs-per-segment','10000000', + '--optimize-concurrency','8','--num-concurrency','8','--concurrency-duration','30s', + '--serial-cooldown','3s','--payload-profile','ids_only','--enable-mmap=true', + '--is-using-refiner=false','--output',str(root/(name+'.json'))] + if precision != 'fp32': args += ['--quantize-type',precision] + env = dict(os.environ,GOMAXPROCS='8',GOMEMLIMIT='24GiB') + print('START',name,flush=True) + start = time.monotonic() + with (root/(name+'.log')).open('w') as log: + proc = subprocess.Popen(args,stdout=log,stderr=subprocess.STDOUT,env=env,cwd=repo) + _, status, usage = os.wait4(proc.pid,0) + proc.returncode = os.waitstatus_to_exitcode(status) + elapsed = time.monotonic()-start + metrics = {'precision':precision,'command':args,'exit_code':proc.returncode, + 'peak_rss_kib':usage.ru_maxrss,'peak_rss_mib':usage.ru_maxrss/1024, + 'wall_seconds':elapsed,'user_cpu_seconds':usage.ru_utime,'system_cpu_seconds':usage.ru_stime} + (root/(name+'.resources.json')).write_text(json.dumps(metrics,indent=2)+'\n') + metadata['runs'].append(metrics) + (root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') + print('FINISH',name,json.dumps(metrics),flush=True) + if proc.returncode: raise SystemExit(proc.returncode) + print((root/(name+'.log')).read_text()[-2000:],flush=True) diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/source.patch b/docs/benchmark-runs/vamana-borrowed-20260928/source.patch new file mode 100644 index 0000000..2222bdb --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/source.patch @@ -0,0 +1,1043 @@ +diff --git a/collection.go b/collection.go +index d62eb29..230745c 100644 +--- a/collection.go ++++ b/collection.go +@@ -584,9 +584,22 @@ func openCollectionDenseArtifact( + return core.OpenScalarQuantizedIVFIndex(ctx, path, kind, reformer) + case IndexTypeVamana: + if spec.quantize == QuantizeTypeUndefined { +- return core.OpenVamanaIndex(ctx, path) ++ return core.OpenVamanaIndexWithMmap(ctx, path, useMmap) + } +- return core.OpenScalarQuantizedVamanaIndex(ctx, path, kind, reformer) ++ if field.DataType == DataTypeVectorFP32 { ++ reader, keys, err := collectionEncodedDenseReader(ctx, field, documents) ++ if err != nil { ++ return nil, err ++ } ++ if reader != nil { ++ originals := make(map[uint64][]byte, len(keys)) ++ for position, key := range keys { ++ originals[key] = reader.rows[position] ++ } ++ return core.OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx, path, kind, reformer, originals, useMmap) ++ } ++ } ++ return core.OpenScalarQuantizedVamanaIndexWithMmap(ctx, path, kind, reformer, useMmap) + case IndexTypeDiskANN: + if spec.quantize == QuantizeTypeUndefined { + candidateCount, candidateErr := collectionDenseCandidateCount(ctx, field, documents) +@@ -664,7 +677,7 @@ func (c *Collection) segmentDocumentsLocked(ctx context.Context) ([]collectionSe + if err != nil { + return nil, err + } +- if spec.indexType == IndexTypeHNSW || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { ++ if spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { + borrowedFields[field.Name] = struct{}{} + } + } +@@ -1027,7 +1040,7 @@ func buildCollectionIndexes( + } + if field.DataType.IsDenseVector() { + var exact collectionDenseIndex +- useLazyExact := spec.indexType == IndexTypeHNSW || ++ useLazyExact := spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || + (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) || + (spec.indexType == IndexTypeDiskANN && spec.quantize == QuantizeTypeUndefined) + if field.DataType == DataTypeVectorFP32 && useLazyExact { +@@ -1054,8 +1067,8 @@ func buildCollectionIndexes( + var flat collectionDenseIndex + if spec.quantize == QuantizeTypeUndefined || spec.indexType == IndexTypeHNSWRaBitQ || spec.indexType == IndexTypeIVFRaBitQ { + flat = exact +- } else if spec.indexType != IndexTypeHNSW { +- // Quantized HNSW supplies a shared Flat view after opening the graph. ++ } else if spec.indexType != IndexTypeHNSW && spec.indexType != IndexTypeVamana { ++ // Quantized graphs supply a shared Flat view after opening. + flat, err = buildCollectionDenseFlat(ctx, schema.Name, field, documents, spec) + if err != nil { + return fail(err) +@@ -1081,9 +1094,11 @@ func buildCollectionIndexes( + } + indexes.denseNative[field.Name] = native + if flat == nil { +- quantized, ok := native.(*core.ScalarQuantizedHNSWIndex) ++ quantized, ok := native.(interface { ++ FlatIndex() *core.ScalarQuantizedFlatIndex ++ }) + if !ok { +- return fail(fmt.Errorf("quantized HNSW field %q has an incompatible native index", field.Name)) ++ return fail(fmt.Errorf("quantized graph field %q has an incompatible native index", field.Name)) + } + indexes.denseFlat[field.Name] = quantized.FlatIndex() + } +@@ -2529,7 +2544,7 @@ func buildCollectionDenseVamana( + spec collectionVectorIndex, + workers int, + ) (collectionVamanaIndex, error) { +- candidates, err := collectionDenseCandidates(ctx, field, documents) ++ count, err := collectionDenseCandidateCount(ctx, field, documents) + if err != nil { + return nil, err + } +@@ -2542,6 +2557,24 @@ func buildCollectionDenseVamana( + options.MaxOcclusionSize = core.DefaultVamanaMaxOcclusionSize + } + options.SaturateGraph = spec.vamana.SaturateGraph ++ if field.DataType == DataTypeVectorFP32 { ++ candidates, err := collectionDenseBorrowedCandidates(ctx, field, documents) ++ if err != nil { ++ return nil, err ++ } ++ if spec.quantize == QuantizeTypeUndefined { ++ return core.BuildVamanaWithBorrowedVectors(ctx, int(field.Dimension), options, candidates, workers) ++ } ++ kind, err := toCoreQuantization(spec.quantize) ++ if err != nil { ++ return nil, err ++ } ++ reformer, err := collectionReformer(schemaName, field, spec) ++ if err != nil { ++ return nil, err ++ } ++ return core.BuildScalarQuantizedVamanaWithBorrowedVectors(ctx, int(field.Dimension), options, candidates, workers, kind, reformer) ++ } + var builder *core.VamanaBuilder + if field.DataType == DataTypeVectorFP16 && spec.quantize == QuantizeTypeUndefined { + builder, err = core.NewVamanaBuilderFP16(int(field.Dimension), options) +@@ -2551,17 +2584,27 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- for _, candidate := range candidates { +- if err := builder.Add(ctx, candidate.Key, candidate.Vector); err != nil { ++ if err := builder.Reserve(count); err != nil { ++ return nil, err ++ } ++ for _, document := range documents { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ value, found := document.Fields[field.Name] ++ if !found || value == nil { ++ continue ++ } ++ vector, err := denseValueToFloat32Borrowed(value) ++ if err != nil { ++ return nil, fmt.Errorf("document %d field %q: %w", document.DocID, field.Name, err) ++ } ++ if err := builder.Add(ctx, document.DocID, vector); err != nil { + return nil, err + } +- } +- base, err := builder.BuildInterleavedWithWorkers(ctx, workers) +- if err != nil { +- return nil, err + } + if spec.quantize == QuantizeTypeUndefined { +- return base, nil ++ return builder.BuildInterleavedWithWorkers(ctx, workers) + } + kind, err := toCoreQuantization(spec.quantize) + if err != nil { +@@ -2571,7 +2614,7 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- return core.NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) + } + + func buildCollectionDenseDiskANN( +diff --git a/internal/ailego/container/heap.go b/internal/ailego/container/heap.go +index dd981ef..5e88123 100644 +--- a/internal/ailego/container/heap.go ++++ b/internal/ailego/container/heap.go +@@ -41,6 +41,12 @@ func NewHeapWithCapacity[T any](capacity int, less func(a, b T) bool) *Heap[T] { + // Len returns the number of values in h. + func (h *Heap[T]) Len() int { return len(h.values) } + ++// Clear removes all values while retaining capacity for the next traversal. ++func (h *Heap[T]) Clear() { ++ clear(h.values) ++ h.values = h.values[:0] ++} ++ + // Push inserts value into h. + func (h *Heap[T]) Push(value T) { + h.values = append(h.values, value) +diff --git a/internal/core/algorithm/vamana_algorithm.go b/internal/core/algorithm/vamana_algorithm.go +index 84cba77..3461221 100644 +--- a/internal/core/algorithm/vamana_algorithm.go ++++ b/internal/core/algorithm/vamana_algorithm.go +@@ -22,9 +22,11 @@ import ( + "errors" + "fmt" + "math" ++ "os" + "slices" + "sync" + ++ mmap "github.com/blevesearch/mmap-go" + "github.com/gorse-io/xvec/internal/ailego/container" + "github.com/gorse-io/xvec/internal/ailego/hash" + "github.com/gorse-io/xvec/internal/ailego/io" +@@ -100,6 +102,7 @@ type VamanaBuilder struct { + options VamanaBuildOptions + keys []uint64 + vectors []float32 ++ vectorRows [][]float32 + vectorsFP16 []uint16 + fp16 bool + positions map[uint64]int +@@ -148,6 +151,46 @@ func newBorrowedVamanaBuilder( + return builder, nil + } + ++// Reserve preallocates storage for at least count total vectors. It does not ++// change the builder length or the ownership guarantees of Add. ++func (b *VamanaBuilder) Reserve(count int) error { ++ if b == nil { ++ return errors.New("core: nil Vamana builder") ++ } ++ if count < 0 || uint64(count) >= math.MaxUint32 || (count > 0 && count > maxPlatformInt()/b.dimension) { ++ return ErrVamanaCapacity ++ } ++ b.mu.Lock() ++ defer b.mu.Unlock() ++ if b.built { ++ return ErrBuilderClosed ++ } ++ vectorCapacity := cap(b.vectors) ++ if b.fp16 { ++ vectorCapacity = cap(b.vectorsFP16) ++ } ++ if count <= cap(b.keys) && count*b.dimension <= vectorCapacity { ++ return nil ++ } ++ reservedKeys := make([]uint64, len(b.keys), max(count, len(b.keys))) ++ copy(reservedKeys, b.keys) ++ var reservedVectors []float32 ++ var reservedVectorsFP16 []uint16 ++ if b.fp16 { ++ reservedVectorsFP16 = make([]uint16, len(b.vectorsFP16), max(count*b.dimension, len(b.vectorsFP16))) ++ copy(reservedVectorsFP16, b.vectorsFP16) ++ } else { ++ reservedVectors = make([]float32, len(b.vectors), max(count*b.dimension, len(b.vectors))) ++ copy(reservedVectors, b.vectors) ++ } ++ reservedPositions := make(map[uint64]int, max(count, len(b.positions))) ++ for key, position := range b.positions { ++ reservedPositions[key] = position ++ } ++ b.keys, b.vectors, b.vectorsFP16, b.positions = reservedKeys, reservedVectors, reservedVectorsFP16, reservedPositions ++ return nil ++} ++ + // Add validates and clones one unique vector while the builder is open. + func (b *VamanaBuilder) Add(ctx context.Context, key uint64, vector []float32) error { + if b == nil { +@@ -253,7 +296,7 @@ func (b *VamanaBuilder) build(ctx context.Context, workers int) (*VamanaIndex, e + index := &VamanaIndex{ + dimension: b.dimension, options: b.options, + distance: distance, distanceFP16: distanceFP16, +- keys: b.keys, vectors: b.vectors, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, ++ keys: b.keys, vectors: b.vectors, vectorRows: b.vectorRows, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, + neighbors: make([][]int, len(b.keys)), neighborDistances: make([][]float32, len(b.keys)), + entryPoint: -1, + } +@@ -310,6 +353,7 @@ func (b *VamanaBuilder) build(ctx context.Context, workers int) (*VamanaIndex, e + b.built = true + b.keys = nil + b.vectors = nil ++ b.vectorRows = nil + b.vectorsFP16 = nil + b.positions = nil + return index, nil +@@ -348,7 +392,7 @@ func (b *VamanaBuilder) buildInterleaved(ctx context.Context, workers int) (*Vam + index := &VamanaIndex{ + dimension: b.dimension, options: b.options, + distance: distance, distanceFP16: distanceFP16, +- keys: b.keys, vectors: b.vectors, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, ++ keys: b.keys, vectors: b.vectors, vectorRows: b.vectorRows, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, + neighbors: make([][]int, len(b.keys)), neighborDistances: make([][]float32, len(b.keys)), + entryPoint: -1, + } +@@ -410,6 +454,7 @@ func (b *VamanaBuilder) buildInterleaved(ctx context.Context, workers int) (*Vam + b.built = true + b.keys = nil + b.vectors = nil ++ b.vectorRows = nil + b.vectorsFP16 = nil + b.positions = nil + return index, nil +@@ -438,6 +483,8 @@ type VamanaIndex struct { + distanceFP16 mathutil.DenseDistanceFP16 + keys []uint64 + vectors []float32 ++ vectorRows [][]float32 // Immutable FP32 rows borrowed during collection builds. ++ encodedVectors [][]byte // Immutable originals borrowed by scalar-quantized readers. + vectorsFP16 []uint16 + fp16 bool + vectorMagnitudes []float32 +@@ -562,6 +609,8 @@ type vamanaPruneScratch struct { + } + + type vamanaSearchScratch struct { ++ frontier *container.Heap[vamanaDistanceNode] ++ retained *container.Heap[vamanaDistanceNode] + visited []uint32 + generation uint32 + neighbors []int +@@ -572,12 +621,11 @@ func (i *VamanaIndex) searchBuildCandidates(ctx context.Context, queryPosition, + if limit <= 0 { + return []vamanaDistanceNode{}, nil + } +- better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } +- worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } +- frontier := container.NewHeap(better) +- retained := container.NewHeap(worse) + scratch := i.acquireVamanaSearchScratch() + defer i.releaseVamanaSearchScratch(scratch) ++ frontier, retained := scratch.frontier, scratch.retained ++ frontier.Clear() ++ retained.Clear() + visited, generation := scratch.visited, scratch.generation + distance, err := i.graphDistanceAt(queryPosition, entry) + if err != nil { +@@ -631,12 +679,11 @@ func (i *VamanaIndex) searchBuildCandidatesInterleaved( + if limit <= 0 { + return []vamanaDistanceNode{}, nil + } +- better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } +- worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } +- frontier := container.NewHeap(better) +- retained := container.NewHeap(worse) + scratch := i.acquireVamanaSearchScratch() + defer i.releaseVamanaSearchScratch(scratch) ++ frontier, retained := scratch.frontier, scratch.retained ++ frontier.Clear() ++ retained.Clear() + visited, generation := scratch.visited, scratch.generation + distance, err := i.graphDistanceAt(queryPosition, entry) + if err != nil { +@@ -688,7 +735,10 @@ func (i *VamanaIndex) acquireVamanaSearchScratch() *vamanaSearchScratch { + value := i.searchScratch.Get() + var scratch *vamanaSearchScratch + if value == nil { +- scratch = &vamanaSearchScratch{} ++ scratch = &vamanaSearchScratch{ ++ frontier: container.NewHeap(vamanaDistanceBetter), ++ retained: container.NewHeap(func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) }), ++ } + } else { + scratch = value.(*vamanaSearchScratch) + } +@@ -1030,6 +1080,17 @@ func (i *VamanaIndex) cacheCosineMagnitudes(ctx context.Context, workers int) er + return nil + } + i.vectorMagnitudes = make([]float32, len(i.keys)) ++ if i.encodedVectors != nil { ++ vector := make([]float32, i.dimension) ++ for position, encoded := range i.encodedVectors { ++ if err := ctx.Err(); err != nil { ++ return err ++ } ++ decodeHNSWVector(encoded, vector) ++ i.vectorMagnitudes[position] = mathutil.L2Magnitude(vector) ++ } ++ return nil ++ } + if err := parallel.ParallelFor(ctx, len(i.keys), workers, func(_ context.Context, position int) error { + if i.fp16 { + i.vectorMagnitudes[position] = mathutil.L2MagnitudeFP16(i.vectorFP16At(position)) +@@ -1095,6 +1156,14 @@ func (i *VamanaIndex) calculateMedoid(ctx context.Context) (int, error) { + } + + func (i *VamanaIndex) vectorAt(position int) []float32 { ++ if i.vectorRows != nil { ++ return i.vectorRows[position] ++ } ++ if i.encodedVectors != nil { ++ vector := make([]float32, i.dimension) ++ decodeHNSWVector(i.encodedVectors[position], vector) ++ return vector ++ } + start := position * i.dimension + return i.vectors[start : start+i.dimension] + } +@@ -1134,7 +1203,21 @@ func cloneVamanaIndex(ctx context.Context, source *VamanaIndex) (*VamanaIndex, e + keys: slices.Clone(source.keys), vectors: slices.Clone(source.vectors), vectorsFP16: slices.Clone(source.vectorsFP16), + vectorMagnitudes: slices.Clone(source.vectorMagnitudes), + positions: cloneUint64Positions(source.positions), entryPoint: source.entryPoint, +- neighbors: make([][]int, len(source.neighbors)), neighborDistances: make([][]float32, len(source.neighborDistances)), ++ neighbors: make([][]int, len(source.neighbors)), neighborDistances: make([][]float32, len(source.neighbors)), ++ } ++ if source.vectorRows != nil || source.encodedVectors != nil { ++ clone.vectors = make([]float32, len(source.keys)*source.dimension) ++ for position := range source.keys { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ destination := clone.vectors[position*source.dimension : (position+1)*source.dimension] ++ if source.encodedVectors != nil { ++ decodeHNSWVector(source.encodedVectors[position], destination) ++ } else { ++ copy(destination, source.vectorRows[position]) ++ } ++ } + } + for position := range clone.neighbors { + if position&255 == 0 { +@@ -1143,7 +1226,18 @@ func cloneVamanaIndex(ctx context.Context, source *VamanaIndex) (*VamanaIndex, e + } + } + clone.neighbors[position] = slices.Clone(source.neighbors[position]) +- clone.neighborDistances[position] = slices.Clone(source.neighborDistances[position]) ++ if source.neighborDistances != nil { ++ clone.neighborDistances[position] = slices.Clone(source.neighborDistances[position]) ++ } else { ++ clone.neighborDistances[position] = make([]float32, len(clone.neighbors[position])) ++ for offset, neighbor := range clone.neighbors[position] { ++ distance, err := clone.graphDistanceAt(position, neighbor) ++ if err != nil { ++ return nil, err ++ } ++ clone.neighborDistances[position][offset] = distance ++ } ++ } + } + return clone, nil + } +@@ -1168,8 +1262,22 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + wantVectors := count * index.dimension + validVectors := (!index.fp16 && len(index.vectors) == wantVectors && len(index.vectorsFP16) == 0) || + (index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == wantVectors) ++ if index.vectorRows != nil { ++ validVectors = !index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == 0 && index.encodedVectors == nil && len(index.vectorRows) == count ++ } ++ if index.encodedVectors != nil { ++ validVectors = !index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == 0 && index.vectorRows == nil && len(index.encodedVectors) == count ++ for _, vector := range index.encodedVectors { ++ if len(vector) != index.dimension*4 { ++ validVectors = false ++ break ++ } ++ } ++ } ++ // Edge-distance caches are unnecessary for immutable quantized readers. ++ missingDistances := index.encodedVectors != nil && index.neighborDistances == nil + if !validVectors || len(index.positions) != count || +- len(index.neighbors) != count || len(index.neighborDistances) != count { ++ len(index.neighbors) != count || (!missingDistances && len(index.neighborDistances) != count) { + return errors.New("core: inconsistent Vamana storage") + } + if (index.options.Metric == MetricCosine && len(index.vectorMagnitudes) != count) || +@@ -1179,28 +1287,40 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + if (count == 0 && index.entryPoint != -1) || (count > 0 && (index.entryPoint < 0 || index.entryPoint >= count)) { + return errors.New("core: invalid Vamana entry point") + } +- seenKeys := make(map[uint64]struct{}, count) ++ // The key-to-position bijection also detects duplicate keys. ++ seenNeighbors := make(map[int]struct{}, index.options.MaxDegree) ++ var vectorScratch []float32 ++ if index.fp16 || index.encodedVectors != nil { ++ vectorScratch = make([]float32, index.dimension) ++ } + for position, key := range index.keys { + if position&255 == 0 { + if err := ctx.Err(); err != nil { + return err + } + } +- if _, found := seenKeys[key]; found || index.positions[key] != position { ++ if mapped, found := index.positions[key]; !found || mapped != position { + return errors.New("core: invalid Vamana key map") + } +- seenKeys[key] = struct{}{} + if index.fp16 { +- if err := validateTrainingVector(float32VectorFromFP16(index.vectorFP16At(position)), index.dimension); err != nil { ++ for offset, value := range index.vectorFP16At(position) { ++ vectorScratch[offset] = utility.Float16BitsToFloat32(value) ++ } ++ if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { ++ return err ++ } ++ } else if index.encodedVectors != nil { ++ decodeHNSWVector(index.encodedVectors[position], vectorScratch) ++ if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { + return err + } + } else if err := validateTrainingVector(index.vectorAt(position), index.dimension); err != nil { + return err + } +- if len(index.neighbors[position]) > index.options.MaxDegree || len(index.neighbors[position]) != len(index.neighborDistances[position]) { ++ if len(index.neighbors[position]) > index.options.MaxDegree || (!missingDistances && len(index.neighbors[position]) != len(index.neighborDistances[position])) { + return errors.New("core: invalid Vamana degree") + } +- seenNeighbors := make(map[int]struct{}, len(index.neighbors[position])) ++ clear(seenNeighbors) + for offset, neighbor := range index.neighbors[position] { + if neighbor < 0 || neighbor >= count || neighbor == position { + return errors.New("core: invalid Vamana neighbor") +@@ -1209,6 +1329,9 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + return errors.New("core: duplicate Vamana neighbor") + } + seenNeighbors[neighbor] = struct{}{} ++ if missingDistances { ++ continue ++ } + distance := index.neighborDistances[position][offset] + if math.IsNaN(float64(distance)) || math.IsInf(float64(distance), 0) { + return errors.New("core: invalid Vamana neighbor distance") +@@ -1369,7 +1492,13 @@ func (i *VamanaIndex) searchVamana(ctx context.Context, query []float32, options + return denseDistances(i.options.Metric, query, batch.vectors, queryMagnitude, batch.magnitudes, scores) + } + prefetch := func(neighbors []int) { +- if !i.fp16 { ++ if i.vectorRows != nil { ++ batch.vectors = batch.vectors[:0] ++ for _, position := range neighbors { ++ batch.vectors = append(batch.vectors, i.vectorRows[position]) ++ } ++ prefetchDenseHNSWRows(batch.vectors, options.PrefetchOffset, options.PrefetchLines) ++ } else if !i.fp16 && i.encodedVectors == nil { + prefetchDenseHNSWNeighbors(i.vectors, i.dimension, neighbors, options.PrefetchOffset, options.PrefetchLines) + } + } +@@ -1402,13 +1531,14 @@ func searchVamanaGraph( + worse := func(left, right hnswScoredNode) bool { return resultBetter(right, left) } + frontier := container.NewHeap(better) + accepted := container.NewHeap(worse) +- visited := make([]bool, len(keys)) ++ visited := acquireHNSWVisited(len(keys)) ++ defer releaseHNSWVisited(visited) + score, err := scoreAt(entry) + if err != nil { + return nil, fmt.Errorf("core: score Vamana entry point: %w", err) + } + start := hnswScoredNode{position: entry, score: score} +- visited[entry] = true ++ visited.mark(entry) + frontier.Push(start) + if acceptVamanaNode(metric, keys, start, options.SearchOptions) { + accepted.Push(start) +@@ -1429,10 +1559,10 @@ func searchVamanaGraph( + batch.positions = batch.positions[:0] + batch.scores = batch.scores[:0] + for _, neighbor := range adjacent { +- if visited[neighbor] { ++ if visited.seen(neighbor) { + continue + } +- visited[neighbor] = true ++ visited.mark(neighbor) + batch.positions = append(batch.positions, neighbor) + batch.scores = append(batch.scores, 0) + } +@@ -1508,7 +1638,8 @@ func searchVamanaGraphBlockHeap( + ) ([]Result, error) { + capacity := min(len(keys), max(options.TopK, options.EFSearch)) + batch.blockHeap.Reset(capacity, cap(batch.positions)) +- states := make([]uint8, len(keys)) ++ states := acquireHNSWVisited(len(keys)) ++ defer releaseHNSWVisited(states) + score, err := scoreAt(entry) + if err != nil { + return nil, fmt.Errorf("core: score Vamana entry point: %w", err) +@@ -1519,7 +1650,7 @@ func searchVamanaGraphBlockHeap( + batch.blockHeap.pushBlockWithTies([]float32{entryDistance}, []uint32{entryID}, []uint64{entryTie}) + batch.overflow = batch.overflow[:0] + overflowCursor := 0 +- states[entry] = 1 ++ states.mark(entry) + + for batch.blockHeap.HasNext() || overflowCursor < len(batch.overflow) { + if err := ctx.Err(); err != nil { +@@ -1532,10 +1663,10 @@ func searchVamanaGraphBlockHeap( + current = batch.overflow[overflowCursor] + overflowCursor++ + } +- if states[current] == 2 { ++ if states.expanded(int(current)) { + continue + } +- states[current] = 2 ++ states.markExpanded(int(current)) + adjacent := neighbors[int(current)] + if prefetch != nil { + prefetch(adjacent) +@@ -1545,10 +1676,10 @@ func searchVamanaGraphBlockHeap( + batch.ties = batch.ties[:0] + batch.scores = batch.scores[:0] + for _, neighbor := range adjacent { +- if states[neighbor] != 0 { ++ if states.seen(neighbor) { + continue + } +- states[neighbor] = 1 ++ states.mark(neighbor) + batch.positions = append(batch.positions, neighbor) + batch.ids = append(batch.ids, uint32(neighbor)) + batch.ties = append(batch.ties, keys[neighbor]) +@@ -1630,7 +1761,7 @@ func (i *VamanaIndex) Add(ctx context.Context, key uint64, vector []float32) err + i.mu.RUnlock() + return fmt.Errorf("%w: %d", ErrDuplicateKey, key) + } +- vectorCount := len(i.vectors) ++ vectorCount := len(i.keys) * i.dimension + if i.fp16 { + vectorCount = len(i.vectorsFP16) + } +@@ -1683,6 +1814,7 @@ func (i *VamanaIndex) Add(ctx context.Context, key uint64, vector []float32) err + i.mu.Unlock() + return err + } ++ i.vectorRows, i.encodedVectors = nil, nil + i.keys, i.vectors, i.vectorsFP16, i.vectorMagnitudes, i.positions = working.keys, working.vectors, working.vectorsFP16, working.vectorMagnitudes, working.positions + i.neighbors, i.neighborDistances, i.entryPoint = working.neighbors, working.neighborDistances, working.entryPoint + i.mu.Unlock() +@@ -1724,17 +1856,24 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { + if i == nil { + return fmt.Errorf("%w: nil index", ErrInvalidVamanaFile) + } ++ // Add publishes a new generation without mutating existing storage. Holding ++ // the read lock keeps this generation stable while it is streamed to disk. + i.mu.RLock() +- snapshot, err := cloneVamanaIndex(ctx, i) +- i.mu.RUnlock() +- if err != nil { +- return err +- } +- encoded, err := encodeVamanaIndex(ctx, snapshot) +- if err != nil { ++ defer i.mu.RUnlock() ++ if err := ioutil.WriteFileAtomicFunc(ctx, path, 0o600, func(file *os.File) error { ++ if _, err := file.Write(make([]byte, vamanaHeaderSize)); err != nil { ++ return err ++ } ++ header, err := writeVamanaPayload(ctx, i, func(data []byte) error { ++ _, err := file.Write(data) ++ return err ++ }) ++ if err != nil { ++ return err ++ } ++ _, err = file.WriteAt(header, 0) + return err +- } +- if err := ioutil.WriteFileAtomic(ctx, path, encoded, 0o600); err != nil { ++ }); err != nil { + return fmt.Errorf("core: save Vamana file: %w", err) + } + return nil +@@ -1742,6 +1881,17 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { + + // OpenVamanaIndex reads and verifies a native Go Vamana artifact. + func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { ++ return OpenVamanaIndexWithMmap(ctx, path, false) ++} ++ ++// OpenVamanaIndexWithMmap decodes an owned index using an optional temporary ++// read-only mapping, avoiding a serialized full-file copy in the Go heap. ++// No references to the mapping survive the call. ++func OpenVamanaIndexWithMmap(ctx context.Context, path string, useMmap bool) (*VamanaIndex, error) { ++ return openVamanaIndexWithStorage(ctx, path, useMmap, nil) ++} ++ ++func openVamanaIndexWithStorage(ctx context.Context, path string, useMmap bool, originals map[uint64][]byte) (index *VamanaIndex, resultErr error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana open context") + } +@@ -1751,11 +1901,24 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + if path == "" { + return nil, fmt.Errorf("%w: empty path", ErrInvalidVamanaFile) + } ++ if useMmap { ++ file, err := os.Open(path) ++ if err != nil { ++ return nil, err ++ } ++ defer func() { resultErr = errors.Join(resultErr, file.Close()) }() ++ encoded, err := mmap.Map(file, mmap.RDONLY, 0) ++ if err != nil { ++ return nil, err ++ } ++ defer func() { resultErr = errors.Join(resultErr, encoded.Unmap()) }() ++ return decodeVamanaIndexWithStorage(ctx, encoded, originals) ++ } + encoded, err := readHNSWFile(ctx, path) + if err != nil { + return nil, fmt.Errorf("core: read Vamana file: %w", err) + } +- index, err := decodeVamanaIndex(ctx, encoded) ++ index, err = decodeVamanaIndexWithStorage(ctx, encoded, originals) + if err != nil { + return nil, fmt.Errorf("core: open Vamana file: %w", err) + } +@@ -1763,6 +1926,21 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + } + + func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) { ++ encoded := make([]byte, vamanaHeaderSize) ++ header, err := writeVamanaPayload(ctx, index, func(data []byte) error { ++ encoded = append(encoded, data...) ++ return nil ++ }) ++ if err != nil { ++ return nil, err ++ } ++ copy(encoded, header) ++ return encoded, nil ++} ++ ++// writeVamanaPayload bounds serialization scratch space independently of vector ++// count. The returned header contains the checksum of the streamed payload. ++func writeVamanaPayload(ctx context.Context, index *VamanaIndex, write func([]byte) error) ([]byte, error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana encode context") + } +@@ -1786,7 +1964,21 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + if err != nil { + return nil, err + } +- payload := make([]byte, 0, payloadSize) ++ payload := make([]byte, 0, 64<<10) ++ var checksum uint32 ++ written := 0 ++ flush := func() error { ++ if err := ctx.Err(); err != nil { ++ return err ++ } ++ if err := write(payload); err != nil { ++ return err ++ } ++ checksum = hashutil.UpdateCRC32C(checksum, payload) ++ written += len(payload) ++ payload = payload[:0] ++ return nil ++ } + for position, key := range index.keys { + if position&1023 == 0 { + if err := ctx.Err(); err != nil { +@@ -1794,6 +1986,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint64(payload, key) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + if index.fp16 { + for position, value := range index.vectorsFP16 { +@@ -1803,17 +2000,41 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint16(payload, value) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } ++ } ++ } else if index.encodedVectors != nil { ++ for _, raw := range index.encodedVectors { ++ for len(raw) != 0 { ++ n := min(len(raw), (64<<10)-len(payload)) ++ payload = append(payload, raw[:n]...) ++ raw = raw[n:] ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } ++ } + } + } else { +- for position, value := range index.vectors { +- if position&16383 == 0 { +- if err := ctx.Err(); err != nil { +- return nil, err ++ for position := range index.keys { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ for _, value := range index.vectorAt(position) { ++ payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } + } + } +- payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) + } + } ++ + for position, adjacent := range index.neighbors { + if position&255 == 0 { + if err := ctx.Err(); err != nil { +@@ -1821,11 +2042,24 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint32(payload, uint32(len(adjacent))) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + for _, neighbor := range adjacent { + payload = binary.LittleEndian.AppendUint32(payload, uint32(neighbor)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } +- if len(payload) != payloadSize { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ if written != payloadSize { + return nil, fmt.Errorf("%w: internal payload length", ErrInvalidVamanaFile) + } + header := make([]byte, vamanaHeaderSize) +@@ -1853,12 +2087,16 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + entry = uint64(index.entryPoint) + } + binary.LittleEndian.PutUint64(header[72:80], entry) +- binary.LittleEndian.PutUint32(header[80:84], hashutil.CRC32C(payload)) ++ binary.LittleEndian.PutUint32(header[80:84], checksum) + binary.LittleEndian.PutUint32(header[124:128], hashutil.CRC32C(header[:124])) +- return append(header, payload...), nil ++ return header, nil + } + + func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error) { ++ return decodeVamanaIndexWithStorage(ctx, encoded, nil) ++} ++ ++func decodeVamanaIndexWithStorage(ctx context.Context, encoded []byte, originals map[uint64][]byte) (*VamanaIndex, error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana decode context") + } +@@ -1964,7 +2202,24 @@ func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error + } + index.keys[position], index.positions[key] = key, position + } +- if fp16 { ++ if originals != nil { ++ if fp16 || len(originals) != count { ++ return nil, fmt.Errorf("%w: incompatible encoded originals", ErrInvalidVamanaFile) ++ } ++ index.encodedVectors = make([][]byte, count) ++ index.neighborDistances = nil ++ for position, key := range index.keys { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ original, found := originals[key] ++ if !found || len(original) != dimension*4 || !bytes.Equal(original, payload[offset:offset+dimension*4]) { ++ return nil, fmt.Errorf("%w: encoded vector %d differs from artifact", ErrInvalidVamanaFile, key) ++ } ++ index.encodedVectors[position] = original[:len(original):len(original)] ++ offset += dimension * 4 ++ } ++ } else if fp16 { + index.distanceFP16, err = denseDistanceFP16(options.Metric) + if err != nil { + return nil, err +@@ -2022,6 +2277,9 @@ func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error + return nil, fmt.Errorf("%w: inconsistent edge payload", ErrInvalidVamanaFile) + } + for position, adjacent := range index.neighbors { ++ if index.neighborDistances == nil { ++ break ++ } + if position&255 == 0 { + if err := ctx.Err(); err != nil { + return nil, err +diff --git a/internal/core/algorithm/vamana_quantized_searcher.go b/internal/core/algorithm/vamana_quantized_searcher.go +index 6cbba85..456e1fd 100644 +--- a/internal/core/algorithm/vamana_quantized_searcher.go ++++ b/internal/core/algorithm/vamana_quantized_searcher.go +@@ -47,8 +47,26 @@ func NewScalarQuantizedVamanaIndex( + if err != nil { + return nil, err + } +- vectors, err := newOwnedScalarQuantizedVectors( +- ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, ++ return newOwnedScalarQuantizedVamanaIndex(ctx, snapshot, kind, reformer) ++} ++ ++// BuildScalarQuantizedInterleavedWithWorkers transfers the completed graph to ++// an immutable quantized index without cloning its vectors and adjacency. ++func (b *VamanaBuilder) BuildScalarQuantizedInterleavedWithWorkers(ctx context.Context, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { ++ base, err := b.BuildInterleavedWithWorkers(ctx, workers) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} ++ ++func newOwnedScalarQuantizedVamanaIndex(ctx context.Context, snapshot *VamanaIndex, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { ++ var reader DenseVectorReader ++ if snapshot.encodedVectors != nil { ++ reader = encodedHNSWVectorReader(snapshot.encodedVectors) ++ } ++ vectors, err := newScalarQuantizedVectorStorageWithReader( ++ ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, snapshot.vectorRows, reader, + ) + if err != nil { + return nil, err +@@ -72,7 +90,26 @@ func OpenScalarQuantizedVamanaIndex(ctx context.Context, path string, kind Quant + if err != nil { + return nil, err + } +- return NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} ++ ++// FlatIndex returns an immutable linear-search view sharing scalar codes and ++// original vectors with the graph. No vectors are copied or quantized again. ++func (i *ScalarQuantizedVamanaIndex) FlatIndex() *ScalarQuantizedFlatIndex { ++ if i == nil { ++ return nil ++ } ++ return &ScalarQuantizedFlatIndex{vectors: i.vectors} ++} ++ ++// OpenScalarQuantizedVamanaIndexWithMmap uses a temporary read-only file mapping ++// while decoding, releasing it before scalar codes are reconstructed. ++func OpenScalarQuantizedVamanaIndexWithMmap(ctx context.Context, path string, kind Quantization, reformer DenseReformer, useMmap bool) (*ScalarQuantizedVamanaIndex, error) { ++ base, err := OpenVamanaIndexWithMmap(ctx, path, useMmap) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) + } + + func (i *ScalarQuantizedVamanaIndex) Dimension() int { +@@ -188,3 +225,22 @@ var ( + _ DenseSearcher = (*ScalarQuantizedVamanaIndex)(nil) + _ DenseQuerySearcher = (*ScalarQuantizedVamanaIndex)(nil) + ) ++ ++// OpenScalarQuantizedVamanaIndexWithEncodedVectors verifies persisted originals ++// against immutable collection-owned little-endian FP32 bytes, then retains ++// those bytes for refinement instead of a second full decoded vector array. ++// The caller must keep the bytes immutable and alive for the index's lifetime. ++// The map and temporary artifact mapping are not retained. ++func OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx context.Context, path string, kind Quantization, reformer DenseReformer, originals map[uint64][]byte, useMmap bool) (*ScalarQuantizedVamanaIndex, error) { ++ if originals == nil { ++ return nil, errors.New("core: nil encoded Vamana originals") ++ } ++ if !kind.valid() { ++ return nil, ErrInvalidQuantization ++ } ++ base, err := openVamanaIndexWithStorage(ctx, path, useMmap, originals) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} +diff --git a/internal/db/collection.go b/internal/db/collection.go +index a13c666..9e0ff97 100644 +--- a/internal/db/collection.go ++++ b/internal/db/collection.go +@@ -538,7 +538,7 @@ func (c *CollectionStore) OptimizationNeeded(ctx context.Context) (bool, error) + if c.closed { + return false, ErrCollectionClosed + } +- if writing := c.manager.Writing(); writing != nil && len(writing.Documents()) != 0 { ++ if writing := c.manager.Writing(); writing != nil && writing.Metadata().DocCount != 0 { + return true, nil + } + if c.manager.Deletes().Count() != 0 { +diff --git a/internal/core/algorithm/vamana_borrowed_vectors.go b/internal/core/algorithm/vamana_borrowed_vectors.go +new file mode 100644 +index 0000000..51236e4 +--- /dev/null ++++ b/internal/core/algorithm/vamana_borrowed_vectors.go +@@ -0,0 +1,79 @@ ++// Copyright 2026-present the xvec project ++// ++// Licensed under the Apache License, Version 2.0 (the "License"); ++// you may not use this file except in compliance with the License. ++// You may obtain a copy of the License at ++// ++// http://www.apache.org/licenses/LICENSE-2.0 ++// ++// Unless required by applicable law or agreed to in writing, software ++// distributed under the License is distributed on an "AS IS" BASIS, ++// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. ++// See the License for the specific language governing permissions and ++// limitations under the License. ++ ++package core ++ ++import ( ++ "context" ++ "errors" ++ "fmt" ++) ++ ++// BuildVamanaWithBorrowedVectors builds an index over immutable FP32 rows. ++// The caller must keep the rows immutable and alive for the index's lifetime. ++// The candidate slice itself is not retained. Add uses copy-on-write storage. ++func BuildVamanaWithBorrowedVectors(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate, workers int) (*VamanaIndex, error) { ++ builder, err := newVamanaBuilderWithBorrowedRows(ctx, dimension, options, candidates) ++ if err != nil { ++ return nil, err ++ } ++ return builder.BuildInterleavedWithWorkers(ctx, workers) ++} ++ ++// BuildScalarQuantizedVamanaWithBorrowedVectors builds and quantizes immutable ++// FP32 rows without cloning the originals. Ownership matches ++// BuildVamanaWithBorrowedVectors; codes and topology are owned by the result. ++func BuildScalarQuantizedVamanaWithBorrowedVectors(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { ++ if !kind.valid() { ++ return nil, ErrInvalidQuantization ++ } ++ builder, err := newVamanaBuilderWithBorrowedRows(ctx, dimension, options, candidates) ++ if err != nil { ++ return nil, err ++ } ++ return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) ++} ++ ++func newVamanaBuilderWithBorrowedRows(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate) (*VamanaBuilder, error) { ++ if ctx == nil { ++ return nil, errors.New("core: nil borrowed Vamana context") ++ } ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ builder, err := NewVamanaBuilder(dimension, options) ++ if err != nil { ++ return nil, err ++ } ++ if len(candidates) > maxPlatformInt()/dimension { ++ return nil, ErrVamanaCapacity ++ } ++ builder.keys = make([]uint64, len(candidates)) ++ builder.vectorRows = make([][]float32, len(candidates)) ++ builder.positions = make(map[uint64]int, len(candidates)) ++ for position, candidate := range candidates { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ if err := validateTrainingVector(candidate.Vector, dimension); err != nil { ++ return nil, err ++ } ++ if _, duplicate := builder.positions[candidate.Key]; duplicate { ++ return nil, fmt.Errorf("%w: %d", ErrDuplicateKey, candidate.Key) ++ } ++ builder.keys[position], builder.positions[candidate.Key] = candidate.Key, position ++ builder.vectorRows[position] = candidate.Vector[:dimension:dimension] ++ } ++ return builder, nil ++} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/update_csv.py b/docs/benchmark-runs/vamana-borrowed-20260928/update_csv.py new file mode 100644 index 0000000..673d7d3 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/update_csv.py @@ -0,0 +1,43 @@ +import csv, json, pathlib, shutil +root=pathlib.Path(__file__).parent +repo=pathlib.Path('/home/zhenghaoz/xvec') +archive=repo/'docs/benchmark-runs/vamana-borrowed-20260928' +meta=json.loads((root/'metadata.json').read_text()) +assert len(meta['runs'])==4 and all(r['exit_code']==0 for r in meta['runs']) +with (archive/'previous.csv').open() as f: + reader=csv.DictReader(f); fields=reader.fieldnames; rows=list(reader) +for row in rows: + if row['backend']!='xvec': continue + kind=row['quantize_type']; name='xvec-'+kind + report=json.loads((root/(name+'.json')).read_text()); resource=json.loads((root/(name+'.resources.json')).read_text()) + c=report['config']; load=report['load']; serial=report['serial']; conc=report['concurrent'][0] + assert load['rows']==100000 and serial['queries']==1000 and conc['concurrency']==8 + assert len(report['concurrent'])==1 and report['case']['name']==row['case'] + assert report['system']['go_version']==row['go_version'] + assert c['backend']=='xvec' and c['index_type']=='vamana' + assert c['quantize_type']==('none' if kind=='fp32' else kind) + assert c['ef_search']==200 and c['concurrency_duration']=='30s' and c['serial_cooldown']=='3s' + for key in ['use_refiner','enable_mmap','k','batch_size','max_docs_per_segment','optimize_concurrency','payload_profile']: + assert str(c[key]).lower()==row[key],(key,c[key],row[key]) + row['backend_version']=meta['backend_version'] + row['inserted_count']=load['rows'] + for key in ['insert_duration_sec','optimize_duration_sec','load_duration_sec']: row[key]=load[key] + row['insert_rows_per_sec']=load['rows_per_second'] + row['recall_at_k_pct']=serial['recall']*100 + for prefix,metrics in [('serial',serial),('concurrent',conc)]: + for key in ['queries','qps','latency_avg_ms','latency_p95_ms','latency_p99_ms']: row[prefix+'_'+key]=metrics[key] + for key in ['peak_rss_kib','peak_rss_mib','wall_seconds','user_cpu_seconds','system_cpu_seconds']: row[key]=resource[key] + for key,value in row.items(): + if isinstance(value,float): row[key]=f'{value:.6f}'.rstrip('0').rstrip('.') + for suffix in ['.json','.resources.json','.log']: shutil.copy2(root/(name+suffix),archive/(name+suffix)) +with (repo/'docs/benchmark-vamana.csv').open('w') as f: + writer=csv.DictWriter(f,fields,lineterminator='\n'); writer.writeheader(); writer.writerows(rows) +shutil.copy2(root/'metadata.json',archive/'metadata.json') +# Store the exact commands and wait4 implementation for reproducibility. +shutil.copy2(root/'run.py',archive/'run.py') +shutil.copy2(root/'update_csv.py',archive/'update_csv.py') +with (archive/'previous.csv').open() as f: old={r['quantize_type']:r for r in csv.DictReader(f) if r['backend']=='xvec'} +for row in rows: + if row['backend']=='xvec': + prior=old[row['quantize_type']] + print(row['quantize_type'],'RSS MiB',prior['peak_rss_mib'],'->',row['peak_rss_mib'], 'change%',round((float(row['peak_rss_mib'])/float(prior['peak_rss_mib'])-1)*100,2),'recall',row['recall_at_k_pct']) diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.json new file mode 100644 index 0000000..17c030f --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:29:52.88551989Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "fp16", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.00939112, + "optimize_duration_sec": 77.357544103, + "load_duration_sec": 83.374320352, + "rows_per_second": 16640.620988570303, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 137.83846789039637, + "recall": 0.9934900000000035, + "latency_avg_ms": 6.962300731999995, + "latency_p95_ms": 9.219942999999999, + "latency_p99_ms": 10.0609182 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 26480, + "qps": 882.0812992867798, + "latency_avg_ms": 9.066431377756748, + "latency_p95_ms": 12.23844695, + "latency_p99_ms": 14.063463599999995 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.00939112, + "optimize_duration": 77.357544103, + "load_duration": 83.374320352, + "qps": 882.0812992867798, + "recall": 0.9934900000000035, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0100609182, + "serial_latency_p95": 0.009219943, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 882.0812992867798 + ], + "conc_latency_p99_list": [ + 0.014063463599999996 + ], + "conc_latency_p95_list": [ + 0.01223844695 + ], + "conc_latency_avg_list": [ + 0.009066431377756748 + ] + } +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.log b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.log new file mode 100644 index 0000000..4185f69 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (16640.9 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-borrowed-20260928/dataset +load rows=100000 duration=83.374s insert=6.009s optimize=77.358s rows/s=16640.6 +serial queries=1000 qps=137.84 recall=0.9935 mrr=0.0000 ndcg=0.0000 avg=6.962ms p95=9.220ms p99=10.061ms +concurrent workers=8 queries=26480 qps=882.08 avg=9.066ms p95=12.238ms p99=14.063ms +result=/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.json diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.resources.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.resources.json new file mode 100644 index 0000000..3c3d557 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp16.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 1521700, + "peak_rss_mib": 1486.03515625, + "wall_seconds": 127.39010814100038, + "user_cpu_seconds": 785.333968, + "system_cpu_seconds": 5.479874 +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.json new file mode 100644 index 0000000..c276d00 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:32:00.275013351Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "none", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 4.6837359549999995, + "optimize_duration_sec": 75.022232053, + "load_duration_sec": 79.714969667, + "rows_per_second": 21350.477687207713, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 204.62903553545596, + "recall": 0.9937400000000033, + "latency_avg_ms": 4.537134413999999, + "latency_p95_ms": 5.77674705, + "latency_p99_ms": 6.1920792 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 29375, + "qps": 978.7741153764767, + "latency_avg_ms": 8.168969618314856, + "latency_p95_ms": 11.059548999999999, + "latency_p99_ms": 12.92071079999997 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 4.6837359549999995, + "optimize_duration": 75.022232053, + "load_duration": 79.714969667, + "qps": 978.7741153764767, + "recall": 0.9937400000000033, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.006192079200000001, + "serial_latency_p95": 0.00577674705, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 978.7741153764767 + ], + "conc_latency_p99_list": [ + 0.01292071079999997 + ], + "conc_latency_p95_list": [ + 0.011059548999999998 + ], + "conc_latency_avg_list": [ + 0.008168969618314856 + ] + } +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.log b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.log new file mode 100644 index 0000000..b037e71 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (21350.9 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-borrowed-20260928/dataset +load rows=100000 duration=79.715s insert=4.684s optimize=75.022s rows/s=21350.5 +serial queries=1000 qps=204.63 recall=0.9937 mrr=0.0000 ndcg=0.0000 avg=4.537ms p95=5.777ms p99=6.192ms +concurrent workers=8 queries=29375 qps=978.77 avg=8.169ms p95=11.060ms p99=12.921ms +result=/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.json diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.resources.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.resources.json new file mode 100644 index 0000000..6f76cdb --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-fp32.resources.json @@ -0,0 +1,47 @@ +{ + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2003564, + "peak_rss_mib": 1956.60546875, + "wall_seconds": 123.51656048799987, + "user_cpu_seconds": 773.204417, + "system_cpu_seconds": 8.120851 +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.json new file mode 100644 index 0000000..9cab649 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:25:35.445852461Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int4", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 8.996762556, + "optimize_duration_sec": 81.640320801, + "load_duration_sec": 90.645780026, + "rows_per_second": 11115.109393801811, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 212.11322255530092, + "recall": 0.8701300000000001, + "latency_avg_ms": 4.4289291179999974, + "latency_p95_ms": 5.76836345, + "latency_p99_ms": 6.701301900000001 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 42322, + "qps": 1410.056421647557, + "latency_avg_ms": 5.67052223533857, + "latency_p95_ms": 7.61011645, + "latency_p99_ms": 9.019854610000001 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 8.996762556, + "optimize_duration": 81.640320801, + "load_duration": 90.645780026, + "qps": 1410.056421647557, + "recall": 0.8701300000000001, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0067013019000000005, + "serial_latency_p95": 0.00576836345, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1410.056421647557 + ], + "conc_latency_p99_list": [ + 0.009019854610000001 + ], + "conc_latency_p95_list": [ + 0.007610116449999999 + ], + "conc_latency_avg_list": [ + 0.00567052223533857 + ] + } +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.log b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.log new file mode 100644 index 0000000..78f7c2f --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (11115.2 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-borrowed-20260928/dataset +load rows=100000 duration=90.646s insert=8.997s optimize=81.640s rows/s=11115.1 +serial queries=1000 qps=212.11 recall=0.8701 mrr=0.0000 ndcg=0.0000 avg=4.429ms p95=5.768ms p99=6.701ms +concurrent workers=8 queries=42322 qps=1410.06 avg=5.671ms p95=7.610ms p99=9.020ms +result=/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.json diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.resources.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.resources.json new file mode 100644 index 0000000..51475e4 --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int4.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 1613092, + "peak_rss_mib": 1575.28515625, + "wall_seconds": 132.75145816099985, + "user_cpu_seconds": 814.904934, + "system_cpu_seconds": 9.082705 +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.json new file mode 100644 index 0000000..ed64f7d --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:27:48.199060822Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int8", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.900676102, + "optimize_duration_sec": 75.83793905, + "load_duration_sec": 82.746722591, + "rows_per_second": 14491.333678306873, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 201.40214980364527, + "recall": 0.9872600000000076, + "latency_avg_ms": 4.6701495090000105, + "latency_p95_ms": 6.22249045, + "latency_p99_ms": 6.9524939 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 39706, + "qps": 1322.8135344901111, + "latency_avg_ms": 6.045327178335768, + "latency_p95_ms": 8.3432715, + "latency_p99_ms": 9.731152999999994 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.900676102, + "optimize_duration": 75.83793905, + "load_duration": 82.746722591, + "qps": 1322.8135344901111, + "recall": 0.9872600000000076, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0069524939000000004, + "serial_latency_p95": 0.00622249045, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1322.8135344901111 + ], + "conc_latency_p99_list": [ + 0.009731152999999994 + ], + "conc_latency_p95_list": [ + 0.0083432715 + ], + "conc_latency_avg_list": [ + 0.006045327178335768 + ] + } +} diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.log b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.log new file mode 100644 index 0000000..6406cfd --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (14491.5 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-borrowed-20260928/dataset +load rows=100000 duration=82.747s insert=6.901s optimize=75.838s rows/s=14491.3 +serial queries=1000 qps=201.40 recall=0.9873 mrr=0.0000 ndcg=0.0000 avg=4.670ms p95=6.222ms p99=6.952ms +concurrent workers=8 queries=39706 qps=1322.81 avg=6.045ms p95=8.343ms p99=9.731ms +result=/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.json diff --git a/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.resources.json b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.resources.json new file mode 100644 index 0000000..cdbc38e --- /dev/null +++ b/docs/benchmark-runs/vamana-borrowed-20260928/xvec-int8.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-borrowed-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-borrowed-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-borrowed-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 1544248, + "peak_rss_mib": 1508.0546875, + "wall_seconds": 124.68303719800042, + "user_cpu_seconds": 775.998832, + "system_cpu_seconds": 5.110706 +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/README.md b/docs/benchmark-runs/vamana-memory-20260928/README.md new file mode 100644 index 0000000..fdb9da5 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/README.md @@ -0,0 +1,23 @@ +# Vamana memory rerun, 2026-09-28 + +These artifacts support the four updated xvec rows in +[`../../benchmark-vamana.csv`](../../benchmark-vamana.csv). +The zvec rows are unchanged historical measurements. + +- `xvec-*.json`: original benchmark reports and separate `*.resources.json` + files from `wait4` (KiB RSS, seconds for wall/user/system time). +- `xvec-*.log`: benchmark output. +- `metadata.json`: build version, command lines, environment settings, binary, + source-patch and dataset hashes, and successful exit statuses. +- `source.patch`: the exact production-code patch applied to the recorded base. +- `previous.csv`: measurements before this rerun. +- `run.py`: the sequential runner and resource measurement implementation. +- `update_csv.py`: validation and mapping of raw metrics into CSV fields. + +The scripts record this machine's paths; adjust `root` and `repo` when +reproducing elsewhere. Use the source patch with the recorded base commit, +build with `CGO_ENABLED=0`, and download the three files from +`https://assets.zilliz.com/benchmark/cohere_small_100k/` into `dataset/` before +running. Downloads and compilation occur outside the measured child processes. + +See [the analysis](../../benchmark-vamana-memory.md) for results and limitations. diff --git a/docs/benchmark-runs/vamana-memory-20260928/metadata.json b/docs/benchmark-runs/vamana-memory-20260928/metadata.json new file mode 100644 index 0000000..ebf44c3 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/metadata.json @@ -0,0 +1,213 @@ +{ + "machine": "e2-standard-8", + "cpu": "AMD EPYC 7B12", + "cpu_affinity": "0-7", + "gomaxprocs": 8, + "gomemlimit": "24GiB", + "cgo_enabled": 0, + "base_commit": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997", + "source_patch_sha256": "7e9600d19f64b61a8b741bf20f0f533b7d1724988102b998fed40b32e6c2b0f6", + "backend_version": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-memory.7e9600d19f64", + "binary_sha256": "e9a49f21cdb30732ad79a072eb80b19d2b1257a352a3847df14dfc093a901019", + "dataset_sha256": { + "neighbors.parquet": "ee07cdb43a7919bc1ad0525b3bf646036b43be5b7a0aee3ddcef06c21a42594a", + "shuffle_train.parquet": "9590e29f947e21cea6caae8a40e5a8f656937cbecd47859492bfae3fd9806540", + "test.parquet": "252a25003060713a268cd2bf5c5f8fab6159f13772924859487907bef64f1391" + }, + "runs": [ + { + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 3191984, + "peak_rss_mib": 3117.171875, + "wall_seconds": 119.64811540799997, + "user_cpu_seconds": 695.18229, + "system_cpu_seconds": 11.090022 + }, + { + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 3227792, + "peak_rss_mib": 3152.140625, + "wall_seconds": 119.18220637699983, + "user_cpu_seconds": 695.121586, + "system_cpu_seconds": 11.657617 + }, + { + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 3357600, + "peak_rss_mib": 3278.90625, + "wall_seconds": 116.86802954000018, + "user_cpu_seconds": 685.473934, + "system_cpu_seconds": 10.137268 + }, + { + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2534380, + "peak_rss_mib": 2474.98046875, + "wall_seconds": 112.62573622000014, + "user_cpu_seconds": 704.518379, + "system_cpu_seconds": 7.482471 + } + ] +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/previous.csv b/docs/benchmark-runs/vamana-memory-20260928/previous.csv new file mode 100644 index 0000000..45d6fff --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/previous.csv @@ -0,0 +1,9 @@ +machine,backend,backend_version,case,index_type,quantize_type,rotate,use_refiner,enable_mmap,vamana_max_degree,vamana_build_list,vamana_query_list,vamana_alpha,vamana_max_occlusion_size,vamana_saturate_graph,vamana_two_pass_build,vamana_use_contiguous_memory,vamana_use_id_map,k,batch_size,max_docs_per_segment,optimize_concurrency,query_concurrency,concurrency_duration_sec,serial_cooldown_sec,payload_profile,gomaxprocs,gomemlimit,cpu_affinity,go_version,inserted_count,insert_duration_sec,optimize_duration_sec,load_duration_sec,insert_rows_per_sec,serial_queries,serial_qps,recall_at_k_pct,serial_latency_avg_ms,serial_latency_p95_ms,serial_latency_p99_ms,concurrent_queries,concurrent_qps,concurrent_latency_avg_ms,concurrent_latency_p95_ms,concurrent_latency_p99_ms,peak_rss_kib,peak_rss_mib,wall_seconds,user_cpu_seconds,system_cpu_seconds +e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.450255,49.679974,56.135672,15503.263734,1000,330.151683,87.012,2.796118,3.577116,3.940061,60109,2003.053308,3.992393,5.781849,7.197208,3166476,3092.261719,100.682135,552.081874,9.451192 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.726085,31.247228,37.006987,17463.938926,1000,388.831267,81.039,2.355339,3.555305,3.973072,79847,2661.173599,3.004251,4.884632,6.706344,580228,566.628906,73.133591,448.838842,10.488299 +e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.464561,55.287267,61.7595,15468.955289,1000,276.881767,98.73,3.33458,4.513977,5.536994,57449,1913.679863,4.178597,6.025895,7.338456,3221456,3145.953125,106.379817,577.352607,14.930293 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.096447,28.252815,34.383313,16402.995921,1000,551.288631,98.355,1.622462,2.329426,2.686843,94285,3142.44868,2.5443,3.962252,5.503418,659776,644.3125,69.623703,420.824085,12.125019 +e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,8.872388,94.384487,103.266854,11270.922316,1000,152.714245,99.363,6.258708,10.97716,15.707088,38405,1279.504062,6.250287,9.105754,11.102209,3828872,3739.132812,154.680377,810.329211,25.327972 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.294007,77.133349,82.463938,18889.284977,1000,382.938654,99.198,2.399853,3.193495,3.536057,68113,2270.108375,3.522075,5.363077,7.251954,802112,783.3125,118.577794,784.075142,17.826853 +e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.953501,50.42262,55.383697,20187.742056,1000,317.498132,99.375,2.912781,3.781935,4.197044,48226,1607.41641,4.974847,6.636764,7.634434,2864312,2797.179688,95.84813,545.816312,12.665932 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.376264,57.116215,61.51392,22850.542618,1000,357.309009,99.392,2.593317,3.435763,3.753755,55841,1860.602216,4.297414,6.194429,8.125008,788484,770.003906,97.844471,638.635022,14.392509 diff --git a/docs/benchmark-runs/vamana-memory-20260928/run.py b/docs/benchmark-runs/vamana-memory-20260928/run.py new file mode 100644 index 0000000..34671cc --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/run.py @@ -0,0 +1,48 @@ +import hashlib, json, os, pathlib, subprocess, time +def file_hash(path): + with path.open('rb') as f: return hashlib.file_digest(f, 'sha256').hexdigest() + +root = pathlib.Path(__file__).parent +repo = pathlib.Path('/home/zhenghaoz/xvec') +source_files = ['collection.go','internal/core/algorithm/vamana_algorithm.go','internal/core/algorithm/vamana_quantized_searcher.go'] +patch = subprocess.check_output(['git','diff','HEAD','--',*source_files], cwd=repo) +(root/'source.patch').write_bytes(patch) +revision = subprocess.check_output(['git','rev-parse','HEAD'],cwd=repo,text=True).strip() +patch_hash = hashlib.sha256(patch).hexdigest() +metadata = { + 'machine':'e2-standard-8', 'cpu':'AMD EPYC 7B12', 'cpu_affinity':'0-7', + 'gomaxprocs':8, 'gomemlimit':'24GiB', 'cgo_enabled':0, + 'base_commit': revision, 'source_patch_sha256': patch_hash, + 'backend_version': revision+'+vamana-memory.'+patch_hash[:12], + 'binary_sha256':file_hash(root/'vector-db-bench'), + 'dataset_sha256':{p.name:file_hash(p) for p in sorted((root/'dataset').glob('*.parquet'))}, + 'runs':[] +} +(root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') +for precision in ['int4','int8','fp16','fp32']: + name = 'xvec-'+precision + args = ['taskset','-c','0-7',str(root/'vector-db-bench'),'xvec', + '--path',str(root/(name+'.collection')),'--case-type','Performance768D100K', + '--dataset-dir',str(root/'dataset'),'--skip-download','--index-type','vamana', + '--ef-search','200','--k','100','--batch-size','100','--max-docs-per-segment','10000000', + '--optimize-concurrency','8','--num-concurrency','8','--concurrency-duration','30s', + '--serial-cooldown','3s','--payload-profile','ids_only','--enable-mmap=true', + '--is-using-refiner=false','--output',str(root/(name+'.json'))] + if precision != 'fp32': args += ['--quantize-type',precision] + env = dict(os.environ,GOMAXPROCS='8',GOMEMLIMIT='24GiB') + print('START',name,flush=True) + start = time.monotonic() + with (root/(name+'.log')).open('w') as log: + proc = subprocess.Popen(args,stdout=log,stderr=subprocess.STDOUT,env=env,cwd=repo) + _, status, usage = os.wait4(proc.pid,0) + proc.returncode = os.waitstatus_to_exitcode(status) + elapsed = time.monotonic()-start + metrics = {'precision':precision,'command':args,'exit_code':proc.returncode, + 'peak_rss_kib':usage.ru_maxrss,'peak_rss_mib':usage.ru_maxrss/1024, + 'wall_seconds':elapsed,'user_cpu_seconds':usage.ru_utime,'system_cpu_seconds':usage.ru_stime} + (root/(name+'.resources.json')).write_text(json.dumps(metrics,indent=2)+'\n') + metadata['runs'].append(metrics) + (root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') + print('FINISH',name,json.dumps(metrics),flush=True) + if proc.returncode: raise SystemExit(proc.returncode) + print((root/(name+'.log')).read_text()[-2000:],flush=True) diff --git a/docs/benchmark-runs/vamana-memory-20260928/source.patch b/docs/benchmark-runs/vamana-memory-20260928/source.patch new file mode 100644 index 0000000..baaa3e3 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/source.patch @@ -0,0 +1,343 @@ +diff --git a/collection.go b/collection.go +index d62eb29..e74a040 100644 +--- a/collection.go ++++ b/collection.go +@@ -2529,7 +2529,7 @@ func buildCollectionDenseVamana( + spec collectionVectorIndex, + workers int, + ) (collectionVamanaIndex, error) { +- candidates, err := collectionDenseCandidates(ctx, field, documents) ++ count, err := collectionDenseCandidateCount(ctx, field, documents) + if err != nil { + return nil, err + } +@@ -2551,17 +2551,27 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- for _, candidate := range candidates { +- if err := builder.Add(ctx, candidate.Key, candidate.Vector); err != nil { ++ if err := builder.Reserve(count); err != nil { ++ return nil, err ++ } ++ for _, document := range documents { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ value, found := document.Fields[field.Name] ++ if !found || value == nil { ++ continue ++ } ++ vector, err := denseValueToFloat32Borrowed(value) ++ if err != nil { ++ return nil, fmt.Errorf("document %d field %q: %w", document.DocID, field.Name, err) ++ } ++ if err := builder.Add(ctx, document.DocID, vector); err != nil { + return nil, err + } +- } +- base, err := builder.BuildInterleavedWithWorkers(ctx, workers) +- if err != nil { +- return nil, err + } + if spec.quantize == QuantizeTypeUndefined { +- return base, nil ++ return builder.BuildInterleavedWithWorkers(ctx, workers) + } + kind, err := toCoreQuantization(spec.quantize) + if err != nil { +@@ -2571,7 +2581,7 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- return core.NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) + } + + func buildCollectionDenseDiskANN( +diff --git a/internal/core/algorithm/vamana_algorithm.go b/internal/core/algorithm/vamana_algorithm.go +index 84cba77..b9e65de 100644 +--- a/internal/core/algorithm/vamana_algorithm.go ++++ b/internal/core/algorithm/vamana_algorithm.go +@@ -22,6 +22,7 @@ import ( + "errors" + "fmt" + "math" ++ "os" + "slices" + "sync" + +@@ -148,6 +149,46 @@ func newBorrowedVamanaBuilder( + return builder, nil + } + ++// Reserve preallocates storage for at least count total vectors. It does not ++// change the builder length or the ownership guarantees of Add. ++func (b *VamanaBuilder) Reserve(count int) error { ++ if b == nil { ++ return errors.New("core: nil Vamana builder") ++ } ++ if count < 0 || uint64(count) >= math.MaxUint32 || (count > 0 && count > maxPlatformInt()/b.dimension) { ++ return ErrVamanaCapacity ++ } ++ b.mu.Lock() ++ defer b.mu.Unlock() ++ if b.built { ++ return ErrBuilderClosed ++ } ++ vectorCapacity := cap(b.vectors) ++ if b.fp16 { ++ vectorCapacity = cap(b.vectorsFP16) ++ } ++ if count <= cap(b.keys) && count*b.dimension <= vectorCapacity { ++ return nil ++ } ++ reservedKeys := make([]uint64, len(b.keys), max(count, len(b.keys))) ++ copy(reservedKeys, b.keys) ++ var reservedVectors []float32 ++ var reservedVectorsFP16 []uint16 ++ if b.fp16 { ++ reservedVectorsFP16 = make([]uint16, len(b.vectorsFP16), max(count*b.dimension, len(b.vectorsFP16))) ++ copy(reservedVectorsFP16, b.vectorsFP16) ++ } else { ++ reservedVectors = make([]float32, len(b.vectors), max(count*b.dimension, len(b.vectors))) ++ copy(reservedVectors, b.vectors) ++ } ++ reservedPositions := make(map[uint64]int, max(count, len(b.positions))) ++ for key, position := range b.positions { ++ reservedPositions[key] = position ++ } ++ b.keys, b.vectors, b.vectorsFP16, b.positions = reservedKeys, reservedVectors, reservedVectorsFP16, reservedPositions ++ return nil ++} ++ + // Add validates and clones one unique vector while the builder is open. + func (b *VamanaBuilder) Add(ctx context.Context, key uint64, vector []float32) error { + if b == nil { +@@ -1179,19 +1220,26 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + if (count == 0 && index.entryPoint != -1) || (count > 0 && (index.entryPoint < 0 || index.entryPoint >= count)) { + return errors.New("core: invalid Vamana entry point") + } +- seenKeys := make(map[uint64]struct{}, count) ++ // The key-to-position bijection also detects duplicate keys. ++ seenNeighbors := make(map[int]struct{}, index.options.MaxDegree) ++ var vectorScratch []float32 ++ if index.fp16 { ++ vectorScratch = make([]float32, index.dimension) ++ } + for position, key := range index.keys { + if position&255 == 0 { + if err := ctx.Err(); err != nil { + return err + } + } +- if _, found := seenKeys[key]; found || index.positions[key] != position { ++ if mapped, found := index.positions[key]; !found || mapped != position { + return errors.New("core: invalid Vamana key map") + } +- seenKeys[key] = struct{}{} + if index.fp16 { +- if err := validateTrainingVector(float32VectorFromFP16(index.vectorFP16At(position)), index.dimension); err != nil { ++ for offset, value := range index.vectorFP16At(position) { ++ vectorScratch[offset] = utility.Float16BitsToFloat32(value) ++ } ++ if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { + return err + } + } else if err := validateTrainingVector(index.vectorAt(position), index.dimension); err != nil { +@@ -1200,7 +1248,7 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + if len(index.neighbors[position]) > index.options.MaxDegree || len(index.neighbors[position]) != len(index.neighborDistances[position]) { + return errors.New("core: invalid Vamana degree") + } +- seenNeighbors := make(map[int]struct{}, len(index.neighbors[position])) ++ clear(seenNeighbors) + for offset, neighbor := range index.neighbors[position] { + if neighbor < 0 || neighbor >= count || neighbor == position { + return errors.New("core: invalid Vamana neighbor") +@@ -1724,17 +1772,24 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { + if i == nil { + return fmt.Errorf("%w: nil index", ErrInvalidVamanaFile) + } ++ // Add publishes a new generation without mutating existing storage. Holding ++ // the read lock keeps this generation stable while it is streamed to disk. + i.mu.RLock() +- snapshot, err := cloneVamanaIndex(ctx, i) +- i.mu.RUnlock() +- if err != nil { +- return err +- } +- encoded, err := encodeVamanaIndex(ctx, snapshot) +- if err != nil { ++ defer i.mu.RUnlock() ++ if err := ioutil.WriteFileAtomicFunc(ctx, path, 0o600, func(file *os.File) error { ++ if _, err := file.Write(make([]byte, vamanaHeaderSize)); err != nil { ++ return err ++ } ++ header, err := writeVamanaPayload(ctx, i, func(data []byte) error { ++ _, err := file.Write(data) ++ return err ++ }) ++ if err != nil { ++ return err ++ } ++ _, err = file.WriteAt(header, 0) + return err +- } +- if err := ioutil.WriteFileAtomic(ctx, path, encoded, 0o600); err != nil { ++ }); err != nil { + return fmt.Errorf("core: save Vamana file: %w", err) + } + return nil +@@ -1763,6 +1818,21 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + } + + func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) { ++ encoded := make([]byte, vamanaHeaderSize) ++ header, err := writeVamanaPayload(ctx, index, func(data []byte) error { ++ encoded = append(encoded, data...) ++ return nil ++ }) ++ if err != nil { ++ return nil, err ++ } ++ copy(encoded, header) ++ return encoded, nil ++} ++ ++// writeVamanaPayload bounds serialization scratch space independently of vector ++// count. The returned header contains the checksum of the streamed payload. ++func writeVamanaPayload(ctx context.Context, index *VamanaIndex, write func([]byte) error) ([]byte, error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana encode context") + } +@@ -1786,7 +1856,21 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + if err != nil { + return nil, err + } +- payload := make([]byte, 0, payloadSize) ++ payload := make([]byte, 0, 64<<10) ++ var checksum uint32 ++ written := 0 ++ flush := func() error { ++ if err := ctx.Err(); err != nil { ++ return err ++ } ++ if err := write(payload); err != nil { ++ return err ++ } ++ checksum = hashutil.UpdateCRC32C(checksum, payload) ++ written += len(payload) ++ payload = payload[:0] ++ return nil ++ } + for position, key := range index.keys { + if position&1023 == 0 { + if err := ctx.Err(); err != nil { +@@ -1794,6 +1878,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint64(payload, key) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + if index.fp16 { + for position, value := range index.vectorsFP16 { +@@ -1803,6 +1892,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint16(payload, value) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } else { + for position, value := range index.vectors { +@@ -1812,6 +1906,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } + for position, adjacent := range index.neighbors { +@@ -1821,11 +1920,24 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint32(payload, uint32(len(adjacent))) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + for _, neighbor := range adjacent { + payload = binary.LittleEndian.AppendUint32(payload, uint32(neighbor)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } +- if len(payload) != payloadSize { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ if written != payloadSize { + return nil, fmt.Errorf("%w: internal payload length", ErrInvalidVamanaFile) + } + header := make([]byte, vamanaHeaderSize) +@@ -1853,9 +1965,9 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + entry = uint64(index.entryPoint) + } + binary.LittleEndian.PutUint64(header[72:80], entry) +- binary.LittleEndian.PutUint32(header[80:84], hashutil.CRC32C(payload)) ++ binary.LittleEndian.PutUint32(header[80:84], checksum) + binary.LittleEndian.PutUint32(header[124:128], hashutil.CRC32C(header[:124])) +- return append(header, payload...), nil ++ return header, nil + } + + func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error) { +diff --git a/internal/core/algorithm/vamana_quantized_searcher.go b/internal/core/algorithm/vamana_quantized_searcher.go +index 6cbba85..1af85db 100644 +--- a/internal/core/algorithm/vamana_quantized_searcher.go ++++ b/internal/core/algorithm/vamana_quantized_searcher.go +@@ -47,6 +47,20 @@ func NewScalarQuantizedVamanaIndex( + if err != nil { + return nil, err + } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, snapshot, kind, reformer) ++} ++ ++// BuildScalarQuantizedInterleavedWithWorkers transfers the completed graph to ++// an immutable quantized index without cloning its vectors and adjacency. ++func (b *VamanaBuilder) BuildScalarQuantizedInterleavedWithWorkers(ctx context.Context, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { ++ base, err := b.BuildInterleavedWithWorkers(ctx, workers) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} ++ ++func newOwnedScalarQuantizedVamanaIndex(ctx context.Context, snapshot *VamanaIndex, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { + vectors, err := newOwnedScalarQuantizedVectors( + ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, + ) +@@ -72,7 +86,7 @@ func OpenScalarQuantizedVamanaIndex(ctx context.Context, path string, kind Quant + if err != nil { + return nil, err + } +- return NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) + } + + func (i *ScalarQuantizedVamanaIndex) Dimension() int { diff --git a/docs/benchmark-runs/vamana-memory-20260928/update_csv.py b/docs/benchmark-runs/vamana-memory-20260928/update_csv.py new file mode 100644 index 0000000..8e10409 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/update_csv.py @@ -0,0 +1,43 @@ +import csv, json, pathlib, shutil +root=pathlib.Path(__file__).parent +repo=pathlib.Path('/home/zhenghaoz/xvec') +archive=repo/'docs/benchmark-runs/vamana-memory-20260928' +meta=json.loads((root/'metadata.json').read_text()) +assert len(meta['runs'])==4 and all(r['exit_code']==0 for r in meta['runs']) +with (archive/'previous.csv').open() as f: + reader=csv.DictReader(f); fields=reader.fieldnames; rows=list(reader) +for row in rows: + if row['backend']!='xvec': continue + kind=row['quantize_type']; name='xvec-'+kind + report=json.loads((root/(name+'.json')).read_text()); resource=json.loads((root/(name+'.resources.json')).read_text()) + c=report['config']; load=report['load']; serial=report['serial']; conc=report['concurrent'][0] + assert load['rows']==100000 and serial['queries']==1000 and conc['concurrency']==8 + assert len(report['concurrent'])==1 and report['case']['name']==row['case'] + assert report['system']['go_version']==row['go_version'] + assert c['backend']=='xvec' and c['index_type']=='vamana' + assert c['quantize_type']==('none' if kind=='fp32' else kind) + assert c['ef_search']==200 and c['concurrency_duration']=='30s' and c['serial_cooldown']=='3s' + for key in ['use_refiner','enable_mmap','k','batch_size','max_docs_per_segment','optimize_concurrency','payload_profile']: + assert str(c[key]).lower()==row[key],(key,c[key],row[key]) + row['backend_version']=meta['backend_version'] + row['inserted_count']=load['rows'] + for key in ['insert_duration_sec','optimize_duration_sec','load_duration_sec']: row[key]=load[key] + row['insert_rows_per_sec']=load['rows_per_second'] + row['recall_at_k_pct']=serial['recall']*100 + for prefix,metrics in [('serial',serial),('concurrent',conc)]: + for key in ['queries','qps','latency_avg_ms','latency_p95_ms','latency_p99_ms']: row[prefix+'_'+key]=metrics[key] + for key in ['peak_rss_kib','peak_rss_mib','wall_seconds','user_cpu_seconds','system_cpu_seconds']: row[key]=resource[key] + for key,value in row.items(): + if isinstance(value,float): row[key]=f'{value:.6f}'.rstrip('0').rstrip('.') + for suffix in ['.json','.resources.json','.log']: shutil.copy2(root/(name+suffix),archive/(name+suffix)) +with (repo/'docs/benchmark-vamana.csv').open('w') as f: + writer=csv.DictWriter(f,fields,lineterminator='\n'); writer.writeheader(); writer.writerows(rows) +shutil.copy2(root/'metadata.json',archive/'metadata.json') +# Store the exact commands and wait4 implementation for reproducibility. +shutil.copy2(root/'run.py',archive/'run.py') +shutil.copy2(root/'update_csv.py',archive/'update_csv.py') +with (archive/'previous.csv').open() as f: old={r['quantize_type']:r for r in csv.DictReader(f) if r['backend']=='xvec'} +for row in rows: + if row['backend']=='xvec': + prior=old[row['quantize_type']] + print(row['quantize_type'],'RSS MiB',prior['peak_rss_mib'],'->',row['peak_rss_mib'], 'change%',round((float(row['peak_rss_mib'])/float(prior['peak_rss_mib'])-1)*100,2),'recall',row['recall_at_k_pct']) diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.json new file mode 100644 index 0000000..deaa648 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T11:53:09.938310973Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-retest-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "fp16", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 5.755098846, + "optimize_duration_sec": 64.093180044, + "load_duration_sec": 69.855302499, + "rows_per_second": 17375.89617066327, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 163.42715068462937, + "recall": 0.9936300000000036, + "latency_avg_ms": 5.845329519999993, + "latency_p95_ms": 7.8192565499999995, + "latency_p99_ms": 8.670611809999999 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 29568, + "qps": 985.3467466718437, + "latency_avg_ms": 8.115983062669077, + "latency_p95_ms": 11.002918649999996, + "latency_p99_ms": 12.513949899999993 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 5.755098846, + "optimize_duration": 64.093180044, + "load_duration": 69.855302499, + "qps": 985.3467466718437, + "recall": 0.9936300000000036, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.00867061181, + "serial_latency_p95": 0.007819256549999999, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 985.3467466718437 + ], + "conc_latency_p99_list": [ + 0.012513949899999993 + ], + "conc_latency_p95_list": [ + 0.011002918649999997 + ], + "conc_latency_avg_list": [ + 0.008115983062669077 + ] + } +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.log b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.log new file mode 100644 index 0000000..8b032af --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (17376.1 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-retest-20260928/dataset +load rows=100000 duration=69.855s insert=5.755s optimize=64.093s rows/s=17375.9 +serial queries=1000 qps=163.43 recall=0.9936 mrr=0.0000 ndcg=0.0000 avg=5.845ms p95=7.819ms p99=8.671ms +concurrent workers=8 queries=29568 qps=985.35 avg=8.116ms p95=11.003ms p99=12.514ms +result=/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.json diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.resources.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.resources.json new file mode 100644 index 0000000..f0264e2 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp16.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 3357600, + "peak_rss_mib": 3278.90625, + "wall_seconds": 116.86802954000018, + "user_cpu_seconds": 685.473934, + "system_cpu_seconds": 10.137268 +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.json new file mode 100644 index 0000000..be1de2a --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T11:55:06.805786015Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-retest-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "none", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 4.129177599, + "optimize_duration_sec": 65.782052629, + "load_duration_sec": 69.918232578, + "rows_per_second": 24217.89753587201, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 239.45485466416665, + "recall": 0.9937800000000035, + "latency_avg_ms": 3.896063934999997, + "latency_p95_ms": 5.0285095, + "latency_p99_ms": 5.41041921 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 30526, + "qps": 1016.9844455730826, + "latency_avg_ms": 7.862629193277848, + "latency_p95_ms": 11.08141, + "latency_p99_ms": 13.527539999999998 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 4.129177599, + "optimize_duration": 65.782052629, + "load_duration": 69.918232578, + "qps": 1016.9844455730826, + "recall": 0.9937800000000035, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.00541041921, + "serial_latency_p95": 0.0050285095, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1016.9844455730826 + ], + "conc_latency_p99_list": [ + 0.013527539999999998 + ], + "conc_latency_p95_list": [ + 0.01108141 + ], + "conc_latency_avg_list": [ + 0.007862629193277848 + ] + } +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.log b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.log new file mode 100644 index 0000000..2083819 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (24218.3 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-retest-20260928/dataset +load rows=100000 duration=69.918s insert=4.129s optimize=65.782s rows/s=24217.9 +serial queries=1000 qps=239.45 recall=0.9938 mrr=0.0000 ndcg=0.0000 avg=3.896ms p95=5.029ms p99=5.410ms +concurrent workers=8 queries=30526 qps=1016.98 avg=7.863ms p95=11.081ms p99=13.528ms +result=/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.json diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.resources.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.resources.json new file mode 100644 index 0000000..e1d4569 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-fp32.resources.json @@ -0,0 +1,47 @@ +{ + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2534380, + "peak_rss_mib": 2474.98046875, + "wall_seconds": 112.62573622000014, + "user_cpu_seconds": 704.518379, + "system_cpu_seconds": 7.482471 +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.json new file mode 100644 index 0000000..d11126f --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T11:49:11.104735389Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-retest-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-retest-20260928/xvec-int4.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int4", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.96541527, + "optimize_duration_sec": 65.869903012, + "load_duration_sec": 72.844356971, + "rows_per_second": 14356.645817041142, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 225.57641702000376, + "recall": 0.8701800000000003, + "latency_avg_ms": 4.1465193559999936, + "latency_p95_ms": 5.38936295, + "latency_p99_ms": 5.959916199999999 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 42323, + "qps": 1409.9883424797058, + "latency_avg_ms": 5.671430123148138, + "latency_p95_ms": 8.0410551, + "latency_p99_ms": 9.507926219999996 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.96541527, + "optimize_duration": 65.869903012, + "load_duration": 72.844356971, + "qps": 1409.9883424797058, + "recall": 0.8701800000000003, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.005959916199999999, + "serial_latency_p95": 0.00538936295, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1409.9883424797058 + ], + "conc_latency_p99_list": [ + 0.009507926219999996 + ], + "conc_latency_p95_list": [ + 0.0080410551 + ], + "conc_latency_avg_list": [ + 0.005671430123148138 + ] + } +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.log b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.log new file mode 100644 index 0000000..ed0adf5 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (14356.8 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-retest-20260928/dataset +load rows=100000 duration=72.844s insert=6.965s optimize=65.870s rows/s=14356.6 +serial queries=1000 qps=225.58 recall=0.8702 mrr=0.0000 ndcg=0.0000 avg=4.147ms p95=5.389ms p99=5.960ms +concurrent workers=8 queries=42323 qps=1409.99 avg=5.671ms p95=8.041ms p99=9.508ms +result=/home/zhenghaoz/vamana-retest-20260928/xvec-int4.json diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.resources.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.resources.json new file mode 100644 index 0000000..eb32c7d --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int4.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 3191984, + "peak_rss_mib": 3117.171875, + "wall_seconds": 119.64811540799997, + "user_cpu_seconds": 695.18229, + "system_cpu_seconds": 11.090022 +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.json new file mode 100644 index 0000000..87b8cc4 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T11:51:10.754006566Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-retest-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-retest-20260928/xvec-int8.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int8", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.623609178, + "optimize_duration_sec": 66.415754632, + "load_duration_sec": 73.04619476, + "rows_per_second": 15097.509124201531, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 227.27686965194587, + "recall": 0.9872000000000077, + "latency_avg_ms": 4.1218248840000005, + "latency_p95_ms": 5.502667499999999, + "latency_p99_ms": 6.4205166999999985 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 40234, + "qps": 1340.8681668203114, + "latency_avg_ms": 5.96374291303374, + "latency_p95_ms": 8.631253849999998, + "latency_p99_ms": 12.2553498 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.623609178, + "optimize_duration": 66.415754632, + "load_duration": 73.04619476, + "qps": 1340.8681668203114, + "recall": 0.9872000000000077, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.006420516699999998, + "serial_latency_p95": 0.005502667499999999, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1340.8681668203114 + ], + "conc_latency_p99_list": [ + 0.012255349799999999 + ], + "conc_latency_p95_list": [ + 0.008631253849999998 + ], + "conc_latency_avg_list": [ + 0.00596374291303374 + ] + } +} diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.log b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.log new file mode 100644 index 0000000..0cb015b --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (15097.7 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-retest-20260928/dataset +load rows=100000 duration=73.046s insert=6.624s optimize=66.416s rows/s=15097.5 +serial queries=1000 qps=227.28 recall=0.9872 mrr=0.0000 ndcg=0.0000 avg=4.122ms p95=5.503ms p99=6.421ms +concurrent workers=8 queries=40234 qps=1340.87 avg=5.964ms p95=8.631ms p99=12.255ms +result=/home/zhenghaoz/vamana-retest-20260928/xvec-int8.json diff --git a/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.resources.json b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.resources.json new file mode 100644 index 0000000..4fc9b88 --- /dev/null +++ b/docs/benchmark-runs/vamana-memory-20260928/xvec-int8.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-retest-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-retest-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-retest-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 3227792, + "peak_rss_mib": 3152.140625, + "wall_seconds": 119.18220637699983, + "user_cpu_seconds": 695.121586, + "system_cpu_seconds": 11.657617 +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/README.md b/docs/benchmark-runs/vamana-runtime-20260928/README.md new file mode 100644 index 0000000..e65cde2 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/README.md @@ -0,0 +1,30 @@ +# Vamana runtime memory rerun, 2026-09-28 + +These artifacts support the four updated xvec rows in +[`../../benchmark-vamana.csv`](../../benchmark-vamana.csv). +The zvec rows are unchanged historical measurements. `previous.csv` contains +the first memory-optimization rerun; the original historical CSV is in +`../vamana-memory-20260928/previous.csv`. + +- `xvec-*.json`: original benchmark reports and separate `*.resources.json` + files from `wait4` (KiB RSS, seconds for wall/user/system time). +- `xvec-*.log`: benchmark output. +- `metadata.json`: build version, command lines, environment settings, binary, + source-patch and dataset hashes, and successful exit statuses. +- `source.patch`: the exact production-code patch applied to the recorded base. +- `previous.csv`: measurements before this rerun. +- `run.py`: the sequential runner and resource measurement implementation. +- `update_csv.py`: validation and mapping of raw metrics into CSV fields. + +The scripts record this machine's paths; adjust `root` and `repo` when +reproducing elsewhere. Use the source patch with the recorded base commit, +build with `CGO_ENABLED=0`, and download the three files from +`https://assets.zilliz.com/benchmark/cohere_small_100k/` into `dataset/` before +running. Downloads and compilation occur outside the measured child processes. + +See [the analysis](../../benchmark-vamana-memory.md) for results and limitations. + +`query-before.json` and `query-after.json` are separate diagnostics: both binaries +search the same FP32 collection, with loading skipped and concurrency duration +reduced to 10 seconds. They do not replace full-workload CSV values. +`compare_query.py` records the exact procedure and binary paths. diff --git a/docs/benchmark-runs/vamana-runtime-20260928/compare_query.py b/docs/benchmark-runs/vamana-runtime-20260928/compare_query.py new file mode 100644 index 0000000..e21c45e --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/compare_query.py @@ -0,0 +1,13 @@ +import pathlib,subprocess,json,os +root=pathlib.Path(__file__).parent +meta=json.loads((root/'metadata.json').read_text()) +base=meta['runs'][-1]['command'] +for label,binary in [('before','/home/zhenghaoz/vamana-retest-20260928/vector-db-bench'),('after',str(root/'vector-db-bench'))]: + args=base.copy();args[3]=binary + args[args.index('--concurrency-duration')+1]='10s' + args[args.index('--output')+1]=str(root/('query-'+label+'.json')) + args+=['--skip-load','--skip-drop-old'] + print('START query',label,flush=True) + with (root/('query-'+label+'.log')).open('w') as f: + subprocess.run(args,env=dict(os.environ,GOMAXPROCS='8',GOMEMLIMIT='24GiB'),stdout=f,stderr=subprocess.STDOUT,check=True) + print((root/('query-'+label+'.log')).read_text(),flush=True) diff --git a/docs/benchmark-runs/vamana-runtime-20260928/metadata.json b/docs/benchmark-runs/vamana-runtime-20260928/metadata.json new file mode 100644 index 0000000..2c5b589 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/metadata.json @@ -0,0 +1,213 @@ +{ + "machine": "e2-standard-8", + "cpu": "AMD EPYC 7B12", + "cpu_affinity": "0-7", + "gomaxprocs": 8, + "gomemlimit": "24GiB", + "cgo_enabled": 0, + "base_commit": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997", + "source_patch_sha256": "47525295f56e9f34bad82a68f710ba7cf4304d33fe447630991c93c035975709", + "backend_version": "5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-runtime.47525295f56e", + "binary_sha256": "66e451998eda73b27458cc1576c165c97ba5b22dd183d0149fc7df50529bc4c3", + "dataset_sha256": { + "neighbors.parquet": "ee07cdb43a7919bc1ad0525b3bf646036b43be5b7a0aee3ddcef06c21a42594a", + "shuffle_train.parquet": "9590e29f947e21cea6caae8a40e5a8f656937cbecd47859492bfae3fd9806540", + "test.parquet": "252a25003060713a268cd2bf5c5f8fab6159f13772924859487907bef64f1391" + }, + "runs": [ + { + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 2295552, + "peak_rss_mib": 2241.75, + "wall_seconds": 118.85539514399989, + "user_cpu_seconds": 713.076651, + "system_cpu_seconds": 5.329071 + }, + { + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 2301332, + "peak_rss_mib": 2247.39453125, + "wall_seconds": 114.82267373400009, + "user_cpu_seconds": 688.447641, + "system_cpu_seconds": 5.169885 + }, + { + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 2310924, + "peak_rss_mib": 2256.76171875, + "wall_seconds": 114.75985079299971, + "user_cpu_seconds": 686.68149, + "system_cpu_seconds": 5.027617 + }, + { + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2199088, + "peak_rss_mib": 2147.546875, + "wall_seconds": 113.34803818499995, + "user_cpu_seconds": 686.835882, + "system_cpu_seconds": 5.929749 + } + ] +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/previous.csv b/docs/benchmark-runs/vamana-runtime-20260928/previous.csv new file mode 100644 index 0000000..949cd1f --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/previous.csv @@ -0,0 +1,9 @@ +machine,backend,backend_version,case,index_type,quantize_type,rotate,use_refiner,enable_mmap,vamana_max_degree,vamana_build_list,vamana_query_list,vamana_alpha,vamana_max_occlusion_size,vamana_saturate_graph,vamana_two_pass_build,vamana_use_contiguous_memory,vamana_use_id_map,k,batch_size,max_docs_per_segment,optimize_concurrency,query_concurrency,concurrency_duration_sec,serial_cooldown_sec,payload_profile,gomaxprocs,gomemlimit,cpu_affinity,go_version,inserted_count,insert_duration_sec,optimize_duration_sec,load_duration_sec,insert_rows_per_sec,serial_queries,serial_qps,recall_at_k_pct,serial_latency_avg_ms,serial_latency_p95_ms,serial_latency_p99_ms,concurrent_queries,concurrent_qps,concurrent_latency_avg_ms,concurrent_latency_p95_ms,concurrent_latency_p99_ms,peak_rss_kib,peak_rss_mib,wall_seconds,user_cpu_seconds,system_cpu_seconds +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-memory.7e9600d19f64,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.965415,65.869903,72.844357,14356.645817,1000,225.576417,87.018,4.146519,5.389363,5.959916,42323,1409.988342,5.67143,8.041055,9.507926,3191984,3117.171875,119.648115,695.18229,11.090022 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.726085,31.247228,37.006987,17463.938926,1000,388.831267,81.039,2.355339,3.555305,3.973072,79847,2661.173599,3.004251,4.884632,6.706344,580228,566.628906,73.133591,448.838842,10.488299 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-memory.7e9600d19f64,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.623609,66.415755,73.046195,15097.509124,1000,227.27687,98.72,4.121825,5.502667,6.420517,40234,1340.868167,5.963743,8.631254,12.25535,3227792,3152.140625,119.182206,695.121586,11.657617 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.096447,28.252815,34.383313,16402.995921,1000,551.288631,98.355,1.622462,2.329426,2.686843,94285,3142.44868,2.5443,3.962252,5.503418,659776,644.3125,69.623703,420.824085,12.125019 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-memory.7e9600d19f64,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.755099,64.09318,69.855302,17375.896171,1000,163.427151,99.363,5.84533,7.819257,8.670612,29568,985.346747,8.115983,11.002919,12.51395,3357600,3278.90625,116.86803,685.473934,10.137268 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.294007,77.133349,82.463938,18889.284977,1000,382.938654,99.198,2.399853,3.193495,3.536057,68113,2270.108375,3.522075,5.363077,7.251954,802112,783.3125,118.577794,784.075142,17.826853 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-memory.7e9600d19f64,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.129178,65.782053,69.918233,24217.897536,1000,239.454855,99.378,3.896064,5.02851,5.410419,30526,1016.984446,7.862629,11.08141,13.52754,2534380,2474.980469,112.625736,704.518379,7.482471 +e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.376264,57.116215,61.51392,22850.542618,1000,357.309009,99.392,2.593317,3.435763,3.753755,55841,1860.602216,4.297414,6.194429,8.125008,788484,770.003906,97.844471,638.635022,14.392509 diff --git a/docs/benchmark-runs/vamana-runtime-20260928/profile-before-alloc.txt b/docs/benchmark-runs/vamana-runtime-20260928/profile-before-alloc.txt new file mode 100644 index 0000000..0153f08 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/profile-before-alloc.txt @@ -0,0 +1,20 @@ +File: bench +Build ID: 1ef83183ffe11bd27dae6e004019fa2d7f349822 +Type: alloc_space +Time: 2026-09-28 12:03:00 UTC +Showing nodes accounting for 6679.51MB, 74.98% of 8908.46MB total +Dropped 223 nodes (cum <= 44.54MB) +Showing top 12 nodes out of 101 + flat flat% sum% cum cum% + 1882.86MB 21.14% 21.14% 1882.86MB 21.14% github.com/gorse-io/xvec/internal/ailego/container.(*Heap[go.shape.struct { github.com/gorse-io/xvec/internal/core/algorithm.position int; github.com/gorse-io/xvec/internal/core/algorithm.distance float32 }]).Push + 705.61MB 7.92% 29.06% 705.61MB 7.92% main.decodeVectorParquetRow + 666.92MB 7.49% 36.54% 666.92MB 7.49% slices.Clone[go.shape.[]uint8,go.shape.uint8] + 596.30MB 6.69% 43.24% 881.12MB 9.89% github.com/gorse-io/xvec.marshalDocumentPayload + 584.18MB 6.56% 49.79% 584.18MB 6.56% github.com/gorse-io/xvec.decodeDocumentValue + 356.21MB 4.00% 53.79% 468.75MB 5.26% github.com/gorse-io/xvec/internal/core/algorithm.newScalarQuantizedVectorStorageWithReader + 350.15MB 3.93% 57.72% 350.15MB 3.93% github.com/gorse-io/xvec/internal/db/index/storage/wal.encodeWALRecord + 335.60MB 3.77% 61.49% 336.32MB 3.78% github.com/gorse-io/xvec/internal/core/algorithm.decodeVamanaIndex + 309.05MB 3.47% 64.96% 309.05MB 3.47% github.com/gorse-io/xvec/internal/core/algorithm.readHNSWFile + 298.87MB 3.35% 68.31% 672.25MB 7.55% github.com/gorse-io/xvec/internal/core/algorithm.NewScalarQuantizedFlatIndex + 297.71MB 3.34% 71.66% 297.71MB 3.34% github.com/gorse-io/xvec/internal/core/algorithm.newDenseFlatIndexFromValidatedCandidates + 296.02MB 3.32% 74.98% 296.02MB 3.32% github.com/gorse-io/xvec/internal/core/algorithm.(*VamanaBuilder).Reserve diff --git a/docs/benchmark-runs/vamana-runtime-20260928/profile-before-inuse.txt b/docs/benchmark-runs/vamana-runtime-20260928/profile-before-inuse.txt new file mode 100644 index 0000000..5ef5cbb --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/profile-before-inuse.txt @@ -0,0 +1,20 @@ +File: bench +Build ID: 1ef83183ffe11bd27dae6e004019fa2d7f349822 +Type: inuse_space +Time: 2026-09-28 12:03:00 UTC +Showing nodes accounting for 1376.71MB, 97.91% of 1406.09MB total +Dropped 67 nodes (cum <= 7.03MB) +Showing top 12 nodes out of 38 + flat flat% sum% cum cum% + 335.60MB 23.87% 23.87% 336.32MB 23.92% github.com/gorse-io/xvec/internal/core/algorithm.decodeVamanaIndex + 309.05MB 21.98% 45.85% 309.05MB 21.98% github.com/gorse-io/xvec/internal/core/algorithm.readHNSWFile + 297.71MB 21.17% 67.02% 297.71MB 21.17% github.com/gorse-io/xvec/internal/core/algorithm.newDenseFlatIndexFromValidatedCandidates + 294.85MB 20.97% 87.99% 294.85MB 20.97% github.com/gorse-io/xvec.decodeDocumentValue + 47.21MB 3.36% 91.35% 57.72MB 4.11% main.readQueryData + 40.01MB 2.85% 94.19% 40.01MB 2.85% github.com/gorse-io/xvec/internal/core/algorithm.quantizeInteger + 36.51MB 2.60% 96.79% 331.36MB 23.57% github.com/gorse-io/xvec.unmarshalDocumentPayloadWithBorrowedVectors + 8.65MB 0.62% 97.41% 48.67MB 3.46% github.com/gorse-io/xvec/internal/core/algorithm.newScalarQuantizedVectorStorageWithReader + 3.82MB 0.27% 97.68% 335.18MB 23.84% github.com/gorse-io/xvec.(*Collection).segmentDocumentsLocked.func1 + 3.29MB 0.23% 97.91% 51.96MB 3.70% github.com/gorse-io/xvec/internal/core/algorithm.NewScalarQuantizedFlatIndexWithBorrowedVectors + 0 0% 97.91% 1334.04MB 94.88% github.com/gorse-io/xvec.(*Collection).Query + 0 0% 97.91% 1334.04MB 94.88% github.com/gorse-io/xvec.(*Collection).acquireQuerySnapshotLocked diff --git a/docs/benchmark-runs/vamana-runtime-20260928/query-after.json b/docs/benchmark-runs/vamana-runtime-20260928/query-after.json new file mode 100644 index 0000000..71b6802 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/query-after.json @@ -0,0 +1,102 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:14:27.095717006Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "none", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "10s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "serial": { + "queries": 1000, + "qps": 211.6744166131789, + "recall": 0.9938300000000034, + "latency_avg_ms": 4.413742982999988, + "latency_p95_ms": 5.6440715, + "latency_p99_ms": 6.14558929 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 9808, + "qps": 979.5135701535369, + "latency_avg_ms": 8.161286299755298, + "latency_p95_ms": 11.170627849999999, + "latency_p99_ms": 12.735827970000004 + } + ], + "vectordbbench_metrics": { + "inserted_count": 0, + "insert_duration": 0, + "optimize_duration": 0, + "load_duration": 0, + "qps": 979.5135701535369, + "recall": 0.9938300000000034, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0061455892900000005, + "serial_latency_p95": 0.0056440715, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 979.5135701535369 + ], + "conc_latency_p99_list": [ + 0.012735827970000004 + ], + "conc_latency_p95_list": [ + 0.011170627849999998 + ], + "conc_latency_avg_list": [ + 0.008161286299755299 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/query-after.log b/docs/benchmark-runs/vamana-runtime-20260928/query-after.log new file mode 100644 index 0000000..588592b --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/query-after.log @@ -0,0 +1,5 @@ +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +serial queries=1000 qps=211.67 recall=0.9938 mrr=0.0000 ndcg=0.0000 avg=4.414ms p95=5.644ms p99=6.146ms +concurrent workers=8 queries=9808 qps=979.51 avg=8.161ms p95=11.171ms p99=12.736ms +result=/home/zhenghaoz/vamana-runtime-20260928/query-after.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/query-before.json b/docs/benchmark-runs/vamana-runtime-20260928/query-before.json new file mode 100644 index 0000000..50018c5 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/query-before.json @@ -0,0 +1,102 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:14:00.309550701Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "none", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "10s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "serial": { + "queries": 1000, + "qps": 211.1087859964726, + "recall": 0.9938300000000034, + "latency_avg_ms": 4.442909714000003, + "latency_p95_ms": 5.889790999999999, + "latency_p99_ms": 6.67937791 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 8838, + "qps": 882.17825175712, + "latency_avg_ms": 9.062737882213174, + "latency_p95_ms": 12.412655999999998, + "latency_p99_ms": 14.026316599999998 + } + ], + "vectordbbench_metrics": { + "inserted_count": 0, + "insert_duration": 0, + "optimize_duration": 0, + "load_duration": 0, + "qps": 882.17825175712, + "recall": 0.9938300000000034, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.00667937791, + "serial_latency_p95": 0.005889790999999999, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 882.17825175712 + ], + "conc_latency_p99_list": [ + 0.014026316599999998 + ], + "conc_latency_p95_list": [ + 0.012412655999999998 + ], + "conc_latency_avg_list": [ + 0.009062737882213174 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/query-before.log b/docs/benchmark-runs/vamana-runtime-20260928/query-before.log new file mode 100644 index 0000000..5eafd1d --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/query-before.log @@ -0,0 +1,5 @@ +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +serial queries=1000 qps=211.11 recall=0.9938 mrr=0.0000 ndcg=0.0000 avg=4.443ms p95=5.890ms p99=6.679ms +concurrent workers=8 queries=8838 qps=882.18 avg=9.063ms p95=12.413ms p99=14.026ms +result=/home/zhenghaoz/vamana-runtime-20260928/query-before.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/run.py b/docs/benchmark-runs/vamana-runtime-20260928/run.py new file mode 100644 index 0000000..1ad19d1 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/run.py @@ -0,0 +1,48 @@ +import hashlib, json, os, pathlib, subprocess, time +def file_hash(path): + with path.open('rb') as f: return hashlib.file_digest(f, 'sha256').hexdigest() + +root = pathlib.Path(__file__).parent +repo = pathlib.Path('/home/zhenghaoz/xvec') +source_files = ['internal/ailego/container/heap.go', 'collection.go','internal/core/algorithm/vamana_algorithm.go','internal/core/algorithm/vamana_quantized_searcher.go'] +patch = subprocess.check_output(['git','diff','HEAD','--',*source_files], cwd=repo) +(root/'source.patch').write_bytes(patch) +revision = subprocess.check_output(['git','rev-parse','HEAD'],cwd=repo,text=True).strip() +patch_hash = hashlib.sha256(patch).hexdigest() +metadata = { + 'machine':'e2-standard-8', 'cpu':'AMD EPYC 7B12', 'cpu_affinity':'0-7', + 'gomaxprocs':8, 'gomemlimit':'24GiB', 'cgo_enabled':0, + 'base_commit': revision, 'source_patch_sha256': patch_hash, + 'backend_version': revision+'+vamana-runtime.'+patch_hash[:12], + 'binary_sha256':file_hash(root/'vector-db-bench'), + 'dataset_sha256':{p.name:file_hash(p) for p in sorted((root/'dataset').glob('*.parquet'))}, + 'runs':[] +} +(root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') +for precision in ['int4','int8','fp16','fp32']: + name = 'xvec-'+precision + args = ['taskset','-c','0-7',str(root/'vector-db-bench'),'xvec', + '--path',str(root/(name+'.collection')),'--case-type','Performance768D100K', + '--dataset-dir',str(root/'dataset'),'--skip-download','--index-type','vamana', + '--ef-search','200','--k','100','--batch-size','100','--max-docs-per-segment','10000000', + '--optimize-concurrency','8','--num-concurrency','8','--concurrency-duration','30s', + '--serial-cooldown','3s','--payload-profile','ids_only','--enable-mmap=true', + '--is-using-refiner=false','--output',str(root/(name+'.json'))] + if precision != 'fp32': args += ['--quantize-type',precision] + env = dict(os.environ,GOMAXPROCS='8',GOMEMLIMIT='24GiB') + print('START',name,flush=True) + start = time.monotonic() + with (root/(name+'.log')).open('w') as log: + proc = subprocess.Popen(args,stdout=log,stderr=subprocess.STDOUT,env=env,cwd=repo) + _, status, usage = os.wait4(proc.pid,0) + proc.returncode = os.waitstatus_to_exitcode(status) + elapsed = time.monotonic()-start + metrics = {'precision':precision,'command':args,'exit_code':proc.returncode, + 'peak_rss_kib':usage.ru_maxrss,'peak_rss_mib':usage.ru_maxrss/1024, + 'wall_seconds':elapsed,'user_cpu_seconds':usage.ru_utime,'system_cpu_seconds':usage.ru_stime} + (root/(name+'.resources.json')).write_text(json.dumps(metrics,indent=2)+'\n') + metadata['runs'].append(metrics) + (root/'metadata.json').write_text(json.dumps(metadata,indent=2)+'\n') + print('FINISH',name,json.dumps(metrics),flush=True) + if proc.returncode: raise SystemExit(proc.returncode) + print((root/(name+'.log')).read_text()[-2000:],flush=True) diff --git a/docs/benchmark-runs/vamana-runtime-20260928/source.patch b/docs/benchmark-runs/vamana-runtime-20260928/source.patch new file mode 100644 index 0000000..90db695 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/source.patch @@ -0,0 +1,606 @@ +diff --git a/collection.go b/collection.go +index d62eb29..95c824e 100644 +--- a/collection.go ++++ b/collection.go +@@ -584,9 +584,9 @@ func openCollectionDenseArtifact( + return core.OpenScalarQuantizedIVFIndex(ctx, path, kind, reformer) + case IndexTypeVamana: + if spec.quantize == QuantizeTypeUndefined { +- return core.OpenVamanaIndex(ctx, path) ++ return core.OpenVamanaIndexWithMmap(ctx, path, useMmap) + } +- return core.OpenScalarQuantizedVamanaIndex(ctx, path, kind, reformer) ++ return core.OpenScalarQuantizedVamanaIndexWithMmap(ctx, path, kind, reformer, useMmap) + case IndexTypeDiskANN: + if spec.quantize == QuantizeTypeUndefined { + candidateCount, candidateErr := collectionDenseCandidateCount(ctx, field, documents) +@@ -664,7 +664,7 @@ func (c *Collection) segmentDocumentsLocked(ctx context.Context) ([]collectionSe + if err != nil { + return nil, err + } +- if spec.indexType == IndexTypeHNSW || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { ++ if spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) { + borrowedFields[field.Name] = struct{}{} + } + } +@@ -1027,7 +1027,7 @@ func buildCollectionIndexes( + } + if field.DataType.IsDenseVector() { + var exact collectionDenseIndex +- useLazyExact := spec.indexType == IndexTypeHNSW || ++ useLazyExact := spec.indexType == IndexTypeHNSW || spec.indexType == IndexTypeVamana || + (spec.indexType == IndexTypeFlat && spec.quantize != QuantizeTypeUndefined) || + (spec.indexType == IndexTypeDiskANN && spec.quantize == QuantizeTypeUndefined) + if field.DataType == DataTypeVectorFP32 && useLazyExact { +@@ -1054,8 +1054,8 @@ func buildCollectionIndexes( + var flat collectionDenseIndex + if spec.quantize == QuantizeTypeUndefined || spec.indexType == IndexTypeHNSWRaBitQ || spec.indexType == IndexTypeIVFRaBitQ { + flat = exact +- } else if spec.indexType != IndexTypeHNSW { +- // Quantized HNSW supplies a shared Flat view after opening the graph. ++ } else if spec.indexType != IndexTypeHNSW && spec.indexType != IndexTypeVamana { ++ // Quantized graphs supply a shared Flat view after opening. + flat, err = buildCollectionDenseFlat(ctx, schema.Name, field, documents, spec) + if err != nil { + return fail(err) +@@ -1081,9 +1081,11 @@ func buildCollectionIndexes( + } + indexes.denseNative[field.Name] = native + if flat == nil { +- quantized, ok := native.(*core.ScalarQuantizedHNSWIndex) ++ quantized, ok := native.(interface { ++ FlatIndex() *core.ScalarQuantizedFlatIndex ++ }) + if !ok { +- return fail(fmt.Errorf("quantized HNSW field %q has an incompatible native index", field.Name)) ++ return fail(fmt.Errorf("quantized graph field %q has an incompatible native index", field.Name)) + } + indexes.denseFlat[field.Name] = quantized.FlatIndex() + } +@@ -2529,7 +2531,7 @@ func buildCollectionDenseVamana( + spec collectionVectorIndex, + workers int, + ) (collectionVamanaIndex, error) { +- candidates, err := collectionDenseCandidates(ctx, field, documents) ++ count, err := collectionDenseCandidateCount(ctx, field, documents) + if err != nil { + return nil, err + } +@@ -2551,17 +2553,27 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- for _, candidate := range candidates { +- if err := builder.Add(ctx, candidate.Key, candidate.Vector); err != nil { ++ if err := builder.Reserve(count); err != nil { ++ return nil, err ++ } ++ for _, document := range documents { ++ if err := ctx.Err(); err != nil { ++ return nil, err ++ } ++ value, found := document.Fields[field.Name] ++ if !found || value == nil { ++ continue ++ } ++ vector, err := denseValueToFloat32Borrowed(value) ++ if err != nil { ++ return nil, fmt.Errorf("document %d field %q: %w", document.DocID, field.Name, err) ++ } ++ if err := builder.Add(ctx, document.DocID, vector); err != nil { + return nil, err + } +- } +- base, err := builder.BuildInterleavedWithWorkers(ctx, workers) +- if err != nil { +- return nil, err + } + if spec.quantize == QuantizeTypeUndefined { +- return base, nil ++ return builder.BuildInterleavedWithWorkers(ctx, workers) + } + kind, err := toCoreQuantization(spec.quantize) + if err != nil { +@@ -2571,7 +2583,7 @@ func buildCollectionDenseVamana( + if err != nil { + return nil, err + } +- return core.NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) + } + + func buildCollectionDenseDiskANN( +diff --git a/internal/ailego/container/heap.go b/internal/ailego/container/heap.go +index dd981ef..5e88123 100644 +--- a/internal/ailego/container/heap.go ++++ b/internal/ailego/container/heap.go +@@ -41,6 +41,12 @@ func NewHeapWithCapacity[T any](capacity int, less func(a, b T) bool) *Heap[T] { + // Len returns the number of values in h. + func (h *Heap[T]) Len() int { return len(h.values) } + ++// Clear removes all values while retaining capacity for the next traversal. ++func (h *Heap[T]) Clear() { ++ clear(h.values) ++ h.values = h.values[:0] ++} ++ + // Push inserts value into h. + func (h *Heap[T]) Push(value T) { + h.values = append(h.values, value) +diff --git a/internal/core/algorithm/vamana_algorithm.go b/internal/core/algorithm/vamana_algorithm.go +index 84cba77..1c65006 100644 +--- a/internal/core/algorithm/vamana_algorithm.go ++++ b/internal/core/algorithm/vamana_algorithm.go +@@ -22,9 +22,11 @@ import ( + "errors" + "fmt" + "math" ++ "os" + "slices" + "sync" + ++ mmap "github.com/blevesearch/mmap-go" + "github.com/gorse-io/xvec/internal/ailego/container" + "github.com/gorse-io/xvec/internal/ailego/hash" + "github.com/gorse-io/xvec/internal/ailego/io" +@@ -148,6 +150,46 @@ func newBorrowedVamanaBuilder( + return builder, nil + } + ++// Reserve preallocates storage for at least count total vectors. It does not ++// change the builder length or the ownership guarantees of Add. ++func (b *VamanaBuilder) Reserve(count int) error { ++ if b == nil { ++ return errors.New("core: nil Vamana builder") ++ } ++ if count < 0 || uint64(count) >= math.MaxUint32 || (count > 0 && count > maxPlatformInt()/b.dimension) { ++ return ErrVamanaCapacity ++ } ++ b.mu.Lock() ++ defer b.mu.Unlock() ++ if b.built { ++ return ErrBuilderClosed ++ } ++ vectorCapacity := cap(b.vectors) ++ if b.fp16 { ++ vectorCapacity = cap(b.vectorsFP16) ++ } ++ if count <= cap(b.keys) && count*b.dimension <= vectorCapacity { ++ return nil ++ } ++ reservedKeys := make([]uint64, len(b.keys), max(count, len(b.keys))) ++ copy(reservedKeys, b.keys) ++ var reservedVectors []float32 ++ var reservedVectorsFP16 []uint16 ++ if b.fp16 { ++ reservedVectorsFP16 = make([]uint16, len(b.vectorsFP16), max(count*b.dimension, len(b.vectorsFP16))) ++ copy(reservedVectorsFP16, b.vectorsFP16) ++ } else { ++ reservedVectors = make([]float32, len(b.vectors), max(count*b.dimension, len(b.vectors))) ++ copy(reservedVectors, b.vectors) ++ } ++ reservedPositions := make(map[uint64]int, max(count, len(b.positions))) ++ for key, position := range b.positions { ++ reservedPositions[key] = position ++ } ++ b.keys, b.vectors, b.vectorsFP16, b.positions = reservedKeys, reservedVectors, reservedVectorsFP16, reservedPositions ++ return nil ++} ++ + // Add validates and clones one unique vector while the builder is open. + func (b *VamanaBuilder) Add(ctx context.Context, key uint64, vector []float32) error { + if b == nil { +@@ -562,6 +604,8 @@ type vamanaPruneScratch struct { + } + + type vamanaSearchScratch struct { ++ frontier *container.Heap[vamanaDistanceNode] ++ retained *container.Heap[vamanaDistanceNode] + visited []uint32 + generation uint32 + neighbors []int +@@ -572,12 +616,11 @@ func (i *VamanaIndex) searchBuildCandidates(ctx context.Context, queryPosition, + if limit <= 0 { + return []vamanaDistanceNode{}, nil + } +- better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } +- worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } +- frontier := container.NewHeap(better) +- retained := container.NewHeap(worse) + scratch := i.acquireVamanaSearchScratch() + defer i.releaseVamanaSearchScratch(scratch) ++ frontier, retained := scratch.frontier, scratch.retained ++ frontier.Clear() ++ retained.Clear() + visited, generation := scratch.visited, scratch.generation + distance, err := i.graphDistanceAt(queryPosition, entry) + if err != nil { +@@ -631,12 +674,11 @@ func (i *VamanaIndex) searchBuildCandidatesInterleaved( + if limit <= 0 { + return []vamanaDistanceNode{}, nil + } +- better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } +- worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } +- frontier := container.NewHeap(better) +- retained := container.NewHeap(worse) + scratch := i.acquireVamanaSearchScratch() + defer i.releaseVamanaSearchScratch(scratch) ++ frontier, retained := scratch.frontier, scratch.retained ++ frontier.Clear() ++ retained.Clear() + visited, generation := scratch.visited, scratch.generation + distance, err := i.graphDistanceAt(queryPosition, entry) + if err != nil { +@@ -688,7 +730,10 @@ func (i *VamanaIndex) acquireVamanaSearchScratch() *vamanaSearchScratch { + value := i.searchScratch.Get() + var scratch *vamanaSearchScratch + if value == nil { +- scratch = &vamanaSearchScratch{} ++ scratch = &vamanaSearchScratch{ ++ frontier: container.NewHeap(vamanaDistanceBetter), ++ retained: container.NewHeap(func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) }), ++ } + } else { + scratch = value.(*vamanaSearchScratch) + } +@@ -1179,19 +1224,26 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + if (count == 0 && index.entryPoint != -1) || (count > 0 && (index.entryPoint < 0 || index.entryPoint >= count)) { + return errors.New("core: invalid Vamana entry point") + } +- seenKeys := make(map[uint64]struct{}, count) ++ // The key-to-position bijection also detects duplicate keys. ++ seenNeighbors := make(map[int]struct{}, index.options.MaxDegree) ++ var vectorScratch []float32 ++ if index.fp16 { ++ vectorScratch = make([]float32, index.dimension) ++ } + for position, key := range index.keys { + if position&255 == 0 { + if err := ctx.Err(); err != nil { + return err + } + } +- if _, found := seenKeys[key]; found || index.positions[key] != position { ++ if mapped, found := index.positions[key]; !found || mapped != position { + return errors.New("core: invalid Vamana key map") + } +- seenKeys[key] = struct{}{} + if index.fp16 { +- if err := validateTrainingVector(float32VectorFromFP16(index.vectorFP16At(position)), index.dimension); err != nil { ++ for offset, value := range index.vectorFP16At(position) { ++ vectorScratch[offset] = utility.Float16BitsToFloat32(value) ++ } ++ if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { + return err + } + } else if err := validateTrainingVector(index.vectorAt(position), index.dimension); err != nil { +@@ -1200,7 +1252,7 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { + if len(index.neighbors[position]) > index.options.MaxDegree || len(index.neighbors[position]) != len(index.neighborDistances[position]) { + return errors.New("core: invalid Vamana degree") + } +- seenNeighbors := make(map[int]struct{}, len(index.neighbors[position])) ++ clear(seenNeighbors) + for offset, neighbor := range index.neighbors[position] { + if neighbor < 0 || neighbor >= count || neighbor == position { + return errors.New("core: invalid Vamana neighbor") +@@ -1402,13 +1454,14 @@ func searchVamanaGraph( + worse := func(left, right hnswScoredNode) bool { return resultBetter(right, left) } + frontier := container.NewHeap(better) + accepted := container.NewHeap(worse) +- visited := make([]bool, len(keys)) ++ visited := acquireHNSWVisited(len(keys)) ++ defer releaseHNSWVisited(visited) + score, err := scoreAt(entry) + if err != nil { + return nil, fmt.Errorf("core: score Vamana entry point: %w", err) + } + start := hnswScoredNode{position: entry, score: score} +- visited[entry] = true ++ visited.mark(entry) + frontier.Push(start) + if acceptVamanaNode(metric, keys, start, options.SearchOptions) { + accepted.Push(start) +@@ -1429,10 +1482,10 @@ func searchVamanaGraph( + batch.positions = batch.positions[:0] + batch.scores = batch.scores[:0] + for _, neighbor := range adjacent { +- if visited[neighbor] { ++ if visited.seen(neighbor) { + continue + } +- visited[neighbor] = true ++ visited.mark(neighbor) + batch.positions = append(batch.positions, neighbor) + batch.scores = append(batch.scores, 0) + } +@@ -1508,7 +1561,8 @@ func searchVamanaGraphBlockHeap( + ) ([]Result, error) { + capacity := min(len(keys), max(options.TopK, options.EFSearch)) + batch.blockHeap.Reset(capacity, cap(batch.positions)) +- states := make([]uint8, len(keys)) ++ states := acquireHNSWVisited(len(keys)) ++ defer releaseHNSWVisited(states) + score, err := scoreAt(entry) + if err != nil { + return nil, fmt.Errorf("core: score Vamana entry point: %w", err) +@@ -1519,7 +1573,7 @@ func searchVamanaGraphBlockHeap( + batch.blockHeap.pushBlockWithTies([]float32{entryDistance}, []uint32{entryID}, []uint64{entryTie}) + batch.overflow = batch.overflow[:0] + overflowCursor := 0 +- states[entry] = 1 ++ states.mark(entry) + + for batch.blockHeap.HasNext() || overflowCursor < len(batch.overflow) { + if err := ctx.Err(); err != nil { +@@ -1532,10 +1586,10 @@ func searchVamanaGraphBlockHeap( + current = batch.overflow[overflowCursor] + overflowCursor++ + } +- if states[current] == 2 { ++ if states.expanded(int(current)) { + continue + } +- states[current] = 2 ++ states.markExpanded(int(current)) + adjacent := neighbors[int(current)] + if prefetch != nil { + prefetch(adjacent) +@@ -1545,10 +1599,10 @@ func searchVamanaGraphBlockHeap( + batch.ties = batch.ties[:0] + batch.scores = batch.scores[:0] + for _, neighbor := range adjacent { +- if states[neighbor] != 0 { ++ if states.seen(neighbor) { + continue + } +- states[neighbor] = 1 ++ states.mark(neighbor) + batch.positions = append(batch.positions, neighbor) + batch.ids = append(batch.ids, uint32(neighbor)) + batch.ties = append(batch.ties, keys[neighbor]) +@@ -1724,17 +1778,24 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { + if i == nil { + return fmt.Errorf("%w: nil index", ErrInvalidVamanaFile) + } ++ // Add publishes a new generation without mutating existing storage. Holding ++ // the read lock keeps this generation stable while it is streamed to disk. + i.mu.RLock() +- snapshot, err := cloneVamanaIndex(ctx, i) +- i.mu.RUnlock() +- if err != nil { +- return err +- } +- encoded, err := encodeVamanaIndex(ctx, snapshot) +- if err != nil { ++ defer i.mu.RUnlock() ++ if err := ioutil.WriteFileAtomicFunc(ctx, path, 0o600, func(file *os.File) error { ++ if _, err := file.Write(make([]byte, vamanaHeaderSize)); err != nil { ++ return err ++ } ++ header, err := writeVamanaPayload(ctx, i, func(data []byte) error { ++ _, err := file.Write(data) ++ return err ++ }) ++ if err != nil { ++ return err ++ } ++ _, err = file.WriteAt(header, 0) + return err +- } +- if err := ioutil.WriteFileAtomic(ctx, path, encoded, 0o600); err != nil { ++ }); err != nil { + return fmt.Errorf("core: save Vamana file: %w", err) + } + return nil +@@ -1742,6 +1803,13 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { + + // OpenVamanaIndex reads and verifies a native Go Vamana artifact. + func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { ++ return OpenVamanaIndexWithMmap(ctx, path, false) ++} ++ ++// OpenVamanaIndexWithMmap decodes an owned index using an optional temporary ++// read-only mapping, avoiding a serialized full-file copy in the Go heap. ++// No references to the mapping survive the call. ++func OpenVamanaIndexWithMmap(ctx context.Context, path string, useMmap bool) (index *VamanaIndex, resultErr error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana open context") + } +@@ -1751,11 +1819,24 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + if path == "" { + return nil, fmt.Errorf("%w: empty path", ErrInvalidVamanaFile) + } ++ if useMmap { ++ file, err := os.Open(path) ++ if err != nil { ++ return nil, err ++ } ++ defer func() { resultErr = errors.Join(resultErr, file.Close()) }() ++ encoded, err := mmap.Map(file, mmap.RDONLY, 0) ++ if err != nil { ++ return nil, err ++ } ++ defer func() { resultErr = errors.Join(resultErr, encoded.Unmap()) }() ++ return decodeVamanaIndex(ctx, encoded) ++ } + encoded, err := readHNSWFile(ctx, path) + if err != nil { + return nil, fmt.Errorf("core: read Vamana file: %w", err) + } +- index, err := decodeVamanaIndex(ctx, encoded) ++ index, err = decodeVamanaIndex(ctx, encoded) + if err != nil { + return nil, fmt.Errorf("core: open Vamana file: %w", err) + } +@@ -1763,6 +1844,21 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + } + + func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) { ++ encoded := make([]byte, vamanaHeaderSize) ++ header, err := writeVamanaPayload(ctx, index, func(data []byte) error { ++ encoded = append(encoded, data...) ++ return nil ++ }) ++ if err != nil { ++ return nil, err ++ } ++ copy(encoded, header) ++ return encoded, nil ++} ++ ++// writeVamanaPayload bounds serialization scratch space independently of vector ++// count. The returned header contains the checksum of the streamed payload. ++func writeVamanaPayload(ctx context.Context, index *VamanaIndex, write func([]byte) error) ([]byte, error) { + if ctx == nil { + return nil, errors.New("core: nil Vamana encode context") + } +@@ -1786,7 +1882,21 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + if err != nil { + return nil, err + } +- payload := make([]byte, 0, payloadSize) ++ payload := make([]byte, 0, 64<<10) ++ var checksum uint32 ++ written := 0 ++ flush := func() error { ++ if err := ctx.Err(); err != nil { ++ return err ++ } ++ if err := write(payload); err != nil { ++ return err ++ } ++ checksum = hashutil.UpdateCRC32C(checksum, payload) ++ written += len(payload) ++ payload = payload[:0] ++ return nil ++ } + for position, key := range index.keys { + if position&1023 == 0 { + if err := ctx.Err(); err != nil { +@@ -1794,6 +1904,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint64(payload, key) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + if index.fp16 { + for position, value := range index.vectorsFP16 { +@@ -1803,6 +1918,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint16(payload, value) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } else { + for position, value := range index.vectors { +@@ -1812,6 +1932,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } + for position, adjacent := range index.neighbors { +@@ -1821,11 +1946,24 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + } + } + payload = binary.LittleEndian.AppendUint32(payload, uint32(len(adjacent))) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + for _, neighbor := range adjacent { + payload = binary.LittleEndian.AppendUint32(payload, uint32(neighbor)) ++ if len(payload) >= (64<<10)-8 { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ } + } + } +- if len(payload) != payloadSize { ++ if err := flush(); err != nil { ++ return nil, err ++ } ++ if written != payloadSize { + return nil, fmt.Errorf("%w: internal payload length", ErrInvalidVamanaFile) + } + header := make([]byte, vamanaHeaderSize) +@@ -1853,9 +1991,9 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) + entry = uint64(index.entryPoint) + } + binary.LittleEndian.PutUint64(header[72:80], entry) +- binary.LittleEndian.PutUint32(header[80:84], hashutil.CRC32C(payload)) ++ binary.LittleEndian.PutUint32(header[80:84], checksum) + binary.LittleEndian.PutUint32(header[124:128], hashutil.CRC32C(header[:124])) +- return append(header, payload...), nil ++ return header, nil + } + + func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error) { +diff --git a/internal/core/algorithm/vamana_quantized_searcher.go b/internal/core/algorithm/vamana_quantized_searcher.go +index 6cbba85..4ee42a9 100644 +--- a/internal/core/algorithm/vamana_quantized_searcher.go ++++ b/internal/core/algorithm/vamana_quantized_searcher.go +@@ -47,6 +47,20 @@ func NewScalarQuantizedVamanaIndex( + if err != nil { + return nil, err + } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, snapshot, kind, reformer) ++} ++ ++// BuildScalarQuantizedInterleavedWithWorkers transfers the completed graph to ++// an immutable quantized index without cloning its vectors and adjacency. ++func (b *VamanaBuilder) BuildScalarQuantizedInterleavedWithWorkers(ctx context.Context, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { ++ base, err := b.BuildInterleavedWithWorkers(ctx, workers) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} ++ ++func newOwnedScalarQuantizedVamanaIndex(ctx context.Context, snapshot *VamanaIndex, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { + vectors, err := newOwnedScalarQuantizedVectors( + ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, + ) +@@ -72,7 +86,26 @@ func OpenScalarQuantizedVamanaIndex(ctx context.Context, path string, kind Quant + if err != nil { + return nil, err + } +- return NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) ++} ++ ++// FlatIndex returns an immutable linear-search view sharing scalar codes and ++// original vectors with the graph. No vectors are copied or quantized again. ++func (i *ScalarQuantizedVamanaIndex) FlatIndex() *ScalarQuantizedFlatIndex { ++ if i == nil { ++ return nil ++ } ++ return &ScalarQuantizedFlatIndex{vectors: i.vectors} ++} ++ ++// OpenScalarQuantizedVamanaIndexWithMmap uses a temporary read-only file mapping ++// while decoding, releasing it before scalar codes are reconstructed. ++func OpenScalarQuantizedVamanaIndexWithMmap(ctx context.Context, path string, kind Quantization, reformer DenseReformer, useMmap bool) (*ScalarQuantizedVamanaIndex, error) { ++ base, err := OpenVamanaIndexWithMmap(ctx, path, useMmap) ++ if err != nil { ++ return nil, err ++ } ++ return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) + } + + func (i *ScalarQuantizedVamanaIndex) Dimension() int { diff --git a/docs/benchmark-runs/vamana-runtime-20260928/update_csv.py b/docs/benchmark-runs/vamana-runtime-20260928/update_csv.py new file mode 100644 index 0000000..f644a9f --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/update_csv.py @@ -0,0 +1,43 @@ +import csv, json, pathlib, shutil +root=pathlib.Path(__file__).parent +repo=pathlib.Path('/home/zhenghaoz/xvec') +archive=repo/'docs/benchmark-runs/vamana-runtime-20260928' +meta=json.loads((root/'metadata.json').read_text()) +assert len(meta['runs'])==4 and all(r['exit_code']==0 for r in meta['runs']) +with (archive/'previous.csv').open() as f: + reader=csv.DictReader(f); fields=reader.fieldnames; rows=list(reader) +for row in rows: + if row['backend']!='xvec': continue + kind=row['quantize_type']; name='xvec-'+kind + report=json.loads((root/(name+'.json')).read_text()); resource=json.loads((root/(name+'.resources.json')).read_text()) + c=report['config']; load=report['load']; serial=report['serial']; conc=report['concurrent'][0] + assert load['rows']==100000 and serial['queries']==1000 and conc['concurrency']==8 + assert len(report['concurrent'])==1 and report['case']['name']==row['case'] + assert report['system']['go_version']==row['go_version'] + assert c['backend']=='xvec' and c['index_type']=='vamana' + assert c['quantize_type']==('none' if kind=='fp32' else kind) + assert c['ef_search']==200 and c['concurrency_duration']=='30s' and c['serial_cooldown']=='3s' + for key in ['use_refiner','enable_mmap','k','batch_size','max_docs_per_segment','optimize_concurrency','payload_profile']: + assert str(c[key]).lower()==row[key],(key,c[key],row[key]) + row['backend_version']=meta['backend_version'] + row['inserted_count']=load['rows'] + for key in ['insert_duration_sec','optimize_duration_sec','load_duration_sec']: row[key]=load[key] + row['insert_rows_per_sec']=load['rows_per_second'] + row['recall_at_k_pct']=serial['recall']*100 + for prefix,metrics in [('serial',serial),('concurrent',conc)]: + for key in ['queries','qps','latency_avg_ms','latency_p95_ms','latency_p99_ms']: row[prefix+'_'+key]=metrics[key] + for key in ['peak_rss_kib','peak_rss_mib','wall_seconds','user_cpu_seconds','system_cpu_seconds']: row[key]=resource[key] + for key,value in row.items(): + if isinstance(value,float): row[key]=f'{value:.6f}'.rstrip('0').rstrip('.') + for suffix in ['.json','.resources.json','.log']: shutil.copy2(root/(name+suffix),archive/(name+suffix)) +with (repo/'docs/benchmark-vamana.csv').open('w') as f: + writer=csv.DictWriter(f,fields,lineterminator='\n'); writer.writeheader(); writer.writerows(rows) +shutil.copy2(root/'metadata.json',archive/'metadata.json') +# Store the exact commands and wait4 implementation for reproducibility. +shutil.copy2(root/'run.py',archive/'run.py') +shutil.copy2(root/'update_csv.py',archive/'update_csv.py') +with (archive/'previous.csv').open() as f: old={r['quantize_type']:r for r in csv.DictReader(f) if r['backend']=='xvec'} +for row in rows: + if row['backend']=='xvec': + prior=old[row['quantize_type']] + print(row['quantize_type'],'RSS MiB',prior['peak_rss_mib'],'->',row['peak_rss_mib'], 'change%',round((float(row['peak_rss_mib'])/float(prior['peak_rss_mib'])-1)*100,2),'recall',row['recall_at_k_pct']) diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.json new file mode 100644 index 0000000..790b1a2 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:09:28.275742403Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "fp16", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 5.589116436, + "optimize_duration_sec": 63.622920624, + "load_duration_sec": 69.21988089, + "rows_per_second": 17891.91568024796, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 162.31162699935655, + "recall": 0.9934700000000035, + "latency_avg_ms": 5.882405586000005, + "latency_p95_ms": 7.740576, + "latency_p99_ms": 8.3007524 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 30785, + "qps": 1025.8162494429691, + "latency_avg_ms": 7.795802401949016, + "latency_p95_ms": 10.396235800000001, + "latency_p99_ms": 11.7814722 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 5.589116436, + "optimize_duration": 63.622920624, + "load_duration": 69.21988089, + "qps": 1025.8162494429691, + "recall": 0.9934700000000035, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0083007524, + "serial_latency_p95": 0.007740576, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1025.8162494429691 + ], + "conc_latency_p99_list": [ + 0.0117814722 + ], + "conc_latency_p95_list": [ + 0.0103962358 + ], + "conc_latency_avg_list": [ + 0.007795802401949017 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.log b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.log new file mode 100644 index 0000000..7a627f3 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (17892.1 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +load rows=100000 duration=69.220s insert=5.589s optimize=63.623s rows/s=17891.9 +serial queries=1000 qps=162.31 recall=0.9935 mrr=0.0000 ndcg=0.0000 avg=5.882ms p95=7.741ms p99=8.301ms +concurrent workers=8 queries=30785 qps=1025.82 avg=7.796ms p95=10.396ms p99=11.781ms +result=/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.resources.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.resources.json new file mode 100644 index 0000000..2fab11e --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp16.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "fp16", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp16.json", + "--quantize-type", + "fp16" + ], + "exit_code": 0, + "peak_rss_kib": 2310924, + "peak_rss_mib": 2256.76171875, + "wall_seconds": 114.75985079299971, + "user_cpu_seconds": 686.68149, + "system_cpu_seconds": 5.027617 +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.json new file mode 100644 index 0000000..522addc --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:11:23.038642037Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "none", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 4.152613319, + "optimize_duration_sec": 63.988752889, + "load_duration_sec": 68.150493298, + "rows_per_second": 24081.221225789744, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 140.19690565111893, + "recall": 0.9938300000000034, + "latency_avg_ms": 6.6578310660000035, + "latency_p95_ms": 8.64638755, + "latency_p99_ms": 9.5597882 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 25369, + "qps": 845.5232712027275, + "latency_avg_ms": 9.456017884898863, + "latency_p95_ms": 14.404440999999995, + "latency_p99_ms": 19.1552378 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 4.152613319, + "optimize_duration": 63.988752889, + "load_duration": 68.150493298, + "qps": 845.5232712027275, + "recall": 0.9938300000000034, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0095597882, + "serial_latency_p95": 0.00864638755, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 845.5232712027275 + ], + "conc_latency_p99_list": [ + 0.0191552378 + ], + "conc_latency_p95_list": [ + 0.014404440999999995 + ], + "conc_latency_avg_list": [ + 0.009456017884898863 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.log b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.log new file mode 100644 index 0000000..4a45632 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (24081.6 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +load rows=100000 duration=68.150s insert=4.153s optimize=63.989s rows/s=24081.2 +serial queries=1000 qps=140.20 recall=0.9938 mrr=0.0000 ndcg=0.0000 avg=6.658ms p95=8.646ms p99=9.560ms +concurrent workers=8 queries=25369 qps=845.52 avg=9.456ms p95=14.404ms p99=19.155ms +result=/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.resources.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.resources.json new file mode 100644 index 0000000..a8f6a70 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-fp32.resources.json @@ -0,0 +1,47 @@ +{ + "precision": "fp32", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-fp32.json" + ], + "exit_code": 0, + "peak_rss_kib": 2199088, + "peak_rss_mib": 2147.546875, + "wall_seconds": 113.34803818499995, + "user_cpu_seconds": 686.835882, + "system_cpu_seconds": 5.929749 +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.json new file mode 100644 index 0000000..9c7f329 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:05:34.595840156Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int4", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.9355164160000005, + "optimize_duration_sec": 67.633822034, + "load_duration_sec": 74.576680209, + "rows_per_second": 14418.536991607922, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 228.35749914274766, + "recall": 0.8702200000000001, + "latency_avg_ms": 4.101897902000005, + "latency_p95_ms": 5.3139164999999995, + "latency_p99_ms": 5.992039 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 46307, + "qps": 1542.9666854596956, + "latency_avg_ms": 5.183015929405923, + "latency_p95_ms": 6.933552299999999, + "latency_p99_ms": 8.076526140000002 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.9355164160000005, + "optimize_duration": 67.633822034, + "load_duration": 74.576680209, + "qps": 1542.9666854596956, + "recall": 0.8702200000000001, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.005992039, + "serial_latency_p95": 0.0053139165, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1542.9666854596956 + ], + "conc_latency_p99_list": [ + 0.008076526140000002 + ], + "conc_latency_p95_list": [ + 0.006933552299999999 + ], + "conc_latency_avg_list": [ + 0.005183015929405923 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.log b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.log new file mode 100644 index 0000000..ed4b291 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (14418.7 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +load rows=100000 duration=74.577s insert=6.936s optimize=67.634s rows/s=14418.5 +serial queries=1000 qps=228.36 recall=0.8702 mrr=0.0000 ndcg=0.0000 avg=4.102ms p95=5.314ms p99=5.992ms +concurrent workers=8 queries=46307 qps=1542.97 avg=5.183ms p95=6.934ms p99=8.077ms +result=/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.resources.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.resources.json new file mode 100644 index 0000000..ca6eb41 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int4.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int4", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int4.json", + "--quantize-type", + "int4" + ], + "exit_code": 0, + "peak_rss_kib": 2295552, + "peak_rss_mib": 2241.75, + "wall_seconds": 118.85539514399989, + "user_cpu_seconds": 713.076651, + "system_cpu_seconds": 5.329071 +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.json new file mode 100644 index 0000000..d0b8c84 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.json @@ -0,0 +1,111 @@ +{ + "schema_version": "vector-db-bench/v1", + "tool": "xvec/cmd/vector-db-bench", + "timestamp": "2026-09-28T12:07:33.452670379Z", + "case": { + "name": "Performance768D100K", + "workload": "vector", + "dataset_name": "cohere", + "dataset_folder": "cohere_small_100k", + "size": 100000, + "dimension": 768, + "metric": "cosine", + "train_files": [ + "shuffle_train.parquet" + ] + }, + "dataset_dir": "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "config": { + "backend": "xvec", + "path": "/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.collection", + "db_label": "xvec-go", + "index_type": "vamana", + "m": 50, + "ef_construction": 500, + "ef_search": 200, + "ivf_n_list": 1024, + "ivf_n_iterations": 10, + "ivf_use_soar": false, + "ivf_n_probe": 10, + "ivf_scale_factor": 10, + "diskann_max_degree": 100, + "diskann_build_list": 50, + "diskann_pq_chunks": 0, + "diskann_query_list": 300, + "quantize_type": "int8", + "use_refiner": false, + "k": 100, + "batch_size": 100, + "concurrency_duration": "30s", + "serial_cooldown": "3s", + "num_concurrency": [ + 8 + ], + "optimize_concurrency": 8, + "max_docs_per_segment": 10000000, + "enable_mmap": true, + "payload_profile": "ids_only" + }, + "system": { + "goos": "linux", + "goarch": "amd64", + "go_version": "go1.27.1", + "num_cpu": 8, + "compiler": "gc" + }, + "load": { + "rows": 100000, + "insert_duration_sec": 6.592063777, + "optimize_duration_sec": 64.45318582, + "load_duration_sec": 71.052367847, + "rows_per_second": 15169.756146611384, + "immutable_segments": 1, + "storage_bytes": 317988890 + }, + "serial": { + "queries": 1000, + "qps": 228.6498825532904, + "recall": 0.9872200000000078, + "latency_avg_ms": 4.098632620000001, + "latency_p95_ms": 5.387367, + "latency_p99_ms": 6.1195283 + }, + "concurrent": [ + { + "concurrency": 8, + "queries": 44668, + "qps": 1488.734158865434, + "latency_avg_ms": 5.371760687024299, + "latency_p95_ms": 7.20476485, + "latency_p99_ms": 8.442133730000002 + } + ], + "vectordbbench_metrics": { + "inserted_count": 100000, + "insert_duration": 6.592063777, + "optimize_duration": 64.45318582, + "load_duration": 71.052367847, + "qps": 1488.734158865434, + "recall": 0.9872200000000078, + "mrr": 0, + "ndcg": 0, + "payload_profile": "ids_only", + "serial_latency_p99": 0.0061195283, + "serial_latency_p95": 0.0053873670000000005, + "conc_num_list": [ + 8 + ], + "conc_qps_list": [ + 1488.734158865434 + ], + "conc_latency_p99_list": [ + 0.008442133730000002 + ], + "conc_latency_p95_list": [ + 0.00720476485 + ], + "conc_latency_avg_list": [ + 0.005371760687024299 + ] + } +} diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.log b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.log new file mode 100644 index 0000000..4aad14d --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.log @@ -0,0 +1,7 @@ +inserted 100000 vectors (15169.9 rows/s) +cooling down for 3s before serial search +case=Performance768D100K dataset=/home/zhenghaoz/vamana-runtime-20260928/dataset +load rows=100000 duration=71.052s insert=6.592s optimize=64.453s rows/s=15169.8 +serial queries=1000 qps=228.65 recall=0.9872 mrr=0.0000 ndcg=0.0000 avg=4.099ms p95=5.387ms p99=6.120ms +concurrent workers=8 queries=44668 qps=1488.73 avg=5.372ms p95=7.205ms p99=8.442ms +result=/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.json diff --git a/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.resources.json b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.resources.json new file mode 100644 index 0000000..77b7ec0 --- /dev/null +++ b/docs/benchmark-runs/vamana-runtime-20260928/xvec-int8.resources.json @@ -0,0 +1,49 @@ +{ + "precision": "int8", + "command": [ + "taskset", + "-c", + "0-7", + "/home/zhenghaoz/vamana-runtime-20260928/vector-db-bench", + "xvec", + "--path", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.collection", + "--case-type", + "Performance768D100K", + "--dataset-dir", + "/home/zhenghaoz/vamana-runtime-20260928/dataset", + "--skip-download", + "--index-type", + "vamana", + "--ef-search", + "200", + "--k", + "100", + "--batch-size", + "100", + "--max-docs-per-segment", + "10000000", + "--optimize-concurrency", + "8", + "--num-concurrency", + "8", + "--concurrency-duration", + "30s", + "--serial-cooldown", + "3s", + "--payload-profile", + "ids_only", + "--enable-mmap=true", + "--is-using-refiner=false", + "--output", + "/home/zhenghaoz/vamana-runtime-20260928/xvec-int8.json", + "--quantize-type", + "int8" + ], + "exit_code": 0, + "peak_rss_kib": 2301332, + "peak_rss_mib": 2247.39453125, + "wall_seconds": 114.82267373400009, + "user_cpu_seconds": 688.447641, + "system_cpu_seconds": 5.169885 +} diff --git a/docs/benchmark-vamana-memory.md b/docs/benchmark-vamana-memory.md new file mode 100644 index 0000000..1948ad8 --- /dev/null +++ b/docs/benchmark-vamana-memory.md @@ -0,0 +1,199 @@ +# Vamana memory optimization + +The original `benchmark-vamana.csv` reported xvec peak RSS of 2,797–3,739 MiB +versus 567–783 MiB for zvec on the 100K × 768 workload. The four xvec rows +were replaced with the latest shared-vector rerun on 2026-09-28; zvec retains its historical +measurements. The [previous CSV](benchmark-runs/vamana-memory-20260928/previous.csv) +is preserved for comparison. + +The local zvec implementation separates vector and graph storage and dumps +vector/neighbor segments independently (`VamanaEntity::dump_vectors`, +`dump_neighbors` in `src/core/algorithm/vamana/vamana_entity.cc`). Its +`VamanaStreamerEntity` also allocates vector and graph regions from known +record counts. The initial Go changes applied explicit capacity and ownership management +to avoid multiple simultaneous copies: + +- Reserve Vamana builder storage using the collection's vector count. +- Feed documents directly to the builder, borrowing FP32 input until `Add` + copies it; converted FP16 inputs are temporary per document. +- Transfer freshly built/reopened graphs into scalar quantization. The public + constructor still clones caller-owned graphs to preserve snapshot isolation. +- Stream persistence through a 64 KiB buffer, with incremental CRC32C and + atomic replacement. This removes the full graph clone and full-file buffers. +- Reuse validation scratch, including the FP16 conversion buffer and neighbor + deduplication map. + +The file version and bytes are unchanged. Regression tests compare SHA-256 +checksums captured from the previous encoder for FP32 and FP16 artifacts. +Saving holds the index read lock through persistence, so concurrent Add +publication waits for the save to finish; searches remain concurrent. + +## Local allocation measurement + +Baseline: `5d9b8f5`; Linux amd64, AMD EPYC 7B12, GOMAXPROCS=8. +`BenchmarkVamanaSaveMemory` uses 2,048 vectors of dimension 768 and excludes +construction from the timed/allocation region. Each measurement ran three +save operations with the same benchmark source on the baseline and new code. + +| Storage | Before B/op | After B/op | Before allocations/op | After allocations/op | +| --- | ---: | ---: | ---: | ---: | +| FP32 | 19,807,328 | 68,952 | 4,877 | 18 | +| FP16 | 16,670,208 | 72,024 | 6,955 | 19 | + +These are allocated bytes per save, not retained heap or process peak RSS. +The separate 100K rerun below measures end-to-end behavior. + +```sh +go test ./internal/core/algorithm -run '^$' \ + -bench BenchmarkVamanaSaveMemory -benchtime=3x -count=1 +go test ./... +``` + +The full test suite passed, including concurrent Add/Search/Save/Open tests +and collection quantization/optimization/reopen tests. The race detector +could not run in this environment because no C compiler is installed. + +## First 100K end-to-end rerun (2026-09-28) + +The four xvec runs used fresh collections on e2-standard-8 (AMD EPYC 7B12), +Go 1.27.1, CGO_ENABLED=0, GOMAXPROCS=8, GOMEMLIMIT=24GiB, and CPU affinity +0–7. The Cohere shuffled training data, all 1,000 test queries, K=100, +construction degree/list 64/100, alpha 1.2, occlusion limit 750, search list +200, rotation for INT4/INT8, and other settings match the prior CSV. +Concurrent search ran for 30 seconds with 8 workers, followed by a 3-second +cooldown before serial search. Dataset downloads and binary compilation were +excluded from process resource measurements. + +| Precision | Previous peak RSS (MiB) | Rerun peak RSS (MiB) | Change | Recall@100 (%) | Serial QPS | Concurrent QPS | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| INT4 | 3092.26 | 3117.17 | +0.81% | 87.018 | 225.58 | 1409.99 | +| INT8 | 3145.95 | 3152.14 | +0.20% | 98.720 | 227.28 | 1340.87 | +| FP16 | 3739.13 | 3278.91 | -12.31% | 99.363 | 163.43 | 985.35 | +| FP32 | 2797.18 | 2474.98 | -11.52% | 99.378 | 239.45 | 1016.98 | + +FP16/FP32 show lower process peak RSS; INT4/INT8 do not. This confirms that +the save-allocation improvement alone does not establish an equivalent reduction +in the whole benchmark's high-water mark. Concurrent QPS is lower in all four +reruns. These are single-run historical comparisons, not a controlled isolation +of the patch: the current base commit also includes later changes, including +SIMD kernels. Graph construction is interleaved and can vary between runs. + +[Raw reports and provenance](benchmark-runs/vamana-memory-20260928/) include +JSON results, wait4 resource measurements, exact commands, binary and dataset +SHA-256 hashes, the source patch, and the previous CSV. `backend_version` +identifies base commit `5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997` plus patch +SHA-256 prefix `7e9600d19f64`; it does not claim the uncommitted changes are +part of that base commit. + +## Runtime-memory follow-up (2026-09-28) + +Profiling the first patch on the complete INT4 workload exposed three roughly +300 MiB allocations during query startup: the serialized artifact read buffer, +an independently built exact Flat index, and decoded document vectors. The +native graph's decoded vectors added another 336 MiB including topology. +Construction also allocated about 1.8 GiB cumulatively in candidate heap growth; +this is allocation volume, not live heap or peak RSS. + +The follow-up removes the first three copies and the repeated quantized Flat +encoding: Vamana uses a lazy exact fallback, a Flat view sharing graph scalar +codes, and encoded immutable document vectors in read-only collections. +`enable_mmap` now uses a temporary mapping when reopening Vamana artifacts; +owned graph data remains valid after unmapping. Construction heaps and query +visit marks are reused between traversals. No forced GC, GOGC change, or memory +limit change was introduced. Linear search, refinement, projected vectors, +read-only reopen, and concurrent operations remain covered by the tests. + +The following table compares the first patch with the runtime follow-up. +The four new complete runs used the same environment and parameters as above. + +| Precision | First patch RSS (MiB) | Runtime patch RSS (MiB) | Change | zvec RSS (MiB, historical) | Recall@100 (%) | Concurrent QPS | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| INT4 | 3117.17 | 2241.75 | -28.08% | 566.63 | 87.022 | 1542.97 | +| INT8 | 3152.14 | 2247.39 | -28.70% | 644.31 | 98.722 | 1488.73 | +| FP16 | 3278.91 | 2256.76 | -31.17% | 783.31 | 99.347 | 1025.82 | +| FP32 | 2474.98 | 2147.55 | -13.23% | 770.00 | 99.383 | 845.52 | + +Peak RSS is lower by 13–31% versus the first patch, but is still about 2.8–4.0 +times the historical zvec measurements. This does not establish memory parity. +The graph still owns original FP32 scoring vectors, and construction and storage +also contribute to the process-wide peak. Further reductions need profiling +of those remaining paths; the diagnostic heap samples above are from before +this follow-up. + +INT4/INT8/FP16 concurrent QPS improved versus the first rerun. FP32 full-run QPS +was lower (serial 140.20, concurrent 845.52), so an additional controlled query +comparison reused exactly the same persisted graph with both binaries: + +| Same FP32 graph, query-only diagnostic | First patch | Runtime patch | +| --- | ---: | ---: | +| Serial QPS (1,000 queries) | 211.11 | 211.67 | +| Concurrent QPS (8 workers, 10 seconds) | 882.18 | 979.51 | + +This diagnostic did not reproduce the large serial regression. It is a separate +single-run check, not proof of invariant performance. The CSV retains the +original complete run, including the lower FP32 QPS, rather than substituting +query-only results. Interleaved graph construction and machine variability limit +causal conclusions from single-run measurements. + +[Follow-up artifacts](benchmark-runs/vamana-runtime-20260928/) preserve both +kinds of reports, commands, checksums, the production source patch, and the prior +CSV. That version is base `5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997` plus +runtime patch `47525295f56e`. The full Go suite and added mmap/storage-sharing +regressions passed; the race-detector limitation noted above still applies. + +## Shared-vector follow-up (2026-09-28) + +The next patch addresses the remaining original-vector copies and an unrelated +whole-document copy in the optimization predicate: + +- Build Vamana directly over immutable collection FP32 rows. This removes the + builder's additional contiguous FP32 array; quantization shares these rows + too. Mutable `Add` operations detach using copy-on-write storage. +- Reopen scalar-quantized Vamana against the collection's existing encoded FP32 + originals. The loader verifies each row against the persisted artifact before + retaining it, then releases the temporary mapping. Original-vector access and + refinement decode on demand instead of retaining a second FP32 array. +- Omit edge-distance caches for that immutable quantized reader. Graph traversal + uses topology and scalar codes; cloning for mutation reconstructs the cache. +- Check the writing segment's document count through metadata when deciding + whether optimization is needed. Previously this predicate cloned all stored + document payloads through `Documents()` merely to check for an empty segment. + +The patch keeps FP32 construction distances, quantization, graph parameters, +refinement behavior, and the on-disk format. It introduces no GC tuning or forced +collection. Regression tests build more than 1,000 nodes to exercise graph +traversal and prefetching, compare borrowed and owned artifact bytes, and cover +L2/IP/cosine, three scalar precisions, mmap on/off, filtered searches, original +vector isolation, resave, and copy-on-write mutation. Missing or mismatched +encoded originals are rejected. + +All four complete 100K runs used the same settings and fresh collections. The +CSV now contains these measurements: + +| Precision | Previous RSS (MiB) | New RSS (MiB) | Change | zvec RSS (MiB, historical) | Recall@100 (%) | Concurrent QPS | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| INT4 | 2241.75 | 1575.29 | -29.73% | 566.63 | 87.013 | 1410.06 | +| INT8 | 2247.39 | 1508.05 | -32.90% | 644.31 | 98.726 | 1322.81 | +| FP16 | 2256.76 | 1486.04 | -34.15% | 783.31 | 99.349 | 882.08 | +| FP32 | 2147.55 | 1956.61 | -8.89% | 770.00 | 99.374 | 978.77 | + +The quantized runs reduce peak RSS by 30–34%; unquantized FP32 improves by 9%. +The encoded-original reopen path applies only to scalar-quantized indexes; +unquantized reopen still owns a contiguous FP32 scoring array. Peak RSS remains +1.9–2.8 times the historical zvec values, so the gap is smaller but substantial. +zvec was not rerun in this round. + +Recall changes are within 0.009 percentage points of the previous rerun. INT4, +INT8, and FP16 concurrent QPS decline by 8.6%, 11.1%, and 14.0%, respectively; +FP32 improves by 15.8%. The Optimize phase is slower in all four runs. The patch +therefore demonstrates a memory improvement, not a general speed improvement. +These are single-run comparisons with interleaved construction, not controlled +attribution of the performance changes. All measured timings, including the +regressions, are retained in the CSV. + +The full Go suite and the additional Vamana collection snapshot/mutation test +passed. [Raw reports and provenance](benchmark-runs/vamana-borrowed-20260928/) +include all four reports, the previous CSV, process resource measurements, +commands, and source/binary/dataset hashes. The latest version is base +`5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997` plus production patch +`586bbd53868e`. diff --git a/docs/benchmark-vamana.csv b/docs/benchmark-vamana.csv index 45d6fff..c3ce91c 100644 --- a/docs/benchmark-vamana.csv +++ b/docs/benchmark-vamana.csv @@ -1,9 +1,9 @@ machine,backend,backend_version,case,index_type,quantize_type,rotate,use_refiner,enable_mmap,vamana_max_degree,vamana_build_list,vamana_query_list,vamana_alpha,vamana_max_occlusion_size,vamana_saturate_graph,vamana_two_pass_build,vamana_use_contiguous_memory,vamana_use_id_map,k,batch_size,max_docs_per_segment,optimize_concurrency,query_concurrency,concurrency_duration_sec,serial_cooldown_sec,payload_profile,gomaxprocs,gomemlimit,cpu_affinity,go_version,inserted_count,insert_duration_sec,optimize_duration_sec,load_duration_sec,insert_rows_per_sec,serial_queries,serial_qps,recall_at_k_pct,serial_latency_avg_ms,serial_latency_p95_ms,serial_latency_p99_ms,concurrent_queries,concurrent_qps,concurrent_latency_avg_ms,concurrent_latency_p95_ms,concurrent_latency_p99_ms,peak_rss_kib,peak_rss_mib,wall_seconds,user_cpu_seconds,system_cpu_seconds -e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.450255,49.679974,56.135672,15503.263734,1000,330.151683,87.012,2.796118,3.577116,3.940061,60109,2003.053308,3.992393,5.781849,7.197208,3166476,3092.261719,100.682135,552.081874,9.451192 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,8.996763,81.640321,90.64578,11115.109394,1000,212.113223,87.013,4.428929,5.768363,6.701302,42322,1410.056422,5.670522,7.610116,9.019855,1613092,1575.285156,132.751458,814.904934,9.082705 e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int4,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.726085,31.247228,37.006987,17463.938926,1000,388.831267,81.039,2.355339,3.555305,3.973072,79847,2661.173599,3.004251,4.884632,6.706344,580228,566.628906,73.133591,448.838842,10.488299 -e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.464561,55.287267,61.7595,15468.955289,1000,276.881767,98.73,3.33458,4.513977,5.536994,57449,1913.679863,4.178597,6.025895,7.338456,3221456,3145.953125,106.379817,577.352607,14.930293 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.900676,75.837939,82.746723,14491.333678,1000,201.40215,98.726,4.67015,6.22249,6.952494,39706,1322.813534,6.045327,8.343272,9.731153,1544248,1508.054688,124.683037,775.998832,5.110706 e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,int8,true,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.096447,28.252815,34.383313,16402.995921,1000,551.288631,98.355,1.622462,2.329426,2.686843,94285,3142.44868,2.5443,3.962252,5.503418,659776,644.3125,69.623703,420.824085,12.125019 -e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,8.872388,94.384487,103.266854,11270.922316,1000,152.714245,99.363,6.258708,10.97716,15.707088,38405,1279.504062,6.250287,9.105754,11.102209,3828872,3739.132812,154.680377,810.329211,25.327972 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,6.009391,77.357544,83.37432,16640.620989,1000,137.838468,99.349,6.962301,9.219943,10.060918,26480,882.081299,9.066431,12.238447,14.063464,1521700,1486.035156,127.390108,785.333968,5.479874 e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp16,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,5.294007,77.133349,82.463938,18889.284977,1000,382.938654,99.198,2.399853,3.193495,3.536057,68113,2270.108375,3.522075,5.363077,7.251954,802112,783.3125,118.577794,784.075142,17.826853 -e2-standard-8,xvec,6a8b120d4284bf16853b2b4ad18465c599e72d82,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.953501,50.42262,55.383697,20187.742056,1000,317.498132,99.375,2.912781,3.781935,4.197044,48226,1607.41641,4.974847,6.636764,7.634434,2864312,2797.179688,95.84813,545.816312,12.665932 +e2-standard-8,xvec,5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.683736,75.022232,79.71497,21350.477687,1000,204.629036,99.374,4.537134,5.776747,6.192079,29375,978.774115,8.16897,11.059549,12.920711,2003564,1956.605469,123.51656,773.204417,8.120851 e2-standard-8,zvec,v0.7.0+rotate,Performance768D100K,vamana,fp32,false,false,true,64,100,200,1.2,750,false,false,false,false,100,100,10000000,8,8,30,3,ids_only,8,24GiB,0-7,go1.27.1,100000,4.376264,57.116215,61.51392,22850.542618,1000,357.309009,99.392,2.593317,3.435763,3.753755,55841,1860.602216,4.297414,6.194429,8.125008,788484,770.003906,97.844471,638.635022,14.392509 diff --git a/docs/tests/benchmark.test.ts b/docs/tests/benchmark.test.ts index dee024a..c53bb02 100644 --- a/docs/tests/benchmark.test.ts +++ b/docs/tests/benchmark.test.ts @@ -182,7 +182,7 @@ test('Vamana includes all four precisions and stays separate from other indexes' for (const record of records) { assert.equal(record.inserted_count, 100000); assert.equal(record.serial_queries, 1000); - assert.equal(record.backend_version, record.backend === 'xvec' ? '6a8b120d4284bf16853b2b4ad18465c599e72d82' : 'v0.7.0+rotate'); + assert.equal(record.backend_version, record.backend === 'xvec' ? '5d9b8f5ff98f13ba0a1d5ee65f9d53dd57456997+vamana-borrowed.586bbd53868e' : 'v0.7.0+rotate'); } }); diff --git a/encoded_vector_test.go b/encoded_vector_test.go index fc8dc6f..8c29b54 100644 --- a/encoded_vector_test.go +++ b/encoded_vector_test.go @@ -30,6 +30,10 @@ func TestReadOnlyQuantizedFlatEncodedVectors(t *testing.T) { testReadOnlyQuantizedEncodedVectors(t, IndexTypeFlat) } +func TestReadOnlyQuantizedVamanaEncodedVectors(t *testing.T) { + testReadOnlyQuantizedEncodedVectors(t, IndexTypeVamana) +} + func TestReadOnlyQuantizedHNSWEncodedVectors(t *testing.T) { testReadOnlyQuantizedEncodedVectors(t, IndexTypeHNSW) } @@ -52,7 +56,17 @@ func testReadOnlyQuantizedEncodedVectors(t *testing.T, indexType IndexType) { hp.M, hp.EFConstruction, hp.Quantize, hp.Quantizer = 8, 40, quantize, params.Quantizer indexParams = hp } + if indexType == IndexTypeVamana { + vp := NewVamanaIndexParams(MetricTypeL2) + vp.MaxDegree, vp.SearchListSize, vp.Quantize, vp.Quantizer = 8, 40, quantize, params.Quantizer + indexParams = vp + } newQueryParams := func(refine bool) QueryParams { + if indexType == IndexTypeVamana { + qp := NewVamanaQueryParams() + qp.UseRefiner = refine + return qp + } if indexType == IndexTypeHNSW { qp := NewHNSWQueryParams() qp.UseRefiner = refine @@ -95,6 +109,10 @@ func testReadOnlyQuantizedEncodedVectors(t *testing.T, indexType IndexType) { hnsw.Linear = true groupParams = hnsw } + if vamana, ok := groupParams.(VamanaQueryParams); ok { + vamana.Linear = true + groupParams = vamana + } groupQuery := GroupByVectorQuery{Field: "embedding", DenseVector: VectorFP32{.3, -.5, 1, .7}, Params: groupParams, GroupByField: "rating", GroupCount: 3, TopKPerGroup: 2, Projection: Projection{IncludeVectors: true}} wantGroups, err := writer.GroupByQuery(ctx, groupQuery) require.NoError(t, err) diff --git a/internal/ailego/container/heap.go b/internal/ailego/container/heap.go index dd981ef..5e88123 100644 --- a/internal/ailego/container/heap.go +++ b/internal/ailego/container/heap.go @@ -41,6 +41,12 @@ func NewHeapWithCapacity[T any](capacity int, less func(a, b T) bool) *Heap[T] { // Len returns the number of values in h. func (h *Heap[T]) Len() int { return len(h.values) } +// Clear removes all values while retaining capacity for the next traversal. +func (h *Heap[T]) Clear() { + clear(h.values) + h.values = h.values[:0] +} + // Push inserts value into h. func (h *Heap[T]) Push(value T) { h.values = append(h.values, value) diff --git a/internal/ailego/container/heap_test.go b/internal/ailego/container/heap_test.go index b49060c..b1a33ac 100644 --- a/internal/ailego/container/heap_test.go +++ b/internal/ailego/container/heap_test.go @@ -53,3 +53,22 @@ func TestHeapWithCapacityRejectsInvalidArguments(t *testing.T) { require.Panics(t, func() { NewHeapWithCapacity(-1, func(a, b int) bool { return a < b }) }) require.Panics(t, func() { NewHeapWithCapacity[int](0, nil) }) } + +func TestHeapClearReusesStorageAndReleasesValues(t *testing.T) { + first, second := 1, 2 + heap := NewHeapWithCapacity(4, func(a, b *int) bool { return *a < *b }) + heap.Push(&first) + heap.Push(&second) + storage := heap.values[:cap(heap.values)] + heap.Clear() + require.Zero(t, heap.Len()) + require.Equal(t, 4, cap(heap.values)) + for _, value := range storage { + require.Nil(t, value) + } + heap.Push(&second) + heap.Push(&first) + value, found := heap.Pop() + require.True(t, found) + require.Same(t, &first, value) +} diff --git a/internal/core/algorithm/vamana_algorithm.go b/internal/core/algorithm/vamana_algorithm.go index 84cba77..3461221 100644 --- a/internal/core/algorithm/vamana_algorithm.go +++ b/internal/core/algorithm/vamana_algorithm.go @@ -22,9 +22,11 @@ import ( "errors" "fmt" "math" + "os" "slices" "sync" + mmap "github.com/blevesearch/mmap-go" "github.com/gorse-io/xvec/internal/ailego/container" "github.com/gorse-io/xvec/internal/ailego/hash" "github.com/gorse-io/xvec/internal/ailego/io" @@ -100,6 +102,7 @@ type VamanaBuilder struct { options VamanaBuildOptions keys []uint64 vectors []float32 + vectorRows [][]float32 vectorsFP16 []uint16 fp16 bool positions map[uint64]int @@ -148,6 +151,46 @@ func newBorrowedVamanaBuilder( return builder, nil } +// Reserve preallocates storage for at least count total vectors. It does not +// change the builder length or the ownership guarantees of Add. +func (b *VamanaBuilder) Reserve(count int) error { + if b == nil { + return errors.New("core: nil Vamana builder") + } + if count < 0 || uint64(count) >= math.MaxUint32 || (count > 0 && count > maxPlatformInt()/b.dimension) { + return ErrVamanaCapacity + } + b.mu.Lock() + defer b.mu.Unlock() + if b.built { + return ErrBuilderClosed + } + vectorCapacity := cap(b.vectors) + if b.fp16 { + vectorCapacity = cap(b.vectorsFP16) + } + if count <= cap(b.keys) && count*b.dimension <= vectorCapacity { + return nil + } + reservedKeys := make([]uint64, len(b.keys), max(count, len(b.keys))) + copy(reservedKeys, b.keys) + var reservedVectors []float32 + var reservedVectorsFP16 []uint16 + if b.fp16 { + reservedVectorsFP16 = make([]uint16, len(b.vectorsFP16), max(count*b.dimension, len(b.vectorsFP16))) + copy(reservedVectorsFP16, b.vectorsFP16) + } else { + reservedVectors = make([]float32, len(b.vectors), max(count*b.dimension, len(b.vectors))) + copy(reservedVectors, b.vectors) + } + reservedPositions := make(map[uint64]int, max(count, len(b.positions))) + for key, position := range b.positions { + reservedPositions[key] = position + } + b.keys, b.vectors, b.vectorsFP16, b.positions = reservedKeys, reservedVectors, reservedVectorsFP16, reservedPositions + return nil +} + // Add validates and clones one unique vector while the builder is open. func (b *VamanaBuilder) Add(ctx context.Context, key uint64, vector []float32) error { if b == nil { @@ -253,7 +296,7 @@ func (b *VamanaBuilder) build(ctx context.Context, workers int) (*VamanaIndex, e index := &VamanaIndex{ dimension: b.dimension, options: b.options, distance: distance, distanceFP16: distanceFP16, - keys: b.keys, vectors: b.vectors, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, + keys: b.keys, vectors: b.vectors, vectorRows: b.vectorRows, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, neighbors: make([][]int, len(b.keys)), neighborDistances: make([][]float32, len(b.keys)), entryPoint: -1, } @@ -310,6 +353,7 @@ func (b *VamanaBuilder) build(ctx context.Context, workers int) (*VamanaIndex, e b.built = true b.keys = nil b.vectors = nil + b.vectorRows = nil b.vectorsFP16 = nil b.positions = nil return index, nil @@ -348,7 +392,7 @@ func (b *VamanaBuilder) buildInterleaved(ctx context.Context, workers int) (*Vam index := &VamanaIndex{ dimension: b.dimension, options: b.options, distance: distance, distanceFP16: distanceFP16, - keys: b.keys, vectors: b.vectors, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, + keys: b.keys, vectors: b.vectors, vectorRows: b.vectorRows, vectorsFP16: b.vectorsFP16, fp16: b.fp16, positions: b.positions, neighbors: make([][]int, len(b.keys)), neighborDistances: make([][]float32, len(b.keys)), entryPoint: -1, } @@ -410,6 +454,7 @@ func (b *VamanaBuilder) buildInterleaved(ctx context.Context, workers int) (*Vam b.built = true b.keys = nil b.vectors = nil + b.vectorRows = nil b.vectorsFP16 = nil b.positions = nil return index, nil @@ -438,6 +483,8 @@ type VamanaIndex struct { distanceFP16 mathutil.DenseDistanceFP16 keys []uint64 vectors []float32 + vectorRows [][]float32 // Immutable FP32 rows borrowed during collection builds. + encodedVectors [][]byte // Immutable originals borrowed by scalar-quantized readers. vectorsFP16 []uint16 fp16 bool vectorMagnitudes []float32 @@ -562,6 +609,8 @@ type vamanaPruneScratch struct { } type vamanaSearchScratch struct { + frontier *container.Heap[vamanaDistanceNode] + retained *container.Heap[vamanaDistanceNode] visited []uint32 generation uint32 neighbors []int @@ -572,12 +621,11 @@ func (i *VamanaIndex) searchBuildCandidates(ctx context.Context, queryPosition, if limit <= 0 { return []vamanaDistanceNode{}, nil } - better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } - worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } - frontier := container.NewHeap(better) - retained := container.NewHeap(worse) scratch := i.acquireVamanaSearchScratch() defer i.releaseVamanaSearchScratch(scratch) + frontier, retained := scratch.frontier, scratch.retained + frontier.Clear() + retained.Clear() visited, generation := scratch.visited, scratch.generation distance, err := i.graphDistanceAt(queryPosition, entry) if err != nil { @@ -631,12 +679,11 @@ func (i *VamanaIndex) searchBuildCandidatesInterleaved( if limit <= 0 { return []vamanaDistanceNode{}, nil } - better := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(left, right) } - worse := func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) } - frontier := container.NewHeap(better) - retained := container.NewHeap(worse) scratch := i.acquireVamanaSearchScratch() defer i.releaseVamanaSearchScratch(scratch) + frontier, retained := scratch.frontier, scratch.retained + frontier.Clear() + retained.Clear() visited, generation := scratch.visited, scratch.generation distance, err := i.graphDistanceAt(queryPosition, entry) if err != nil { @@ -688,7 +735,10 @@ func (i *VamanaIndex) acquireVamanaSearchScratch() *vamanaSearchScratch { value := i.searchScratch.Get() var scratch *vamanaSearchScratch if value == nil { - scratch = &vamanaSearchScratch{} + scratch = &vamanaSearchScratch{ + frontier: container.NewHeap(vamanaDistanceBetter), + retained: container.NewHeap(func(left, right vamanaDistanceNode) bool { return vamanaDistanceBetter(right, left) }), + } } else { scratch = value.(*vamanaSearchScratch) } @@ -1030,6 +1080,17 @@ func (i *VamanaIndex) cacheCosineMagnitudes(ctx context.Context, workers int) er return nil } i.vectorMagnitudes = make([]float32, len(i.keys)) + if i.encodedVectors != nil { + vector := make([]float32, i.dimension) + for position, encoded := range i.encodedVectors { + if err := ctx.Err(); err != nil { + return err + } + decodeHNSWVector(encoded, vector) + i.vectorMagnitudes[position] = mathutil.L2Magnitude(vector) + } + return nil + } if err := parallel.ParallelFor(ctx, len(i.keys), workers, func(_ context.Context, position int) error { if i.fp16 { i.vectorMagnitudes[position] = mathutil.L2MagnitudeFP16(i.vectorFP16At(position)) @@ -1095,6 +1156,14 @@ func (i *VamanaIndex) calculateMedoid(ctx context.Context) (int, error) { } func (i *VamanaIndex) vectorAt(position int) []float32 { + if i.vectorRows != nil { + return i.vectorRows[position] + } + if i.encodedVectors != nil { + vector := make([]float32, i.dimension) + decodeHNSWVector(i.encodedVectors[position], vector) + return vector + } start := position * i.dimension return i.vectors[start : start+i.dimension] } @@ -1134,7 +1203,21 @@ func cloneVamanaIndex(ctx context.Context, source *VamanaIndex) (*VamanaIndex, e keys: slices.Clone(source.keys), vectors: slices.Clone(source.vectors), vectorsFP16: slices.Clone(source.vectorsFP16), vectorMagnitudes: slices.Clone(source.vectorMagnitudes), positions: cloneUint64Positions(source.positions), entryPoint: source.entryPoint, - neighbors: make([][]int, len(source.neighbors)), neighborDistances: make([][]float32, len(source.neighborDistances)), + neighbors: make([][]int, len(source.neighbors)), neighborDistances: make([][]float32, len(source.neighbors)), + } + if source.vectorRows != nil || source.encodedVectors != nil { + clone.vectors = make([]float32, len(source.keys)*source.dimension) + for position := range source.keys { + if err := ctx.Err(); err != nil { + return nil, err + } + destination := clone.vectors[position*source.dimension : (position+1)*source.dimension] + if source.encodedVectors != nil { + decodeHNSWVector(source.encodedVectors[position], destination) + } else { + copy(destination, source.vectorRows[position]) + } + } } for position := range clone.neighbors { if position&255 == 0 { @@ -1143,7 +1226,18 @@ func cloneVamanaIndex(ctx context.Context, source *VamanaIndex) (*VamanaIndex, e } } clone.neighbors[position] = slices.Clone(source.neighbors[position]) - clone.neighborDistances[position] = slices.Clone(source.neighborDistances[position]) + if source.neighborDistances != nil { + clone.neighborDistances[position] = slices.Clone(source.neighborDistances[position]) + } else { + clone.neighborDistances[position] = make([]float32, len(clone.neighbors[position])) + for offset, neighbor := range clone.neighbors[position] { + distance, err := clone.graphDistanceAt(position, neighbor) + if err != nil { + return nil, err + } + clone.neighborDistances[position][offset] = distance + } + } } return clone, nil } @@ -1168,8 +1262,22 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { wantVectors := count * index.dimension validVectors := (!index.fp16 && len(index.vectors) == wantVectors && len(index.vectorsFP16) == 0) || (index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == wantVectors) + if index.vectorRows != nil { + validVectors = !index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == 0 && index.encodedVectors == nil && len(index.vectorRows) == count + } + if index.encodedVectors != nil { + validVectors = !index.fp16 && len(index.vectors) == 0 && len(index.vectorsFP16) == 0 && index.vectorRows == nil && len(index.encodedVectors) == count + for _, vector := range index.encodedVectors { + if len(vector) != index.dimension*4 { + validVectors = false + break + } + } + } + // Edge-distance caches are unnecessary for immutable quantized readers. + missingDistances := index.encodedVectors != nil && index.neighborDistances == nil if !validVectors || len(index.positions) != count || - len(index.neighbors) != count || len(index.neighborDistances) != count { + len(index.neighbors) != count || (!missingDistances && len(index.neighborDistances) != count) { return errors.New("core: inconsistent Vamana storage") } if (index.options.Metric == MetricCosine && len(index.vectorMagnitudes) != count) || @@ -1179,28 +1287,40 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { if (count == 0 && index.entryPoint != -1) || (count > 0 && (index.entryPoint < 0 || index.entryPoint >= count)) { return errors.New("core: invalid Vamana entry point") } - seenKeys := make(map[uint64]struct{}, count) + // The key-to-position bijection also detects duplicate keys. + seenNeighbors := make(map[int]struct{}, index.options.MaxDegree) + var vectorScratch []float32 + if index.fp16 || index.encodedVectors != nil { + vectorScratch = make([]float32, index.dimension) + } for position, key := range index.keys { if position&255 == 0 { if err := ctx.Err(); err != nil { return err } } - if _, found := seenKeys[key]; found || index.positions[key] != position { + if mapped, found := index.positions[key]; !found || mapped != position { return errors.New("core: invalid Vamana key map") } - seenKeys[key] = struct{}{} if index.fp16 { - if err := validateTrainingVector(float32VectorFromFP16(index.vectorFP16At(position)), index.dimension); err != nil { + for offset, value := range index.vectorFP16At(position) { + vectorScratch[offset] = utility.Float16BitsToFloat32(value) + } + if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { + return err + } + } else if index.encodedVectors != nil { + decodeHNSWVector(index.encodedVectors[position], vectorScratch) + if err := validateTrainingVector(vectorScratch, index.dimension); err != nil { return err } } else if err := validateTrainingVector(index.vectorAt(position), index.dimension); err != nil { return err } - if len(index.neighbors[position]) > index.options.MaxDegree || len(index.neighbors[position]) != len(index.neighborDistances[position]) { + if len(index.neighbors[position]) > index.options.MaxDegree || (!missingDistances && len(index.neighbors[position]) != len(index.neighborDistances[position])) { return errors.New("core: invalid Vamana degree") } - seenNeighbors := make(map[int]struct{}, len(index.neighbors[position])) + clear(seenNeighbors) for offset, neighbor := range index.neighbors[position] { if neighbor < 0 || neighbor >= count || neighbor == position { return errors.New("core: invalid Vamana neighbor") @@ -1209,6 +1329,9 @@ func validateVamanaIndex(ctx context.Context, index *VamanaIndex) error { return errors.New("core: duplicate Vamana neighbor") } seenNeighbors[neighbor] = struct{}{} + if missingDistances { + continue + } distance := index.neighborDistances[position][offset] if math.IsNaN(float64(distance)) || math.IsInf(float64(distance), 0) { return errors.New("core: invalid Vamana neighbor distance") @@ -1369,7 +1492,13 @@ func (i *VamanaIndex) searchVamana(ctx context.Context, query []float32, options return denseDistances(i.options.Metric, query, batch.vectors, queryMagnitude, batch.magnitudes, scores) } prefetch := func(neighbors []int) { - if !i.fp16 { + if i.vectorRows != nil { + batch.vectors = batch.vectors[:0] + for _, position := range neighbors { + batch.vectors = append(batch.vectors, i.vectorRows[position]) + } + prefetchDenseHNSWRows(batch.vectors, options.PrefetchOffset, options.PrefetchLines) + } else if !i.fp16 && i.encodedVectors == nil { prefetchDenseHNSWNeighbors(i.vectors, i.dimension, neighbors, options.PrefetchOffset, options.PrefetchLines) } } @@ -1402,13 +1531,14 @@ func searchVamanaGraph( worse := func(left, right hnswScoredNode) bool { return resultBetter(right, left) } frontier := container.NewHeap(better) accepted := container.NewHeap(worse) - visited := make([]bool, len(keys)) + visited := acquireHNSWVisited(len(keys)) + defer releaseHNSWVisited(visited) score, err := scoreAt(entry) if err != nil { return nil, fmt.Errorf("core: score Vamana entry point: %w", err) } start := hnswScoredNode{position: entry, score: score} - visited[entry] = true + visited.mark(entry) frontier.Push(start) if acceptVamanaNode(metric, keys, start, options.SearchOptions) { accepted.Push(start) @@ -1429,10 +1559,10 @@ func searchVamanaGraph( batch.positions = batch.positions[:0] batch.scores = batch.scores[:0] for _, neighbor := range adjacent { - if visited[neighbor] { + if visited.seen(neighbor) { continue } - visited[neighbor] = true + visited.mark(neighbor) batch.positions = append(batch.positions, neighbor) batch.scores = append(batch.scores, 0) } @@ -1508,7 +1638,8 @@ func searchVamanaGraphBlockHeap( ) ([]Result, error) { capacity := min(len(keys), max(options.TopK, options.EFSearch)) batch.blockHeap.Reset(capacity, cap(batch.positions)) - states := make([]uint8, len(keys)) + states := acquireHNSWVisited(len(keys)) + defer releaseHNSWVisited(states) score, err := scoreAt(entry) if err != nil { return nil, fmt.Errorf("core: score Vamana entry point: %w", err) @@ -1519,7 +1650,7 @@ func searchVamanaGraphBlockHeap( batch.blockHeap.pushBlockWithTies([]float32{entryDistance}, []uint32{entryID}, []uint64{entryTie}) batch.overflow = batch.overflow[:0] overflowCursor := 0 - states[entry] = 1 + states.mark(entry) for batch.blockHeap.HasNext() || overflowCursor < len(batch.overflow) { if err := ctx.Err(); err != nil { @@ -1532,10 +1663,10 @@ func searchVamanaGraphBlockHeap( current = batch.overflow[overflowCursor] overflowCursor++ } - if states[current] == 2 { + if states.expanded(int(current)) { continue } - states[current] = 2 + states.markExpanded(int(current)) adjacent := neighbors[int(current)] if prefetch != nil { prefetch(adjacent) @@ -1545,10 +1676,10 @@ func searchVamanaGraphBlockHeap( batch.ties = batch.ties[:0] batch.scores = batch.scores[:0] for _, neighbor := range adjacent { - if states[neighbor] != 0 { + if states.seen(neighbor) { continue } - states[neighbor] = 1 + states.mark(neighbor) batch.positions = append(batch.positions, neighbor) batch.ids = append(batch.ids, uint32(neighbor)) batch.ties = append(batch.ties, keys[neighbor]) @@ -1630,7 +1761,7 @@ func (i *VamanaIndex) Add(ctx context.Context, key uint64, vector []float32) err i.mu.RUnlock() return fmt.Errorf("%w: %d", ErrDuplicateKey, key) } - vectorCount := len(i.vectors) + vectorCount := len(i.keys) * i.dimension if i.fp16 { vectorCount = len(i.vectorsFP16) } @@ -1683,6 +1814,7 @@ func (i *VamanaIndex) Add(ctx context.Context, key uint64, vector []float32) err i.mu.Unlock() return err } + i.vectorRows, i.encodedVectors = nil, nil i.keys, i.vectors, i.vectorsFP16, i.vectorMagnitudes, i.positions = working.keys, working.vectors, working.vectorsFP16, working.vectorMagnitudes, working.positions i.neighbors, i.neighborDistances, i.entryPoint = working.neighbors, working.neighborDistances, working.entryPoint i.mu.Unlock() @@ -1724,17 +1856,24 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { if i == nil { return fmt.Errorf("%w: nil index", ErrInvalidVamanaFile) } + // Add publishes a new generation without mutating existing storage. Holding + // the read lock keeps this generation stable while it is streamed to disk. i.mu.RLock() - snapshot, err := cloneVamanaIndex(ctx, i) - i.mu.RUnlock() - if err != nil { - return err - } - encoded, err := encodeVamanaIndex(ctx, snapshot) - if err != nil { + defer i.mu.RUnlock() + if err := ioutil.WriteFileAtomicFunc(ctx, path, 0o600, func(file *os.File) error { + if _, err := file.Write(make([]byte, vamanaHeaderSize)); err != nil { + return err + } + header, err := writeVamanaPayload(ctx, i, func(data []byte) error { + _, err := file.Write(data) + return err + }) + if err != nil { + return err + } + _, err = file.WriteAt(header, 0) return err - } - if err := ioutil.WriteFileAtomic(ctx, path, encoded, 0o600); err != nil { + }); err != nil { return fmt.Errorf("core: save Vamana file: %w", err) } return nil @@ -1742,6 +1881,17 @@ func (i *VamanaIndex) Save(ctx context.Context, path string) error { // OpenVamanaIndex reads and verifies a native Go Vamana artifact. func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { + return OpenVamanaIndexWithMmap(ctx, path, false) +} + +// OpenVamanaIndexWithMmap decodes an owned index using an optional temporary +// read-only mapping, avoiding a serialized full-file copy in the Go heap. +// No references to the mapping survive the call. +func OpenVamanaIndexWithMmap(ctx context.Context, path string, useMmap bool) (*VamanaIndex, error) { + return openVamanaIndexWithStorage(ctx, path, useMmap, nil) +} + +func openVamanaIndexWithStorage(ctx context.Context, path string, useMmap bool, originals map[uint64][]byte) (index *VamanaIndex, resultErr error) { if ctx == nil { return nil, errors.New("core: nil Vamana open context") } @@ -1751,11 +1901,24 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { if path == "" { return nil, fmt.Errorf("%w: empty path", ErrInvalidVamanaFile) } + if useMmap { + file, err := os.Open(path) + if err != nil { + return nil, err + } + defer func() { resultErr = errors.Join(resultErr, file.Close()) }() + encoded, err := mmap.Map(file, mmap.RDONLY, 0) + if err != nil { + return nil, err + } + defer func() { resultErr = errors.Join(resultErr, encoded.Unmap()) }() + return decodeVamanaIndexWithStorage(ctx, encoded, originals) + } encoded, err := readHNSWFile(ctx, path) if err != nil { return nil, fmt.Errorf("core: read Vamana file: %w", err) } - index, err := decodeVamanaIndex(ctx, encoded) + index, err = decodeVamanaIndexWithStorage(ctx, encoded, originals) if err != nil { return nil, fmt.Errorf("core: open Vamana file: %w", err) } @@ -1763,6 +1926,21 @@ func OpenVamanaIndex(ctx context.Context, path string) (*VamanaIndex, error) { } func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) { + encoded := make([]byte, vamanaHeaderSize) + header, err := writeVamanaPayload(ctx, index, func(data []byte) error { + encoded = append(encoded, data...) + return nil + }) + if err != nil { + return nil, err + } + copy(encoded, header) + return encoded, nil +} + +// writeVamanaPayload bounds serialization scratch space independently of vector +// count. The returned header contains the checksum of the streamed payload. +func writeVamanaPayload(ctx context.Context, index *VamanaIndex, write func([]byte) error) ([]byte, error) { if ctx == nil { return nil, errors.New("core: nil Vamana encode context") } @@ -1786,7 +1964,21 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) if err != nil { return nil, err } - payload := make([]byte, 0, payloadSize) + payload := make([]byte, 0, 64<<10) + var checksum uint32 + written := 0 + flush := func() error { + if err := ctx.Err(); err != nil { + return err + } + if err := write(payload); err != nil { + return err + } + checksum = hashutil.UpdateCRC32C(checksum, payload) + written += len(payload) + payload = payload[:0] + return nil + } for position, key := range index.keys { if position&1023 == 0 { if err := ctx.Err(); err != nil { @@ -1794,6 +1986,11 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) } } payload = binary.LittleEndian.AppendUint64(payload, key) + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } + } } if index.fp16 { for position, value := range index.vectorsFP16 { @@ -1803,17 +2000,41 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) } } payload = binary.LittleEndian.AppendUint16(payload, value) + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } + } + } + } else if index.encodedVectors != nil { + for _, raw := range index.encodedVectors { + for len(raw) != 0 { + n := min(len(raw), (64<<10)-len(payload)) + payload = append(payload, raw[:n]...) + raw = raw[n:] + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } + } + } } } else { - for position, value := range index.vectors { - if position&16383 == 0 { - if err := ctx.Err(); err != nil { - return nil, err + for position := range index.keys { + if err := ctx.Err(); err != nil { + return nil, err + } + for _, value := range index.vectorAt(position) { + payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } } } - payload = binary.LittleEndian.AppendUint32(payload, math.Float32bits(value)) } } + for position, adjacent := range index.neighbors { if position&255 == 0 { if err := ctx.Err(); err != nil { @@ -1821,11 +2042,24 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) } } payload = binary.LittleEndian.AppendUint32(payload, uint32(len(adjacent))) + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } + } for _, neighbor := range adjacent { payload = binary.LittleEndian.AppendUint32(payload, uint32(neighbor)) + if len(payload) >= (64<<10)-8 { + if err := flush(); err != nil { + return nil, err + } + } } } - if len(payload) != payloadSize { + if err := flush(); err != nil { + return nil, err + } + if written != payloadSize { return nil, fmt.Errorf("%w: internal payload length", ErrInvalidVamanaFile) } header := make([]byte, vamanaHeaderSize) @@ -1853,12 +2087,16 @@ func encodeVamanaIndex(ctx context.Context, index *VamanaIndex) ([]byte, error) entry = uint64(index.entryPoint) } binary.LittleEndian.PutUint64(header[72:80], entry) - binary.LittleEndian.PutUint32(header[80:84], hashutil.CRC32C(payload)) + binary.LittleEndian.PutUint32(header[80:84], checksum) binary.LittleEndian.PutUint32(header[124:128], hashutil.CRC32C(header[:124])) - return append(header, payload...), nil + return header, nil } func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error) { + return decodeVamanaIndexWithStorage(ctx, encoded, nil) +} + +func decodeVamanaIndexWithStorage(ctx context.Context, encoded []byte, originals map[uint64][]byte) (*VamanaIndex, error) { if ctx == nil { return nil, errors.New("core: nil Vamana decode context") } @@ -1964,7 +2202,24 @@ func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error } index.keys[position], index.positions[key] = key, position } - if fp16 { + if originals != nil { + if fp16 || len(originals) != count { + return nil, fmt.Errorf("%w: incompatible encoded originals", ErrInvalidVamanaFile) + } + index.encodedVectors = make([][]byte, count) + index.neighborDistances = nil + for position, key := range index.keys { + if err := ctx.Err(); err != nil { + return nil, err + } + original, found := originals[key] + if !found || len(original) != dimension*4 || !bytes.Equal(original, payload[offset:offset+dimension*4]) { + return nil, fmt.Errorf("%w: encoded vector %d differs from artifact", ErrInvalidVamanaFile, key) + } + index.encodedVectors[position] = original[:len(original):len(original)] + offset += dimension * 4 + } + } else if fp16 { index.distanceFP16, err = denseDistanceFP16(options.Metric) if err != nil { return nil, err @@ -2022,6 +2277,9 @@ func decodeVamanaIndex(ctx context.Context, encoded []byte) (*VamanaIndex, error return nil, fmt.Errorf("%w: inconsistent edge payload", ErrInvalidVamanaFile) } for position, adjacent := range index.neighbors { + if index.neighborDistances == nil { + break + } if position&255 == 0 { if err := ctx.Err(); err != nil { return nil, err diff --git a/internal/core/algorithm/vamana_borrowed_vectors.go b/internal/core/algorithm/vamana_borrowed_vectors.go new file mode 100644 index 0000000..51236e4 --- /dev/null +++ b/internal/core/algorithm/vamana_borrowed_vectors.go @@ -0,0 +1,79 @@ +// Copyright 2026-present the xvec project +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package core + +import ( + "context" + "errors" + "fmt" +) + +// BuildVamanaWithBorrowedVectors builds an index over immutable FP32 rows. +// The caller must keep the rows immutable and alive for the index's lifetime. +// The candidate slice itself is not retained. Add uses copy-on-write storage. +func BuildVamanaWithBorrowedVectors(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate, workers int) (*VamanaIndex, error) { + builder, err := newVamanaBuilderWithBorrowedRows(ctx, dimension, options, candidates) + if err != nil { + return nil, err + } + return builder.BuildInterleavedWithWorkers(ctx, workers) +} + +// BuildScalarQuantizedVamanaWithBorrowedVectors builds and quantizes immutable +// FP32 rows without cloning the originals. Ownership matches +// BuildVamanaWithBorrowedVectors; codes and topology are owned by the result. +func BuildScalarQuantizedVamanaWithBorrowedVectors(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { + if !kind.valid() { + return nil, ErrInvalidQuantization + } + builder, err := newVamanaBuilderWithBorrowedRows(ctx, dimension, options, candidates) + if err != nil { + return nil, err + } + return builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, workers, kind, reformer) +} + +func newVamanaBuilderWithBorrowedRows(ctx context.Context, dimension int, options VamanaBuildOptions, candidates []Candidate) (*VamanaBuilder, error) { + if ctx == nil { + return nil, errors.New("core: nil borrowed Vamana context") + } + if err := ctx.Err(); err != nil { + return nil, err + } + builder, err := NewVamanaBuilder(dimension, options) + if err != nil { + return nil, err + } + if len(candidates) > maxPlatformInt()/dimension { + return nil, ErrVamanaCapacity + } + builder.keys = make([]uint64, len(candidates)) + builder.vectorRows = make([][]float32, len(candidates)) + builder.positions = make(map[uint64]int, len(candidates)) + for position, candidate := range candidates { + if err := ctx.Err(); err != nil { + return nil, err + } + if err := validateTrainingVector(candidate.Vector, dimension); err != nil { + return nil, err + } + if _, duplicate := builder.positions[candidate.Key]; duplicate { + return nil, fmt.Errorf("%w: %d", ErrDuplicateKey, candidate.Key) + } + builder.keys[position], builder.positions[candidate.Key] = candidate.Key, position + builder.vectorRows[position] = candidate.Vector[:dimension:dimension] + } + return builder, nil +} diff --git a/internal/core/algorithm/vamana_memory_test.go b/internal/core/algorithm/vamana_memory_test.go new file mode 100644 index 0000000..6d96d6e --- /dev/null +++ b/internal/core/algorithm/vamana_memory_test.go @@ -0,0 +1,265 @@ +// Copyright 2026-present the xvec project +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package core + +import ( + "context" + "crypto/sha256" + "encoding/binary" + "fmt" + "math" + "os" + "path/filepath" + "slices" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestVamanaReservedStorageAndQuantizedTransfer(t *testing.T) { + ctx := context.Background() + for _, fp16 := range []bool{false, true} { + t.Run(fmt.Sprint(fp16), func(t *testing.T) { + builder, err := newVamanaBuilder(4, DefaultVamanaBuildOptions(MetricL2), fp16) + require.NoError(t, err) + require.ErrorIs(t, builder.Reserve(-1), ErrVamanaCapacity) + require.ErrorIs(t, builder.Reserve(maxPlatformInt()), ErrVamanaCapacity) + require.NoError(t, builder.Reserve(40)) + for _, candidate := range quantizedIndexCandidates(40) { + require.NoError(t, builder.Add(ctx, candidate.Key, candidate.Vector)) + } + require.NoError(t, builder.Reserve(1)) + if fp16 { + owned := &builder.vectorsFP16[0] + index, err := builder.BuildInterleavedWithWorkers(ctx, 2) + require.NoError(t, err) + require.Same(t, owned, &index.vectorsFP16[0]) + } else { + owned := &builder.vectors[0] + index, err := builder.BuildScalarQuantizedInterleavedWithWorkers(ctx, 2, QuantizationInt8, nil) + require.NoError(t, err) + require.Same(t, owned, &index.base.vectors[0]) + require.Same(t, owned, &index.vectors.originals[0]) + path := filepath.Join(t.TempDir(), "quantized") + require.NoError(t, index.Save(ctx, path)) + reopened, err := OpenScalarQuantizedVamanaIndex(ctx, path, QuantizationInt8, nil) + require.NoError(t, err) + want, err := index.Search(ctx, []float32{1, 2, 3, 4}, 10) + require.NoError(t, err) + got, err := reopened.Search(ctx, []float32{1, 2, 3, 4}, 10) + require.NoError(t, err) + require.Equal(t, want, got) + } + require.ErrorIs(t, builder.Reserve(80), ErrBuilderClosed) + }) + } +} + +func TestVamanaStreamedSave(t *testing.T) { + ctx := context.Background() + for _, fp16 := range []bool{false, true} { + t.Run(fmt.Sprint(fp16), func(t *testing.T) { + builder, err := newVamanaBuilder(768, DefaultVamanaBuildOptions(MetricL2), fp16) + require.NoError(t, err) + require.NoError(t, builder.Reserve(48)) + for n := range 48 { + vector := make([]float32, 768) + vector[n] = float32(n + 1) + require.NoError(t, builder.Add(ctx, uint64(n), vector)) + } + index, err := builder.Build(ctx) + require.NoError(t, err) + encoded, err := encodeVamanaIndex(ctx, index) + require.NoError(t, err) + path := filepath.Join(t.TempDir(), "index") + require.NoError(t, index.Save(ctx, path)) + got, err := os.ReadFile(path) + require.NoError(t, err) + require.Equal(t, encoded, got) + // Checksums captured from the pre-streaming encoder at 5d9b8f5. + wantHash := "00f9cc48998de8bf9f4546e7edaa25e8db9b1cc822200c0d0f2afcf16c770fa6" + if fp16 { + wantHash = "879bb875ea8b6016c242c8583ef8ec72d19ecc4ec485279260c1fcd533287b71" + } + require.Equal(t, wantHash, fmt.Sprintf("%x", sha256.Sum256(got))) + reopened, err := OpenVamanaIndex(ctx, path) + require.NoError(t, err) + require.Equal(t, index.neighbors, reopened.neighbors) + require.Equal(t, index.vectors, reopened.vectors) + require.Equal(t, index.vectorsFP16, reopened.vectorsFP16) + mapped, err := OpenVamanaIndexWithMmap(ctx, path, true) + require.NoError(t, err) + require.Equal(t, reopened.vectors, mapped.vectors) + require.Equal(t, reopened.vectorsFP16, mapped.vectorsFP16) + require.Equal(t, reopened.neighbors, mapped.neighbors) + // The decoded index owns its data after the temporary mapping closes. + require.NoError(t, mapped.Add(ctx, 1000, make([]float32, 768))) + require.Equal(t, 49, mapped.Len()) + require.Equal(t, 48, reopened.Len()) + canceled, cancel := context.WithCancel(ctx) + cancel() + require.ErrorIs(t, index.Save(canceled, path), context.Canceled) + unchanged, err := os.ReadFile(path) + require.NoError(t, err) + require.Equal(t, got, unchanged) + }) + } +} + +// BenchmarkVamanaSaveMemory isolates persistence allocations from graph build. +func BenchmarkVamanaSaveMemory(b *testing.B) { + for _, fp16 := range []bool{false, true} { + b.Run(fmt.Sprint(fp16), func(b *testing.B) { + ctx := context.Background() + builder, err := newVamanaBuilder(768, DefaultVamanaBuildOptions(MetricL2), fp16) + if err != nil { + b.Fatal(err) + } + for n := range 2048 { + vector := make([]float32, 768) + vector[n%768] = float32(n%100 + 1) + if err := builder.Add(ctx, uint64(n), vector); err != nil { + b.Fatal(err) + } + } + index, err := builder.BuildInterleavedWithWorkers(ctx, 2) + if err != nil { + b.Fatal(err) + } + path := filepath.Join(b.TempDir(), "index") + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + if err := index.Save(ctx, path); err != nil { + b.Fatal(err) + } + } + }) + } +} + +func TestVamanaBorrowedRowsAndEncodedOriginals(t *testing.T) { + ctx := context.Background() + for _, metric := range []Metric{MetricL2, MetricIP, MetricCosine} { + t.Run(fmt.Sprint(metric), func(t *testing.T) { + candidates := quantizedIndexCandidates(1100) + options := DefaultVamanaBuildOptions(metric) + options.MaxDegree, options.SearchListSize = 8, 32 + builder, err := NewVamanaBuilder(4, options) + require.NoError(t, err) + require.NoError(t, builder.Reserve(len(candidates))) + originals := make(map[uint64][]byte, len(candidates)) + for _, c := range candidates { + require.NoError(t, builder.Add(ctx, c.Key, c.Vector)) + for _, value := range c.Vector { + originals[c.Key] = binary.LittleEndian.AppendUint32(originals[c.Key], math.Float32bits(value)) + } + } + owned, err := builder.BuildInterleavedWithWorkers(ctx, 1) + require.NoError(t, err) + borrowed, err := BuildVamanaWithBorrowedVectors(ctx, 4, options, candidates, 1) + require.NoError(t, err) + require.Nil(t, borrowed.vectors) + require.Same(t, &candidates[0].Vector[0], &borrowed.vectorRows[0][0]) + wantBytes, err := encodeVamanaIndex(ctx, owned) + require.NoError(t, err) + gotBytes, err := encodeVamanaIndex(ctx, borrowed) + require.NoError(t, err) + require.Equal(t, wantBytes, gotBytes) + query := candidates[7].Vector + want, err := owned.Search(ctx, query, 20) + require.NoError(t, err) + got, err := borrowed.Search(ctx, query, 20) + require.NoError(t, err) + require.Equal(t, want, got) + path := filepath.Join(t.TempDir(), "index") + require.NoError(t, borrowed.Save(ctx, path)) + require.NoError(t, borrowed.Add(ctx, 9999999, []float32{1, 2, 3, 4})) + require.Nil(t, borrowed.vectorRows) + require.Equal(t, candidates[0].Vector, borrowed.vectorAt(0)) + for _, kind := range []Quantization{QuantizationInt4, QuantizationInt8, QuantizationFP16} { + for _, useMmap := range []bool{false, true} { + t.Run(fmt.Sprintf("%v/%v", kind, useMmap), func(t *testing.T) { + expected, err := NewScalarQuantizedVamanaIndex(ctx, owned, kind, nil) + require.NoError(t, err) + encoded, err := OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx, path, kind, nil, originals, useMmap) + require.NoError(t, err) + require.Nil(t, encoded.base.vectors) + require.Nil(t, encoded.base.neighborDistances) + require.Nil(t, encoded.vectors.originals) + require.NotNil(t, encoded.vectors.reader) + require.Same(t, &originals[candidates[0].Key][0], &encoded.base.encodedVectors[0][0]) + require.Same(t, encoded.vectors, encoded.FlatIndex().vectors) + for _, filter := range []CandidateFilter{nil, func(key uint64) bool { return key%3 != 0 }} { + search := VamanaSearchOptions{SearchOptions: SearchOptions{TopK: 20, Filter: filter}, EFSearch: 64} + want, err := expected.SearchVamana(ctx, query, search) + require.NoError(t, err) + got, err := encoded.SearchVamana(ctx, query, search) + require.NoError(t, err) + require.Equal(t, want, got) + } + vector, found := encoded.Vector(candidates[0].Key) + require.True(t, found) + require.Equal(t, candidates[0].Vector, vector) + vector[0]++ + again, _ := encoded.Vector(candidates[0].Key) + require.Equal(t, candidates[0].Vector, again) + saved := filepath.Join(t.TempDir(), "resaved") + require.NoError(t, encoded.Save(ctx, saved)) + savedBytes, err := os.ReadFile(saved) + require.NoError(t, err) + require.Equal(t, wantBytes, savedBytes) + // Copy-on-write rebuilds the omitted build caches and detaches originals. + detached, err := cloneVamanaIndex(ctx, encoded.base) + require.NoError(t, err) + require.NoError(t, detached.Add(ctx, 9999999, []float32{1, 2, 3, 4})) + require.Nil(t, detached.encodedVectors) + require.NoError(t, validateVamanaIndex(ctx, detached)) + }) + } + } + key := candidates[0].Key + original := originals[key] + originals[key] = slices.Clone(original) + originals[key][0] ^= 1 + _, err = OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx, path, QuantizationInt4, nil, originals, true) + require.ErrorIs(t, err, ErrInvalidVamanaFile) + delete(originals, key) + _, err = OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx, path, QuantizationInt4, nil, originals, false) + require.ErrorIs(t, err, ErrInvalidVamanaFile) + }) + } +} + +func TestVamanaBorrowedRowsValidation(t *testing.T) { + ctx := context.Background() + options := DefaultVamanaBuildOptions(MetricL2) + _, err := BuildVamanaWithBorrowedVectors(nil, 4, options, nil, 1) + require.Error(t, err) + canceled, cancel := context.WithCancel(ctx) + cancel() + _, err = BuildVamanaWithBorrowedVectors(canceled, 4, options, nil, 1) + require.ErrorIs(t, err, context.Canceled) + candidates := quantizedIndexCandidates(2) + candidates[1].Key = candidates[0].Key + _, err = BuildVamanaWithBorrowedVectors(ctx, 4, options, candidates, 1) + require.ErrorIs(t, err, ErrDuplicateKey) + _, err = BuildVamanaWithBorrowedVectors(ctx, 3, options, candidates[:1], 1) + require.Error(t, err) + _, err = BuildVamanaWithBorrowedVectors(ctx, 4, options, nil, 0) + require.ErrorIs(t, err, ErrInvalidVamanaWorkers) + _, err = BuildScalarQuantizedVamanaWithBorrowedVectors(ctx, 4, options, nil, 1, Quantization(255), nil) + require.ErrorIs(t, err, ErrInvalidQuantization) +} diff --git a/internal/core/algorithm/vamana_quantized_searcher.go b/internal/core/algorithm/vamana_quantized_searcher.go index 6cbba85..456e1fd 100644 --- a/internal/core/algorithm/vamana_quantized_searcher.go +++ b/internal/core/algorithm/vamana_quantized_searcher.go @@ -47,8 +47,26 @@ func NewScalarQuantizedVamanaIndex( if err != nil { return nil, err } - vectors, err := newOwnedScalarQuantizedVectors( - ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, + return newOwnedScalarQuantizedVamanaIndex(ctx, snapshot, kind, reformer) +} + +// BuildScalarQuantizedInterleavedWithWorkers transfers the completed graph to +// an immutable quantized index without cloning its vectors and adjacency. +func (b *VamanaBuilder) BuildScalarQuantizedInterleavedWithWorkers(ctx context.Context, workers int, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { + base, err := b.BuildInterleavedWithWorkers(ctx, workers) + if err != nil { + return nil, err + } + return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) +} + +func newOwnedScalarQuantizedVamanaIndex(ctx context.Context, snapshot *VamanaIndex, kind Quantization, reformer DenseReformer) (*ScalarQuantizedVamanaIndex, error) { + var reader DenseVectorReader + if snapshot.encodedVectors != nil { + reader = encodedHNSWVectorReader(snapshot.encodedVectors) + } + vectors, err := newScalarQuantizedVectorStorageWithReader( + ctx, snapshot.dimension, snapshot.options.Metric, kind, reformer, snapshot.keys, snapshot.vectors, snapshot.vectorRows, reader, ) if err != nil { return nil, err @@ -72,7 +90,26 @@ func OpenScalarQuantizedVamanaIndex(ctx context.Context, path string, kind Quant if err != nil { return nil, err } - return NewScalarQuantizedVamanaIndex(ctx, base, kind, reformer) + return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) +} + +// FlatIndex returns an immutable linear-search view sharing scalar codes and +// original vectors with the graph. No vectors are copied or quantized again. +func (i *ScalarQuantizedVamanaIndex) FlatIndex() *ScalarQuantizedFlatIndex { + if i == nil { + return nil + } + return &ScalarQuantizedFlatIndex{vectors: i.vectors} +} + +// OpenScalarQuantizedVamanaIndexWithMmap uses a temporary read-only file mapping +// while decoding, releasing it before scalar codes are reconstructed. +func OpenScalarQuantizedVamanaIndexWithMmap(ctx context.Context, path string, kind Quantization, reformer DenseReformer, useMmap bool) (*ScalarQuantizedVamanaIndex, error) { + base, err := OpenVamanaIndexWithMmap(ctx, path, useMmap) + if err != nil { + return nil, err + } + return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) } func (i *ScalarQuantizedVamanaIndex) Dimension() int { @@ -188,3 +225,22 @@ var ( _ DenseSearcher = (*ScalarQuantizedVamanaIndex)(nil) _ DenseQuerySearcher = (*ScalarQuantizedVamanaIndex)(nil) ) + +// OpenScalarQuantizedVamanaIndexWithEncodedVectors verifies persisted originals +// against immutable collection-owned little-endian FP32 bytes, then retains +// those bytes for refinement instead of a second full decoded vector array. +// The caller must keep the bytes immutable and alive for the index's lifetime. +// The map and temporary artifact mapping are not retained. +func OpenScalarQuantizedVamanaIndexWithEncodedVectors(ctx context.Context, path string, kind Quantization, reformer DenseReformer, originals map[uint64][]byte, useMmap bool) (*ScalarQuantizedVamanaIndex, error) { + if originals == nil { + return nil, errors.New("core: nil encoded Vamana originals") + } + if !kind.valid() { + return nil, ErrInvalidQuantization + } + base, err := openVamanaIndexWithStorage(ctx, path, useMmap, originals) + if err != nil { + return nil, err + } + return newOwnedScalarQuantizedVamanaIndex(ctx, base, kind, reformer) +} diff --git a/internal/db/collection.go b/internal/db/collection.go index a13c666..9e0ff97 100644 --- a/internal/db/collection.go +++ b/internal/db/collection.go @@ -538,7 +538,7 @@ func (c *CollectionStore) OptimizationNeeded(ctx context.Context) (bool, error) if c.closed { return false, ErrCollectionClosed } - if writing := c.manager.Writing(); writing != nil && len(writing.Documents()) != 0 { + if writing := c.manager.Writing(); writing != nil && writing.Metadata().DocCount != 0 { return true, nil } if c.manager.Deletes().Count() != 0 {