Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -169,7 +169,7 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/AcceleratedHNSWU
public static List<byte[]> quantizeFloatVectorsToScalar(List<float[]> floatVectors)
```

Scalar quantization.
Scalar quantization to unsigned 7-bit values stored in Java bytes.

**Parameters**

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -57,7 +57,7 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAccelerate

Build the indexes and writes it to the disk.

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:258`_
_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:240`_

### mergeOneField

Expand All @@ -67,7 +67,7 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAccelerate

Write field for merging.

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:341`_
_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:323`_

### finish

Expand All @@ -77,7 +77,7 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAccelerate

Called once at the end before close.

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:372`_
_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:354`_

### close

Expand All @@ -87,7 +87,7 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAccelerate

Closes the resources.

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:392`_
_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:374`_

### ramBytesUsed

Expand All @@ -97,6 +97,6 @@ _Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAccelerate

Returns the memory usage of this object in bytes.

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:401`_
_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:383`_

_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/LuceneAcceleratedHNSWScalarQuantizedVectorsWriter.java:56`_
Original file line number Diff line number Diff line change
Expand Up @@ -540,7 +540,7 @@ public static List<byte[]> quantizeFloatVectorsToBinary(List<float[]> floatVecto
}

/**
* Scalar quantization.
* Scalar quantization to unsigned 7-bit values stored in Java bytes.
*
* @param floatVectors A list of float vectors
* @return A list of byte scalar representation for the input vectors
Expand All @@ -550,13 +550,14 @@ public static List<byte[]> quantizeFloatVectorsToScalar(List<float[]> floatVecto
return new ArrayList<>();
}

final int maximumUnsignedSevenBitValue = 127;
int dimensions = floatVectors.get(0).length;
int numVectors = floatVectors.size();

float[] minPerDim = new float[dimensions];
float[] maxPerDim = new float[dimensions];
Arrays.fill(minPerDim, Float.MAX_VALUE);
Arrays.fill(maxPerDim, Float.MIN_VALUE);
Arrays.fill(minPerDim, Float.POSITIVE_INFINITY);
Arrays.fill(maxPerDim, Float.NEGATIVE_INFINITY);

for (float[] vector : floatVectors) {
for (int d = 0; d < dimensions; d++) {
Expand All @@ -572,8 +573,8 @@ public static List<byte[]> quantizeFloatVectorsToScalar(List<float[]> floatVecto
float range = maxPerDim[d] - minPerDim[d];
if (range > 0) {
float normalized = (vector[d] - minPerDim[d]) / range;
int quantizedValue = Math.round(normalized * 127.0f) - 64;
quantized[d] = (byte) Math.max(-64, Math.min(63, quantizedValue));
int quantizedValue = Math.round(normalized * maximumUnsignedSevenBitValue);
quantized[d] = (byte) Math.max(0, Math.min(maximumUnsignedSevenBitValue, quantizedValue));
} else {
quantized[d] = 0;
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -148,26 +148,14 @@ public KnnFieldVectorsWriter<?> addField(FieldInfo fieldInfo) throws IOException
return writer;
}

private static byte signedToUnsignedByte(byte signedByte) {
return (byte) (signedByte & 0xFF);
}

private static byte[] convertSignedToUnsigned(byte[] signedVector) {
byte[] unsignedVector = new byte[signedVector.length];
for (int i = 0; i < signedVector.length; i++) {
unsignedVector[i] = signedToUnsignedByte(signedVector[i]);
}
return unsignedVector;
}

/**
* Builds the intermediate CAGRA index and builds and writes the HNSW index.
*
* @param fieldInfo instance of FieldInfo that has the field description
* @param vectors quantized vectors
* @throws IOException
*/
private void writeFieldInternal(FieldInfo fieldInfo, List<?> vectors) throws IOException {
private void writeFieldInternal(FieldInfo fieldInfo, List<byte[]> vectors) throws IOException {
int size = vectors.size();
if (writeTrivialField(fieldInfo, size)) {
return;
Expand All @@ -176,20 +164,14 @@ private void writeFieldInternal(FieldInfo fieldInfo, List<?> vectors) throws IOE
try {
int dimensions = fieldInfo.getVectorDimension();

// Convert 7-bit signed bytes to 8-bit unsigned bytes for cuVS compatibility
List<byte[]> unsignedVectors = new ArrayList<>(vectors.size());
for (Object signedVector : vectors) {
unsignedVectors.add(convertSignedToUnsigned((byte[]) signedVector));
}

// Create CuVSMatrix with BYTE data type (unsigned bytes)
// The scalar quantizer already emits nonnegative bytes for cuVS's unsigned BYTE type.
hostInputMemory.withMatrix(
fieldInfo.name,
size,
dimensions,
CuVSMatrix.DataType.BYTE,
builder -> {
for (byte[] vector : unsignedVectors) {
for (byte[] vector : vectors) {
builder.addVector(vector);
}
writeNonTrivialField(fieldInfo, builder.build());
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
/*
* SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: Apache-2.0
*/
package com.nvidia.cuvs.lucene;

import static org.junit.Assert.assertArrayEquals;

import java.util.List;
import org.junit.Test;

public class TestScalarQuantization {

@Test
public void quantizesNegativeOnlyDimensionsAcrossTheFullRange() {
List<byte[]> quantized =
AcceleratedHNSWUtils.quantizeFloatVectorsToScalar(
List.of(new float[] {-10.0f, -4.0f}, new float[] {-5.0f, -2.0f}));

assertArrayEquals(new byte[] {0, 0}, quantized.get(0));
assertArrayEquals(new byte[] {127, 127}, quantized.get(1));
}

@Test
public void quantizesIntermediateValuesMonotonically() {
List<byte[]> quantized =
AcceleratedHNSWUtils.quantizeFloatVectorsToScalar(
List.of(
new float[] {-1.0f},
new float[] {-0.5f},
new float[] {0.0f},
new float[] {0.5f},
new float[] {1.0f}));

// Quarter-step inputs have exact expected codes after rounding across the 0..127 range.
assertArrayEquals(new byte[] {0}, quantized.get(0));
assertArrayEquals(new byte[] {32}, quantized.get(1));
assertArrayEquals(new byte[] {64}, quantized.get(2));
assertArrayEquals(new byte[] {95}, quantized.get(3));
assertArrayEquals(new byte[] {127}, quantized.get(4));
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,209 @@
/*
* SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: Apache-2.0
*/
package com.nvidia.cuvs.lucene;

import static com.nvidia.cuvs.lucene.ThreadLocalCuVSResourcesProvider.isSupported;
import static org.apache.lucene.index.VectorSimilarityFunction.EUCLIDEAN;
import static org.junit.Assume.assumeTrue;

import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.concurrent.CopyOnWriteArrayList;
import java.util.stream.IntStream;
import org.apache.lucene.document.Document;
import org.apache.lucene.document.Field;
import org.apache.lucene.document.KnnFloatVectorField;
import org.apache.lucene.document.StringField;
import org.apache.lucene.index.DirectoryReader;
import org.apache.lucene.index.IndexWriter;
import org.apache.lucene.index.IndexWriterConfig;
import org.apache.lucene.index.NoMergePolicy;
import org.apache.lucene.index.SerialMergeScheduler;
import org.apache.lucene.index.TieredMergePolicy;
import org.apache.lucene.search.IndexSearcher;
import org.apache.lucene.search.KnnFloatVectorQuery;
import org.apache.lucene.store.Directory;
import org.apache.lucene.tests.util.LuceneTestCase;
import org.apache.lucene.tests.util.TestUtil;
import org.apache.lucene.util.InfoStream;
import org.junit.Test;

/** Checks the GPU-built scalar graph with CPU HNSW search against exact Euclidean neighbors. */
public class TestScalarQuantizedCagraGraph extends LuceneTestCase {
private static final String VECTOR_FIELD = "vector";
private static final String ID_FIELD = "id";
private static final int VECTOR_COUNT = 512;
private static final int GRID_COLUMNS = 32;
private static final int INITIAL_SEGMENT_COUNT = 2;
private static final int DOCUMENTS_PER_SEGMENT = VECTOR_COUNT / INITIAL_SEGMENT_COUNT;
private static final int MERGED_SEGMENT_COUNT = 1;
private static final int DIMENSIONS = 128;
private static final int DIMENSION_PATTERN_COUNT = 4;
private static final int TOP_K = 10;
// CPU HNSW may be approximate; allow at most two misses while byte-level tests stay exact.
private static final int MIN_EXACT_NEIGHBORS = 8;
private static final int CAGRA_GRAPH_DEGREE = 32;
private static final int CAGRA_INTERMEDIATE_GRAPH_DEGREE = 64;
private static final int HNSW_LAYER_COUNT = 1;

@Test
public void testGpuBuiltScalarHnswRetainsRecallAcrossMerge() throws Exception {
assumeTrue("cuVS is not supported", isSupported());

float[][] vectors = vectorsWithNegativeAndMixedSignDimensions();
RecordingInfoStream buildLog = new RecordingInfoStream();
AcceleratedHNSWParams params =
new AcceleratedHNSWParams.Builder()
.withStrategy(AcceleratedHNSWParams.Strategy.CUSTOM)
.withGraphDegree(CAGRA_GRAPH_DEGREE)
.withIntermediateGraphDegree(CAGRA_INTERMEDIATE_GRAPH_DEGREE)
.withHNSWLayer(HNSW_LAYER_COUNT)
.build();
int maxBufferedDocsWithoutAutomaticFlush = VECTOR_COUNT + 1;
IndexWriterConfig config =
new IndexWriterConfig()
.setCodec(
TestUtil.alwaysKnnVectorsFormat(
new LuceneAcceleratedHNSWScalarQuantizedVectorsFormat(params)))
.setMergePolicy(NoMergePolicy.INSTANCE)
.setMergeScheduler(new SerialMergeScheduler())
.setMaxBufferedDocs(maxBufferedDocsWithoutAutomaticFlush)
.setRAMBufferSizeMB(IndexWriterConfig.DISABLE_AUTO_FLUSH)
.setInfoStream(buildLog);

try (Directory directory = newDirectory()) {
try (IndexWriter writer = new IndexWriter(directory, config)) {
for (int id = 0; id < vectors.length; id++) {
Document document = new Document();
document.add(new StringField(ID_FIELD, Integer.toString(id), Field.Store.YES));
document.add(new KnnFloatVectorField(VECTOR_FIELD, vectors[id], EUCLIDEAN));
writer.addDocument(document);
if (id == DOCUMENTS_PER_SEGMENT - 1) {
writer.commit();
}
}
writer.commit();

try (DirectoryReader reader = DirectoryReader.open(directory)) {
assertEquals(INITIAL_SEGMENT_COUNT, reader.leaves().size());
assertHnswRecallAgainstExactNeighbors(reader, vectors);
}

long gpuBuildsBeforeMerge = buildLog.gpuWriterOpenCount();
assertTrue(
"Insufficient scalar GPU writer openings for initial segments: " + buildLog.messages,
gpuBuildsBeforeMerge >= INITIAL_SEGMENT_COUNT);

writer.getConfig().setMergePolicy(new TieredMergePolicy());
writer.forceMerge(MERGED_SEGMENT_COUNT);
writer.commit();
assertTrue(
"The merge did not open a scalar GPU writer: " + buildLog.messages,
buildLog.gpuWriterOpenCount() > gpuBuildsBeforeMerge);
}

try (DirectoryReader reader = DirectoryReader.open(directory)) {
assertEquals(MERGED_SEGMENT_COUNT, reader.leaves().size());
assertHnswRecallAgainstExactNeighbors(reader, vectors);
}
}
}

private static void assertHnswRecallAgainstExactNeighbors(
DirectoryReader reader, float[][] vectors) throws Exception {
IndexSearcher searcher = new IndexSearcher(reader);
int centerColumn = GRID_COLUMNS / 2;
int lastRowOfFirstSegmentCenter = DOCUMENTS_PER_SEGMENT - GRID_COLUMNS + centerColumn;
int firstRowOfSecondSegmentCenter = DOCUMENTS_PER_SEGMENT + centerColumn;
// Probe opposite grid corners and adjacent center rows across the segment split. The center
// neighborhoods cross the scalar midpoint where signed codes wrapped when read as unsigned.
for (int queryId :
new int[] {
0, lastRowOfFirstSegmentCenter, firstRowOfSecondSegmentCenter, VECTOR_COUNT - 1
}) {
var results =
searcher.search(new KnnFloatVectorQuery(VECTOR_FIELD, vectors[queryId], TOP_K), TOP_K);
List<Integer> actual = new ArrayList<>();
for (var hit : results.scoreDocs) {
actual.add(Integer.parseInt(searcher.storedFields().document(hit.doc).get(ID_FIELD)));
}

assertEquals("Query " + queryId, TOP_K, actual.size());
assertEquals("Query " + queryId, queryId, actual.get(0).intValue());
assertEquals(
"Query " + queryId + " returned duplicates", TOP_K, new HashSet<>(actual).size());

List<Integer> exact =
IntStream.range(0, vectors.length)
.boxed()
.sorted(
Comparator.comparingDouble(
(Integer id) -> squaredDistance(vectors[queryId], vectors[id]))
.thenComparingInt(Integer::intValue))
.limit(TOP_K)
.toList();
Set<Integer> expected = new HashSet<>(exact);
long overlap = actual.stream().filter(expected::contains).count();
assertTrue(
"Query " + queryId + " found " + overlap + "/" + TOP_K + " exact neighbors: " + actual,
overlap >= MIN_EXACT_NEIGHBORS);
}
}

private static double squaredDistance(float[] left, float[] right) {
double distance = 0;
for (int dimension = 0; dimension < left.length; dimension++) {
double difference = left[dimension] - right[dimension];
distance += difference * difference;
}
return distance;
}

private static float[][] vectorsWithNegativeAndMixedSignDimensions() {
float[][] vectors = new float[VECTOR_COUNT][DIMENSIONS];
// Dominant row and column terms create grid-local neighborhoods; the small cross-axis terms
// keep both coordinates represented in each dimension pattern.
for (int id = 0; id < VECTOR_COUNT; id++) {
int column = id % GRID_COLUMNS;
int row = id / GRID_COLUMNS;
for (int dimension = 0; dimension < DIMENSIONS; dimension++) {
vectors[id][dimension] =
switch (dimension % DIMENSION_PATTERN_COUNT) {
case 0 -> -20.0f + 0.4f * column + 0.006f * row; // all negative
case 1 -> -7.5f + 0.9f * row + 0.002f * column; // crosses zero
case 2 -> 4.0f + 0.25f * column + 0.003f * row; // all positive
default -> -6.0f + 0.21f * (column + row); // crosses zero
};
}
}
return vectors;
}

private static final class RecordingInfoStream extends InfoStream {
private static final String GPU_WRITER_OPENED =
"Lucene99AcceleratedHNSWQuantizedVectorsWriter opened";
private final List<String> messages = new CopyOnWriteArrayList<>();

@Override
public void message(String component, String message) {
messages.add(component + ": " + message);
}

@Override
public boolean isEnabled(String component) {
return true;
}

@Override
public void close() {}

private long gpuWriterOpenCount() {
return messages.stream().filter(message -> message.contains(GPU_WRITER_OPENED)).count();
}
}
}
Loading