diff --git a/fern/docs.yml b/fern/docs.yml index c5b3fd6842..4a1395a202 100644 --- a/fern/docs.yml +++ b/fern/docs.yml @@ -580,6 +580,10 @@ navigation: path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-gpusearchparams.md" - page: "IndexSearcherTimingBridge" path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md" + - page: "IndexWriterConfigRAMLimitBridge" + path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge.md" + - page: "Lucene101AcceleratedHNSWCodecFactory" + path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory.md" - page: "LuceneProvider" path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-luceneprovider.md" - page: "ThreadLocalCuVSResourcesProvider" diff --git a/fern/pages/cuvs_bench/lucene_backend.md b/fern/pages/cuvs_bench/lucene_backend.md index 3be47ec9ac..d282702a56 100644 --- a/fern/pages/cuvs_bench/lucene_backend.md +++ b/fern/pages/cuvs_bench/lucene_backend.md @@ -26,6 +26,113 @@ Selecting this backend is explicit. Commands that do not select it with explicit `--algorithms` value. Select `lucene_cpu_hnsw` explicitly when a CPU control is needed. +The accelerated-HNSW build accepts Lucene's `m` and `beam_width` parameters. +Both must be integers in the range 1 through 512 when either is specified; +omitting both preserves the codec defaults of 32 and 32. cuVS derives the +CAGRA build parameters with the `SAME_GRAPH_FOOTPRINT` heuristic, so `m: 16` +produces `graph_degree=32` and `intermediate_graph_degree=48`. For example, an +algorithm configuration for `m=16` and `beam_width=80` is: + +```yaml +name: lucene_accelerated_hnsw +groups: + m16_bw80: + build: + codec: ["Lucene101AcceleratedHNSWCodec"] + m: [16] + beam_width: [80] + search: {} +``` + +Pass that file with `--configuration`, select +`--algorithms lucene_accelerated_hnsw --groups m16_bw80`, and use `--build`. +The normalized HNSW parameters are recorded in the index manifest. Result +metadata records those requested parameters together with graph degrees derived +from the selected heuristic; the graph-degree fields are not direct native +observations. The machine-readable +`graph_degree_source=requested_hnsw_same_graph_footprint_derivation` field makes +that provenance explicit, including when the writer uses its CPU fallback. + +Both cuVS-backed algorithms can control sequential ingestion partitions and +the final segment topology. `premerge_segment_count`, +`force_merge_segment_count`, and `ram_per_thread_hard_limit_mb` must be +specified together. The partition count must be a positive integer, and +`force_merge_segment_count` must be either `0` (retain the partition segments) +or `1` (produce one final segment). CAGRA builds require +`force_merge_segment_count: 0`; they publish the segments built directly +rather than invoking the codec's vector-merge path. + +Each equal, contiguous partition uses a separate writer lifecycle. Automatic +merges are disabled during ingestion, and the document-count flush boundary is +derived from the partition size. Values from 1 through 2047 MiB use Lucene's +public per-thread RAM-limit setter. If that limit causes an earlier flush, the +backend rejects the physical topology mismatch instead of silently reporting +the requested segment shape. Use smaller partitions when a partition cannot +fit under the public safety limit. + +Larger direct-built segments are an explicit unsupported mode. A +`ram_per_thread_hard_limit_mb` value of 2048 MiB or greater also requires +`allow_unsupported_lucene_ram_limit: true` and +`force_merge_segment_count: 0`. The thin JAR applies the requested value to +Lucene 10.2's protected, non-public +`LiveIndexWriterConfig.perThreadHardLimitMB` field and reads the value back. +The bridge rejects a missing field, a field with the wrong type, an access or +write failure, or a getter-readback mismatch. The override only changes +Lucene's flush threshold: it does not reserve memory, make construction +out-of-core, guarantee that the host/JVM/GPU can hold the segment, or make a +later force merge safe. + +When a final count of one is requested from more than one pre-merge segment, a +serial `forceMerge(1)` exercises the codec's vector-merge path; an index already +containing one segment does not invoke a merge and records +`runtime_force_merge_seconds=0` and `final_merge_policy=null`. This control does +not by itself make that merge out-of-core. For example, a four-partition build +that retains all four segments uses: + +```yaml +name: lucene_accelerated_hnsw +groups: + m16_bw80_four_partitions: + build: + codec: ["Lucene101AcceleratedHNSWCodec"] + m: [16] + beam_width: [80] + premerge_segment_count: [4] + force_merge_segment_count: [0] + ram_per_thread_hard_limit_mb: [1945] + search: {} +``` + +This produces and retains four equal segments. Set +`force_merge_segment_count: [1]` to merge them serially into one segment after +ingestion. Results record the requested and observed pre-merge segment counts, +exact segment vector counts, derived `max_buffered_docs`, applied hard limit, +RAM-limit application mode, merge policies, and final segment count. Both RAM +limit paths verify the applied value at runtime. The manifest stores the +canonical request and runtime topology evidence. Reuse validates the evidence +against the request, dataset row count, and final physical segment count, and +reused build/search results surface that same evidence. + +For example, this CAGRA configuration requests one directly built segment with +a 6144 MiB flush threshold: + +```yaml +name: lucene_cuvs_cagra +groups: + one_large_segment: + build: + codec: ["CuVS2510GPUSearchCodec"] + premerge_segment_count: [1] + force_merge_segment_count: [0] + ram_per_thread_hard_limit_mb: [6144] + allow_unsupported_lucene_ram_limit: [true] + search: {} +``` + +Use this opt-in only after sizing the process and device for the complete +segment. The same topology keys are available to +`lucene_accelerated_hnsw`. + ```bash python -m cuvs_bench.run \ --backend lucene \ @@ -40,6 +147,12 @@ python -m cuvs_bench.run \ ## Runtime requirements +Current indexes use manifest schema 4. Rebuild older manifests with +`--build --force`. Accelerated-HNSW index names now include the canonical `m` +and `beam_width` values, even when defaults are used. Old indexes are not +automatically migrated, and differently named indexes are not automatically +deleted or replaced. + The Lucene backend is opt-in because its runtime is not provisioned by the ordinary cuVS Bench installation. Provisioning the required custom PyLucene build is currently external to cuVS Bench. Every algorithm requires PyLucene @@ -61,12 +174,38 @@ classpath inconsistent. The initial backend accepts nonempty, finite, `float32` Euclidean/L2 vectors and supports latency-mode sweeps. The CPU HNSW algorithm accepts at most 1024 dimensions; both cuVS-backed algorithms accept at most 4096. CAGRA uses the -codec's fixed defaults and supports `k <= 1024`. The backend validates the +codec's fixed defaults and supports `k <= 1024`. Accelerated HNSW accepts the +build parameters described above. The backend validates the physical segment codec and every persisted vector field before searching, and fails if CAGRA construction silently produced a brute-force index. Both HNSW algorithms support an explicit `num_candidates` value greater than or equal to `k`; their CPU search path is not subject to CAGRA's `k <= 1024` limit. +### Large-build memory and ingestion + +The initial build path materializes the complete training-vector file as a +NumPy array. It then indexes one document at a time through PyLucene: every row +is converted to a Python list and Java `float[]`, a Lucene `Document` is +created, and `IndexWriter.addDocument` crosses the JCC boundary. This path is +not a streaming, bulk-FBIN, or out-of-core ingestion path. + +Size the Python process and JVM heap for the dataset and codec being tested. +Additional JVM arguments can be supplied through the Lucene backend +configuration, for example: + +```yaml +backend: lucene +jvm_args: + - -Xms16g + - -Xmx64g + - -XX:+ExitOnOutOfMemoryError +``` + +Pass the file with `--backend-config`. These values are illustrative, not +defaults: choose them from the available memory and expected workload. JVM +arguments are immutable after PyLucene initializes the process-global JVM, so +use a new Python process when changing them. + ### Build PyLucene 10.2.0 from source The ordinary cuVS Bench wheel, conda package, and container do not include the @@ -173,6 +312,24 @@ JVM arguments cannot be changed after `lucene.initVM(...)`. ## Timing contract +`build_time_seconds` and `index_build_call_seconds` cover the in-process Lucene +build call: directory open/close, writer setup, document ingestion, an optional +synchronous force merge, writer commit/close, and the post-build reader check. +They exclude dataset loading, index verification, manifest publication, +installation, and size measurement; use +`backend_build_total_seconds` for that complete backend lifecycle. + +In particular, `runtime_document_ingest_seconds` includes NumPy-to-Python and +Python-to-Java conversion, Java object creation, JCC dispatch, and Lucene work +performed by `addDocument`. It is not a measurement of cuvs-lucene or GPU graph +construction alone. `runtime_force_merge_seconds` measures an explicitly +requested synchronous `forceMerge` as a nested sub-timer; it is already included +in the enclosing build-call timings and must not be added to them. +`runtime_writer_commit_close_seconds` includes only work that the selected +codec defers until flush, commit, or close. Absolute build times from this +document-at-a-time path are therefore not directly comparable with a +Java-native or bulk-FBIN benchmark harness. + This initial backend invokes one Lucene query at a time. The common cuVS Bench `--batch-size` value is retained for configuration and result-file compatibility, but it does not introduce bulk or concurrent execution. Results @@ -231,3 +388,29 @@ GPU; missing prerequisites fail rather than skip. Accelerated-HNSW GPU-intended cases fail if the codec logs its CPU-writer fallback. A separate GPU-hidden negative control verifies that this warning remains observable and attributable to the case that produced it. + +The direct PyLucene greater-than-2-GiB cases are independently gated because +they generate a 3 GiB FBIN and build one retained segment by issuing every +document through the Python/JCC boundary. Run them alone, in a fresh process, +on a local filesystem with at least 16 GiB free: + +```bash +python -m pytest -q -s -x \ + python/cuvs_bench/cuvs_bench/tests/test_lucene_large_segment_python_integration.py \ + --run-lucene-large-segment-e2e \ + --basetemp=/path/to/local-disk/lucene-large-pytest +``` + +Run this suite serially, without pytest-xdist. The `-x` option stops after the +first failure while pytest unwinds the fixture and removes its 3 GiB source +file. + +The test configures a 12 GiB JVM maximum heap and requires at least 16 GiB of +currently available host memory. It maps the generated vectors read-only to +avoid a second 3 GiB Python allocation, but still exercises the production +per-document PyLucene conversion and `IndexWriter.addDocument` loop. Both the +CAGRA-built HNSW and GPU CAGRA cases require one physical segment and verify +the explicit 6144 MiB override. The HNSW case rejects logged CPU-writer +fallback; both cases reject brute-force-index fallback and graph-parameter +clamp warnings. The large-suite flag does not select the ordinary +live module, and `--run-lucene-e2e` does not select these capacity cases. diff --git a/fern/pages/lucene_api/index.md b/fern/pages/lucene_api/index.md index 90733c62ec..9863785016 100644 --- a/fern/pages/lucene_api/index.md +++ b/fern/pages/lucene_api/index.md @@ -14,6 +14,8 @@ For an introduction to the codecs, configuration, and tuning, see the [Lucene In - [GPUIndex](/api-reference/lucene-api-com-nvidia-cuvs-lucene-gpuindex) - [GPUSearchParams](/api-reference/lucene-api-com-nvidia-cuvs-lucene-gpusearchparams) - [IndexSearcherTimingBridge](/api-reference/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge) +- [IndexWriterConfigRAMLimitBridge](/api-reference/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge) +- [Lucene101AcceleratedHNSWCodecFactory](/api-reference/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory) - [LuceneProvider](/api-reference/lucene-api-com-nvidia-cuvs-lucene-luceneprovider) - [ThreadLocalCuVSResourcesProvider](/api-reference/lucene-api-com-nvidia-cuvs-lucene-threadlocalcuvsresourcesprovider) - [Utils](/api-reference/lucene-api-com-nvidia-cuvs-lucene-utils) diff --git a/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge.md b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge.md new file mode 100644 index 0000000000..54040183a1 --- /dev/null +++ b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge.md @@ -0,0 +1,32 @@ +--- +slug: api-reference/lucene-api-com-nvidia-cuvs-lucene-indexwriterconfigramlimitbridge +--- + +# IndexWriterConfigRAMLimitBridge + +_Java package: `com.nvidia.cuvs.lucene`_ + +```java +public final class IndexWriterConfigRAMLimitBridge implements Function, Map> +``` + +Applies and verifies Lucene's per-thread indexing-memory limit. + +Lucene 10.2 accepts limits below 2048 MiB through its public setter. Larger limits require an +explicit opt-in and are applied to Lucene's non-public `perThreadHardLimitMB` field. That +unsupported override is intended only for controlled cuVS Bench vector-only builds that disable +automatic merges and validate their final segment topology. It must not be treated as a general +replacement for Lucene's safety limit. + +The standard `Function` and `Map` types provide a narrow bridge for generated Java +bindings that do not expose reflection. The request must contain `config` (an +`IndexWriterConfig`), `per_thread_hard_limit_mb` (a positive `Integer`), and +`allow_unsupported_lucene_ram_limit` (a `Boolean`). The response returns the same +config, the verified limit, and `application_mode`, which is either `public_setter` +or `unsupported_field_override`. + +The non-public path deliberately depends on Lucene's field name and type. It fails if the +field cannot be found, made accessible, written, or read back, so callers never silently continue +with a different limit. + +_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.java:35`_ diff --git a/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory.md b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory.md new file mode 100644 index 0000000000..29cc62c973 --- /dev/null +++ b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory.md @@ -0,0 +1,54 @@ +--- +slug: api-reference/lucene-api-com-nvidia-cuvs-lucene-lucene101acceleratedhnswcodecfactory +--- + +# Lucene101AcceleratedHNSWCodecFactory + +_Java package: `com.nvidia.cuvs.lucene`_ + +```java +public final class Lucene101AcceleratedHNSWCodecFactory implements Function, Map> +``` + +Constructs an accelerated HNSW codec from one self-contained parameter request. + +The standard `Function` and `Map` types form a narrow bridge for generated Java +bindings that do not wrap parameterized constructors. The request has exactly two entries, +max_conn and beam_width. Both must be `Integer` values in the inclusive range 1 through +512. The response contains the configured `codec` and the verified applied values under the +same parameter keys. Invalid requests throw `IllegalArgumentException`; codec construction +or initialization failures throw `IllegalStateException`. + +The factory is stateless. In particular, it does not use JVM system properties, so concurrent +callers cannot observe or overwrite one another's configuration. + +## Public Members + +### apply + +```java +@Override public Map apply(Map request) +``` + +Constructs one codec from the complete request. + +**Parameters** + +| Name | Description | +| --- | --- | +| `request` | exactly the `max_conn` and `beam_width` integer entries | + +**Returns** + +the codec and its verified applied parameter values + +**Throws** + +| Type | Description | +| --- | --- | +| `IllegalArgumentException` | if the request is null, incomplete, has extra keys, contains non-integer values, or contains values outside the inclusive range 1 through 512 | +| `IllegalStateException` | if codec construction or vector-format initialization fails | + +_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.java:42`_ + +_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.java:28`_ diff --git a/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.java b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.java new file mode 100644 index 0000000000..6012cfd58d --- /dev/null +++ b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.java @@ -0,0 +1,145 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import java.lang.reflect.Field; +import java.util.HashMap; +import java.util.Map; +import java.util.Set; +import java.util.function.Function; +import org.apache.lucene.index.IndexWriterConfig; +import org.apache.lucene.index.LiveIndexWriterConfig; + +/** + * Applies and verifies Lucene's per-thread indexing-memory limit. + * + *

Lucene 10.2 accepts limits below 2048 MiB through its public setter. Larger limits require an + * explicit opt-in and are applied to Lucene's non-public {@code perThreadHardLimitMB} field. That + * unsupported override is intended only for controlled cuVS Bench vector-only builds that disable + * automatic merges and validate their final segment topology. It must not be treated as a general + * replacement for Lucene's safety limit. + * + *

The standard {@link Function} and {@link Map} types provide a narrow bridge for generated Java + * bindings that do not expose reflection. The request must contain {@code config} (an + * {@code IndexWriterConfig}), {@code per_thread_hard_limit_mb} (a positive {@link Integer}), and + * {@code allow_unsupported_lucene_ram_limit} (a {@link Boolean}). The response returns the same + * config, the verified limit, and {@code application_mode}, which is either {@code public_setter} + * or {@code unsupported_field_override}. + * + *

The non-public path deliberately depends on Lucene's field name and type. It fails if the + * field cannot be found, made accessible, written, or read back, so callers never silently continue + * with a different limit. + */ +public final class IndexWriterConfigRAMLimitBridge + implements Function, Map> { + public static final String CONFIG_KEY = "config"; + public static final String PER_THREAD_HARD_LIMIT_MB_KEY = "per_thread_hard_limit_mb"; + public static final String ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY = + "allow_unsupported_lucene_ram_limit"; + public static final String APPLICATION_MODE_KEY = "application_mode"; + public static final String PUBLIC_SETTER_MODE = "public_setter"; + public static final String UNSUPPORTED_FIELD_OVERRIDE_MODE = "unsupported_field_override"; + + private static final int FIRST_UNSUPPORTED_LIMIT_MB = 2048; + private static final String PER_THREAD_HARD_LIMIT_MB_FIELD = "perThreadHardLimitMB"; + private static final Set REQUEST_KEYS = + Set.of(CONFIG_KEY, PER_THREAD_HARD_LIMIT_MB_KEY, ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY); + + @Override + public Map apply(Map request) { + validateRequestKeys(request); + IndexWriterConfig config = requiredValue(request, CONFIG_KEY, IndexWriterConfig.class); + Integer requestedLimit = requiredValue(request, PER_THREAD_HARD_LIMIT_MB_KEY, Integer.class); + Boolean allowUnsupported = + requiredValue(request, ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY, Boolean.class); + + String applicationMode = applyAndVerify(config, requestedLimit, allowUnsupported); + + Map response = new HashMap<>(); + response.put(CONFIG_KEY, config); + response.put(PER_THREAD_HARD_LIMIT_MB_KEY, config.getRAMPerThreadHardLimitMB()); + response.put(APPLICATION_MODE_KEY, applicationMode); + return response; + } + + static String applyAndVerify( + IndexWriterConfig config, int requestedLimit, boolean allowUnsupported) { + if (config == null) { + throw new IllegalArgumentException(CONFIG_KEY + " must not be null"); + } + if (requestedLimit <= 0) { + throw new IllegalArgumentException(PER_THREAD_HARD_LIMIT_MB_KEY + " must be positive"); + } + + String applicationMode; + if (requestedLimit < FIRST_UNSUPPORTED_LIMIT_MB) { + config.setRAMPerThreadHardLimitMB(requestedLimit); + applicationMode = PUBLIC_SETTER_MODE; + } else { + if (!allowUnsupported) { + throw new IllegalArgumentException( + PER_THREAD_HARD_LIMIT_MB_KEY + + " values of 2048 MiB or greater require " + + ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY + + "=true"); + } + setUnsupportedLimit(config, requestedLimit); + applicationMode = UNSUPPORTED_FIELD_OVERRIDE_MODE; + } + + int actualLimit = config.getRAMPerThreadHardLimitMB(); + if (actualLimit != requestedLimit) { + throw new IllegalStateException( + "Failed to verify LiveIndexWriterConfig." + + PER_THREAD_HARD_LIMIT_MB_FIELD + + ": requested " + + requestedLimit + + " but read " + + actualLimit); + } + return applicationMode; + } + + private static void setUnsupportedLimit(LiveIndexWriterConfig config, int requestedLimit) { + Field limitField; + try { + limitField = LiveIndexWriterConfig.class.getDeclaredField(PER_THREAD_HARD_LIMIT_MB_FIELD); + } catch (NoSuchFieldException | SecurityException error) { + throw new IllegalStateException( + "Unable to locate LiveIndexWriterConfig." + PER_THREAD_HARD_LIMIT_MB_FIELD, error); + } + + if (limitField.getType() != int.class) { + throw new IllegalStateException( + "LiveIndexWriterConfig." + PER_THREAD_HARD_LIMIT_MB_FIELD + " is not an int field"); + } + + try { + limitField.setAccessible(true); + limitField.setInt(config, requestedLimit); + } catch (IllegalAccessException | RuntimeException error) { + throw new IllegalStateException( + "Unable to set LiveIndexWriterConfig." + PER_THREAD_HARD_LIMIT_MB_FIELD, error); + } + } + + private static void validateRequestKeys(Map request) { + if (request == null) { + throw new IllegalArgumentException("request must not be null"); + } + if (!request.keySet().equals(REQUEST_KEYS)) { + throw new IllegalArgumentException("request must contain exactly " + REQUEST_KEYS); + } + } + + private static T requiredValue( + Map values, String key, Class expectedType) { + Object value = values.get(key); + if (!expectedType.isInstance(value)) { + throw new IllegalArgumentException(key + " must have type " + expectedType.getSimpleName()); + } + return expectedType.cast(value); + } +} diff --git a/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.java b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.java new file mode 100644 index 0000000000..10093b7547 --- /dev/null +++ b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.java @@ -0,0 +1,107 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import com.nvidia.cuvs.CagraIndexParams.HnswHeuristicType; +import java.util.HashMap; +import java.util.Map; +import java.util.function.Function; +import org.apache.lucene.codecs.Codec; + +/** + * Constructs an accelerated HNSW codec from one self-contained parameter request. + * + *

The standard {@link Function} and {@link Map} types form a narrow bridge for generated Java + * bindings that do not wrap parameterized constructors. The request has exactly two entries, + * max_conn and beam_width. Both must be {@link Integer} values in the inclusive range 1 through + * 512. The response contains the configured {@code codec} and the verified applied values under the + * same parameter keys. Invalid requests throw {@link IllegalArgumentException}; codec construction + * or initialization failures throw {@link IllegalStateException}. + * + *

The factory is stateless. In particular, it does not use JVM system properties, so concurrent + * callers cannot observe or overwrite one another's configuration. + * + * @since 26.12 + */ +public final class Lucene101AcceleratedHNSWCodecFactory + implements Function, Map> { + public static final String MAX_CONN_KEY = "max_conn"; + public static final String BEAM_WIDTH_KEY = "beam_width"; + public static final String CODEC_KEY = "codec"; + + /** + * Constructs one codec from the complete request. + * + * @param request exactly the {@code max_conn} and {@code beam_width} integer entries + * @return the codec and its verified applied parameter values + * @throws IllegalArgumentException if the request is null, incomplete, has extra keys, contains + * non-integer values, or contains values outside the inclusive range 1 through 512 + * @throws IllegalStateException if codec construction or vector-format initialization fails + */ + @Override + public Map apply(Map request) { + int maxConn = + requiredInteger( + request, + MAX_CONN_KEY, + AcceleratedHNSWParams.MIN_MAX_CONN, + AcceleratedHNSWParams.MAX_MAX_CONN); + int beamWidth = + requiredInteger( + request, + BEAM_WIDTH_KEY, + AcceleratedHNSWParams.MIN_BEAM_WIDTH, + AcceleratedHNSWParams.MAX_BEAM_WIDTH); + if (request.size() != 2) { + throw new IllegalArgumentException( + "request must contain only " + MAX_CONN_KEY + " and " + BEAM_WIDTH_KEY); + } + AcceleratedHNSWParams parameters = createParameters(maxConn, beamWidth); + + final Codec codec; + try { + codec = new Lucene101AcceleratedHNSWCodec(parameters); + if (codec.knnVectorsFormat() == null) { + throw new IllegalStateException( + "Accelerated HNSW codec did not initialize a vector format"); + } + } catch (IllegalStateException error) { + throw error; + } catch (Exception | LinkageError error) { + throw new IllegalStateException("Could not construct the accelerated HNSW codec", error); + } + + Map response = new HashMap<>(); + response.put(CODEC_KEY, codec); + response.put(MAX_CONN_KEY, parameters.getMaxConn()); + response.put(BEAM_WIDTH_KEY, parameters.getBeamWidth()); + return response; + } + + static AcceleratedHNSWParams createParameters(int maxConn, int beamWidth) { + return new AcceleratedHNSWParams.Builder() + .withStrategy(AcceleratedHNSWParams.Strategy.HEURISTIC) + .withHnswHeuristicType(HnswHeuristicType.SAME_GRAPH_FOOTPRINT) + .withMaxConn(maxConn) + .withBeamWidth(beamWidth) + .build(); + } + + private static int requiredInteger( + Map values, String key, int minimum, int maximum) { + if (values == null) { + throw new IllegalArgumentException("request must not be null"); + } + Object value = values.get(key); + if (!(value instanceof Integer integer)) { + throw new IllegalArgumentException(key + " must have type Integer"); + } + if (integer < minimum || integer > maximum) { + throw new IllegalArgumentException( + key + " must be in range [" + minimum + ", " + maximum + "], but was: " + integer); + } + return integer; + } +} diff --git a/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexWriterConfigRAMLimitBridge.java b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexWriterConfigRAMLimitBridge.java new file mode 100644 index 0000000000..cd2c5ef905 --- /dev/null +++ b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexWriterConfigRAMLimitBridge.java @@ -0,0 +1,122 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertSame; +import static org.junit.Assert.assertThrows; + +import java.util.HashMap; +import java.util.Map; +import java.util.function.Function; +import org.apache.lucene.index.IndexWriterConfig; +import org.junit.Test; + +public class TestIndexWriterConfigRAMLimitBridge { + + @Test + public void testUsesPublicSetterForSupportedLimit() { + var config = new IndexWriterConfig(); + + Map response = bridge().apply(request(config, 2047, false)); + + assertApplied(response, config, 2047, IndexWriterConfigRAMLimitBridge.PUBLIC_SETTER_MODE); + } + + @Test + public void testExplicitOptInAppliesLimitAbovePublicSetterCap() { + var config = new IndexWriterConfig(); + assertThrows(IllegalArgumentException.class, () -> config.setRAMPerThreadHardLimitMB(6144)); + + Map response = bridge().apply(request(config, 6144, true)); + + assertApplied( + response, config, 6144, IndexWriterConfigRAMLimitBridge.UNSUPPORTED_FIELD_OVERRIDE_MODE); + } + + @Test + public void testRejectsUnsupportedLimitWithoutOptIn() { + var config = new IndexWriterConfig(); + int originalLimit = config.getRAMPerThreadHardLimitMB(); + + IllegalArgumentException error = + assertThrows( + IllegalArgumentException.class, () -> bridge().apply(request(config, 2048, false))); + + assertEquals( + "per_thread_hard_limit_mb values of 2048 MiB or greater require " + + "allow_unsupported_lucene_ram_limit=true", + error.getMessage()); + assertEquals(originalLimit, config.getRAMPerThreadHardLimitMB()); + } + + @Test + public void testRejectsIncompleteOrUnexpectedRequest() { + var request = request(new IndexWriterConfig(), 2047, false); + request.remove(IndexWriterConfigRAMLimitBridge.CONFIG_KEY); + assertInvalidRequest(request, "request must contain exactly"); + + request = request(new IndexWriterConfig(), 2047, false); + request.put("unexpected", true); + assertInvalidRequest(request, "request must contain exactly"); + } + + @Test + public void testRejectsIncorrectRequestTypes() { + var wrongConfig = request(new IndexWriterConfig(), 2047, false); + wrongConfig.put(IndexWriterConfigRAMLimitBridge.CONFIG_KEY, new Object()); + assertInvalidRequest(wrongConfig, "config must have type IndexWriterConfig"); + + var wrongLimit = request(new IndexWriterConfig(), 2047, false); + wrongLimit.put(IndexWriterConfigRAMLimitBridge.PER_THREAD_HARD_LIMIT_MB_KEY, 6144L); + assertInvalidRequest(wrongLimit, "per_thread_hard_limit_mb must have type Integer"); + + var wrongOptIn = request(new IndexWriterConfig(), 2047, false); + wrongOptIn.put(IndexWriterConfigRAMLimitBridge.ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY, "true"); + assertInvalidRequest(wrongOptIn, "allow_unsupported_lucene_ram_limit must have type Boolean"); + } + + @Test + public void testRejectsNonPositiveLimitWithoutChangingConfig() { + var config = new IndexWriterConfig(); + int originalLimit = config.getRAMPerThreadHardLimitMB(); + + assertInvalidRequest(request(config, 0, false), "per_thread_hard_limit_mb must be positive"); + + assertEquals(originalLimit, config.getRAMPerThreadHardLimitMB()); + } + + private static Function, Map> bridge() { + return new IndexWriterConfigRAMLimitBridge(); + } + + private static Map request( + IndexWriterConfig config, int limit, boolean allowUnsupported) { + var request = new HashMap(); + request.put(IndexWriterConfigRAMLimitBridge.CONFIG_KEY, config); + request.put(IndexWriterConfigRAMLimitBridge.PER_THREAD_HARD_LIMIT_MB_KEY, limit); + request.put( + IndexWriterConfigRAMLimitBridge.ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT_KEY, allowUnsupported); + return request; + } + + private static void assertApplied( + Map response, + IndexWriterConfig config, + int expectedLimit, + String expectedMode) { + assertEquals(expectedLimit, config.getRAMPerThreadHardLimitMB()); + assertSame(config, response.get(IndexWriterConfigRAMLimitBridge.CONFIG_KEY)); + assertEquals( + expectedLimit, response.get(IndexWriterConfigRAMLimitBridge.PER_THREAD_HARD_LIMIT_MB_KEY)); + assertEquals(expectedMode, response.get(IndexWriterConfigRAMLimitBridge.APPLICATION_MODE_KEY)); + } + + private static void assertInvalidRequest(Map request, String expectedMessage) { + IllegalArgumentException error = + assertThrows(IllegalArgumentException.class, () -> bridge().apply(request)); + org.junit.Assert.assertTrue(error.getMessage(), error.getMessage().contains(expectedMessage)); + } +} diff --git a/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestLucene101AcceleratedHNSWCodecFactory.java b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestLucene101AcceleratedHNSWCodecFactory.java new file mode 100644 index 0000000000..405b5f493e --- /dev/null +++ b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestLucene101AcceleratedHNSWCodecFactory.java @@ -0,0 +1,153 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import static com.nvidia.cuvs.lucene.Lucene101AcceleratedHNSWCodecFactory.BEAM_WIDTH_KEY; +import static com.nvidia.cuvs.lucene.Lucene101AcceleratedHNSWCodecFactory.CODEC_KEY; +import static com.nvidia.cuvs.lucene.Lucene101AcceleratedHNSWCodecFactory.MAX_CONN_KEY; +import static com.nvidia.cuvs.lucene.ThreadLocalCuVSResourcesProvider.isSupported; + +import com.nvidia.cuvs.CagraIndexParams; +import com.nvidia.cuvs.CagraIndexParams.CagraGraphBuildAlgo; +import com.nvidia.cuvs.CagraIndexParams.HnswHeuristicType; +import java.lang.reflect.Modifier; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.concurrent.Executors; +import java.util.concurrent.Future; +import java.util.concurrent.TimeUnit; +import java.util.function.Function; +import org.apache.lucene.codecs.Codec; +import org.apache.lucene.tests.util.LuceneTestCase; +import org.junit.Test; + +/** Tests the binding-compatible accelerated-HNSW codec factory contract. */ +public class TestLucene101AcceleratedHNSWCodecFactory extends LuceneTestCase { + + private static final String ACCELERATED_CODEC_NAME = "Lucene101AcceleratedHNSWCodec"; + private static final String ACCELERATED_FORMAT_NAME = "Lucene99AcceleratedHNSWVectorsFormat"; + + @Test + public void testCreatesCodecAndReportsAppliedConfiguration() throws Exception { + var factory = new Lucene101AcceleratedHNSWCodecFactory(); + + Map response = factory.apply(request(16, 80)); + + Codec codec = (Codec) response.get(CODEC_KEY); + assertTrue(Modifier.isFinal(factory.getClass().getModifiers())); + assertTrue(Modifier.isPublic(factory.getClass().getConstructors()[0].getModifiers())); + assertEquals(16, response.get(MAX_CONN_KEY)); + assertEquals(80, response.get(BEAM_WIDTH_KEY)); + assertEquals(ACCELERATED_CODEC_NAME, codec.getName()); + assertNotNull(codec.knnVectorsFormat()); + assertEquals(ACCELERATED_FORMAT_NAME, codec.knnVectorsFormat().getName()); + assertCodecParameters(codec, 16, 80); + assertEquals( + Lucene101AcceleratedHNSWCodec.class, Codec.forName(ACCELERATED_CODEC_NAME).getClass()); + } + + @Test + public void testCreatesExpectedCagraBuildParameters() { + assumeTrue("cuVS not supported", isSupported()); + AcceleratedHNSWParams parameters = + Lucene101AcceleratedHNSWCodecFactory.createParameters(16, 80); + + CagraIndexParams cagraParameters = CagraIndexParamsFactory.create(parameters, 10_000_000, 96); + + assertEquals(AcceleratedHNSWParams.Strategy.HEURISTIC, parameters.getStrategy()); + assertEquals(HnswHeuristicType.SAME_GRAPH_FOOTPRINT, parameters.getHnswHeuristicType()); + assertEquals(AcceleratedHNSWParams.DEFAULT_WRITER_THREADS, parameters.getWriterThreads()); + assertEquals(32, cagraParameters.getGraphDegree()); + assertEquals(48, cagraParameters.getIntermediateGraphDegree()); + assertEquals( + AcceleratedHNSWParams.DEFAULT_WRITER_THREADS, cagraParameters.getNumWriterThreads()); + assertEquals(CagraGraphBuildAlgo.IVF_PQ, cagraParameters.getCagraGraphBuildAlgo()); + } + + @Test + public void testRejectsMissingWrongTypeAndOutOfRangeParameters() { + var factory = new Lucene101AcceleratedHNSWCodecFactory(); + + assertInvalidRequest(factory, null, "request must not be null"); + assertInvalidRequest(factory, new HashMap<>(), "max_conn must have type Integer"); + + var wrongType = new HashMap(); + wrongType.put(MAX_CONN_KEY, 16L); + wrongType.put(BEAM_WIDTH_KEY, 80); + assertInvalidRequest(factory, wrongType, "max_conn must have type Integer"); + + var unexpected = request(16, 80); + unexpected.put("writer_threads", 2); + assertInvalidRequest(factory, unexpected, "request must contain only"); + + assertInvalidRequest( + factory, request(AcceleratedHNSWParams.MIN_MAX_CONN - 1, 80), "max_conn must be in range"); + assertInvalidRequest( + factory, + request(16, AcceleratedHNSWParams.MAX_BEAM_WIDTH + 1), + "beam_width must be in range"); + } + + @Test + public void testConcurrentRequestsKeepIndependentParameterPairs() throws Exception { + Function, Map> factory = + new Lucene101AcceleratedHNSWCodecFactory(); + var executor = Executors.newFixedThreadPool(4); + try { + List>> responses = new ArrayList<>(); + for (int i = 0; i < 32; i++) { + int maxConn = i % 2 == 0 ? 16 : 24; + int beamWidth = i % 2 == 0 ? 80 : 96; + responses.add(executor.submit(() -> factory.apply(request(maxConn, beamWidth)))); + } + + for (int i = 0; i < responses.size(); i++) { + Map response = responses.get(i).get(); + assertEquals(i % 2 == 0 ? 16 : 24, response.get(MAX_CONN_KEY)); + assertEquals(i % 2 == 0 ? 80 : 96, response.get(BEAM_WIDTH_KEY)); + assertEquals(ACCELERATED_CODEC_NAME, ((Codec) response.get(CODEC_KEY)).getName()); + assertCodecParameters( + (Codec) response.get(CODEC_KEY), i % 2 == 0 ? 16 : 24, i % 2 == 0 ? 80 : 96); + } + } finally { + executor.shutdownNow(); + assertTrue( + "codec-factory executor did not terminate", + executor.awaitTermination(10, TimeUnit.SECONDS)); + } + } + + private static void assertCodecParameters(Codec codec, int maxConn, int beamWidth) + throws ReflectiveOperationException { + // Inspect the returned format, not the response map: echoed input cannot prove configuration. + // Keep this reflection test-only rather than adding a production accessor for the test. + var format = codec.knnVectorsFormat(); + assertTrue(format instanceof Lucene99AcceleratedHNSWVectorsFormat); + var parametersField = + Lucene99AcceleratedHNSWVectorsFormat.class.getDeclaredField("acceleratedHNSWParams"); + parametersField.setAccessible(true); + var parameters = (AcceleratedHNSWParams) parametersField.get(format); + assertEquals(maxConn, parameters.getMaxConn()); + assertEquals(beamWidth, parameters.getBeamWidth()); + } + + private static Map request(int maxConn, int beamWidth) { + var request = new HashMap(); + request.put(MAX_CONN_KEY, maxConn); + request.put(BEAM_WIDTH_KEY, beamWidth); + return request; + } + + private static void assertInvalidRequest( + Lucene101AcceleratedHNSWCodecFactory factory, + Map request, + String expectedMessage) { + IllegalArgumentException error = + expectThrows(IllegalArgumentException.class, () -> factory.apply(request)); + assertTrue(error.getMessage().contains(expectedMessage)); + } +} diff --git a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py index 65bd18ec9a..4aa73caa85 100644 --- a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py +++ b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py @@ -24,6 +24,23 @@ CPU_HNSW_CODEC = "Lucene101" ACCELERATED_HNSW_CODEC = "Lucene101AcceleratedHNSWCodec" CAGRA_CODEC = "CuVS2510GPUSearchCodec" +CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY = ( + "com.nvidia.cuvs.lucene.Lucene101AcceleratedHNSWCodecFactory" +) +INDEX_WRITER_CONFIG_RAM_LIMIT_BRIDGE = ( + "com.nvidia.cuvs.lucene.IndexWriterConfigRAMLimitBridge" +) +_CONFIGURED_CODEC_RESPONSE_KEY = "codec" +_HNSW_MAX_CONN_REQUEST_KEY = "max_conn" +_HNSW_BEAM_WIDTH_REQUEST_KEY = "beam_width" +_MAX_RAM_PER_THREAD_HARD_LIMIT_MB = 2_147_483_647 +_FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB = 2048 +_RAM_LIMIT_CONFIG_KEY = "config" +_RAM_LIMIT_VALUE_KEY = "per_thread_hard_limit_mb" +_RAM_LIMIT_ALLOW_UNSUPPORTED_KEY = "allow_unsupported_lucene_ram_limit" +_RAM_LIMIT_APPLICATION_MODE_KEY = "application_mode" +_PUBLIC_RAM_LIMIT_SETTER_MODE = "public_setter" +_UNSUPPORTED_RAM_LIMIT_OVERRIDE_MODE = "unsupported_field_override" MAX_CAGRA_TOP_K = 1024 REQUIRED_PYLUCENE_VERSION = "10.2.0" _PYLUCENE_SETUP_GUIDANCE = ( @@ -276,6 +293,8 @@ def _validate_artifacts( "com/nvidia/cuvs/lucene/CuVS2510GPUVectorsFormat.class", "com/nvidia/cuvs/lucene/CuVS2510GPUSearchCodec.class", "com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.class", + "com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.class", + "com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodecFactory.class", "com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodec.class", "META-INF/services/org.apache.lucene.codecs.Codec", } @@ -945,6 +964,7 @@ class RuntimeBuildTiming: directory_open_ns: int writer_setup_ns: int document_ingest_ns: int + force_merge_ns: int writer_commit_close_ns: int post_build_reader_ns: int directory_close_ns: int @@ -955,6 +975,166 @@ class RuntimeBuildTiming: class RuntimeBuildResult: segment_count: int timing: RuntimeBuildTiming + topology: "RuntimeBuildTopology | None" = None + + +@dataclass(frozen=True) +class RuntimeBuildTopology: + """Requested and observed topology for a controlled Lucene build.""" + + requested_premerge_segment_count: int + observed_premerge_segment_count: int + requested_force_merge_segment_count: int + premerge_segment_vector_counts: tuple[int, ...] + max_buffered_docs: int + applied_ram_per_thread_hard_limit_mb: int + ram_per_thread_hard_limit_application: str + ingest_merge_policy: str + final_merge_policy: str | None + + def manifest(self) -> dict[str, Any]: + """Return structured, JSON-safe evidence for index reuse.""" + return { + "requested_premerge_segment_count": self.requested_premerge_segment_count, + "observed_premerge_segment_count": ( + self.observed_premerge_segment_count + ), + "requested_force_merge_segment_count": ( + self.requested_force_merge_segment_count + ), + "premerge_segment_vector_counts": list( + self.premerge_segment_vector_counts + ), + "max_buffered_docs": self.max_buffered_docs, + "applied_ram_per_thread_hard_limit_mb": ( + self.applied_ram_per_thread_hard_limit_mb + ), + "ram_per_thread_hard_limit_application": ( + self.ram_per_thread_hard_limit_application + ), + "ingest_merge_policy": self.ingest_merge_policy, + "final_merge_policy": self.final_merge_policy, + } + + +@dataclass(frozen=True) +class _IndexBuildPhase: + """Measured result of one Lucene index-construction strategy.""" + + writer_setup_ns: int + document_ingest_ns: int + force_merge_ns: int + writer_commit_close_ns: int + post_build_reader_ns: int + segment_count: int + topology: RuntimeBuildTopology | None = None + + +@dataclass(frozen=True) +class _ControlledBuildTopology: + premerge_segment_count: int + force_merge_segment_count: int + ram_per_thread_hard_limit_mb: int + allow_unsupported_lucene_ram_limit: bool + chunk_size: int + max_buffered_docs: int + + +def _controlled_build_topology( + build_parameters: Mapping[str, Any] | None, vector_count: int +) -> _ControlledBuildTopology | None: + """Validate the fail-closed topology contract at the JVM boundary.""" + required = { + "premerge_segment_count", + "force_merge_segment_count", + "ram_per_thread_hard_limit_mb", + } + allow_key = "allow_unsupported_lucene_ram_limit" + parameters = build_parameters or {} + present = (required | {allow_key}) & set(parameters) + if not present: + return None + missing_required = required - set(parameters) + if missing_required: + missing = ", ".join(sorted(missing_required)) + raise RuntimeError( + "Controlled Lucene builds require all topology parameters; " + f"missing: {missing}" + ) + values = {name: parameters[name] for name in required} + for name in ("premerge_segment_count", "ram_per_thread_hard_limit_mb"): + value = values[name] + if type(value) is not int or value < 1: + raise RuntimeError( + f"Controlled Lucene build parameter {name} must be a " + f"positive integer, got {value!r}" + ) + premerge = int(values["premerge_segment_count"]) + hard_limit = int(values["ram_per_thread_hard_limit_mb"]) + if hard_limit > _MAX_RAM_PER_THREAD_HARD_LIMIT_MB: + raise RuntimeError( + "Controlled Lucene build parameter " + "ram_per_thread_hard_limit_mb must be in range [1, " + f"{_MAX_RAM_PER_THREAD_HARD_LIMIT_MB}], got {hard_limit}" + ) + allow_unsupported = parameters.get(allow_key, False) + if type(allow_unsupported) is not bool: + raise RuntimeError( + "Controlled Lucene build parameter " + "allow_unsupported_lucene_ram_limit must be a boolean" + ) + force_merge_value = values["force_merge_segment_count"] + if type(force_merge_value) is not int or force_merge_value not in (0, 1): + raise RuntimeError( + "Controlled Lucene build parameter force_merge_segment_count " + "must be 0 (disabled) or 1" + ) + force_merge = force_merge_value + uses_unsupported_limit = ( + hard_limit >= _FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB + ) + if uses_unsupported_limit and not allow_unsupported: + raise RuntimeError( + "Controlled Lucene ram_per_thread_hard_limit_mb values of 2048 " + "MiB or greater require " + "allow_unsupported_lucene_ram_limit=true" + ) + if allow_unsupported and not uses_unsupported_limit: + raise RuntimeError( + "Controlled Lucene allow_unsupported_lucene_ram_limit=true " + "requires ram_per_thread_hard_limit_mb >= 2048" + ) + if uses_unsupported_limit and force_merge != 0: + raise RuntimeError( + "Controlled Lucene force_merge_segment_count must be 0 when " + "using an unsupported per-thread RAM hard limit" + ) + if premerge > vector_count: + raise RuntimeError( + "premerge_segment_count cannot exceed the vector count: " + f"{premerge} > {vector_count}" + ) + if vector_count % premerge: + raise RuntimeError( + "Controlled Lucene builds require equal partitions: vector count " + f"{vector_count} is not divisible by premerge_segment_count " + f"{premerge}" + ) + chunk_size = vector_count // premerge + max_buffered_docs = chunk_size + 1 + if max_buffered_docs > 2_147_483_647: + raise RuntimeError( + "Controlled Lucene segment exceeds IndexWriter's integer " + f"maxBufferedDocs range: {max_buffered_docs}" + ) + return _ControlledBuildTopology( + premerge_segment_count=premerge, + force_merge_segment_count=force_merge, + ram_per_thread_hard_limit_mb=hard_limit, + allow_unsupported_lucene_ram_limit=allow_unsupported, + chunk_size=chunk_size, + max_buffered_docs=max_buffered_docs, + ) @dataclass(frozen=True) @@ -994,7 +1174,7 @@ class LuceneRuntime: """Own the generated bindings and the narrow Lucene operations Bench uses.""" def __init__(self, lucene: Any): - from java.lang import Class, Integer, Long + from java.lang import Boolean, Class, Integer, Long, String from java.nio.file import Paths from java.util import HashMap, Map from java.util.function import Function @@ -1009,8 +1189,11 @@ def __init__(self, lucene: Any): FieldInfo, IndexWriter, IndexWriterConfig, + NoMergePolicy, SegmentCommitInfo, SegmentInfos, + SerialMergeScheduler, + TieredMergePolicy, VectorEncoding, VectorSimilarityFunction, ) @@ -1022,9 +1205,11 @@ def __init__(self, lucene: Any): from org.apache.lucene.store import FSDirectory, IOContext self.lucene = lucene + self.Boolean = Boolean self.Class = Class self.Integer = Integer self.Long = Long + self.String = String self.HashMap = HashMap self.Map = Map self.Function = Function @@ -1038,8 +1223,11 @@ def __init__(self, lucene: Any): self.FieldInfo = FieldInfo self.IndexWriter = IndexWriter self.IndexWriterConfig = IndexWriterConfig + self.NoMergePolicy = NoMergePolicy self.SegmentCommitInfo = SegmentCommitInfo self.SegmentInfos = SegmentInfos + self.SerialMergeScheduler = SerialMergeScheduler + self.TieredMergePolicy = TieredMergePolicy self.VectorEncoding = VectorEncoding self.VectorSimilarityFunction = VectorSimilarityFunction self.IndexSearcher = IndexSearcher @@ -1053,6 +1241,8 @@ def __init__(self, lucene: Any): self.artifact_provenance: dict[str, str] = {} self._artifact_tokens: dict[str, tuple[int, ...]] = {} self._java_search_timer: Any | None = None + self._java_ram_limit_bridge: Any | None = None + self._java_configured_codec_factory: Any | None = None @classmethod def create(cls, config: Mapping[str, Any]) -> "LuceneRuntime": @@ -1080,6 +1270,127 @@ def _load_java_search_timer(self) -> Any: f"{type(error).__name__}: {error}" ) from error + def _load_java_configured_codec_factory(self) -> Any: + """Load the codec factory through JCC's wrapped Function interface.""" + try: + instance = self.Class.forName( + CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY + ).newInstance() + return self.Function.cast_(instance) + except Exception as error: + raise RuntimeError( + "Could not load or adapt " + f"{CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY} " + "through PyLucene/JCC: " + f"{type(error).__name__}: {error}" + ) from error + + def _load_java_ram_limit_bridge(self) -> Any: + """Load the RAM-limit bridge through JCC's wrapped Function.""" + try: + instance = self.Class.forName( + INDEX_WRITER_CONFIG_RAM_LIMIT_BRIDGE + ).newInstance() + return self.Function.cast_(instance) + except Exception as error: + raise RuntimeError( + "Could not load or adapt " + f"{INDEX_WRITER_CONFIG_RAM_LIMIT_BRIDGE} through " + "PyLucene/JCC: " + f"{type(error).__name__}: {error}" + ) from error + + def _set_ram_per_thread_hard_limit_mb( + self, + config: Any, + requested_mb: int, + *, + allow_unsupported: bool, + ) -> str: + if type(requested_mb) is not int or not ( + 1 <= requested_mb <= _MAX_RAM_PER_THREAD_HARD_LIMIT_MB + ): + raise RuntimeError( + "Lucene per-thread RAM hard limit must be in range [1, " + f"{_MAX_RAM_PER_THREAD_HARD_LIMIT_MB}], got {requested_mb}" + ) + if type(allow_unsupported) is not bool: + raise RuntimeError( + "Lucene allow_unsupported_lucene_ram_limit must be a boolean" + ) + if requested_mb >= _FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB: + return self._set_unsupported_ram_per_thread_hard_limit_mb( + config, requested_mb, allow_unsupported + ) + try: + config.setRAMPerThreadHardLimitMB(requested_mb) + except Exception as error: + raise RuntimeError( + "Could not apply Lucene's supported per-thread RAM hard " + "limit: " + f"{type(error).__name__}: {error}" + ) from error + observed = int(config.getRAMPerThreadHardLimitMB()) + if observed != requested_mb: + raise RuntimeError( + "Lucene per-thread RAM hard limit did not retain the " + f"requested value: requested {requested_mb}, config " + f"reported {observed}" + ) + return _PUBLIC_RAM_LIMIT_SETTER_MODE + + def _set_unsupported_ram_per_thread_hard_limit_mb( + self, config: Any, requested_mb: int, allow_unsupported: bool + ) -> str: + """Apply an explicitly authorized limit through the thin-JAR bridge.""" + if not allow_unsupported: + raise RuntimeError( + "Lucene per-thread RAM hard limits of 2048 MiB or greater " + "require allow_unsupported_lucene_ram_limit=true" + ) + bridge = self._java_ram_limit_bridge + if bridge is None: + bridge = self._load_java_ram_limit_bridge() + self._java_ram_limit_bridge = bridge + request = self.HashMap() + request.put(_RAM_LIMIT_CONFIG_KEY, config) + request.put(_RAM_LIMIT_VALUE_KEY, self.Integer.valueOf(requested_mb)) + request.put( + _RAM_LIMIT_ALLOW_UNSUPPORTED_KEY, + self.Boolean.valueOf(allow_unsupported), + ) + try: + response = self.Map.cast_(bridge.apply(request)) + applied = int( + self.Integer.cast_( + response.get(_RAM_LIMIT_VALUE_KEY) + ).intValue() + ) + application_mode = str( + self.String.cast_( + response.get(_RAM_LIMIT_APPLICATION_MODE_KEY) + ) + ) + except Exception as error: + raise RuntimeError( + "Could not apply the explicitly authorized unsupported " + "Lucene per-thread RAM hard limit: " + f"{type(error).__name__}: {error}" + ) from error + observed = int(config.getRAMPerThreadHardLimitMB()) + if applied != requested_mb or observed != requested_mb: + raise RuntimeError( + "Lucene per-thread RAM hard-limit bridge did not retain the " + f"requested value: requested {requested_mb}, bridge returned " + f"{applied}, config reported {observed}" + ) + if application_mode != _UNSUPPORTED_RAM_LIMIT_OVERRIDE_MODE: + raise RuntimeError( + "Lucene per-thread RAM hard-limit bridge returned an " + f"unexpected application mode: {application_mode!r}" + ) + return application_mode + @property def pylucene_version(self) -> str: return str(self.lucene.VERSION) @@ -1117,6 +1428,62 @@ def resolve_codec(self, name: str) -> Any: self._codecs[name] = codec return codec + def resolve_configured_hnsw_codec( + self, max_conn: int, beam_width: int + ) -> Any: + """Construct an accelerated-HNSW codec from one atomic request.""" + self.attach_current_thread() + factory = self._java_configured_codec_factory + if factory is None: + factory = self._load_java_configured_codec_factory() + self._java_configured_codec_factory = factory + request = self.HashMap() + request.put(_HNSW_MAX_CONN_REQUEST_KEY, self.Integer.valueOf(max_conn)) + request.put( + _HNSW_BEAM_WIDTH_REQUEST_KEY, self.Integer.valueOf(beam_width) + ) + try: + raw_response = factory.apply(request) + response = self.Map.cast_(raw_response) + codec = self.Codec.cast_( + response.get(_CONFIGURED_CODEC_RESPONSE_KEY) + ) + applied_max_conn = int( + self.Integer.cast_( + response.get(_HNSW_MAX_CONN_REQUEST_KEY) + ).intValue() + ) + applied_beam_width = int( + self.Integer.cast_( + response.get(_HNSW_BEAM_WIDTH_REQUEST_KEY) + ).intValue() + ) + except Exception as error: + raise RuntimeError( + "Could not construct a configured accelerated-HNSW codec " + f"through {CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY}: " + f"{type(error).__name__}: {error}" + ) from error + + self._validate_codec(codec, ACCELERATED_HNSW_CODEC) + if (applied_max_conn, applied_beam_width) != (max_conn, beam_width): + raise RuntimeError( + "Configured accelerated-HNSW codec did not retain the " + "requested parameters: requested " + f"({max_conn}, {beam_width}), applied " + f"({applied_max_conn}, {applied_beam_width})" + ) + return codec + + @staticmethod + def _validate_codec(codec: Any, name: str) -> None: + if str(codec.getName()) != name: + raise RuntimeError( + f"Requested codec {name}, resolved {codec.getName()}" + ) + if codec.knnVectorsFormat() is None: + raise RuntimeError(f"{name} did not initialize a vector format") + def _java_vector(self, vector: np.ndarray) -> Any: return self.lucene.JArray("float")(vector.tolist()) @@ -1132,20 +1499,362 @@ def _document(self, document_id: int, vector: np.ndarray) -> Any: ) return document + def _resolve_build_codec( + self, + codec_name: str, + build_parameters: Mapping[str, Any] | None, + ) -> Any: + if codec_name != ACCELERATED_HNSW_CODEC: + return self.resolve_codec(codec_name) + if build_parameters is None or not { + "m", + "beam_width", + }.issubset(build_parameters): + raise RuntimeError( + "Accelerated-HNSW builds require canonical m and " + "beam_width parameters" + ) + return self.resolve_configured_hnsw_codec( + int(build_parameters["m"]), + int(build_parameters["beam_width"]), + ) + + def _controlled_ingest_config( + self, + codec: Any, + topology: _ControlledBuildTopology, + *, + create: bool, + ) -> tuple[Any, str]: + config = self.IndexWriterConfig() + config.setOpenMode( + self.IndexWriterConfig.OpenMode.CREATE + if create + else self.IndexWriterConfig.OpenMode.APPEND + ) + config.setCodec(codec) + config.setUseCompoundFile(False) + config.setCommitOnClose(False) + config.setMergePolicy(self.NoMergePolicy.INSTANCE) + config.setMergeScheduler(self.SerialMergeScheduler()) + # Lucene requires one automatic flush trigger to remain enabled. Set + # maxBufferedDocs first, then disable the RAM trigger deliberately. + config.setMaxBufferedDocs(topology.max_buffered_docs) + config.setRAMBufferSizeMB( + float(self.IndexWriterConfig.DISABLE_AUTO_FLUSH) + ) + application_mode = self._set_ram_per_thread_hard_limit_mb( + config, + topology.ram_per_thread_hard_limit_mb, + allow_unsupported=(topology.allow_unsupported_lucene_ram_limit), + ) + return config, application_mode + + def _tiered_merge_policy(self) -> Any: + merge_policy = self.TieredMergePolicy() + merge_policy.setNoCFSRatio(0.0) + return merge_policy + + def _controlled_merge_config( + self, codec: Any, topology: _ControlledBuildTopology + ) -> tuple[Any, str]: + config = self.IndexWriterConfig() + config.setOpenMode(self.IndexWriterConfig.OpenMode.APPEND) + config.setCodec(codec) + config.setUseCompoundFile(False) + config.setCommitOnClose(False) + config.setMergeScheduler(self.SerialMergeScheduler()) + config.setMergePolicy(self._tiered_merge_policy()) + application_mode = self._set_ram_per_thread_hard_limit_mb( + config, + topology.ram_per_thread_hard_limit_mb, + allow_unsupported=(topology.allow_unsupported_lucene_ram_limit), + ) + return config, application_mode + + def _reader_segment_vector_counts(self, reader: Any) -> tuple[int, ...]: + if int(reader.numDocs()) != int(reader.maxDoc()): + raise RuntimeError( + "Controlled Lucene build unexpectedly contains deletions" + ) + counts = [] + document_count = 0 + for leaf in reader.leaves(): + leaf_reader = leaf.reader() + leaf_documents = int(leaf_reader.numDocs()) + if leaf_documents != int(leaf_reader.maxDoc()): + raise RuntimeError( + "Controlled Lucene segment unexpectedly contains deletions" + ) + values = leaf_reader.getFloatVectorValues(_VECTOR_FIELD) + if values is None: + raise RuntimeError( + "Controlled Lucene segment is missing vector values" + ) + vector_count = int(values.size()) + if vector_count != leaf_documents: + raise RuntimeError( + "Controlled Lucene segment vector and document counts " + f"differ: {vector_count} != {leaf_documents}" + ) + counts.append(vector_count) + document_count += leaf_documents + if document_count != int(reader.numDocs()): + raise RuntimeError( + "Controlled Lucene leaf document counts do not match the " + f"reader: {document_count} != {reader.numDocs()}" + ) + return tuple(counts) + + def _committed_segment_vector_counts( + self, directory: Any + ) -> tuple[int, ...]: + reader = self.DirectoryReader.open(directory) + with _CleanupStack() as cleanups: + cleanups.add("close Lucene topology reader", reader.close) + return self._reader_segment_vector_counts(reader) + + def _write_controlled_chunk( + self, + directory: Any, + vectors: np.ndarray, + codec: Any, + topology: _ControlledBuildTopology, + *, + start: int, + stop: int, + create: bool, + ) -> tuple[int, int, int, str]: + setup_started = time.perf_counter_ns() + config, application_mode = self._controlled_ingest_config( + codec, topology, create=create + ) + writer = self.IndexWriter(directory, config) + setup_ns = time.perf_counter_ns() - setup_started + try: + ingest_started = time.perf_counter_ns() + for document_id in range(start, stop): + writer.addDocument( + self._document(document_id, vectors[document_id]) + ) + ingest_ns = time.perf_counter_ns() - ingest_started + commit_started = time.perf_counter_ns() + writer.flush() + writer.commit() + writer.close() + commit_close_ns = time.perf_counter_ns() - commit_started + except BaseException as error: + _rollback_writer(writer, error) + raise + return setup_ns, ingest_ns, commit_close_ns, application_mode + + def _force_merge_controlled_index( + self, + directory: Any, + codec: Any, + topology: _ControlledBuildTopology, + ) -> tuple[int, int, int, str]: + setup_started = time.perf_counter_ns() + config, application_mode = self._controlled_merge_config( + codec, topology + ) + writer = self.IndexWriter(directory, config) + setup_ns = time.perf_counter_ns() - setup_started + try: + merge_started = time.perf_counter_ns() + writer.forceMerge(topology.force_merge_segment_count, True) + force_merge_ns = time.perf_counter_ns() - merge_started + commit_started = time.perf_counter_ns() + writer.commit() + writer.close() + commit_close_ns = time.perf_counter_ns() - commit_started + except BaseException as error: + _rollback_writer(writer, error) + raise + return setup_ns, force_merge_ns, commit_close_ns, application_mode + + def _build_ordinary_index( + self, + directory: Any, + vectors: np.ndarray, + codec: Any, + ) -> _IndexBuildPhase: + setup_started = time.perf_counter_ns() + config = self.IndexWriterConfig() + config.setOpenMode(self.IndexWriterConfig.OpenMode.CREATE) + config.setCodec(codec) + writer = self.IndexWriter(directory, config) + writer_setup_ns = time.perf_counter_ns() - setup_started + try: + ingest_started = time.perf_counter_ns() + for document_id, vector in enumerate(vectors): + writer.addDocument(self._document(document_id, vector)) + document_ingest_ns = time.perf_counter_ns() - ingest_started + commit_started = time.perf_counter_ns() + writer.commit() + writer.close() + writer_commit_close_ns = time.perf_counter_ns() - commit_started + except BaseException as error: + _rollback_writer(writer, error) + raise + + post_build_started = time.perf_counter_ns() + reader = self.DirectoryReader.open(directory) + with _CleanupStack() as reader_cleanups: + reader_cleanups.add("close Lucene reader", reader.close) + segment_count = int(reader.leaves().size()) + return _IndexBuildPhase( + writer_setup_ns=writer_setup_ns, + document_ingest_ns=document_ingest_ns, + force_merge_ns=0, + writer_commit_close_ns=writer_commit_close_ns, + post_build_reader_ns=(time.perf_counter_ns() - post_build_started), + segment_count=segment_count, + ) + + def _build_controlled_index( + self, + directory: Any, + vectors: np.ndarray, + codec: Any, + controlled: _ControlledBuildTopology, + ) -> _IndexBuildPhase: + writer_setup_ns = 0 + document_ingest_ns = 0 + force_merge_ns = 0 + writer_commit_close_ns = 0 + post_build_reader_ns = 0 + ram_limit_application_modes: set[str] = set() + + # Separate writer lifecycles make the requested serial partitions + # explicit and keep their physical topology independently verifiable. + for chunk_number in range(controlled.premerge_segment_count): + start = chunk_number * controlled.chunk_size + stop = start + controlled.chunk_size + setup_ns, ingest_ns, commit_close_ns, application_mode = ( + self._write_controlled_chunk( + directory, + vectors, + codec, + controlled, + start=start, + stop=stop, + create=chunk_number == 0, + ) + ) + writer_setup_ns += setup_ns + document_ingest_ns += ingest_ns + writer_commit_close_ns += commit_close_ns + ram_limit_application_modes.add(application_mode) + + post_build_started = time.perf_counter_ns() + observed_premerge_counts = self._committed_segment_vector_counts( + directory + ) + post_build_reader_ns += time.perf_counter_ns() - post_build_started + expected_premerge_counts = ( + controlled.chunk_size, + ) * controlled.premerge_segment_count + if observed_premerge_counts != expected_premerge_counts: + raise RuntimeError( + "Controlled Lucene ingest topology mismatch: expected " + "pre-merge segment vector counts " + f"{expected_premerge_counts}, observed " + f"{observed_premerge_counts}" + ) + + final_merge_policy = None + final_counts = observed_premerge_counts + if ( + controlled.force_merge_segment_count == 1 + and controlled.premerge_segment_count > 1 + ): + setup_ns, merge_ns, commit_close_ns, application_mode = ( + self._force_merge_controlled_index( + directory, codec, controlled + ) + ) + writer_setup_ns += setup_ns + force_merge_ns += merge_ns + writer_commit_close_ns += commit_close_ns + ram_limit_application_modes.add(application_mode) + post_build_started = time.perf_counter_ns() + final_counts = self._committed_segment_vector_counts(directory) + post_build_reader_ns += time.perf_counter_ns() - post_build_started + final_merge_policy = "TieredMergePolicy" + expected_final_counts = ( + (int(vectors.shape[0]),) + if controlled.force_merge_segment_count == 1 + else observed_premerge_counts + ) + if final_counts != expected_final_counts: + raise RuntimeError( + "Controlled Lucene final topology mismatch: expected " + f"segment vector counts {expected_final_counts}, " + f"observed {final_counts}" + ) + if len(ram_limit_application_modes) != 1: + raise RuntimeError( + "Controlled Lucene writers used inconsistent per-thread RAM " + f"limit mechanisms: {sorted(ram_limit_application_modes)}" + ) + ram_limit_application = ram_limit_application_modes.pop() + return _IndexBuildPhase( + writer_setup_ns=writer_setup_ns, + document_ingest_ns=document_ingest_ns, + force_merge_ns=force_merge_ns, + writer_commit_close_ns=writer_commit_close_ns, + post_build_reader_ns=post_build_reader_ns, + segment_count=len(final_counts), + topology=RuntimeBuildTopology( + requested_premerge_segment_count=( + controlled.premerge_segment_count + ), + observed_premerge_segment_count=len(observed_premerge_counts), + requested_force_merge_segment_count=( + controlled.force_merge_segment_count + ), + premerge_segment_vector_counts=observed_premerge_counts, + max_buffered_docs=controlled.max_buffered_docs, + applied_ram_per_thread_hard_limit_mb=( + controlled.ram_per_thread_hard_limit_mb + ), + ram_per_thread_hard_limit_application=(ram_limit_application), + ingest_merge_policy="NoMergePolicy", + final_merge_policy=final_merge_policy, + ), + ) + def build_index( - self, index_path: Path, vectors: np.ndarray, codec_name: str + self, + index_path: Path, + vectors: np.ndarray, + codec_name: str, + build_parameters: Mapping[str, Any] | None = None, ) -> RuntimeBuildResult: self.attach_current_thread() runtime_started = time.perf_counter_ns() + controlled = _controlled_build_topology( + build_parameters, int(vectors.shape[0]) + ) + controlled_codecs = {ACCELERATED_HNSW_CODEC, CAGRA_CODEC} + if controlled is not None and codec_name not in controlled_codecs: + raise RuntimeError( + "Controlled segment topology is only supported for cuVS-backed " + "Lucene builds" + ) + if ( + controlled is not None + and codec_name == CAGRA_CODEC + and controlled.force_merge_segment_count != 0 + ): + raise RuntimeError( + "Controlled CAGRA builds require force_merge_segment_count=0" + ) directory_open_started = time.perf_counter_ns() directory = self.FSDirectory.open(self.Paths.get(str(index_path))) directory_open_ns = time.perf_counter_ns() - directory_open_started directory_close_ns = 0 - writer_setup_ns = 0 - document_ingest_ns = 0 - writer_commit_close_ns = 0 - post_build_reader_ns = 0 - segment_count = 0 def close_directory() -> None: nonlocal directory_close_ns @@ -1158,42 +1867,30 @@ def close_directory() -> None: with _CleanupStack() as cleanups: cleanups.add("close Lucene directory", close_directory) writer_setup_started = time.perf_counter_ns() - config = self.IndexWriterConfig() - config.setOpenMode(self.IndexWriterConfig.OpenMode.CREATE) - config.setCodec(self.resolve_codec(codec_name)) - writer = self.IndexWriter(directory, config) - writer_setup_ns = time.perf_counter_ns() - writer_setup_started - try: - ingest_started = time.perf_counter_ns() - for document_id, vector in enumerate(vectors): - writer.addDocument(self._document(document_id, vector)) - document_ingest_ns = time.perf_counter_ns() - ingest_started - commit_started = time.perf_counter_ns() - writer.commit() - writer.close() - writer_commit_close_ns = ( - time.perf_counter_ns() - commit_started + codec = self._resolve_build_codec(codec_name, build_parameters) + codec_setup_ns = time.perf_counter_ns() - writer_setup_started + if controlled is None: + phase = self._build_ordinary_index(directory, vectors, codec) + else: + phase = self._build_controlled_index( + directory, + vectors, + codec, + controlled, ) - except BaseException as error: - _rollback_writer(writer, error) - raise - post_build_started = time.perf_counter_ns() - reader = self.DirectoryReader.open(directory) - with _CleanupStack() as reader_cleanups: - reader_cleanups.add("close Lucene reader", reader.close) - segment_count = int(reader.leaves().size()) - post_build_reader_ns = time.perf_counter_ns() - post_build_started return RuntimeBuildResult( - segment_count=segment_count, + segment_count=phase.segment_count, timing=RuntimeBuildTiming( directory_open_ns=directory_open_ns, - writer_setup_ns=writer_setup_ns, - document_ingest_ns=document_ingest_ns, - writer_commit_close_ns=writer_commit_close_ns, - post_build_reader_ns=post_build_reader_ns, + writer_setup_ns=codec_setup_ns + phase.writer_setup_ns, + document_ingest_ns=phase.document_ingest_ns, + force_merge_ns=phase.force_merge_ns, + writer_commit_close_ns=phase.writer_commit_close_ns, + post_build_reader_ns=phase.post_build_reader_ns, directory_close_ns=directory_close_ns, runtime_build_wall_ns=time.perf_counter_ns() - runtime_started, ), + topology=phase.topology, ) @staticmethod diff --git a/python/cuvs_bench/cuvs_bench/backends/lucene.py b/python/cuvs_bench/cuvs_bench/backends/lucene.py index 0dd7feac27..fa18996c1a 100644 --- a/python/cuvs_bench/cuvs_bench/backends/lucene.py +++ b/python/cuvs_bench/cuvs_bench/backends/lucene.py @@ -64,6 +64,32 @@ CAGRA_ALGORITHM: MAX_CAGRA_DIMENSIONS, } _CUVS_ALGORITHMS = frozenset((ACCELERATED_HNSW_ALGORITHM, CAGRA_ALGORITHM)) +_CUVS_TOPOLOGY_REQUIRED_KEYS = frozenset( + ( + "premerge_segment_count", + "force_merge_segment_count", + "ram_per_thread_hard_limit_mb", + ) +) +_ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT = "allow_unsupported_lucene_ram_limit" +_CUVS_TOPOLOGY_KEYS = frozenset( + (*_CUVS_TOPOLOGY_REQUIRED_KEYS, _ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT) +) +_ACCELERATED_HNSW_BUILD_KEYS = frozenset( + ( + "m", + "beam_width", + *_CUVS_TOPOLOGY_KEYS, + ) +) +_CAGRA_BUILD_KEYS = _CUVS_TOPOLOGY_KEYS +_MIN_HNSW_BUILD_PARAMETER = 1 +_MAX_HNSW_BUILD_PARAMETER = 512 +_DEFAULT_HNSW_M = 32 +_DEFAULT_HNSW_BEAM_WIDTH = 32 +_GRAPH_DEGREE_SOURCE = "requested_hnsw_same_graph_footprint_derivation" +_MAX_RAM_PER_THREAD_HARD_LIMIT_MB = 2_147_483_647 +_FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB = 2048 _BUILD_ROUTE_POLICY_BY_ALGORITHM = { CPU_HNSW_ALGORITHM: "cpu_hnsw", ACCELERATED_HNSW_ALGORITHM: "gpu_cagra_or_cpu_hnsw_fallback", @@ -75,7 +101,20 @@ CAGRA_ALGORITHM: "gpu_cagra", } _MANIFEST_FILE = ".cuvs-bench-lucene.json" -_MANIFEST_SCHEMA = 2 +_MANIFEST_SCHEMA = 4 +_RUNTIME_BUILD_TOPOLOGY_FIELDS = frozenset( + ( + "requested_premerge_segment_count", + "observed_premerge_segment_count", + "requested_force_merge_segment_count", + "premerge_segment_vector_counts", + "max_buffered_docs", + "applied_ram_per_thread_hard_limit_mb", + "ram_per_thread_hard_limit_application", + "ingest_merge_policy", + "final_merge_policy", + ) +) _SAFE_LABEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.-]*") _RUNTIME_KEYS = ( "cuvs_java_jar", @@ -180,6 +219,7 @@ def _runtime_build_timing_metadata( "runtime_directory_open_seconds": timing.directory_open_ns, "runtime_writer_setup_seconds": timing.writer_setup_ns, "runtime_document_ingest_seconds": timing.document_ingest_ns, + "runtime_force_merge_seconds": timing.force_merge_ns, "runtime_writer_commit_close_seconds": timing.writer_commit_close_ns, "runtime_post_build_reader_seconds": timing.post_build_reader_ns, "runtime_directory_close_seconds": timing.directory_close_ns, @@ -404,14 +444,72 @@ def _safe_label(value: str, kind: str) -> str: return value -def _codec_for(algorithm: str, build_params: Mapping[str, Any]) -> str: +def _bounded_hnsw_build_parameter(value: Any, name: str) -> int: + if type(value) is not int or not ( + _MIN_HNSW_BUILD_PARAMETER <= value <= _MAX_HNSW_BUILD_PARAMETER + ): + raise ValueError( + f"Lucene {name} must be an integer in " + f"[{_MIN_HNSW_BUILD_PARAMETER}, {_MAX_HNSW_BUILD_PARAMETER}], " + f"got {value!r}" + ) + return value + + +def _positive_build_parameter(value: Any, name: str) -> int: + if type(value) is not int or value < 1: + raise ValueError( + f"Lucene {name} must be a positive integer, got {value!r}" + ) + return value + + +def _ram_per_thread_hard_limit_mb(value: Any) -> int: + if type(value) is not int or not ( + 1 <= value <= _MAX_RAM_PER_THREAD_HARD_LIMIT_MB + ): + raise ValueError( + "Lucene ram_per_thread_hard_limit_mb must be an integer in " + f"[1, {_MAX_RAM_PER_THREAD_HARD_LIMIT_MB}], got {value!r}" + ) + return value + + +def _allow_unsupported_lucene_ram_limit(value: Any) -> bool: + if type(value) is not bool: + raise ValueError( + "Lucene allow_unsupported_lucene_ram_limit must be a boolean, " + f"got {value!r}" + ) + return value + + +def _force_merge_segment_count(value: Any) -> int: + if type(value) is not int or value not in (0, 1): + raise ValueError( + "Lucene force_merge_segment_count must be 0 (disabled) or 1, " + f"got {value!r}" + ) + return value + + +def _build_parameters_for( + algorithm: str, build_params: Mapping[str, Any] +) -> dict[str, Any]: + if not isinstance(build_params, Mapping): + raise TypeError("Lucene build parameters must be a mapping") try: expected = _CODEC_BY_ALGORITHM[algorithm] except KeyError as error: raise ValueError( f"Unsupported Lucene algorithm: {algorithm!r}" ) from error - unsupported = set(build_params) - {"codec"} + allowed = {"codec"} + if algorithm == ACCELERATED_HNSW_ALGORITHM: + allowed.update(_ACCELERATED_HNSW_BUILD_KEYS) + elif algorithm == CAGRA_ALGORITHM: + allowed.update(_CAGRA_BUILD_KEYS) + unsupported = set(build_params) - allowed if unsupported: raise ValueError( "Unsupported Lucene build parameters: " @@ -422,7 +520,130 @@ def _codec_for(algorithm: str, build_params: Mapping[str, Any]) -> str: raise ValueError( f"{algorithm} requires codec {expected!r}, got {actual!r}" ) - return expected + normalized: dict[str, Any] = {"codec": expected} + configured_hnsw = set(build_params) & {"m", "beam_width"} + if configured_hnsw and configured_hnsw != {"m", "beam_width"}: + missing = ", ".join(sorted({"m", "beam_width"} - configured_hnsw)) + raise ValueError( + "Configured accelerated-HNSW builds require both m and " + f"beam_width; missing: {missing}" + ) + if algorithm == ACCELERATED_HNSW_ALGORITHM: + m = build_params.get("m", _DEFAULT_HNSW_M) + beam_width = build_params.get("beam_width", _DEFAULT_HNSW_BEAM_WIDTH) + normalized.update( + { + "m": _bounded_hnsw_build_parameter(m, "m"), + "beam_width": _bounded_hnsw_build_parameter( + beam_width, "beam_width" + ), + } + ) + configured_topology = set(build_params) & _CUVS_TOPOLOGY_KEYS + if configured_topology: + missing_topology = sorted( + _CUVS_TOPOLOGY_REQUIRED_KEYS - set(build_params) + ) + if missing_topology: + raise ValueError( + "Configured cuVS-backed segment topology requires " + + ", ".join(sorted(_CUVS_TOPOLOGY_REQUIRED_KEYS)) + + "; missing: " + + ", ".join(missing_topology) + ) + force_merge_segment_count = _force_merge_segment_count( + build_params["force_merge_segment_count"] + ) + ram_per_thread_hard_limit_mb = _ram_per_thread_hard_limit_mb( + build_params["ram_per_thread_hard_limit_mb"] + ) + premerge_segment_count = _positive_build_parameter( + build_params["premerge_segment_count"], + "premerge_segment_count", + ) + allow_unsupported = _allow_unsupported_lucene_ram_limit( + build_params.get(_ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT, False) + ) + uses_unsupported_limit = ( + ram_per_thread_hard_limit_mb + >= _FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB + ) + if uses_unsupported_limit and not allow_unsupported: + raise ValueError( + "Lucene ram_per_thread_hard_limit_mb values of 2048 MiB or " + "greater require allow_unsupported_lucene_ram_limit=true" + ) + if allow_unsupported and not uses_unsupported_limit: + raise ValueError( + "Lucene allow_unsupported_lucene_ram_limit=true requires " + "ram_per_thread_hard_limit_mb >= 2048" + ) + if uses_unsupported_limit and force_merge_segment_count != 0: + raise ValueError( + "Lucene force_merge_segment_count must be 0 when using an " + "unsupported per-thread RAM hard limit" + ) + if algorithm == CAGRA_ALGORITHM and force_merge_segment_count != 0: + raise ValueError( + "lucene_cuvs_cagra currently requires " + "force_merge_segment_count=0" + ) + if premerge_segment_count < force_merge_segment_count: + raise ValueError( + "Lucene premerge_segment_count must be greater than or " + "equal to force_merge_segment_count" + ) + normalized.update( + { + "premerge_segment_count": premerge_segment_count, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": (ram_per_thread_hard_limit_mb), + } + ) + if allow_unsupported: + normalized[_ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT] = True + return normalized + + +def _codec_for(algorithm: str, build_params: Mapping[str, Any]) -> str: + return str(_build_parameters_for(algorithm, build_params)["codec"]) + + +def _build_parameter_metadata( + algorithm: str, build_params: Mapping[str, Any] +) -> dict[str, Any]: + parameters = _build_parameters_for(algorithm, build_params) + metadata: dict[str, Any] = {} + if "m" in parameters: + m = int(parameters["m"]) + metadata.update( + { + "hnsw_m": m, + "hnsw_beam_width": int(parameters["beam_width"]), + "hnsw_heuristic": "SAME_GRAPH_FOOTPRINT", + "graph_degree_source": _GRAPH_DEGREE_SOURCE, + "graph_degree": 2 * m, + "intermediate_graph_degree": 3 * m, + } + ) + if "premerge_segment_count" in parameters: + metadata["requested_premerge_segment_count"] = parameters[ + "premerge_segment_count" + ] + metadata.update( + { + "requested_force_merge_segment_count": parameters[ + "force_merge_segment_count" + ], + "ram_per_thread_hard_limit_mb": parameters[ + "ram_per_thread_hard_limit_mb" + ], + "allow_unsupported_lucene_ram_limit": parameters.get( + _ALLOW_UNSUPPORTED_LUCENE_RAM_LIMIT, False + ), + } + ) + return metadata def _search_parameters( @@ -561,6 +782,8 @@ def _manifest_payload( vectors: np.ndarray, algorithm: str, codec: str, + build_parameters: Mapping[str, Any], + runtime_build_topology: Mapping[str, Any] | None, segment_count: int, build_runtime_artifacts: Mapping[str, str], ) -> dict[str, Any]: @@ -568,12 +791,32 @@ def _manifest_payload( "schema_version": _MANIFEST_SCHEMA, "algorithm": algorithm, "codec": codec, + "build_parameters": dict(build_parameters), + "runtime_build_topology": ( + dict(runtime_build_topology) + if runtime_build_topology is not None + else None + ), "dataset": _dataset_identity(dataset, vectors), "segment_count": segment_count, "build_runtime_artifacts": dict(build_runtime_artifacts), } +def _runtime_topology_result_metadata( + topology: Mapping[str, Any] | None, +) -> dict[str, Any]: + """Flatten persisted topology evidence into result-friendly values.""" + if topology is None: + return {} + metadata = dict(topology) + for name in ("premerge_segment_vector_counts",): + metadata[name] = ( + "[" + ",".join(str(value) for value in topology[name]) + "]" + ) + return metadata + + def _artifact_metadata( role: str, provenance: Mapping[str, str] ) -> dict[str, str]: @@ -594,6 +837,109 @@ def _is_manifest_integer(value: Any, *, minimum: int = 0) -> bool: return type(value) is int and value >= minimum +def _validate_manifest_runtime_topology( + payload: Mapping[str, Any], path: Path +) -> None: + """Reject missing, malformed, or self-contradictory build evidence.""" + + def invalid(detail: str) -> None: + raise RuntimeError( + "Lucene index manifest has invalid runtime build topology " + f"({detail}): {path}" + ) + + build_parameters = payload["build_parameters"] + topology = payload["runtime_build_topology"] + if "premerge_segment_count" not in build_parameters: + if topology is not None: + invalid("unexpected evidence for an uncontrolled build") + return + if not isinstance(topology, dict): + invalid("missing evidence for a controlled build") + if set(topology) != _RUNTIME_BUILD_TOPOLOGY_FIELDS: + invalid("unexpected fields") + + positive_integer_fields = ( + "requested_premerge_segment_count", + "observed_premerge_segment_count", + "max_buffered_docs", + "applied_ram_per_thread_hard_limit_mb", + ) + for name in positive_integer_fields: + if not _is_manifest_integer(topology[name], minimum=1): + invalid(f"{name} must be a positive integer") + if not _is_manifest_integer( + topology["requested_force_merge_segment_count"] + ) or topology["requested_force_merge_segment_count"] not in (0, 1): + invalid("requested_force_merge_segment_count must be 0 or 1") + for name in ("premerge_segment_vector_counts",): + values = topology[name] + if ( + not isinstance(values, list) + or not values + or any( + not _is_manifest_integer(value, minimum=1) for value in values + ) + ): + invalid(f"{name} must be a non-empty list of positive integers") + if not isinstance(topology["ingest_merge_policy"], str): + invalid("ingest_merge_policy must be a string") + if topology["final_merge_policy"] is not None and not isinstance( + topology["final_merge_policy"], str + ): + invalid("final_merge_policy must be null or a string") + application_mode = topology["ram_per_thread_hard_limit_application"] + if not isinstance(application_mode, str) or application_mode not in { + "public_setter", + "unsupported_field_override", + }: + invalid("unexpected RAM hard-limit application mode") + + vector_count = int(payload["dataset"]["vector_count"]) + force_merge = int(build_parameters["force_merge_segment_count"]) + requested_count = int(build_parameters["premerge_segment_count"]) + if vector_count % requested_count: + invalid("vector count is not divisible by the requested count") + if topology["requested_force_merge_segment_count"] != force_merge: + invalid("force-merge request does not match build parameters") + if ( + topology["applied_ram_per_thread_hard_limit_mb"] + != build_parameters["ram_per_thread_hard_limit_mb"] + ): + invalid("applied RAM hard limit does not match the request") + requested_hard_limit = build_parameters["ram_per_thread_hard_limit_mb"] + expected_application_mode = ( + "unsupported_field_override" + if requested_hard_limit + >= _FIRST_UNSUPPORTED_RAM_PER_THREAD_HARD_LIMIT_MB + else "public_setter" + ) + if application_mode != expected_application_mode: + invalid("RAM hard-limit application mode does not match the request") + if topology["ingest_merge_policy"] != "NoMergePolicy": + invalid("unexpected ingest merge policy") + chunk_size = vector_count // requested_count + if ( + topology["requested_premerge_segment_count"] != requested_count + or topology["premerge_segment_vector_counts"] + != [chunk_size] * requested_count + or topology["observed_premerge_segment_count"] != requested_count + or topology["max_buffered_docs"] != chunk_size + 1 + ): + invalid("partitioned sequential evidence does not match the request") + expected_final_merge_policy = ( + "TieredMergePolicy" + if force_merge == 1 and requested_count > 1 + else None + ) + + if topology["final_merge_policy"] != expected_final_merge_policy: + invalid("final merge policy does not match the request") + expected_final_segments = 1 if force_merge == 1 else requested_count + if payload["segment_count"] != expected_final_segments: + invalid("final segment count does not match the request") + + def _validate_manifest_dataset(dataset: Any, path: Path) -> None: required = { "name", @@ -683,6 +1029,8 @@ def _read_manifest(index_path: Path) -> dict[str, Any]: "schema_version", "algorithm", "codec", + "build_parameters", + "runtime_build_topology", "dataset", "segment_count", "build_runtime_artifacts", @@ -695,6 +1043,18 @@ def _read_manifest(index_path: Path) -> dict[str, Any]: raise RuntimeError( f"Lucene index manifest has invalid identifiers: {path}" ) + try: + build_parameters = _build_parameters_for( + payload["algorithm"], payload["build_parameters"] + ) + except (TypeError, ValueError) as error: + raise RuntimeError( + f"Lucene index manifest has invalid build parameters: {path}" + ) from error + if build_parameters != payload["build_parameters"]: + raise RuntimeError( + f"Lucene index manifest has noncanonical build parameters: {path}" + ) _validate_manifest_dataset(payload["dataset"], path) segment_count = payload.get("segment_count") if not _is_manifest_integer(segment_count, minimum=1): @@ -706,6 +1066,7 @@ def _read_manifest(index_path: Path) -> dict[str, Any]: "Lucene index manifest has invalid build-runtime artifact " f"provenance: {path}" ) + _validate_manifest_runtime_topology(payload, path) return payload @@ -825,12 +1186,23 @@ def _build_benchmark_configs( _safe_label(algorithm, "algorithm") _safe_label(group, "group") for build_params in build_combos: - codec = _codec_for(algorithm, build_params) - name = algorithm if group == "base" else f"{algorithm}_{group}" + normalized_build_params = _build_parameters_for( + algorithm, build_params + ) + codec = str(normalized_build_params["codec"]) + prefix = ( + algorithm if group == "base" else f"{algorithm}_{group}" + ) + parameter_labels = [ + f"{name}{value}" + for name, value in normalized_build_params.items() + if name != "codec" + ] + name = ".".join([prefix, *parameter_labels]) index = IndexConfig( name=name, algo=algorithm, - build_param={"codec": codec}, + build_param=normalized_build_params, search_params=[dict(item) for item in search_combos], file=str(root / name), ) @@ -899,8 +1271,16 @@ def _index(self, indexes: list[IndexConfig]) -> IndexConfig: raise ValueError( f"Index algorithm {index.algo!r} does not match {self.algorithm!r}" ) - _codec_for(index.algo, index.build_param) - return index + build_parameters = _build_parameters_for(index.algo, index.build_param) + if dict(index.build_param) == build_parameters: + return index + return IndexConfig( + name=index.name, + algo=index.algo, + build_param=build_parameters, + search_params=index.search_params, + file=index.file, + ) def _index_path(self, index: IndexConfig) -> Path: path = Path(os.path.abspath(Path(index.file).expanduser())) @@ -930,7 +1310,10 @@ def _failure_build( ) def _validate_manifest_identity( - self, payload: Mapping[str, Any], dataset_identity: Mapping[str, Any] + self, + payload: Mapping[str, Any], + dataset_identity: Mapping[str, Any], + build_parameters: Mapping[str, Any], ) -> None: build_runtime_artifacts = payload.get("build_runtime_artifacts") if not isinstance(build_runtime_artifacts, Mapping): @@ -996,11 +1379,15 @@ def _validate_manifest_identity( "schema_version": _MANIFEST_SCHEMA, "algorithm": self.algorithm, "codec": self.codec, + "build_parameters": _build_parameters_for( + self.algorithm, build_parameters + ), "dataset": dict(dataset_identity), "build_runtime_artifacts": dict(build_runtime_artifacts), } actual_without_segment_count = dict(payload) actual_without_segment_count.pop("segment_count", None) + actual_without_segment_count.pop("runtime_build_topology", None) if actual_without_segment_count != expected_without_segment_count: raise RuntimeError( "Existing Lucene index does not match this dataset and configuration; " @@ -1115,6 +1502,45 @@ def _install_staged_index(staged: Path, destination: Path) -> str | None: ) return None + @staticmethod + def _discard_failed_staging( + staged: Path, build_error: BaseException + ) -> None: + """Discard an unpublished build without hiding either failure.""" + try: + shutil.rmtree(staged) + except (KeyboardInterrupt, SystemExit) as cleanup_interrupt: + cleanup_interrupt.add_note( + "Index construction first failed: " + f"{type(build_error).__name__}: {build_error}. " + f"The incomplete staged index remains at {staged}." + ) + raise + except FileNotFoundError as cleanup_error: + try: + staged.lstat() + except FileNotFoundError: + # Installation may already have renamed staging into place. + return + except OSError as inspection_error: + build_error.add_note( + "Failed to discard the incomplete staged index at " + f"{staged}: {type(cleanup_error).__name__}: " + f"{cleanup_error}. Could not confirm whether staging " + f"remains: {type(inspection_error).__name__}: " + f"{inspection_error}" + ) + return + build_error.add_note( + "Failed to discard the incomplete staged index at " + f"{staged}: {type(cleanup_error).__name__}: {cleanup_error}" + ) + except Exception as cleanup_error: + build_error.add_note( + "Failed to discard the incomplete staged index at " + f"{staged}: {type(cleanup_error).__name__}: {cleanup_error}" + ) + def build( self, dataset: Dataset, @@ -1123,6 +1549,9 @@ def build( dry_run: bool = False, ) -> BuildResult: index = self._index(indexes) + parameter_metadata = _build_parameter_metadata( + self.algorithm, index.build_param + ) try: _validate_metric(dataset) path = self._index_path(index) @@ -1138,6 +1567,7 @@ def build( "codec": self.codec, "group": self.group, "index_name": index.name, + **parameter_metadata, }, ) if path.exists() and not force: @@ -1145,7 +1575,9 @@ def build( dataset_identity, vector_count, dimensions = ( self._search_dataset_identity(dataset, payload) ) - self._validate_manifest_identity(payload, dataset_identity) + self._validate_manifest_identity( + payload, dataset_identity, index.build_param + ) runtime = self._get_runtime() runtime.verify_artifacts() metadata = self._verification_metadata( @@ -1155,6 +1587,9 @@ def build( dimensions, ) self._validate_manifest_segment_count(payload, metadata) + runtime_topology_metadata = _runtime_topology_result_metadata( + payload["runtime_build_topology"] + ) return BuildResult( index_path=str(path), build_time_seconds=0.0, @@ -1166,10 +1601,12 @@ def build( "codec": self.codec, "group": self.group, "index_name": index.name, + **parameter_metadata, **_artifact_metadata( "build_runtime", payload["build_runtime_artifacts"], ), + **runtime_topology_metadata, **metadata, }, ) @@ -1205,7 +1642,7 @@ def build( try: build_started = time.perf_counter_ns() runtime_build = runtime.build_index( - staged, vectors, self.codec + staged, vectors, self.codec, index.build_param ) build_elapsed_ns = time.perf_counter_ns() - build_started @@ -1223,14 +1660,35 @@ def build( f"the committed index: {runtime_build.segment_count} != " f"{observed_segments}" ) + requested_topology = ( + "premerge_segment_count" in index.build_param + ) + if requested_topology != (runtime_build.topology is not None): + raise RuntimeError( + "Lucene runtime topology evidence does not match the " + "canonical build parameters" + ) + runtime_topology_payload = ( + runtime_build.topology.manifest() + if runtime_build.topology is not None + else None + ) + runtime_topology_metadata = _runtime_topology_result_metadata( + runtime_topology_payload + ) payload = _manifest_payload( dataset, vectors, self.algorithm, self.codec, + index.build_param, + runtime_topology_payload, observed_segments, runtime.artifact_provenance, ) + _validate_manifest_runtime_topology( + payload, staged / _MANIFEST_FILE + ) _write_manifest(staged, payload) validation_elapsed_ns = ( time.perf_counter_ns() - validation_started @@ -1239,8 +1697,8 @@ def build( install_started = time.perf_counter_ns() cleanup_warning = self._install_staged_index(staged, path) install_elapsed_ns = time.perf_counter_ns() - install_started - except BaseException: - shutil.rmtree(staged, ignore_errors=True) + except BaseException as build_error: + self._discard_failed_staging(staged, build_error) raise index_size_started = time.perf_counter_ns() index_size_bytes = _index_size(path) @@ -1291,10 +1749,12 @@ def build( "codec": self.codec, "group": self.group, "index_name": index.name, + **parameter_metadata, "pylucene_version": runtime.pylucene_version, **_artifact_metadata( "build_runtime", runtime.artifact_provenance ), + **runtime_topology_metadata, **lifecycle_metadata, **metadata, }, @@ -1422,6 +1882,9 @@ def search( dry_run: bool = False, ) -> list[SearchResult]: index = self._index(indexes) + parameter_metadata = _build_parameter_metadata( + self.algorithm, index.build_param + ) try: _validate_metric(dataset) if type(k) is not int or k < 1: @@ -1462,6 +1925,7 @@ def search( "codec": self.codec, "group": self.group, "index_name": index.name, + **parameter_metadata, }, ) ] @@ -1482,7 +1946,9 @@ def search( dataset_identity, vector_count, dimensions = ( self._search_dataset_identity(dataset, payload) ) - self._validate_manifest_identity(payload, dataset_identity) + self._validate_manifest_identity( + payload, dataset_identity, index.build_param + ) if int(queries.shape[1]) != dimensions: raise ValueError( "query vector dimensions do not match the indexed dataset: " @@ -1504,6 +1970,9 @@ def search( runtime, path, vector_count, dimensions ) self._validate_manifest_segment_count(payload, metadata) + runtime_topology_metadata = _runtime_topology_result_metadata( + payload["runtime_build_topology"] + ) index_verification_ns = ( time.perf_counter_ns() - index_verification_started ) @@ -1529,10 +1998,12 @@ def search( "mode": mode, "group": self.group, "index_name": index.name, + **parameter_metadata, **_artifact_metadata( "build_runtime", payload["build_runtime_artifacts"], ), + **runtime_topology_metadata, "expected_search_route": ( _SEARCH_ROUTE_BY_ALGORITHM[self.algorithm] ), diff --git a/python/cuvs_bench/cuvs_bench/tests/_lucene_test_support.py b/python/cuvs_bench/cuvs_bench/tests/_lucene_test_support.py index 60987a3bb6..66a30140f5 100644 --- a/python/cuvs_bench/cuvs_bench/tests/_lucene_test_support.py +++ b/python/cuvs_bench/cuvs_bench/tests/_lucene_test_support.py @@ -23,6 +23,7 @@ QueryTiming, RuntimeBuildResult, RuntimeBuildTiming, + RuntimeBuildTopology, RuntimeSearchResult, RuntimeSearchTiming, SearchHit, @@ -94,7 +95,7 @@ def verify( return LuceneIndexVerification( codec=expected_codec, segment_count=self.runtime.segment_count, - field_count=1, + field_count=self.runtime.segment_count, vector_count=expected_vector_count, dimensions=expected_dimensions, ) @@ -103,7 +104,8 @@ def verify( class RecordingCagraVerifier: """Return deterministic persisted-path evidence and retain each request.""" - def __init__(self) -> None: + def __init__(self, runtime: "RecordingRuntime") -> None: + self.runtime = runtime self.calls: list[tuple[Path, int, int]] = [] def verify( @@ -117,8 +119,8 @@ def verify( (index_path, expected_vector_count, expected_dimensions) ) return CagraVerification( - segment_count=1, - field_count=1, + segment_count=self.runtime.segment_count, + field_count=self.runtime.segment_count, vector_count=expected_vector_count, dimensions=expected_dimensions, ) @@ -132,8 +134,10 @@ class RecordingRuntime: def __init__(self) -> None: self.artifact_provenance: dict[str, str] = {} self.index_verifier = RecordingIndexVerifier(self) - self.cagra_verifier = RecordingCagraVerifier() - self.build_calls: list[tuple[Path, np.ndarray, str]] = [] + self.cagra_verifier = RecordingCagraVerifier(self) + self.build_calls: list[ + tuple[Path, np.ndarray, str, dict[str, Any]] + ] = [] self.search_calls: list[dict[str, Any]] = [] self.build_error: Exception | None = None self.search_error: Exception | None = None @@ -148,24 +152,73 @@ def verify_artifacts(self) -> None: self.artifact_verification_count += 1 def build_index( - self, index_path: Path, vectors: np.ndarray, codec_name: str + self, + index_path: Path, + vectors: np.ndarray, + codec_name: str, + build_parameters: Mapping[str, Any] | None = None, ) -> RuntimeBuildResult: - self.build_calls.append((index_path, vectors.copy(), codec_name)) + self.build_calls.append( + ( + index_path, + vectors.copy(), + codec_name, + dict(build_parameters or {}), + ) + ) if self.build_error is not None: raise self.build_error self.document_count, self.dimensions = vectors.shape + self.segment_count = 1 (index_path / "segments.fake").write_text(codec_name, encoding="utf-8") + parameters = dict(build_parameters or {}) + topology = None + if "premerge_segment_count" in parameters: + premerge = int(parameters["premerge_segment_count"]) + force_merge = int(parameters["force_merge_segment_count"]) + chunk_size = int(vectors.shape[0]) // premerge + topology = RuntimeBuildTopology( + requested_premerge_segment_count=premerge, + observed_premerge_segment_count=premerge, + requested_force_merge_segment_count=force_merge, + premerge_segment_vector_counts=(chunk_size,) * premerge, + max_buffered_docs=chunk_size + 1, + applied_ram_per_thread_hard_limit_mb=int( + parameters["ram_per_thread_hard_limit_mb"] + ), + ram_per_thread_hard_limit_application=( + "unsupported_field_override" + if parameters["ram_per_thread_hard_limit_mb"] >= 2048 + else "public_setter" + ), + ingest_merge_policy="NoMergePolicy", + final_merge_policy=( + "TieredMergePolicy" + if force_merge == 1 and premerge > 1 + else None + ), + ) + self.segment_count = 1 if force_merge == 1 else premerge + force_merge_ns = ( + 350_000 + if topology is not None + and topology.requested_force_merge_segment_count == 1 + and topology.requested_premerge_segment_count > 1 + else 0 + ) return RuntimeBuildResult( - segment_count=1, + segment_count=self.segment_count, timing=RuntimeBuildTiming( directory_open_ns=100_000, writer_setup_ns=200_000, document_ingest_ns=300_000, + force_merge_ns=force_merge_ns, writer_commit_close_ns=400_000, post_build_reader_ns=500_000, directory_close_ns=600_000, - runtime_build_wall_ns=2_100_000, + runtime_build_wall_ns=2_100_000 + force_merge_ns, ), + topology=topology, ) def search_index( diff --git a/python/cuvs_bench/cuvs_bench/tests/conftest.py b/python/cuvs_bench/cuvs_bench/tests/conftest.py index a533df258f..7170943979 100644 --- a/python/cuvs_bench/cuvs_bench/tests/conftest.py +++ b/python/cuvs_bench/cuvs_bench/tests/conftest.py @@ -8,7 +8,7 @@ def pytest_addoption(parser): - """Add the explicit opt-in for live Lucene integration tests.""" + """Add explicit opt-ins for ordinary and large Lucene integration tests.""" parser.addoption( "--run-lucene-e2e", action="store_true", @@ -18,16 +18,33 @@ def pytest_addoption(parser): "GPU-intended cases additionally require cuVS/CUDA/GPU" ), ) + parser.addoption( + "--run-lucene-large-segment-e2e", + action="store_true", + default=False, + help=( + "run only the resource-intensive Lucene cases that build a " + "single segment from more than 2 GiB of vector data" + ), + ) def pytest_collection_modifyitems(config, items): - """Skip live Lucene cases unless the suite was selected explicitly.""" - if config.getoption("--run-lucene-e2e"): - return - skip = pytest.mark.skip(reason="requires --run-lucene-e2e") + """Select ordinary and large Lucene cases through independent opt-ins.""" + run_ordinary = config.getoption("--run-lucene-e2e") + run_large = config.getoption("--run-lucene-large-segment-e2e") for item in items: - if "lucene_e2e" in item.keywords: - item.add_marker(skip) + if "lucene_large_segment_e2e" in item.keywords: + if not run_large: + item.add_marker( + pytest.mark.skip( + reason="requires --run-lucene-large-segment-e2e" + ) + ) + elif "lucene_e2e" in item.keywords and not run_ordinary: + item.add_marker( + pytest.mark.skip(reason="requires --run-lucene-e2e") + ) def pytest_configure(config): diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_build_search.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_build_search.py index 7ba1266658..2b0b2245fa 100644 --- a/python/cuvs_bench/cuvs_bench/tests/test_lucene_build_search.py +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_build_search.py @@ -57,7 +57,10 @@ def test_successful_build_reports_the_verified_persisted_index_kind( assert result.success, result.error_message assert result.algorithm == algorithm - assert result.build_params == {"codec": codec} + expected_build_parameters = {"codec": codec} + if algorithm == ACCELERATED_HNSW_ALGORITHM: + expected_build_parameters.update({"m": 32, "beam_width": 32}) + assert result.build_params == expected_build_parameters assert result.metadata["codec"] == codec expected_persisted_kind = { CPU_HNSW_ALGORITHM: "cpu_hnsw", @@ -78,6 +81,18 @@ def test_successful_build_reports_the_verified_persisted_index_kind( assert result.index_size_bytes > 0 assert len(factory.calls) == 1 assert [call[2] for call in runtime.build_calls] == [codec] + assert [call[3] for call in runtime.build_calls] == [ + expected_build_parameters + ] + if algorithm == ACCELERATED_HNSW_ALGORITHM: + assert result.metadata["hnsw_m"] == 32 + assert result.metadata["hnsw_beam_width"] == 32 + assert result.metadata["hnsw_heuristic"] == "SAME_GRAPH_FOOTPRINT" + assert result.metadata["graph_degree_source"] == ( + "requested_hnsw_same_graph_footprint_derivation" + ) + assert result.metadata["graph_degree"] == 64 + assert result.metadata["intermediate_graph_degree"] == 96 if algorithm == CAGRA_ALGORITHM: [(verified_path, vector_count, dimensions)] = ( runtime.cagra_verifier.calls @@ -91,6 +106,292 @@ def test_successful_build_reports_the_verified_persisted_index_kind( assert runtime.cagra_verifier.calls == [] +def test_accelerated_build_propagates_and_persists_requested_parameters( + tmp_path: Path, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + + result = backend.build(_dataset(), [index]) + + expected = { + "codec": index.build_param["codec"], + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + assert result.success, result.error_message + assert result.build_params == expected + assert runtime.build_calls[0][3] == expected + assert result.metadata["graph_degree"] == 32 + assert result.metadata["intermediate_graph_degree"] == 48 + assert result.metadata["graph_degree_source"] == ( + "requested_hnsw_same_graph_footprint_derivation" + ) + assert result.metadata["requested_premerge_segment_count"] == 4 + assert result.metadata["requested_force_merge_segment_count"] == 1 + assert result.metadata["ram_per_thread_hard_limit_mb"] == 1945 + assert result.metadata["premerge_segment_vector_counts"] == "[1,1,1,1]" + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + assert manifest["schema_version"] == 4 + assert manifest["build_parameters"] == expected + + +@pytest.mark.parametrize( + ("premerge_segment_count", "expected_segment_count"), + ((1, 1), (4, 4)), +) +def test_accelerated_build_preserves_no_force_merge_topology( + tmp_path: Path, + premerge_segment_count: int, + expected_segment_count: int, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": premerge_segment_count, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + + result = backend.build(_dataset(), [index]) + + assert result.success, result.error_message + assert result.metadata["segment_count"] == expected_segment_count + assert result.metadata["requested_force_merge_segment_count"] == 0 + assert result.metadata["final_merge_policy"] is None + expected_counts = ( + "[" + + ",".join([str(4 // premerge_segment_count)] * premerge_segment_count) + + "]" + ) + assert result.metadata["premerge_segment_vector_counts"] == expected_counts + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + assert manifest["segment_count"] == expected_segment_count + assert manifest["build_parameters"]["force_merge_segment_count"] == 0 + + +@pytest.mark.parametrize( + ("premerge_segment_count", "expected_segment_count"), + ((1, 1), (4, 4)), +) +def test_cagra_build_preserves_direct_segment_topology( + tmp_path: Path, + premerge_segment_count: int, + expected_segment_count: int, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, CAGRA_ALGORITHM, runtime + ) + index.build_param.update( + { + "premerge_segment_count": premerge_segment_count, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + + result = backend.build(_dataset(), [index]) + + assert result.success, result.error_message + assert result.metadata["segment_count"] == expected_segment_count + assert result.metadata["requested_premerge_segment_count"] == ( + premerge_segment_count + ) + assert result.metadata["requested_force_merge_segment_count"] == 0 + assert result.metadata["ram_per_thread_hard_limit_application"] == ( + "public_setter" + ) + assert result.metadata["final_merge_policy"] is None + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + assert manifest["segment_count"] == expected_segment_count + assert ( + manifest["runtime_build_topology"]["observed_premerge_segment_count"] + == premerge_segment_count + ) + assert ( + manifest["runtime_build_topology"][ + "ram_per_thread_hard_limit_application" + ] + == "public_setter" + ) + + +@pytest.mark.parametrize( + "algorithm", + ( + pytest.param(ACCELERATED_HNSW_ALGORITHM, id="gpu-cagra-built-hnsw"), + pytest.param(CAGRA_ALGORITHM, id="gpu-cagra-search"), + ), +) +def test_explicit_unsupported_ram_limit_is_persisted_and_reusable( + tmp_path: Path, algorithm: str +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index(tmp_path, algorithm, runtime) + if algorithm == ACCELERATED_HNSW_ALGORITHM: + index.build_param.update({"m": 16, "beam_width": 80}) + index.build_param.update( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + } + ) + + built = backend.build(_dataset(), [index]) + reused = backend.build(_dataset(), [index]) + [searched] = backend.search(_dataset(), [index], k=2) + + assert built.success, built.error_message + assert built.build_params == runtime.build_calls[0][3] + assert built.metadata["applied_ram_per_thread_hard_limit_mb"] == 6144 + assert built.metadata["ram_per_thread_hard_limit_application"] == ( + "unsupported_field_override" + ) + assert built.metadata["runtime_force_merge_seconds"] == 0.0 + assert reused.success, reused.error_message + assert reused.metadata["skipped"] is True + assert searched.success, searched.error_message + for metadata in (reused.metadata, searched.metadata): + assert metadata["ram_per_thread_hard_limit_application"] == ( + "unsupported_field_override" + ) + assert metadata["applied_ram_per_thread_hard_limit_mb"] == 6144 + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + assert ( + manifest["build_parameters"]["allow_unsupported_lucene_ram_limit"] + is True + ) + assert ( + manifest["runtime_build_topology"][ + "ram_per_thread_hard_limit_application" + ] + == "unsupported_field_override" + ) + assert len(runtime.build_calls) == 1 + + +@pytest.mark.parametrize( + ("force_merge_segment_count", "expected_segment_count"), + ((0, 4), (1, 1)), +) +def test_accelerated_build_persists_observed_topology_for_reuse( + tmp_path: Path, + force_merge_segment_count: int, + expected_segment_count: int, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + + result = backend.build(_dataset(), [index]) + + expected = { + "codec": index.build_param["codec"], + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": 1945, + } + assert result.success, result.error_message + assert result.build_params == expected + assert runtime.build_calls[0][3] == expected + assert result.metadata["segment_count"] == expected_segment_count + assert result.metadata["requested_premerge_segment_count"] == 4 + assert result.metadata["observed_premerge_segment_count"] == 4 + assert result.metadata["requested_force_merge_segment_count"] == ( + force_merge_segment_count + ) + assert result.metadata["max_buffered_docs"] == 2 + assert result.metadata["premerge_segment_vector_counts"] == "[1,1,1,1]" + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + assert manifest["schema_version"] == 4 + assert manifest["build_parameters"] == expected + assert manifest["runtime_build_topology"] == { + "requested_premerge_segment_count": 4, + "observed_premerge_segment_count": 4, + "requested_force_merge_segment_count": force_merge_segment_count, + "premerge_segment_vector_counts": [1, 1, 1, 1], + "max_buffered_docs": 2, + "applied_ram_per_thread_hard_limit_mb": 1945, + "ram_per_thread_hard_limit_application": "public_setter", + "ingest_merge_policy": "NoMergePolicy", + "final_merge_policy": ( + "TieredMergePolicy" if force_merge_segment_count == 1 else None + ), + } + + reused = backend.build(_dataset(), [index]) + [search] = backend.search(_dataset(), [index], k=2) + + assert reused.success, reused.error_message + assert reused.metadata["skipped"] is True + assert search.success, search.error_message + for reused_metadata in (reused.metadata, search.metadata): + assert reused_metadata["requested_premerge_segment_count"] == 4 + assert reused_metadata["observed_premerge_segment_count"] == 4 + assert reused_metadata["premerge_segment_vector_counts"] == ( + "[1,1,1,1]" + ) + assert result.metadata["runtime_force_merge_seconds"] == ( + 0.00035 if force_merge_segment_count else 0.0 + ) + assert len(runtime.build_calls) == 1 + + @pytest.mark.parametrize( ("algorithm", "dimensions", "maximum_dimensions", "expected_success"), ( @@ -475,6 +776,7 @@ def test_build_timing_rejects_phases_longer_than_the_enclosing_wall() -> None: directory_open_ns=1, writer_setup_ns=2, document_ingest_ns=3, + force_merge_ns=1, writer_commit_close_ns=4, post_build_reader_ns=5, directory_close_ns=6, @@ -799,19 +1101,28 @@ def test_dry_runs_do_not_resolve_or_start_the_runtime( search_result = backend.search(_dataset(), [index], k=2, dry_run=True)[0] assert build_result.success - assert build_result.metadata == { + expected_metadata = { "dry_run": True, "codec": codec, "group": "test", "index_name": algorithm, } + if algorithm == ACCELERATED_HNSW_ALGORITHM: + expected_metadata.update( + { + "hnsw_m": 32, + "hnsw_beam_width": 32, + "hnsw_heuristic": "SAME_GRAPH_FOOTPRINT", + "graph_degree_source": ( + "requested_hnsw_same_graph_footprint_derivation" + ), + "graph_degree": 64, + "intermediate_graph_degree": 96, + } + ) + assert build_result.metadata == expected_metadata assert search_result.success - assert search_result.metadata == { - "dry_run": True, - "codec": codec, - "group": "test", - "index_name": algorithm, - } + assert search_result.metadata == expected_metadata assert factory.calls == [] assert not index_root.exists() diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_integration.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_integration.py index 4dbcbf7a67..e03d185eca 100644 --- a/python/cuvs_bench/cuvs_bench/tests/test_lucene_integration.py +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_integration.py @@ -241,6 +241,96 @@ def _assert_timing_contract( assert "subsequent_java_index_searcher_search_mean_ms" not in metadata +def _assert_accelerated_hnsw_build_diagnostics( + standard_output: str, standard_error: str, case_name: str +) -> None: + assert not case_used_cpu_hnsw_fallback(standard_error, case_name), ( + f"{case_name} used Lucene's CPU HNSW writer fallback:\n" + f"{lucene_case_output(standard_error, case_name)}" + ) + combined_output = standard_output + standard_error + for warning in _GRAPH_CLAMP_WARNINGS: + assert warning.casefold() not in combined_output.casefold() + + +def _assert_accelerated_hnsw_build_topology( + build, + physical_vector_counts: tuple[int, ...], + *, + expected_premerge_vector_counts: tuple[int, ...], + expected_final_vector_counts: tuple[int, ...], + requested_premerge_segment_count: int, + requested_force_merge_segment_count: int, + performed_force_merge: bool, +) -> None: + assert physical_vector_counts == expected_final_vector_counts + assert build.metadata["segment_count"] == len(expected_final_vector_counts) + assert build.metadata["hnsw_m"] == 16 + assert build.metadata["hnsw_beam_width"] == 80 + assert build.metadata["graph_degree_source"] == ( + "requested_hnsw_same_graph_footprint_derivation" + ) + assert build.metadata["requested_premerge_segment_count"] == ( + requested_premerge_segment_count + ) + assert build.metadata["observed_premerge_segment_count"] == len( + expected_premerge_vector_counts + ) + assert build.metadata["requested_force_merge_segment_count"] == ( + requested_force_merge_segment_count + ) + assert build.metadata["applied_ram_per_thread_hard_limit_mb"] == 1024 + assert build.metadata["ram_per_thread_hard_limit_application"] == ( + "public_setter" + ) + assert build.metadata["premerge_segment_vector_counts"] == json.dumps( + list(expected_premerge_vector_counts), separators=(",", ":") + ) + if performed_force_merge: + assert build.metadata["runtime_force_merge_seconds"] > 0.0 + else: + assert build.metadata["runtime_force_merge_seconds"] == 0.0 + + +def _assert_persisted_accelerated_hnsw_topology( + manifest: dict, + *, + build_params: dict, + expected_premerge_vector_counts: tuple[int, ...], + expected_final_vector_counts: tuple[int, ...], + requested_force_merge_segment_count: int, + performed_force_merge: bool, +) -> None: + assert manifest["build_parameters"] == build_params + assert manifest["segment_count"] == len(expected_final_vector_counts) + topology = manifest["runtime_build_topology"] + assert topology["premerge_segment_vector_counts"] == list( + expected_premerge_vector_counts + ) + assert topology["observed_premerge_segment_count"] == len( + expected_premerge_vector_counts + ) + assert topology["requested_force_merge_segment_count"] == ( + requested_force_merge_segment_count + ) + assert topology["ram_per_thread_hard_limit_application"] == ( + "public_setter" + ) + assert topology["final_merge_policy"] == ( + "TieredMergePolicy" if performed_force_merge else None + ) + + +def _assert_cpu_hnsw_search_quality( + result, query_ids: np.ndarray, expected_neighbors: np.ndarray +) -> None: + assert result.success, result.error_message + assert result.metadata["expected_search_route"] == "cpu_hnsw" + np.testing.assert_array_equal(result.neighbors[:, 0], query_ids) + assert all(len(set(row)) == len(row) for row in result.neighbors.tolist()) + assert _recall(result.neighbors, expected_neighbors) >= _MINIMUM_RECALL + + def test_accelerated_hnsw_builds_on_gpu_and_searches_on_cpu( tmp_path, capfd, @@ -262,14 +352,9 @@ def test_accelerated_hnsw_builds_on_gpu_and_searches_on_cpu( result = backend.search(dataset, [index], k=10, batch_size=2)[0] captured = capfd.readouterr() - assert not case_used_cpu_hnsw_fallback(captured.err, case_name), ( - f"{case_name} used Lucene's CPU HNSW writer fallback:\n" - f"{lucene_case_output(captured.err, case_name)}" + _assert_accelerated_hnsw_build_diagnostics( + captured.out, captured.err, case_name ) - for warning in _GRAPH_CLAMP_WARNINGS: - assert ( - warning.casefold() not in (captured.out + captured.err).casefold() - ) assert build.metadata["codec"] == ACCELERATED_HNSW_CODEC assert build.metadata["persisted_index_kind"] == "hnsw" @@ -277,8 +362,9 @@ def test_accelerated_hnsw_builds_on_gpu_and_searches_on_cpu( "gpu_cagra_or_cpu_hnsw_fallback" ) _assert_artifact_provenance(build.metadata, "build_runtime") - assert result.success, result.error_message - assert result.metadata["expected_search_route"] == "cpu_hnsw" + _assert_cpu_hnsw_search_quality( + result, query_ids, dataset.groundtruth_neighbors + ) assert result.metadata["persisted_index_kind"] == "hnsw" assert result.metadata["build_route_policy"] == ( "gpu_cagra_or_cpu_hnsw_fallback" @@ -291,6 +377,229 @@ def test_accelerated_hnsw_builds_on_gpu_and_searches_on_cpu( ) _assert_artifact_provenance(result.metadata, "build_runtime") _assert_artifact_provenance(result.metadata, "search_runtime") + + +@pytest.mark.parametrize( + ( + "premerge_segment_count", + "force_merge_segment_count", + "expected_premerge_vector_counts", + "expected_final_vector_counts", + ), + ( + pytest.param( + 1, + 1, + (1024,), + (1024,), + id="retain-one-segment-without-force-merge", + ), + pytest.param( + 4, + 0, + (256, 256, 256, 256), + (256, 256, 256, 256), + id="retain-four-segments", + ), + pytest.param( + 4, + 1, + (256, 256, 256, 256), + (1024,), + id="force-merge-four-to-one", + ), + ), +) +def test_accelerated_hnsw_controls_persist_observed_topology( + tmp_path, + capfd, + request, + premerge_segment_count, + force_merge_segment_count, + expected_premerge_vector_counts, + expected_final_vector_counts, +): + dataset, query_ids = _case(ACCELERATED_HNSW_ALGORITHM, 128) + backend, index = _backend_and_index( + tmp_path, + ACCELERATED_HNSW_ALGORITHM, + ACCELERATED_HNSW_CODEC, + [{"num_candidates": 64}], + ) + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": premerge_segment_count, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": 1024, + } + ) + case_name = request.node.nodeid + capfd.readouterr() + + with lucene_log_case(case_name): + build = backend.build(dataset, [index], force=True) + assert build.success, build.error_message + + runtime = backend._get_runtime() + runtime.attach_current_thread() + directory = runtime.FSDirectory.open( + runtime.Paths.get(str(Path(index.file))) + ) + try: + physical_vector_counts = runtime._committed_segment_vector_counts( + directory + ) + finally: + directory.close() + + result = backend.search(dataset, [index], k=10, batch_size=2)[0] + + captured = capfd.readouterr() + _assert_accelerated_hnsw_build_diagnostics( + captured.out, captured.err, case_name + ) + performed_force_merge = ( + force_merge_segment_count == 1 and premerge_segment_count > 1 + ) + _assert_accelerated_hnsw_build_topology( + build, + physical_vector_counts, + expected_premerge_vector_counts=expected_premerge_vector_counts, + expected_final_vector_counts=expected_final_vector_counts, + requested_premerge_segment_count=premerge_segment_count, + requested_force_merge_segment_count=force_merge_segment_count, + performed_force_merge=performed_force_merge, + ) + + manifest = json.loads( + (Path(index.file) / ".cuvs-bench-lucene.json").read_text( + encoding="utf-8" + ) + ) + _assert_persisted_accelerated_hnsw_topology( + manifest, + build_params=build.build_params, + expected_premerge_vector_counts=expected_premerge_vector_counts, + expected_final_vector_counts=expected_final_vector_counts, + requested_force_merge_segment_count=force_merge_segment_count, + performed_force_merge=performed_force_merge, + ) + _assert_cpu_hnsw_search_quality( + result, query_ids, dataset.groundtruth_neighbors + ) + + +def test_cagra_controls_retain_direct_segments(tmp_path, capfd): + dataset, query_ids = _case(CAGRA_ALGORITHM, 128) + backend, index = _backend_and_index( + tmp_path, CAGRA_ALGORITHM, CAGRA_CODEC, [{}] + ) + index.build_param.update( + { + "premerge_segment_count": 4, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1024, + } + ) + + build = backend.build(dataset, [index], force=True) + assert build.success, build.error_message + result = backend.search(dataset, [index], k=10, batch_size=2)[0] + output = "\n".join(capfd.readouterr()) + + assert build.metadata["segment_count"] == 4 + assert build.metadata["premerge_segment_vector_counts"] == ( + "[256,256,256,256]" + ) + assert build.metadata["ram_per_thread_hard_limit_application"] == ( + "public_setter" + ) + assert build.metadata["ingest_merge_policy"] == "NoMergePolicy" + assert build.metadata["final_merge_policy"] is None + assert "falling back to a brute force index" not in output + for warning in _GRAPH_CLAMP_WARNINGS: + assert warning.casefold() not in output.casefold() + assert result.success, result.error_message + assert result.metadata["expected_search_route"] == "gpu_cagra" + np.testing.assert_array_equal(result.neighbors[:, 0], query_ids) + assert all(len(set(row)) == len(row) for row in result.neighbors.tolist()) + assert ( + _recall(result.neighbors, dataset.groundtruth_neighbors) + >= _MINIMUM_RECALL + ) + + +@pytest.mark.parametrize( + ("algorithm", "codec", "search_params", "expected_search_route"), + ( + pytest.param( + ACCELERATED_HNSW_ALGORITHM, + ACCELERATED_HNSW_CODEC, + [{"num_candidates": 64}], + "cpu_hnsw", + id="gpu-cagra-built-hnsw", + ), + pytest.param( + CAGRA_ALGORITHM, + CAGRA_CODEC, + [{}], + "gpu_cagra", + id="gpu-cagra-search", + ), + ), +) +def test_cuvs_builds_apply_explicit_unsupported_ram_limit( + tmp_path, + capfd, + request, + algorithm, + codec, + search_params, + expected_search_route, +): + dataset, query_ids = _case(algorithm, 128) + backend, index = _backend_and_index( + tmp_path, algorithm, codec, search_params + ) + if algorithm == ACCELERATED_HNSW_ALGORITHM: + index.build_param.update({"m": 16, "beam_width": 80}) + index.build_param.update( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + } + ) + case_name = request.node.nodeid + capfd.readouterr() + + with lucene_log_case(case_name): + build = backend.build(dataset, [index], force=True) + assert build.success, build.error_message + result = backend.search(dataset, [index], k=10, batch_size=2)[0] + + captured = capfd.readouterr() + combined_output = captured.out + captured.err + if algorithm == ACCELERATED_HNSW_ALGORITHM: + _assert_accelerated_hnsw_build_diagnostics( + captured.out, captured.err, case_name + ) + else: + assert "falling back to a brute force index" not in combined_output + for warning in _GRAPH_CLAMP_WARNINGS: + assert warning.casefold() not in combined_output.casefold() + + assert build.metadata["segment_count"] == 1 + assert build.metadata["applied_ram_per_thread_hard_limit_mb"] == 6144 + assert build.metadata["ram_per_thread_hard_limit_application"] == ( + "unsupported_field_override" + ) + assert build.metadata["runtime_force_merge_seconds"] == 0.0 + assert result.success, result.error_message + assert result.metadata["expected_search_route"] == expected_search_route np.testing.assert_array_equal(result.neighbors[:, 0], query_ids) assert all(len(set(row)) == len(row) for row in result.neighbors.tolist()) assert ( @@ -572,6 +881,9 @@ def test_accelerated_hnsw_logs_cpu_fallback_when_gpu_is_hidden(tmp_path): build = backend.build(dataset, [index], force=True) assert build.success, build.error_message assert build.metadata["persisted_index_kind"] == "hnsw" + assert build.metadata["graph_degree_source"] == ( + "requested_hnsw_same_graph_footprint_derivation" + ) def test_public_cli_builds_searches_and_exports_cpu_hnsw(tmp_path): diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_large_segment_python_integration.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_large_segment_python_integration.py new file mode 100644 index 0000000000..08fde414da --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_large_segment_python_integration.py @@ -0,0 +1,497 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# + +"""Explicitly selected 3 GiB tests for PyLucene document ingestion.""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import shutil +from collections.abc import Iterator +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +import pytest + +from _lucene_log_capture import ( + case_used_cpu_hnsw_fallback, + lucene_case_output, + lucene_log_case, +) +from cuvs_bench.backends._lucene_runtime import ( + ACCELERATED_HNSW_CODEC, + CAGRA_CODEC, +) +from cuvs_bench.backends._lucene_runtime_config import maven_artifact_version +from cuvs_bench.backends.base import Dataset +from cuvs_bench.backends.lucene import ( + ACCELERATED_HNSW_ALGORITHM, + CAGRA_ALGORITHM, + LuceneBackend, +) +from cuvs_bench.orchestrator.config_loaders import IndexConfig + +pytestmark = [ + pytest.mark.lucene_e2e, + pytest.mark.lucene_large_segment_e2e, + pytest.mark.filterwarnings( + "ignore:builtin type .* has no __module__ attribute:DeprecationWarning" + ), +] + +_VECTOR_COUNT = 786_432 +_DIMENSIONS = 1_024 +_CHUNK_ROWS = 2_048 +_TOP_K = 10 +_HEADER_BYTES = 8 +_GIB = 1024**3 +_PAYLOAD_BYTES = _VECTOR_COUNT * _DIMENSIONS * np.dtype(np.float32).itemsize +_MINIMUM_RECALL = 0.75 +_MINIMUM_FREE_DISK_BYTES = 16 * _GIB +_MINIMUM_AVAILABLE_MEMORY_BYTES = 16 * _GIB +_ARTIFACT_VERSION = maven_artifact_version() +_MANIFEST_FILE = ".cuvs-bench-lucene.json" +_JVM_ARGS = ("-Xmx12g",) +_GRAPH_CLAMP_WARNINGS = ( + "Intermediate graph degree cannot be larger", + "cannot be larger than intermediate graph degree", + "for nn-descent needs to match cagra intermediate graph degree", +) + +assert _PAYLOAD_BYTES == 3 * _GIB + + +@dataclass(frozen=True) +class _LargeFbinCase: + path: Path + query: np.ndarray + neighbors: np.ndarray + squared_distances: np.ndarray + payload_sha256: str + + +def _available_memory_bytes() -> int | None: + meminfo = Path("/proc/meminfo") + if not meminfo.is_file(): + return None + for line in meminfo.read_text(encoding="utf-8").splitlines(): + if line.startswith("MemAvailable:"): + return int(line.split()[1]) * 1024 + return None + + +def _require_generation_resources(directory: Path) -> None: + if os.environ.get("PYTEST_XDIST_WORKER"): + pytest.fail( + "The large Lucene suite must run without pytest-xdist so its " + "3 GiB fixture is generated only once" + ) + free_disk = shutil.disk_usage(directory).free + if free_disk < _MINIMUM_FREE_DISK_BYTES: + pytest.fail( + "The explicitly selected large Lucene suite needs at least " + f"{_MINIMUM_FREE_DISK_BYTES // _GIB} GiB of free temporary " + f"storage; found {free_disk / _GIB:.1f} GiB. Select a larger " + "local filesystem with pytest --basetemp." + ) + available_memory = _available_memory_bytes() + if ( + available_memory is not None + and available_memory < _MINIMUM_AVAILABLE_MEMORY_BYTES + ): + pytest.fail( + "The explicitly selected PyLucene document-ingest suite needs " + "at least " + f"{_MINIMUM_AVAILABLE_MEMORY_BYTES // _GIB} GiB of available " + f"host memory; found {available_memory / _GIB:.1f} GiB" + ) + + +def _nearest_neighbors( + best_ids: np.ndarray, + best_distances: np.ndarray, + candidate_ids: np.ndarray, + candidate_distances: np.ndarray, +) -> tuple[np.ndarray, np.ndarray]: + ids = np.concatenate((best_ids, candidate_ids)) + distances = np.concatenate((best_distances, candidate_distances)) + order = np.lexsort((ids, distances))[:_TOP_K] + return ids[order], distances[order] + + +def _write_large_fbin(path: Path) -> _LargeFbinCase: + rng = np.random.default_rng(2624) + query = np.zeros(_DIMENSIONS, dtype=np.float32) + best_ids = np.empty(0, dtype=np.int64) + best_distances = np.empty(0, dtype=np.float32) + digest = hashlib.sha256() + + with path.open("wb") as stream: + stream.write( + np.asarray([_VECTOR_COUNT, _DIMENSIONS], dtype="i", delta, delta, dtype=np.float32 + ) + candidate_ids = np.arange(start, start + row_count, dtype=np.int64) + best_ids, best_distances = _nearest_neighbors( + best_ids, + best_distances, + candidate_ids, + squared_distances, + ) + + np.testing.assert_array_equal(best_ids, np.arange(_TOP_K)) + assert path.stat().st_size == _HEADER_BYTES + _PAYLOAD_BYTES + return _LargeFbinCase( + path=path, + query=query[np.newaxis, :], + neighbors=best_ids[np.newaxis, :].astype(np.int32), + squared_distances=best_distances[np.newaxis, :], + payload_sha256=digest.hexdigest(), + ) + + +def _backend_and_index( + root: Path, + algorithm: str, + codec: str, + search_params: list[dict[str, int]], +) -> tuple[LuceneBackend, IndexConfig]: + index_path = root / algorithm / "index" + return ( + LuceneBackend( + { + "name": algorithm, + "algo": algorithm, + "codec": codec, + "group": "large-python-ingest-test", + "index_root": str(index_path.parent), + "requires_cuvs": True, + "include_cuvs": True, + "jvm_args": list(_JVM_ARGS), + } + ), + IndexConfig( + name=algorithm, + algo=algorithm, + build_param={"codec": codec}, + search_params=search_params, + file=str(index_path), + ), + ) + + +@pytest.fixture(scope="module") +def _large_suite_root(tmp_path_factory) -> Path: + root = tmp_path_factory.mktemp("lucene-large-python-suite") + _require_generation_resources(root) + return root + + +@pytest.fixture(scope="module") +def _verified_gpu_runtime(_large_suite_root: Path) -> None: + """Fail before allocating 3 GiB when the live cuVS path is unavailable.""" + root = _large_suite_root / "runtime-check" + root.mkdir() + backend, index = _backend_and_index( + root, CAGRA_ALGORITHM, CAGRA_CODEC, [{}] + ) + rng = np.random.default_rng(174) + vectors = rng.standard_normal((512, 128)).astype(np.float32) + dataset = Dataset( + name="lucene-large-runtime-check", + training_vectors=vectors, + query_vectors=vectors[:1].copy(), + distance_metric="euclidean", + ) + index.build_param.update( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + } + ) + try: + build = backend.build(dataset, [index], force=True) + assert build.success, build.error_message + assert build.metadata["persisted_index_kind"] == "gpu_cagra_only" + assert build.metadata["applied_ram_per_thread_hard_limit_mb"] == 6144 + assert build.metadata["ram_per_thread_hard_limit_application"] == ( + "unsupported_field_override" + ) + finally: + if Path(index.file).is_dir(): + shutil.rmtree(Path(index.file)) + + +@pytest.fixture(scope="module") +def large_fbin_case( + _large_suite_root: Path, + _verified_gpu_runtime, +) -> Iterator[_LargeFbinCase]: + root = _large_suite_root / "dataset" + root.mkdir() + path = root / "base.fbin" + partial = root / "base.fbin.partial" + try: + generated = _write_large_fbin(partial) + partial.replace(path) + yield _LargeFbinCase( + path=path, + query=generated.query, + neighbors=generated.neighbors, + squared_distances=generated.squared_distances, + payload_sha256=generated.payload_sha256, + ) + finally: + partial.unlink(missing_ok=True) + path.unlink(missing_ok=True) + + +def _assert_artifact_provenance(metadata: dict, role: str) -> None: + expected = { + "cuvs_java": f"com.nvidia.cuvs:cuvs-java:{_ARTIFACT_VERSION}", + "cuvs_lucene": ( + f"com.nvidia.cuvs.lucene:cuvs-lucene:{_ARTIFACT_VERSION}" + ), + } + for artifact, coordinates in expected.items(): + assert metadata[f"{role}_{artifact}_coordinates"] == coordinates + jar_path = Path(metadata[f"{role}_{artifact}_jar_path"]) + assert jar_path.is_file(), f"Missing {role} {artifact} JAR: {jar_path}" + digest = metadata[f"{role}_{artifact}_jar_sha256"] + assert re.fullmatch(r"[0-9a-f]{64}", digest), ( + f"Invalid {role} {artifact} SHA-256: {digest!r}" + ) + + +def _recall(actual: np.ndarray, expected: np.ndarray) -> float: + return float( + np.mean( + [ + len(set(row).intersection(truth)) / expected.shape[1] + for row, truth in zip(actual, expected) + ] + ) + ) + + +def _squared_distances_for_hits( + source: Path, query: np.ndarray, document_ids: np.ndarray +) -> np.ndarray: + distances = [] + with source.open("rb") as stream: + for document_id in document_ids: + stream.seek( + _HEADER_BYTES + + int(document_id) + * _DIMENSIONS + * np.dtype(np.float32).itemsize + ) + vector = np.frombuffer( + stream.read(_DIMENSIONS * np.dtype(np.float32).itemsize), + dtype=np.float32, + ) + if vector.size != _DIMENSIONS: + raise AssertionError( + f"FBIN row {int(document_id)} is unexpectedly truncated" + ) + delta = vector - query + distances.append(float(np.dot(delta, delta))) + return np.asarray([distances], dtype=np.float32) + + +@pytest.mark.parametrize( + ("algorithm", "codec", "search_params", "expected_search_route"), + ( + pytest.param( + ACCELERATED_HNSW_ALGORITHM, + ACCELERATED_HNSW_CODEC, + [{"num_candidates": 256}], + "cpu_hnsw", + id="cagra-hnsw-single-3gib-segment", + ), + pytest.param( + CAGRA_ALGORITHM, + CAGRA_CODEC, + [{}], + "gpu_cagra", + id="cagra-gpu-single-3gib-segment", + ), + ), +) +def test_pylucene_document_ingest_builds_one_segment_above_two_gibibytes( + tmp_path: Path, + capfd, + request, + large_fbin_case: _LargeFbinCase, + algorithm: str, + codec: str, + search_params: list[dict[str, int]], + expected_search_route: str, +) -> None: + mapping = np.memmap( + large_fbin_case.path, + mode="r", + dtype=np.float32, + offset=_HEADER_BYTES, + shape=(_VECTOR_COUNT, _DIMENSIONS), + ) + array_view = np.asarray(mapping) + assert mapping.flags.c_contiguous + assert np.shares_memory(mapping, array_view) + assert np.shares_memory(mapping, np.ascontiguousarray(array_view)) + dataset = Dataset( + name=f"lucene-large-python-ingest-{algorithm}", + training_vectors=mapping, + query_vectors=large_fbin_case.query, + groundtruth_neighbors=large_fbin_case.neighbors, + groundtruth_distances=large_fbin_case.squared_distances, + distance_metric="euclidean", + base_file=str(large_fbin_case.path), + ) + backend, index = _backend_and_index( + tmp_path, algorithm, codec, search_params + ) + if algorithm == ACCELERATED_HNSW_ALGORITHM: + index.build_param.update({"m": 16, "beam_width": 80}) + index.build_param.update( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + } + ) + case_name = request.node.nodeid + capfd.readouterr() + mapping_is_open = True + + try: + with lucene_log_case(case_name): + build = backend.build(dataset, [index], force=True) + assert build.success, build.error_message + dataset.training_vectors = None + mapping._mmap.close() + mapping_is_open = False + [result] = backend.search(dataset, [index], k=_TOP_K, batch_size=1) + + captured = capfd.readouterr() + combined_output = captured.out + captured.err + if algorithm == ACCELERATED_HNSW_ALGORITHM: + assert not case_used_cpu_hnsw_fallback(captured.err, case_name), ( + f"{case_name} used Lucene's CPU HNSW writer fallback:\n" + f"{lucene_case_output(captured.err, case_name)}" + ) + assert "falling back to a brute force index" not in combined_output + for warning in _GRAPH_CLAMP_WARNINGS: + assert warning.casefold() not in combined_output.casefold() + + assert large_fbin_case.path.stat().st_size == ( + _HEADER_BYTES + _PAYLOAD_BYTES + ) + assert build.index_size_bytes > 2 * _GIB + assert build.metadata["codec"] == codec + assert build.metadata["persisted_index_kind"] == ( + "hnsw" + if algorithm == ACCELERATED_HNSW_ALGORITHM + else "gpu_cagra_only" + ) + assert build.metadata["build_route_policy"] == ( + "gpu_cagra_or_cpu_hnsw_fallback" + if algorithm == ACCELERATED_HNSW_ALGORITHM + else "gpu_cagra" + ) + assert build.metadata["segment_count"] == 1 + assert build.metadata["field_count"] == 1 + assert build.metadata["vector_count"] == _VECTOR_COUNT + assert build.metadata["dimensions"] == _DIMENSIONS + assert build.metadata["requested_premerge_segment_count"] == 1 + assert build.metadata["observed_premerge_segment_count"] == 1 + assert build.metadata["requested_force_merge_segment_count"] == 0 + assert build.metadata["premerge_segment_vector_counts"] == ( + f"[{_VECTOR_COUNT}]" + ) + assert build.metadata["max_buffered_docs"] == _VECTOR_COUNT + 1 + assert build.metadata["runtime_force_merge_seconds"] == 0.0 + assert build.metadata["ingest_merge_policy"] == "NoMergePolicy" + assert build.metadata["final_merge_policy"] is None + assert build.metadata["applied_ram_per_thread_hard_limit_mb"] == 6144 + assert build.metadata["ram_per_thread_hard_limit_application"] == ( + "unsupported_field_override" + ) + _assert_artifact_provenance(build.metadata, "build_runtime") + + manifest = json.loads( + (Path(index.file) / _MANIFEST_FILE).read_text(encoding="utf-8") + ) + assert manifest["schema_version"] == 4 + assert manifest["dataset"]["vector_count"] == _VECTOR_COUNT + assert manifest["dataset"]["dimensions"] == _DIMENSIONS + assert manifest["dataset"]["sha256"] == ( + large_fbin_case.payload_sha256 + ) + assert manifest["dataset"]["source"]["size"] == ( + _HEADER_BYTES + _PAYLOAD_BYTES + ) + assert manifest["segment_count"] == 1 + + assert result.success, result.error_message + assert ( + result.metadata["expected_search_route"] == expected_search_route + ) + assert result.neighbors[0, 0] == 0 + assert len(set(result.neighbors[0].tolist())) == _TOP_K + assert np.all(result.neighbors >= 0) + assert np.all(result.neighbors < _VECTOR_COUNT) + assert ( + _recall(result.neighbors, large_fbin_case.neighbors) + >= _MINIMUM_RECALL + ) + exact_distances = _squared_distances_for_hits( + large_fbin_case.path, + large_fbin_case.query[0], + result.neighbors[0], + ) + np.testing.assert_allclose( + result.distances, exact_distances, rtol=1e-4, atol=1e-4 + ) + _assert_artifact_provenance(result.metadata, "build_runtime") + _assert_artifact_provenance(result.metadata, "search_runtime") + finally: + dataset.training_vectors = None + if mapping_is_open: + mapping._mmap.close() + if Path(index.file).is_dir(): + shutil.rmtree(Path(index.file)) diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_lifecycle.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_lifecycle.py index d959af677f..dcba1c876f 100644 --- a/python/cuvs_bench/cuvs_bench/tests/test_lucene_lifecycle.py +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_lifecycle.py @@ -22,6 +22,7 @@ from cuvs_bench.backends._lucene_runtime import CPU_HNSW_CODEC from cuvs_bench.backends.base import Dataset from cuvs_bench.backends.lucene import ( + ACCELERATED_HNSW_ALGORITHM, CAGRA_ALGORITHM, CPU_HNSW_ALGORITHM, LuceneBackend, @@ -31,6 +32,9 @@ from cuvs_bench.orchestrator.config_loaders import IndexConfig +_MISSING_RAM_LIMIT_APPLICATION = object() + + def test_index_prewarm_reads_every_regular_file(tmp_path: Path) -> None: (tmp_path / "segments_1").write_bytes(b"segments") (tmp_path / "vectors.vec").write_bytes(b"vector-data") @@ -175,6 +179,126 @@ def test_failed_force_rebuild_preserves_the_previous_valid_index( assert list(path.parent.glob(f".{path.name}.build-*")) == [] +def test_failed_build_reports_staging_that_cleanup_could_not_remove( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, CPU_HNSW_ALGORITHM, runtime + ) + assert backend.build(_dataset(), [index]).success + destination = Path(index.file) + original_manifest = (destination / ".cuvs-bench-lucene.json").read_bytes() + original_payload = (destination / "segments.fake").read_bytes() + + def fail_after_writing_partial_index( + index_path: Path, *_args, **_kwargs + ) -> None: + (index_path / "partial-segment").write_text( + "incomplete", encoding="utf-8" + ) + raise RuntimeError("replacement build failed") + + orphaned_paths: list[Path] = [] + + def fail_cleanup(path: Path) -> None: + orphaned_paths.append(path) + raise OSError("cleanup denied") + + monkeypatch.setattr( + runtime, "build_index", fail_after_writing_partial_index + ) + monkeypatch.setattr( + "cuvs_bench.backends.lucene.shutil.rmtree", fail_cleanup + ) + + replacement = backend.build(_dataset(offset=0.25), [index], force=True) + + assert not replacement.success + assert replacement.error_message.startswith( + "RuntimeError: replacement build failed" + ) + assert "Failed to discard the incomplete staged index" in ( + replacement.error_message + ) + assert "cleanup denied" in replacement.error_message + [orphaned] = orphaned_paths + assert str(orphaned) in replacement.error_message + assert (orphaned / "partial-segment").read_text(encoding="utf-8") == ( + "incomplete" + ) + assert (destination / ".cuvs-bench-lucene.json").read_bytes() == ( + original_manifest + ) + assert (destination / "segments.fake").read_bytes() == original_payload + + +def test_missing_failed_staging_is_already_discarded(tmp_path: Path) -> None: + build_error = RuntimeError("build failed") + + LuceneBackend._discard_failed_staging( + tmp_path / "already-renamed-staging", build_error + ) + + assert not hasattr(build_error, "__notes__") + + +def test_descendant_cleanup_race_does_not_hide_remaining_staging( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + staged = tmp_path / "staged" + staged.mkdir() + build_error = RuntimeError("build failed") + + def fail_for_missing_descendant(_path: Path) -> None: + raise FileNotFoundError("a staged child disappeared") + + monkeypatch.setattr( + "cuvs_bench.backends.lucene.shutil.rmtree", + fail_for_missing_descendant, + ) + + LuceneBackend._discard_failed_staging(staged, build_error) + + [note] = build_error.__notes__ + assert "Failed to discard the incomplete staged index" in note + assert str(staged) in note + assert "a staged child disappeared" in note + + +@pytest.mark.parametrize("control_error", (KeyboardInterrupt, SystemExit)) +def test_staging_cleanup_process_control_takes_precedence_over_build_failure( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + control_error: type[BaseException], +) -> None: + runtime = RecordingRuntime() + runtime.build_error = RuntimeError("build failed") + backend, index, _factory = _backend_and_index( + tmp_path, CPU_HNSW_ALGORITHM, runtime + ) + orphaned_paths: list[Path] = [] + + def interrupt_cleanup(path: Path) -> None: + orphaned_paths.append(path) + raise control_error("cleanup interrupted") + + monkeypatch.setattr( + "cuvs_bench.backends.lucene.shutil.rmtree", interrupt_cleanup + ) + + with pytest.raises(control_error, match="cleanup interrupted") as failure: + backend.build(_dataset(), [index]) + + [orphaned] = orphaned_paths + [note] = failure.value.__notes__ + assert ( + "Index construction first failed: RuntimeError: build failed" in note + ) + assert str(orphaned) in note + assert orphaned.is_dir() + + def test_backup_cleanup_failure_does_not_report_a_published_index_as_failed( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -528,7 +652,161 @@ def test_search_rejects_malformed_build_runtime_provenance( assert runtime.search_calls == [] -def test_search_requests_rebuild_for_the_legacy_stored_id_schema( +def test_manifest_build_parameters_must_match_the_requested_index( + tmp_path: Path, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + dataset = _dataset() + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + assert backend.build(dataset, [index]).success + + index.build_param["premerge_segment_count"] = 5 + reuse = backend.build(dataset, [index]) + [search] = backend.search(dataset, [index], k=2) + + assert not reuse.success + assert not search.success + assert "does not match this dataset and configuration" in ( + reuse.error_message + ) + assert "does not match this dataset and configuration" in ( + search.error_message + ) + assert len(runtime.build_calls) == 1 + assert runtime.search_calls == [] + + +def test_reuse_rejects_runtime_topology_evidence_that_conflicts_with_request( + tmp_path: Path, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + dataset = _dataset() + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + assert backend.build(dataset, [index]).success + manifest_path = Path(index.file) / ".cuvs-bench-lucene.json" + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + manifest["runtime_build_topology"]["observed_premerge_segment_count"] = 3 + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + + reuse = backend.build(dataset, [index]) + [search] = backend.search(dataset, [index], k=2) + + assert not reuse.success + assert not search.success + assert "invalid runtime build topology" in reuse.error_message + assert "invalid runtime build topology" in search.error_message + assert len(runtime.build_calls) == 1 + assert runtime.search_calls == [] + + +@pytest.mark.parametrize( + ("hard_limit", "recorded_mode"), + ( + pytest.param( + 1945, "unsupported_field_override", id="wrong-supported-mode" + ), + pytest.param(6144, "public_setter", id="wrong-unsupported-mode"), + pytest.param(1945, "unrecognized_mode", id="unknown-mode"), + pytest.param(1945, [], id="non-string-mode"), + pytest.param(1945, _MISSING_RAM_LIMIT_APPLICATION, id="missing-mode"), + ), +) +def test_reuse_rejects_incorrect_ram_limit_application_evidence( + tmp_path: Path, hard_limit: int, recorded_mode: object +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + dataset = _dataset() + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + } + ) + if hard_limit >= 2048: + index.build_param["allow_unsupported_lucene_ram_limit"] = True + assert backend.build(dataset, [index]).success + manifest_path = Path(index.file) / ".cuvs-bench-lucene.json" + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + if recorded_mode is _MISSING_RAM_LIMIT_APPLICATION: + manifest["runtime_build_topology"].pop( + "ram_per_thread_hard_limit_application" + ) + else: + manifest["runtime_build_topology"][ + "ram_per_thread_hard_limit_application" + ] = recorded_mode + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + + reuse = backend.build(dataset, [index]) + [search] = backend.search(dataset, [index], k=2) + + assert not reuse.success + assert not search.success + assert "invalid runtime build topology" in reuse.error_message + assert "invalid runtime build topology" in search.error_message + assert len(runtime.build_calls) == 1 + assert runtime.search_calls == [] + + +def test_search_rejects_noncanonical_manifest_build_parameters( + tmp_path: Path, +) -> None: + runtime = RecordingRuntime() + backend, index, _factory = _backend_and_index( + tmp_path, ACCELERATED_HNSW_ALGORITHM, runtime + ) + dataset = _dataset() + index.build_param.update( + { + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + ) + assert backend.build(dataset, [index]).success + manifest_path = Path(index.file) / ".cuvs-bench-lucene.json" + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + manifest["build_parameters"].pop("ram_per_thread_hard_limit_mb") + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + + [result] = backend.search(dataset, [index], k=2) + + assert not result.success + assert "invalid build parameters" in result.error_message + assert runtime.search_calls == [] + + +def test_search_requests_rebuild_for_the_previous_manifest_schema( tmp_path: Path, ) -> None: runtime = RecordingRuntime() @@ -539,7 +817,7 @@ def test_search_requests_rebuild_for_the_legacy_stored_id_schema( assert backend.build(dataset, [index]).success manifest_path = Path(index.file) / ".cuvs-bench-lucene.json" manifest = json.loads(manifest_path.read_text(encoding="utf-8")) - manifest["schema_version"] = 1 + manifest["schema_version"] = 3 manifest_path.write_text(json.dumps(manifest), encoding="utf-8") result = backend.search(dataset, [index], k=2)[0] diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_results_config.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_results_config.py index 6edbdc8466..6bb164adcd 100644 --- a/python/cuvs_bench/cuvs_bench/tests/test_lucene_results_config.py +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_results_config.py @@ -31,6 +31,8 @@ CPU_HNSW_ALGORITHM, LuceneBackend, LuceneConfigLoader, + _build_parameter_metadata, + _build_parameters_for, _codec_for, _score_to_squared_euclidean, _search_parameters, @@ -91,6 +93,446 @@ def test_build_parameters_accept_only_the_fixed_codec( _codec_for(algorithm, {"codec": codec, "graph_degree": 32}) +def test_accelerated_hnsw_build_parameters_are_canonical() -> None: + assert _build_parameters_for(ACCELERATED_HNSW_ALGORITHM, {}) == { + "codec": ACCELERATED_HNSW_CODEC, + "m": 32, + "beam_width": 32, + } + assert _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + }, + ) == { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + } + + +def test_accelerated_hnsw_metadata_records_the_heuristic_derivation() -> None: + assert _build_parameter_metadata( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + }, + ) == { + "hnsw_m": 16, + "hnsw_beam_width": 80, + "hnsw_heuristic": "SAME_GRAPH_FOOTPRINT", + "graph_degree_source": ( + "requested_hnsw_same_graph_footprint_derivation" + ), + "graph_degree": 32, + "intermediate_graph_degree": 48, + } + + +def test_accelerated_hnsw_segment_topology_is_canonical_and_reported() -> None: + parameters = _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "ram_per_thread_hard_limit_mb": 1945, + "force_merge_segment_count": 1, + "beam_width": 80, + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 4, + "m": 16, + }, + ) + + assert list(parameters) == [ + "codec", + "m", + "beam_width", + "premerge_segment_count", + "force_merge_segment_count", + "ram_per_thread_hard_limit_mb", + ] + assert parameters == { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + assert _build_parameter_metadata( + ACCELERATED_HNSW_ALGORITHM, parameters + ) == { + "hnsw_m": 16, + "hnsw_beam_width": 80, + "hnsw_heuristic": "SAME_GRAPH_FOOTPRINT", + "graph_degree_source": ( + "requested_hnsw_same_graph_footprint_derivation" + ), + "graph_degree": 32, + "intermediate_graph_degree": 48, + "requested_premerge_segment_count": 4, + "requested_force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + "allow_unsupported_lucene_ram_limit": False, + } + + +def test_accelerated_hnsw_rejects_num_indexing_threads() -> None: + with pytest.raises( + ValueError, + match="Unsupported Lucene build parameters: num_indexing_threads", + ): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "num_indexing_threads": 4, + }, + ) + + +@pytest.mark.parametrize( + "missing", + ( + "premerge_segment_count", + "force_merge_segment_count", + "ram_per_thread_hard_limit_mb", + ), +) +def test_accelerated_hnsw_segment_topology_must_be_all_or_none( + missing: str, +) -> None: + topology = { + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + topology.pop(missing) + + with pytest.raises(ValueError, match="segment topology requires.*missing"): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + {"codec": ACCELERATED_HNSW_CODEC, **topology}, + ) + + +@pytest.mark.parametrize( + "value", + (0, -1, True, 1.5, "4", None), + ids=("zero", "negative", "boolean", "float", "string", "none"), +) +def test_accelerated_hnsw_rejects_invalid_premerge_segment_count( + value: Any, +) -> None: + topology: dict[str, Any] = { + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + } + topology["premerge_segment_count"] = value + + with pytest.raises( + ValueError, + match="Lucene premerge_segment_count must be a positive integer", + ): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + {"codec": ACCELERATED_HNSW_CODEC, **topology}, + ) + + +@pytest.mark.parametrize("hard_limit", (1, 2047)) +def test_accelerated_hnsw_accepts_supported_ram_limits( + hard_limit: int, +) -> None: + parameters = _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + }, + ) + + assert parameters["ram_per_thread_hard_limit_mb"] == hard_limit + + +@pytest.mark.parametrize( + "hard_limit", (0, 2_147_483_648, True, 1.5, "1945", None) +) +def test_accelerated_hnsw_rejects_out_of_range_ram_limits( + hard_limit: Any, +) -> None: + with pytest.raises( + ValueError, + match=( + r"ram_per_thread_hard_limit_mb must be an integer in " + r"\[1, 2147483647\]" + ), + ): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + }, + ) + + +@pytest.mark.parametrize( + ("hard_limit", "algorithm", "codec"), + ( + (2048, ACCELERATED_HNSW_ALGORITHM, ACCELERATED_HNSW_CODEC), + (6144, ACCELERATED_HNSW_ALGORITHM, ACCELERATED_HNSW_CODEC), + (2048, CAGRA_ALGORITHM, CAGRA_CODEC), + (6144, CAGRA_ALGORITHM, CAGRA_CODEC), + ), +) +def test_cuvs_builds_accept_explicit_unsupported_ram_limits( + hard_limit: int, algorithm: str, codec: str +) -> None: + parameters = _build_parameters_for( + algorithm, + { + "codec": codec, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + "allow_unsupported_lucene_ram_limit": True, + }, + ) + + assert parameters["ram_per_thread_hard_limit_mb"] == hard_limit + assert parameters["allow_unsupported_lucene_ram_limit"] is True + + +@pytest.mark.parametrize( + ("hard_limit", "allow_unsupported", "message"), + ( + (2048, False, "require allow_unsupported"), + (2047, True, "requires ram_per_thread_hard_limit_mb >= 2048"), + (6144, 1, "must be a boolean"), + ), +) +def test_cuvs_builds_reject_inconsistent_unsupported_limit_opt_in( + hard_limit: int, allow_unsupported: object, message: str +) -> None: + with pytest.raises(ValueError, match=message): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + "allow_unsupported_lucene_ram_limit": allow_unsupported, + }, + ) + + +def test_unsupported_ram_limit_disallows_force_merge() -> None: + with pytest.raises(ValueError, match="force_merge_segment_count"): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 1, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + }, + ) + + +@pytest.mark.parametrize( + "value", + (-1, 2, True, 1.5, "0", None), + ids=("negative", "above-one", "boolean", "float", "string", "none"), +) +def test_accelerated_hnsw_force_merge_must_be_zero_or_one( + value: Any, +) -> None: + with pytest.raises(ValueError, match=r"must be 0 \(disabled\) or 1"): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 2, + "force_merge_segment_count": value, + "ram_per_thread_hard_limit_mb": 1945, + }, + ) + + +@pytest.mark.parametrize("force_merge_segment_count", (0, 1)) +def test_accelerated_hnsw_accepts_supported_force_merge_modes( + force_merge_segment_count: int, +) -> None: + parameters = _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "premerge_segment_count": 4, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": 1945, + }, + ) + + assert parameters["force_merge_segment_count"] == force_merge_segment_count + + +@pytest.mark.parametrize("name", ("m", "beam_width")) +@pytest.mark.parametrize( + "value", + (0, 513, True, 1.5, "16", None), + ids=("below-min", "above-max", "boolean", "float", "string", "none"), +) +def test_accelerated_hnsw_rejects_invalid_build_parameters( + name: str, value: Any +) -> None: + with pytest.raises(ValueError, match=rf"Lucene {name} must be an integer"): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + name: value, + }, + ) + + +def test_accelerated_hnsw_rejects_removed_writer_threads_parameter() -> None: + with pytest.raises( + ValueError, + match=r"Unsupported Lucene build parameters: cuvs_writer_threads", + ): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + "cuvs_writer_threads": 16, + }, + ) + + +@pytest.mark.parametrize( + "parameters", + ( + {"m": 16}, + {"beam_width": 80}, + ), + ids=("missing-beam-width", "missing-m"), +) +def test_accelerated_hnsw_requires_the_parameter_pair( + parameters: dict[str, int], +) -> None: + with pytest.raises(ValueError, match="require both m and beam_width"): + _build_parameters_for( + ACCELERATED_HNSW_ALGORITHM, + {"codec": ACCELERATED_HNSW_CODEC, **parameters}, + ) + + +@pytest.mark.parametrize( + ("algorithm", "codec", "extra_parameters"), + ( + pytest.param( + CPU_HNSW_ALGORITHM, + CPU_HNSW_CODEC, + {"m": 16, "beam_width": 80}, + id="cpu-hnsw-quality", + ), + pytest.param( + CPU_HNSW_ALGORITHM, + CPU_HNSW_CODEC, + { + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + }, + id="cpu-segment-topology", + ), + pytest.param( + CPU_HNSW_ALGORITHM, + CPU_HNSW_CODEC, + {"cuvs_writer_threads": 16}, + id="cpu-cuvs-writer-threads", + ), + pytest.param( + CAGRA_ALGORITHM, + CAGRA_CODEC, + {"m": 16, "beam_width": 80}, + id="cagra-hnsw-quality", + ), + pytest.param( + CAGRA_ALGORITHM, + CAGRA_CODEC, + {"cuvs_writer_threads": 16}, + id="cagra-cuvs-writer-threads", + ), + ), +) +def test_algorithms_reject_parameters_they_do_not_own( + algorithm: str, codec: str, extra_parameters: dict[str, int] +) -> None: + with pytest.raises( + ValueError, match="Unsupported Lucene build parameters" + ): + _build_parameters_for( + algorithm, + {"codec": codec, **extra_parameters}, + ) + + +def test_cagra_segment_topology_is_canonical_and_reported() -> None: + parameters = _build_parameters_for( + CAGRA_ALGORITHM, + { + "codec": CAGRA_CODEC, + "premerge_segment_count": 4, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + }, + ) + + assert parameters == { + "codec": CAGRA_CODEC, + "premerge_segment_count": 4, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + } + assert _build_parameter_metadata(CAGRA_ALGORITHM, parameters) == { + "requested_premerge_segment_count": 4, + "requested_force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 1945, + "allow_unsupported_lucene_ram_limit": False, + } + + +def test_cagra_segment_topology_disallows_force_merge() -> None: + with pytest.raises( + ValueError, match="requires force_merge_segment_count=0" + ): + _build_parameters_for( + CAGRA_ALGORITHM, + { + "codec": CAGRA_CODEC, + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1945, + }, + ) + + @pytest.mark.parametrize( ("parameters", "k", "expected"), ( @@ -308,8 +750,13 @@ def test_config_loader_maps_each_algorithm_to_its_codec_and_requirement( "codec": CAGRA_CODEC } assert by_algorithm[ACCELERATED_HNSW_ALGORITHM].indexes[0].build_param == { - "codec": ACCELERATED_HNSW_CODEC + "codec": ACCELERATED_HNSW_CODEC, + "m": 32, + "beam_width": 32, } + assert by_algorithm[ACCELERATED_HNSW_ALGORITHM].index_name == ( + f"{ACCELERATED_HNSW_ALGORITHM}_test.m32.beam_width32" + ) assert ( by_algorithm[CPU_HNSW_ALGORITHM].backend_config["requires_cuvs"] is False @@ -339,6 +786,118 @@ def test_config_loader_maps_each_algorithm_to_its_codec_and_requirement( assert by_algorithm[CAGRA_ALGORITHM].backend_config["group"] == "test" +@pytest.mark.parametrize("force_merge_segment_count", (0, 1)) +def test_config_loader_uses_canonical_segment_topology_label_order( + tmp_path: Path, + force_merge_segment_count: int, +) -> None: + dataset_configuration = tmp_path / "datasets.yaml" + dataset_configuration.write_text( + "- name: tiny-l2\n distance: euclidean\n dims: 2\n", + encoding="utf-8", + ) + algorithm_configuration = tmp_path / "accelerated.yaml" + algorithm_configuration.write_text( + f"""\ +name: {ACCELERATED_HNSW_ALGORITHM} +groups: + segmented: + build: + ram_per_thread_hard_limit_mb: [1945] + force_merge_segment_count: [{force_merge_segment_count}] + beam_width: [80] + codec: ["{ACCELERATED_HNSW_CODEC}"] + premerge_segment_count: [4] + m: [16] + search: {{}} +""", + encoding="utf-8", + ) + + _dataset_config, [configuration] = LuceneConfigLoader().load( + dataset="tiny-l2", + dataset_path=str(tmp_path), + dataset_configuration=str(dataset_configuration), + algorithm_configuration=str(algorithm_configuration), + algorithms=ACCELERATED_HNSW_ALGORITHM, + groups="segmented", + ) + + expected_parameters = { + "codec": ACCELERATED_HNSW_CODEC, + "m": 16, + "beam_width": 80, + "premerge_segment_count": 4, + "force_merge_segment_count": force_merge_segment_count, + "ram_per_thread_hard_limit_mb": 1945, + } + expected_name = ( + f"{ACCELERATED_HNSW_ALGORITHM}_segmented" + ".m16.beam_width80.premerge_segment_count4" + f".force_merge_segment_count{force_merge_segment_count}" + ".ram_per_thread_hard_limit_mb1945" + ) + assert configuration.indexes[0].build_param == expected_parameters + assert configuration.index_name == expected_name + assert configuration.index_path == ( + tmp_path / "tiny-l2" / "index" / expected_name + ) + + +def test_config_loader_canonicalizes_cagra_unsupported_ram_limit( + tmp_path: Path, +) -> None: + dataset_configuration = tmp_path / "datasets.yaml" + dataset_configuration.write_text( + "- name: tiny-l2\n distance: euclidean\n dims: 2\n", + encoding="utf-8", + ) + algorithm_configuration = tmp_path / "cagra.yaml" + algorithm_configuration.write_text( + f"""\ +name: {CAGRA_ALGORITHM} +groups: + one_large_segment: + build: + allow_unsupported_lucene_ram_limit: [true] + ram_per_thread_hard_limit_mb: [6144] + force_merge_segment_count: [0] + codec: ["{CAGRA_CODEC}"] + premerge_segment_count: [1] + search: {{}} +""", + encoding="utf-8", + ) + + _dataset_config, [configuration] = LuceneConfigLoader().load( + dataset="tiny-l2", + dataset_path=str(tmp_path), + dataset_configuration=str(dataset_configuration), + algorithm_configuration=str(algorithm_configuration), + algorithms=CAGRA_ALGORITHM, + groups="one_large_segment", + ) + + expected_parameters = { + "codec": CAGRA_CODEC, + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + } + expected_name = ( + f"{CAGRA_ALGORITHM}_one_large_segment" + ".premerge_segment_count1.force_merge_segment_count0" + ".ram_per_thread_hard_limit_mb6144" + ".allow_unsupported_lucene_ram_limitTrue" + ) + assert configuration.indexes[0].build_param == expected_parameters + assert configuration.index_name == expected_name + assert configuration.index_path == ( + tmp_path / "tiny-l2" / "index" / expected_name + ) + + def test_config_loader_rejects_dataset_names_that_can_escape_the_root( tmp_path: Path, ) -> None: diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py index a31818ff7e..e32bb6aa91 100644 --- a/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py @@ -9,13 +9,18 @@ import zipfile from pathlib import Path from types import SimpleNamespace +from unittest.mock import Mock import numpy as np import pytest from cuvs_bench.backends import _lucene_runtime from cuvs_bench.backends._lucene_runtime import ( + ACCELERATED_HNSW_CODEC, + CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY, + INDEX_WRITER_CONFIG_RAM_LIMIT_BRIDGE, _CleanupStack, + _controlled_build_topology, _load_pylucene, _rollback_writer, _validate_artifacts, @@ -29,6 +34,240 @@ _ARTIFACT_VERSION = maven_artifact_version() +@pytest.mark.parametrize( + ( + "vector_count", + "premerge_segments", + "force_merge_segments", + "chunk_size", + ), + ( + (10_000_000, 1, 0, 10_000_000), + (10_000_000, 1, 1, 10_000_000), + (100_000_000, 4, 0, 25_000_000), + (100_000_000, 4, 1, 25_000_000), + ), +) +def test_controlled_topology_derives_equal_partition_plan( + vector_count: int, + premerge_segments: int, + force_merge_segments: int, + chunk_size: int, +) -> None: + topology = _controlled_build_topology( + { + "premerge_segment_count": premerge_segments, + "force_merge_segment_count": force_merge_segments, + "ram_per_thread_hard_limit_mb": 1_945, + }, + vector_count, + ) + + assert topology is not None + assert topology.premerge_segment_count == premerge_segments + assert topology.force_merge_segment_count == force_merge_segments + assert topology.ram_per_thread_hard_limit_mb == 1_945 + assert topology.allow_unsupported_lucene_ram_limit is False + assert topology.chunk_size == chunk_size + assert topology.max_buffered_docs == chunk_size + 1 + + +@pytest.mark.parametrize( + ("parameters", "vector_count", "message"), + ( + ( + {"premerge_segment_count": 4}, + 100, + "require all topology parameters", + ), + ( + { + "premerge_segment_count": True, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + "must be a positive integer", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": 2, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": True, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": None, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": "0", + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": 0.0, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": -1, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 100, + r"must be 0 \(disabled\) or 1", + ), + ( + { + "premerge_segment_count": 4, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 1_945, + }, + 101, + "is not divisible", + ), + ), +) +def test_controlled_topology_rejects_ambiguous_shapes( + parameters: dict[str, int], vector_count: int, message: str +) -> None: + with pytest.raises(RuntimeError, match=message): + _controlled_build_topology(parameters, vector_count) + + +@pytest.mark.parametrize("hard_limit", (1, 2047)) +def test_controlled_topology_accepts_supported_ram_limits( + hard_limit: int, +) -> None: + topology = _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + }, + 100, + ) + + assert topology is not None + assert topology.ram_per_thread_hard_limit_mb == hard_limit + + +@pytest.mark.parametrize("hard_limit", (2048, 6144)) +def test_controlled_topology_accepts_explicit_unsupported_ram_limits( + hard_limit: int, +) -> None: + topology = _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + "allow_unsupported_lucene_ram_limit": True, + }, + 100, + ) + + assert topology is not None + assert topology.ram_per_thread_hard_limit_mb == hard_limit + assert topology.allow_unsupported_lucene_ram_limit is True + + +@pytest.mark.parametrize("hard_limit", (0, 2_147_483_648, True)) +def test_controlled_topology_rejects_out_of_range_ram_limits( + hard_limit: object, +) -> None: + with pytest.raises(RuntimeError, match="ram_per_thread_hard_limit_mb"): + _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": hard_limit, + }, + 100, + ) + + +def test_controlled_topology_rejects_unsupported_limit_without_opt_in() -> ( + None +): + with pytest.raises(RuntimeError, match="require.*allow_unsupported"): + _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 2048, + }, + 100, + ) + + +def test_controlled_topology_rejects_unused_unsupported_limit_opt_in() -> None: + with pytest.raises(RuntimeError, match="requires.*>= 2048"): + _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 2047, + "allow_unsupported_lucene_ram_limit": True, + }, + 100, + ) + + +def test_controlled_topology_rejects_merge_with_unsupported_limit() -> None: + with pytest.raises(RuntimeError, match="force_merge_segment_count"): + _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 1, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": True, + }, + 100, + ) + + +@pytest.mark.parametrize("allow_unsupported", (1, "true", None)) +def test_controlled_topology_rejects_non_boolean_unsupported_limit_opt_in( + allow_unsupported: object, +) -> None: + with pytest.raises(RuntimeError, match="must be a boolean"): + _controlled_build_topology( + { + "premerge_segment_count": 1, + "force_merge_segment_count": 0, + "ram_per_thread_hard_limit_mb": 6144, + "allow_unsupported_lucene_ram_limit": allow_unsupported, + }, + 100, + ) + + def _properties(group: str, artifact: str, version: str) -> bytes: return ( f"groupId={group}\nartifactId={artifact}\nversion={version}\n" @@ -65,6 +304,15 @@ def _write_artifacts( archive.writestr( "com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.class", b"" ) + archive.writestr( + "com/nvidia/cuvs/lucene/IndexWriterConfigRAMLimitBridge.class", + b"", + ) + archive.writestr( + "com/nvidia/cuvs/lucene/" + "Lucene101AcceleratedHNSWCodecFactory.class", + b"", + ) archive.writestr( "com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodec.class", b"" ) @@ -161,6 +409,322 @@ def cast_(_instance): assert isinstance(failure.value.__cause__, TypeError) +class _JavaInteger: + def __init__(self, value: int) -> None: + self.value = value + + @classmethod + def valueOf(cls, value: int) -> "_JavaInteger": + return cls(value) + + @classmethod + def cast_(cls, value: object) -> "_JavaInteger": + assert isinstance(value, cls) + return value + + def intValue(self) -> int: + return self.value + + +class _JavaBoolean: + def __init__(self, value: bool) -> None: + self.value = value + + @classmethod + def valueOf(cls, value: bool) -> "_JavaBoolean": + return cls(value) + + def booleanValue(self) -> bool: + return self.value + + +class _JavaString: + def __init__(self, value: str) -> None: + self.value = value + + @classmethod + def cast_(cls, value: object) -> "_JavaString": + assert isinstance(value, cls) + return value + + def __str__(self) -> str: + return self.value + + +class _JavaHashMap(dict): + def put(self, key: str, value: object) -> object | None: + previous = self.get(key) + self[key] = value + return previous + + +class _ConfiguredCodec: + @staticmethod + def getName() -> str: + return ACCELERATED_HNSW_CODEC + + @staticmethod + def knnVectorsFormat() -> object: + return object() + + +def _configured_codec_runtime( + *, factory_error: Exception | None = None +) -> tuple[LuceneRuntime, list[dict[str, int]], list[str]]: + requests: list[dict[str, int]] = [] + class_names: list[str] = [] + + class Factory: + def apply(self, request: _JavaHashMap) -> _JavaHashMap: + values = { + name: _JavaInteger.cast_(value).intValue() + for name, value in request.items() + } + requests.append(values) + if factory_error is not None: + raise factory_error + return _JavaHashMap( + { + "codec": _ConfiguredCodec(), + "max_conn": _JavaInteger(values["max_conn"]), + "beam_width": _JavaInteger(values["beam_width"]), + } + ) + + class ReflectedFactory: + @staticmethod + def newInstance() -> Factory: + return Factory() + + class JavaClass: + @staticmethod + def forName(name: str): + class_names.append(name) + return ReflectedFactory() + + class FunctionBinding: + @staticmethod + def cast_(factory: object) -> object: + return factory + + class MapBinding: + @staticmethod + def cast_(response: object) -> object: + return response + + class CodecBinding: + @staticmethod + def cast_(codec: object): + return codec + + @staticmethod + def availableCodecs(): + raise AssertionError("configured codec must not use Lucene SPI") + + @staticmethod + def forName(_name: str): + raise AssertionError("configured codec must not use Lucene SPI") + + runtime = object.__new__(LuceneRuntime) + runtime.Class = JavaClass + runtime.Function = FunctionBinding + runtime.HashMap = _JavaHashMap + runtime.Integer = _JavaInteger + runtime.Map = MapBinding + runtime.Codec = CodecBinding + runtime._java_configured_codec_factory = None + runtime.attach_current_thread = lambda: None + return runtime, requests, class_names + + +def test_configured_hnsw_codec_uses_one_atomic_factory_request() -> None: + runtime, requests, class_names = _configured_codec_runtime() + + codec = runtime.resolve_configured_hnsw_codec(16, 80) + + assert isinstance(codec, _ConfiguredCodec) + assert requests == [{"max_conn": 16, "beam_width": 80}] + assert class_names == [CONFIGURED_ACCELERATED_HNSW_CODEC_FACTORY] + + +def test_configured_hnsw_codec_reports_factory_failure() -> None: + runtime, requests, _class_names = _configured_codec_runtime( + factory_error=RuntimeError("constructor failed") + ) + + with pytest.raises(RuntimeError, match="constructor failed"): + runtime.resolve_configured_hnsw_codec(16, 80) + + assert requests == [{"max_conn": 16, "beam_width": 80}] + + +class _WriterConfig: + def __init__(self, *, fail: bool = False, retain: bool = True) -> None: + self.fail = fail + self.retain = retain + self.limit = 1_945 + + def setRAMPerThreadHardLimitMB(self, value: int) -> None: + if self.fail: + raise ValueError("setter rejected value") + if self.retain: + self.limit = value + + def getRAMPerThreadHardLimitMB(self) -> int: + return self.limit + + +def _ram_limit_runtime() -> tuple[ + LuceneRuntime, list[dict[str, object]], list[str] +]: + requests: list[dict[str, object]] = [] + class_names: list[str] = [] + + class Bridge: + def apply(self, request: _JavaHashMap) -> _JavaHashMap: + config = request["config"] + hard_limit = _JavaInteger.cast_( + request["per_thread_hard_limit_mb"] + ).intValue() + allow_unsupported = request[ + "allow_unsupported_lucene_ram_limit" + ].booleanValue() + requests.append( + { + "config": config, + "per_thread_hard_limit_mb": hard_limit, + "allow_unsupported_lucene_ram_limit": allow_unsupported, + } + ) + config.limit = hard_limit + return _JavaHashMap( + { + "config": config, + "per_thread_hard_limit_mb": _JavaInteger(hard_limit), + "application_mode": _JavaString( + "unsupported_field_override" + ), + } + ) + + class ReflectedBridge: + @staticmethod + def newInstance() -> Bridge: + return Bridge() + + class JavaClass: + @staticmethod + def forName(name: str) -> ReflectedBridge: + class_names.append(name) + return ReflectedBridge() + + class FunctionBinding: + @staticmethod + def cast_(bridge: object) -> object: + return bridge + + class MapBinding: + @staticmethod + def cast_(response: object) -> object: + return response + + runtime = object.__new__(LuceneRuntime) + runtime.Boolean = _JavaBoolean + runtime.Class = JavaClass + runtime.Function = FunctionBinding + runtime.HashMap = _JavaHashMap + runtime.Integer = _JavaInteger + runtime.Map = MapBinding + runtime.String = _JavaString + runtime._java_ram_limit_bridge = None + return runtime, requests, class_names + + +@pytest.mark.parametrize("hard_limit", (1, 2047)) +def test_ram_per_thread_hard_limit_uses_supported_public_setter( + hard_limit: int, +) -> None: + config = _WriterConfig() + + application_mode = LuceneRuntime._set_ram_per_thread_hard_limit_mb( + object.__new__(LuceneRuntime), + config, + hard_limit, + allow_unsupported=False, + ) + + assert config.getRAMPerThreadHardLimitMB() == hard_limit + assert application_mode == "public_setter" + + +@pytest.mark.parametrize("hard_limit", (2048, 6144)) +def test_ram_per_thread_hard_limit_uses_explicit_unsupported_bridge( + hard_limit: int, +) -> None: + runtime, requests, class_names = _ram_limit_runtime() + config = _WriterConfig() + + application_mode = runtime._set_ram_per_thread_hard_limit_mb( + config, hard_limit, allow_unsupported=True + ) + + assert config.getRAMPerThreadHardLimitMB() == hard_limit + assert application_mode == "unsupported_field_override" + assert requests == [ + { + "config": config, + "per_thread_hard_limit_mb": hard_limit, + "allow_unsupported_lucene_ram_limit": True, + } + ] + assert class_names == [INDEX_WRITER_CONFIG_RAM_LIMIT_BRIDGE] + + +@pytest.mark.parametrize("hard_limit", (0, 2_147_483_648, True)) +def test_ram_per_thread_hard_limit_rejects_out_of_range_values( + hard_limit: object, +) -> None: + with pytest.raises(RuntimeError, match=r"range \[1, 2147483647\]"): + LuceneRuntime._set_ram_per_thread_hard_limit_mb( + object.__new__(LuceneRuntime), + _WriterConfig(), + hard_limit, + allow_unsupported=False, + ) + + +def test_ram_per_thread_hard_limit_rejects_unsupported_value_without_opt_in() -> ( + None +): + with pytest.raises(RuntimeError, match="require.*allow_unsupported"): + LuceneRuntime._set_ram_per_thread_hard_limit_mb( + object.__new__(LuceneRuntime), + _WriterConfig(), + 2048, + allow_unsupported=False, + ) + + +def test_ram_per_thread_hard_limit_reports_setter_failure() -> None: + with pytest.raises(RuntimeError, match="setter rejected value"): + LuceneRuntime._set_ram_per_thread_hard_limit_mb( + object.__new__(LuceneRuntime), + _WriterConfig(fail=True), + 1_024, + allow_unsupported=False, + ) + + +def test_ram_per_thread_hard_limit_rejects_failed_readback() -> None: + with pytest.raises(RuntimeError, match="did not retain"): + LuceneRuntime._set_ram_per_thread_hard_limit_mb( + object.__new__(LuceneRuntime), + _WriterConfig(retain=False), + 1_024, + allow_unsupported=False, + ) + + def test_float32_vectors_are_converted_to_a_jcc_compatible_sequence() -> None: received = [] @@ -504,6 +1068,66 @@ def rollback() -> None: ] +def _controlled_runtime_with_writer(writer: Mock) -> LuceneRuntime: + """Isolate JVM setup while exercising the real write/merge failure paths.""" + runtime = object.__new__(LuceneRuntime) + runtime.IndexWriter = lambda _directory, _config: writer + runtime._controlled_ingest_config = lambda *args, **kwargs: ( + object(), + "public_setter", + ) + runtime._controlled_merge_config = lambda *args: ( + object(), + "public_setter", + ) + runtime._document = lambda _document_id, _vector: object() + return runtime + + +@pytest.mark.parametrize("failed_operation", ("addDocument", "flush")) +def test_controlled_ingestion_rolls_back_without_commit_on_failure( + failed_operation: str, +) -> None: + writer = Mock(spec=["addDocument", "flush", "commit", "close", "rollback"]) + original_error = RuntimeError(f"{failed_operation} failed") + getattr(writer, failed_operation).side_effect = original_error + runtime = _controlled_runtime_with_writer(writer) + + with pytest.raises(RuntimeError) as failure: + runtime._write_controlled_chunk( + object(), + np.zeros((1, 2), dtype=np.float32), + object(), + SimpleNamespace(), + start=0, + stop=1, + create=True, + ) + + assert failure.value is original_error + writer.rollback.assert_called_once_with() + writer.commit.assert_not_called() + writer.close.assert_not_called() + + +def test_controlled_merge_rolls_back_without_commit_on_failure() -> None: + writer = Mock(spec=["forceMerge", "commit", "close", "rollback"]) + original_error = RuntimeError("forceMerge failed") + writer.forceMerge.side_effect = original_error + runtime = _controlled_runtime_with_writer(writer) + + with pytest.raises(RuntimeError) as failure: + runtime._force_merge_controlled_index( + object(), object(), SimpleNamespace(force_merge_segment_count=1) + ) + + assert failure.value is original_error + writer.forceMerge.assert_called_once_with(1, True) + writer.rollback.assert_called_once_with() + writer.commit.assert_not_called() + writer.close.assert_not_called() + + def test_reused_jvm_reports_the_initialized_artifact_identity( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/python/cuvs_bench/pyproject.toml b/python/cuvs_bench/pyproject.toml index 2d2064b19e..b00b16933a 100644 --- a/python/cuvs_bench/pyproject.toml +++ b/python/cuvs_bench/pyproject.toml @@ -69,6 +69,7 @@ lucene = "cuvs_bench.backends.lucene:register" [tool.pytest.ini_options] markers = [ "lucene_e2e: live tests; all require PyLucene/Java, and GPU-intended cases also require cuVS/CUDA/GPU", + "lucene_large_segment_e2e: resource-intensive Lucene tests that build a segment from more than 2 GiB of vectors", "opensearch: tests that require a live OpenSearch node (run with '-m opensearch')", ]