From 79ebb4d6f94b3cc591486f047bc322c16c5c2a62 Mon Sep 17 00:00:00 2001 From: nvzm123 Date: Tue, 6 Oct 2026 04:23:29 +0000 Subject: [PATCH 1/2] Add PyLucene runtime integration --- .../recipes/cuvs-bench/build_pylucene_10_2.sh | 553 +++++++ .../recipes/cuvs-bench/pylucene-10.2.0.patch | 50 + fern/docs.yml | 2 + fern/pages/lucene_api/index.md | 1 + ...a-cuvs-lucene-indexsearchertimingbridge.md | 25 + .../lucene/IndexSearcherTimingBridge.java | 72 + .../lucene/TestIndexSearcherTimingBridge.java | 65 + .../cuvs_bench/backends/_lucene_runtime.py | 1424 +++++++++++++++++ .../backends/_lucene_runtime_config.py | 312 ++++ .../cuvs_bench/tests/test_lucene_runtime.py | 603 +++++++ .../tests/test_lucene_runtime_config.py | 309 ++++ .../cuvs_bench/tests/test_pylucene_builder.py | 76 + 12 files changed, 3492 insertions(+) create mode 100755 conda/recipes/cuvs-bench/build_pylucene_10_2.sh create mode 100644 conda/recipes/cuvs-bench/pylucene-10.2.0.patch create mode 100644 fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md create mode 100644 java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.java create mode 100644 java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexSearcherTimingBridge.java create mode 100644 python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py create mode 100644 python/cuvs_bench/cuvs_bench/backends/_lucene_runtime_config.py create mode 100644 python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py create mode 100644 python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime_config.py create mode 100644 python/cuvs_bench/cuvs_bench/tests/test_pylucene_builder.py diff --git a/conda/recipes/cuvs-bench/build_pylucene_10_2.sh b/conda/recipes/cuvs-bench/build_pylucene_10_2.sh new file mode 100755 index 0000000000..c17d80d004 --- /dev/null +++ b/conda/recipes/cuvs-bench/build_pylucene_10_2.sh @@ -0,0 +1,553 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +set -euo pipefail + +readonly PYLUCENE_VERSION="10.2.0" +readonly PYLUCENE_SCAFFOLD_VERSION="10.0.0" +readonly PYLUCENE_ARCHIVE_SHA256="100c3d61d6799ac16b7b8c1826cddf07fb1715141ebdb0d7b8119cdd96b24574" +readonly LUCENE_ARCHIVE_SHA512="2e8ad344631031416d96277abb844ca4d4c266086041f6b9339aade1ffb015875bdb34eeb3dfba2563648ce56f8fdafc52839b75ecf716036a3feaf1873fd02f" +readonly PYLUCENE_URL="https://archive.apache.org/dist/lucene/pylucene/pylucene-${PYLUCENE_SCAFFOLD_VERSION}-src.tar.gz" +readonly LUCENE_URL="https://archive.apache.org/dist/lucene/java/${PYLUCENE_VERSION}/lucene-${PYLUCENE_VERSION}-src.tgz" +readonly SOURCE_RECIPE_REVISION="1" +readonly NUM_GENERATED_FILES="18" +readonly UPSTREAM_MAKEFILE_SHA256="0bb9ce7d8e473eeebab264aa6dd9515149de67569c29aed2a67954a63c398f65" +readonly UPSTREAM_SETTINGS_GRADLE_SHA256="a799dbfac9fdda839d468fa8edaf2ea483bfa6957b7e29b3711b7f16d4fc7f04" +readonly UPSTREAM_GRADLE_WRAPPER_SHA256="2caec011a749b18ab9a7dea68bf7639a179a3b8316b0e35655d0a62a1d7390fd" +readonly PREPARED_MAKEFILE_SHA256="d9a20059f349f1d2eb346ce63f44b796f7c6b1f688d6c0dc61004ee5fecf45cf" +readonly PREPARED_SETTINGS_GRADLE_SHA256="bc0cbce02a431bca779901af419d62f058f5836d1b2fb2c9dede42f7bb8f3bca" +readonly PREPARED_GRADLE_WRAPPER_SHA256="f0b2060d483a15e115abfa1f99bc271a9a5e09be5c33f2714eda59a3d2e9d7cf" + +readonly SETUPTOOLS_VERSION="80.9.0" +readonly BUILD_VERSION="1.5.0" +readonly WHEEL_VERSION="0.47.0" +readonly PACKAGING_VERSION="26.3" +readonly PYPROJECT_HOOKS_VERSION="1.2.0" +readonly PYTEST_VERSION="9.1.1" +readonly INICONFIG_VERSION="2.3.0" +readonly PLUGGY_VERSION="1.6.0" +readonly PYGMENTS_VERSION="2.20.0" + +script_dir="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P)" +patch_file="$script_dir/pylucene-10.2.0.patch" +user_home="${HOME:-}" +if [[ -n "${XDG_CACHE_HOME:-}" ]]; then + default_cache_root="$XDG_CACHE_HOME" +elif [[ -n "$user_home" ]]; then + default_cache_root="$user_home/.cache" +else + default_cache_root="" +fi +build_root="${CUVS_BENCH_PYLUCENE_BUILD_ROOT:-${default_cache_root:+${default_cache_root}/cuvs-bench/pylucene-${PYLUCENE_VERSION}}}" +python_command="${PYTHON:-python3}" +prepare_only=false + +usage() { + cat <&2 + exit 1 +} + +validate_build_root() { + local candidate="$1" + [[ "$candidate" == /* ]] || fail "--build-root must be an absolute path: $candidate" + if [[ ! "$candidate" =~ ^/[A-Za-z0-9_./-]+$ ]]; then + fail "--build-root may contain only letters, digits, '/', '.', '_', and '-' because JCC and Make interpolate it into shell recipes" + fi +} + +while (($# > 0)); do + case "$1" in + --build-root) + (($# >= 2)) || fail "--build-root requires a path" + build_root="$2" + shift 2 + ;; + --python) + (($# >= 2)) || fail "--python requires an executable" + python_command="$2" + shift 2 + ;; + --prepare-only) + prepare_only=true + shift + ;; + -h | --help) + usage + exit 0 + ;; + *) + fail "unknown argument: $1" + ;; + esac +done + +[[ "$(uname -s)" == "Linux" ]] || fail "this helper currently supports Linux only" +[[ -f "$patch_file" ]] || fail "compatibility patch is missing: $patch_file" +[[ -n "$build_root" ]] || + fail "--build-root is required when HOME and XDG_CACHE_HOME are unset" +validate_build_root "$build_root" + +for command_name in awk cmp cp curl flock gzip mktemp mv patch rm sha256sum sha512sum tar uname; do + command -v "$command_name" >/dev/null || + fail "required command is missing: $command_name" +done + +mkdir -p -- "$build_root" +build_root="$(cd -- "$build_root" && pwd -P)" +validate_build_root "$build_root" +if [[ "$build_root" == "/" || (-n "$user_home" && "$build_root" == "$user_home") ]]; then + fail "refusing unsafe build root: $build_root" +fi + +exec {build_lock_fd}>"$build_root/.build.lock" +flock -n "$build_lock_fd" || fail "another PyLucene build is using $build_root" + +downloads_dir="$build_root/downloads" +source_dir="$build_root/src/pylucene-${PYLUCENE_VERSION}" +source_manifest="$source_dir/.cuvs-pylucene-source-manifest" +venv_dir="$build_root/venv" +build_manifest="$build_root/.cuvs-pylucene-build-manifest" +complete_marker="$build_root/.complete" +activation_file="$build_root/activate.sh" +mkdir -p -- "$downloads_dir" "$(dirname -- "$source_dir")" + +source_manifest_temp="" +build_manifest_temp="" +activation_temp="" +complete_temp="" +staging_dir="" +download_temp="" + +cleanup() { + [[ -z "$source_manifest_temp" ]] || rm -f -- "$source_manifest_temp" + [[ -z "$build_manifest_temp" ]] || rm -f -- "$build_manifest_temp" + [[ -z "$activation_temp" ]] || rm -f -- "$activation_temp" + [[ -z "$complete_temp" ]] || rm -f -- "$complete_temp" + [[ -z "$download_temp" ]] || rm -f -- "$download_temp" + [[ -z "$staging_dir" ]] || rm -rf -- "$staging_dir" +} +trap cleanup EXIT + +verify_checksum() { + local algorithm="$1" + local expected="$2" + local path="$3" + printf '%s %s\n' "$expected" "$path" | + "$algorithm" --check --status - +} + +download_verified() { + local url="$1" + local destination="$2" + local algorithm="$3" + local expected="$4" + + if [[ -f "$destination" ]]; then + verify_checksum "$algorithm" "$expected" "$destination" || + fail "cached download has the wrong checksum; remove $destination or choose a new --build-root" + return + fi + + download_temp="$(mktemp "${destination}.part.XXXXXX")" + printf 'Downloading %s\n' "$url" + if ! curl --fail --location --show-error --proto '=https' --proto-redir '=https' \ + --retry 3 --output "$download_temp" "$url"; then + fail "download failed: $url" + fi + verify_checksum "$algorithm" "$expected" "$download_temp" || + fail "downloaded archive checksum mismatch: $url" + mv -- "$download_temp" "$destination" + download_temp="" +} + +pylucene_archive="$downloads_dir/pylucene-${PYLUCENE_SCAFFOLD_VERSION}-src.tar.gz" +lucene_archive="$downloads_dir/lucene-${PYLUCENE_VERSION}-src.tgz" +download_verified \ + "$PYLUCENE_URL" "$pylucene_archive" sha256sum "$PYLUCENE_ARCHIVE_SHA256" +download_verified \ + "$LUCENE_URL" "$lucene_archive" sha512sum "$LUCENE_ARCHIVE_SHA512" + +patch_sha256="$(sha256sum "$patch_file" | awk '{print $1}')" +source_manifest_temp="$(mktemp "$build_root/source-manifest.XXXXXX")" +cat >"$source_manifest_temp" </dev/null || + fail "required command is missing: $command_name" +done + +python_path="$(command -v "$python_command" 2>/dev/null || true)" +[[ -n "$python_path" ]] || + fail "Python executable is unavailable: $python_command" +python_path="$(readlink -f "$python_path")" + +if ! python_details="$("$python_path" - <<'PY' +import platform +import struct +import sys +import sysconfig + +if sys.implementation.name != "cpython": + raise SystemExit("PyLucene requires CPython for this build") +if not ((3, 11) <= sys.version_info[:2] <= (3, 14)): + raise SystemExit( + "this helper supports CPython 3.11-3.14; " + f"found {platform.python_version()}" + ) +if struct.calcsize("P") != 8: + raise SystemExit("this helper requires a 64-bit CPython") +print(f"python_implementation={sys.implementation.name}") +print(f"python_version={platform.python_version()}") +print(f"python_cache_tag={sys.implementation.cache_tag}") +print(f"python_soabi={sysconfig.get_config_var('SOABI')}") +PY +)"; then + fail "the selected Python interpreter is incompatible: $python_path" +fi + +"$python_path" - <<'PY' +import pathlib +import sysconfig + +header = pathlib.Path(sysconfig.get_paths()["include"]) / "Python.h" +if not header.is_file(): + raise SystemExit(f"Python development header is missing: {header}") +PY + +java_home="${JAVA_HOME:-}" +if [[ -z "$java_home" ]]; then + javac_path="$(command -v javac 2>/dev/null || true)" + [[ -n "$javac_path" ]] || + fail "JAVA_HOME is unset and javac is unavailable; install JDK 22" + java_home="$(dirname -- "$(dirname -- "$(readlink -f "$javac_path")")")" +fi +[[ -d "$java_home" ]] || fail "JAVA_HOME is not a directory: $java_home" +java_home="$(cd -- "$java_home" && pwd -P)" + +for required_path in \ + "$java_home/bin/java" \ + "$java_home/bin/javac" \ + "$java_home/bin/javadoc" \ + "$java_home/include/jni.h" \ + "$java_home/include/linux/jni_md.h" \ + "$java_home/lib/libjava.so" \ + "$java_home/lib/server/libjvm.so" \ + "$java_home/release"; do + [[ -e "$required_path" ]] || + fail "JDK 22 component is missing: $required_path" +done + +java_properties="$("$java_home/bin/java" -XshowSettings:properties -version 2>&1)" +java_specification_version="$(awk -F'= ' '/java.specification.version/{print $2; exit}' <<<"$java_properties")" +java_runtime_version="$(awk -F'= ' '/java.runtime.version/{print $2; exit}' <<<"$java_properties")" +[[ "$java_specification_version" == "22" ]] || + fail "JDK 22 is required; found Java ${java_specification_version:-unknown} at $java_home" + +cc_command="${CC:-cc}" +cxx_command="${CXX:-c++}" +if [[ "$cc_command" =~ [[:space:]] || "$cxx_command" =~ [[:space:]] ]]; then + fail "CC and CXX must name executables without command-line arguments" +fi +cc_path="$(command -v "$cc_command" 2>/dev/null || true)" +cxx_path="$(command -v "$cxx_command" 2>/dev/null || true)" +[[ -n "$cc_path" ]] || fail "C compiler is unavailable: $cc_command" +[[ -n "$cxx_path" ]] || fail "C++ compiler is unavailable: $cxx_command" +cc_path="$(readlink -f "$cc_path")" +cxx_path="$(readlink -f "$cxx_path")" +cc_version="$("$cc_path" --version | awk 'NR == 1 {print; exit}')" +cxx_version="$("$cxx_path" --version | awk 'NR == 1 {print; exit}')" + +for unsupported_path in "$python_path" "$java_home" "$cc_path" "$cxx_path"; do + if [[ "$unsupported_path" =~ [[:space:]|] ]]; then + fail "toolchain paths cannot contain whitespace or '|': $unsupported_path" + fi +done + +script_sha256="$(sha256sum "$script_dir/build_pylucene_10_2.sh" | awk '{print $1}')" +source_manifest_sha256="$(sha256sum "$source_manifest" | awk '{print $1}')" +java_release_sha256="$(sha256sum "$java_home/release" | awk '{print $1}')" +make_version="$(make --version | awk 'NR == 1 {print; exit}')" + +build_manifest_temp="$(mktemp "$build_root/build-manifest.XXXXXX")" +cat >"$build_manifest_temp" </dev/null 2>&1 || + fail "Python venv support is unavailable for $python_path" + "$python_path" -m venv "$venv_dir" || + fail "could not create a virtual environment; install venv support for $python_path" + fi + + "$venv_dir/bin/python" -m pip install --disable-pip-version-check --no-input \ + "setuptools==$SETUPTOOLS_VERSION" \ + "build==$BUILD_VERSION" \ + "wheel==$WHEEL_VERSION" \ + "packaging==$PACKAGING_VERSION" \ + "pyproject-hooks==$PYPROJECT_HOOKS_VERSION" \ + "pytest==$PYTEST_VERSION" \ + "iniconfig==$INICONFIG_VERSION" \ + "pluggy==$PLUGGY_VERSION" \ + "Pygments==$PYGMENTS_VERSION" + + if ! "$venv_dir/bin/python" -c \ + 'import importlib.metadata as metadata; assert metadata.version("JCC") == "3.15"' \ + 2>/dev/null; then + ( + cd "$source_dir/jcc" + "$venv_dir/bin/python" -m build --wheel --no-isolation + ) + shopt -s nullglob + jcc_wheels=("$source_dir"/jcc/dist/[Jj][Cc][Cc]-3.15-*.whl) + shopt -u nullglob + ((${#jcc_wheels[@]} == 1)) || + fail "expected exactly one JCC 3.15 wheel" + "$venv_dir/bin/python" -m pip install --no-deps --force-reinstall \ + "${jcc_wheels[0]}" + fi + + make_variables=( + "PYTHON=$venv_dir/bin/python" + "JCC=$venv_dir/bin/python -m jcc --shared" + "NUM_FILES=$NUM_GENERATED_FILES" + "MODERN_PACKAGING=true" + ) + make -C "$source_dir" "${make_variables[@]}" all + make -C "$source_dir" "${make_variables[@]}" test +fi + +activation_temp="$(mktemp "$build_root/activate.XXXXXX")" +{ + printf '# Generated by %q. Source from Bash; do not execute.\n' "$0" + printf 'source %q\n' "$venv_dir/bin/activate" + printf 'export JAVA_HOME=%q\n' "$java_home" + # shellcheck disable=SC2016 # Expansion is intentionally deferred until source time. + printf 'export PATH=%q${PATH:+":$PATH"}\n' "$java_home/bin" +} >"$activation_temp" +mv -- "$activation_temp" "$activation_file" +activation_temp="" + +CUVS_PYLUCENE_EXPECTED_ROOT="$build_root" "$venv_dir/bin/python" -I - <<'PY' +import importlib.metadata as metadata +import os +from pathlib import Path + +import jcc +import lucene + +root = Path(os.environ["CUVS_PYLUCENE_EXPECTED_ROOT"]).resolve() +assert lucene.VERSION == "10.2.0", lucene.VERSION +assert metadata.version("JCC") == "3.15" +assert Path(lucene.__file__).resolve().is_relative_to(root) +assert Path(jcc.__file__).resolve().is_relative_to(root) +lucene.initVM( + vmargs=[ + "--add-modules=jdk.incubator.vector", + "--enable-native-access=ALL-UNNAMED", + ] +) +from java.lang import Class + +Class.forName("org.apache.lucene.codecs.lucene101.Lucene101Codec") +Class.forName( + "org.apache.lucene.codecs.lucene102.Lucene102HnswBinaryQuantizedVectorsFormat" +) +print(f"PyLucene {lucene.VERSION} JVM smoke test passed") +PY +"$venv_dir/bin/python" -m pip check + +shopt -s nullglob +pylucene_wheels=("${source_dir}/dist/lucene-${PYLUCENE_VERSION}-"*.whl) +jcc_wheels=("$source_dir"/jcc/dist/[Jj][Cc][Cc]-3.15-*.whl) +shopt -u nullglob +((${#pylucene_wheels[@]} == 1)) || + fail "expected exactly one PyLucene ${PYLUCENE_VERSION} wheel" +((${#jcc_wheels[@]} == 1)) || + fail "expected exactly one JCC 3.15 wheel" +printf 'PyLucene wheel: %s\n' "${pylucene_wheels[0]}" +sha256sum "${pylucene_wheels[0]}" +printf 'JCC wheel: %s\n' "${jcc_wheels[0]}" +sha256sum "${jcc_wheels[0]}" + +if [[ "$completed_before" == false ]]; then + complete_temp="$(mktemp "$build_root/complete.XXXXXX")" + printf '%s\n' "$build_manifest_sha256" >"$complete_temp" + mv -- "$complete_temp" "$complete_marker" + complete_temp="" +fi + +printf 'PyLucene %s is ready. Run: source %s\n' \ + "$PYLUCENE_VERSION" "$activation_file" diff --git a/conda/recipes/cuvs-bench/pylucene-10.2.0.patch b/conda/recipes/cuvs-bench/pylucene-10.2.0.patch new file mode 100644 index 0000000000..ea6ee3c9cd --- /dev/null +++ b/conda/recipes/cuvs-bench/pylucene-10.2.0.patch @@ -0,0 +1,50 @@ +--- a/Makefile ++++ b/Makefile +@@ -15,8 +15,8 @@ + # site-packages directory. + # + +-VERSION=10.0.0 +-LUCENE_VER=10.0.0 ++VERSION=10.2.0 ++LUCENE_VER=10.2.0 + PYLUCENE:=$(shell pwd) + LUCENE_SRC=lucene-java-$(LUCENE_VER) + LUCENE=$(LUCENE_SRC)/lucene +@@ -163,7 +163,7 @@ + STEMPEL_JAR=$(LUCENE)/analysis/stempel/build/runtimeJars/lucene-analysis-stempel-$(LUCENE_VER)-SNAPSHOT.jar + SUGGEST_JAR=$(LUCENE)/suggest/build/runtimeJars/lucene-suggest-$(LUCENE_VER)-SNAPSHOT.jar + +-ANTLR_JAR=$(LUCENE)/expressions/build/runtimeJars/antlr4-runtime-4.11.1.jar ++ANTLR_JAR=$(LUCENE)/expressions/build/runtimeJars/antlr4-runtime-4.13.2.jar + ASM_JAR=$(LUCENE)/expressions/build/runtimeJars/asm-9.6.jar + ASM_COMMONS_JAR=$(LUCENE)/expressions/build/runtimeJars/asm-commons-9.6.jar + +@@ -254,6 +254,7 @@ + --package org.antlr.v4.runtime.atn \ + --package org.apache.lucene.analysis.ko.dict \ + --package org.apache.lucene.analysis.ja.dict \ ++ --package org.apache.lucene.sandbox.facet.plain.histograms \ + --exclude org.apache.lucene.sandbox.queries.regex.JakartaRegexpCapabilities \ + --exclude org.apache.regexp.RegexpTunnel \ + --exclude org.apache.lucene.misc.store.WindowsDirectory \ +--- a/lucene-java-10.2.0/settings.gradle ++++ b/lucene-java-10.2.0/settings.gradle +@@ -69,6 +69,7 @@ + include "lucene:distribution.tests" + include "lucene:documentation" + include "lucene:expressions" ++include "lucene:extensions" + include "lucene:facet" + include "lucene:grouping" + include "lucene:highlighter" +--- a/lucene-java-10.2.0/gradle/wrapper/gradle-wrapper.properties ++++ b/lucene-java-10.2.0/gradle/wrapper/gradle-wrapper.properties +@@ -1,6 +1,7 @@ + distributionBase=GRADLE_USER_HOME + distributionPath=wrapper/dists + distributionUrl=https\://services.gradle.org/distributions/gradle-8.10-bin.zip ++distributionSha256Sum=5b9c5eb3f9fc2c94abaea57d90bd78747ca117ddbbf96c859d3741181a12bf2a + networkTimeout=10000 + validateDistributionUrl=true + zipStoreBase=GRADLE_USER_HOME diff --git a/fern/docs.yml b/fern/docs.yml index 77ee0dae2a..921d86ce70 100644 --- a/fern/docs.yml +++ b/fern/docs.yml @@ -575,6 +575,8 @@ navigation: path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-gpuindex.md" - page: "GPUSearchParams" path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-gpusearchparams.md" + - page: "IndexSearcherTimingBridge" + path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md" - page: "LuceneProvider" path: "./pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-luceneprovider.md" - page: "ThreadLocalCuVSResourcesProvider" diff --git a/fern/pages/lucene_api/index.md b/fern/pages/lucene_api/index.md index 3acfc3b798..90733c62ec 100644 --- a/fern/pages/lucene_api/index.md +++ b/fern/pages/lucene_api/index.md @@ -13,6 +13,7 @@ For an introduction to the codecs, configuration, and tuning, see the [Lucene In - [FilterCuVSServiceProvider](/api-reference/lucene-api-com-nvidia-cuvs-lucene-filtercuvsserviceprovider) - [GPUIndex](/api-reference/lucene-api-com-nvidia-cuvs-lucene-gpuindex) - [GPUSearchParams](/api-reference/lucene-api-com-nvidia-cuvs-lucene-gpusearchparams) +- [IndexSearcherTimingBridge](/api-reference/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge) - [LuceneProvider](/api-reference/lucene-api-com-nvidia-cuvs-lucene-luceneprovider) - [ThreadLocalCuVSResourcesProvider](/api-reference/lucene-api-com-nvidia-cuvs-lucene-threadlocalcuvsresourcesprovider) - [Utils](/api-reference/lucene-api-com-nvidia-cuvs-lucene-utils) diff --git a/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md new file mode 100644 index 0000000000..34a598352b --- /dev/null +++ b/fern/pages/lucene_api/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge.md @@ -0,0 +1,25 @@ +--- +slug: api-reference/lucene-api-com-nvidia-cuvs-lucene-indexsearchertimingbridge +--- + +# IndexSearcherTimingBridge + +_Java package: `com.nvidia.cuvs.lucene`_ + +```java +public final class IndexSearcherTimingBridge implements Function, Map> +``` + +Measures one `IndexSearcher#search(Query,int)` invocation inside the JVM. + +The standard `Function` and `Map` types form a narrow bridge for generated Java +bindings that do not wrap this class directly. Each call still represents one ordinary Lucene +query; this class does not add batching or concurrency. + +The request map must contain `searcher` (an `IndexSearcher`), `query` (a +`Query`), and `top_k` (an `Integer`). The response contains `top_docs` (a +`TopDocs`) and `elapsed_nanos` (a `Long`). An `IOException` from Lucene is +exposed as an `UncheckedIOException` because `Function.apply` cannot declare checked +exceptions. + +_Source: `java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.java:29`_ diff --git a/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.java b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.java new file mode 100644 index 0000000000..90bb8c5c4f --- /dev/null +++ b/java/cuvs-lucene/src/main/java/com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.java @@ -0,0 +1,72 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import java.io.IOException; +import java.io.UncheckedIOException; +import java.util.HashMap; +import java.util.Map; +import java.util.function.Function; +import org.apache.lucene.search.IndexSearcher; +import org.apache.lucene.search.Query; +import org.apache.lucene.search.TopDocs; + +/** + * Measures one {@link IndexSearcher#search(Query,int)} invocation inside the JVM. + * + *

The standard {@link Function} and {@link Map} types form a narrow bridge for generated Java + * bindings that do not wrap this class directly. Each call still represents one ordinary Lucene + * query; this class does not add batching or concurrency. + * + *

The request map must contain {@code searcher} (an {@code IndexSearcher}), {@code query} (a + * {@code Query}), and {@code top_k} (an {@code Integer}). The response contains {@code top_docs} (a + * {@code TopDocs}) and {@code elapsed_nanos} (a {@code Long}). An {@code IOException} from Lucene is + * exposed as an {@code UncheckedIOException} because {@code Function.apply} cannot declare checked + * exceptions. + */ +public final class IndexSearcherTimingBridge + implements Function, Map> { + public static final String SEARCHER_KEY = "searcher"; + public static final String QUERY_KEY = "query"; + public static final String TOP_K_KEY = "top_k"; + public static final String TOP_DOCS_KEY = "top_docs"; + public static final String ELAPSED_NANOS_KEY = "elapsed_nanos"; + + @Override + public Map apply(Map request) { + IndexSearcher searcher = requiredValue(request, SEARCHER_KEY, IndexSearcher.class); + Query query = requiredValue(request, QUERY_KEY, Query.class); + Integer topK = requiredValue(request, TOP_K_KEY, Integer.class); + if (topK < 1) { + throw new IllegalArgumentException("top_k must be positive"); + } + + long started = System.nanoTime(); + TopDocs topDocs; + try { + topDocs = searcher.search(query, topK); + } catch (IOException error) { + throw new UncheckedIOException("IndexSearcher.search failed", error); + } + long elapsedNanos = System.nanoTime() - started; + + Map response = new HashMap<>(); + response.put(TOP_DOCS_KEY, topDocs); + response.put(ELAPSED_NANOS_KEY, elapsedNanos); + return response; + } + + private static T requiredValue( + Map values, String key, Class expectedType) { + if (values == null) { + throw new IllegalArgumentException("request must not be null"); + } + Object value = values.get(key); + if (!expectedType.isInstance(value)) { + throw new IllegalArgumentException(key + " must have type " + expectedType.getSimpleName()); + } + return expectedType.cast(value); + } +} diff --git a/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexSearcherTimingBridge.java b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexSearcherTimingBridge.java new file mode 100644 index 0000000000..d62819e4a8 --- /dev/null +++ b/java/cuvs-lucene/src/test/java/com/nvidia/cuvs/lucene/TestIndexSearcherTimingBridge.java @@ -0,0 +1,65 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +package com.nvidia.cuvs.lucene; + +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertNotNull; +import static org.junit.Assert.assertThrows; +import static org.junit.Assert.assertTrue; + +import java.util.HashMap; +import java.util.Map; +import java.util.function.Function; +import org.apache.lucene.document.Document; +import org.apache.lucene.index.DirectoryReader; +import org.apache.lucene.index.IndexWriter; +import org.apache.lucene.index.IndexWriterConfig; +import org.apache.lucene.search.IndexSearcher; +import org.apache.lucene.search.MatchAllDocsQuery; +import org.apache.lucene.search.TopDocs; +import org.apache.lucene.store.ByteBuffersDirectory; +import org.junit.Test; + +public class TestIndexSearcherTimingBridge { + + @Test + public void testReturnsTopDocsAndInJvmElapsedTimeThroughStandardTypes() throws Exception { + try (var directory = new ByteBuffersDirectory(); + var writer = new IndexWriter(directory, new IndexWriterConfig())) { + writer.addDocument(new Document()); + writer.commit(); + + try (var reader = DirectoryReader.open(directory)) { + Function, Map> bridge = new IndexSearcherTimingBridge(); + Map request = new HashMap<>(); + request.put(IndexSearcherTimingBridge.SEARCHER_KEY, new IndexSearcher(reader)); + request.put(IndexSearcherTimingBridge.QUERY_KEY, new MatchAllDocsQuery()); + request.put(IndexSearcherTimingBridge.TOP_K_KEY, 1); + + Map response = bridge.apply(request); + + TopDocs topDocs = (TopDocs) response.get(IndexSearcherTimingBridge.TOP_DOCS_KEY); + Long elapsedNanos = (Long) response.get(IndexSearcherTimingBridge.ELAPSED_NANOS_KEY); + assertNotNull(topDocs); + assertEquals(1, topDocs.scoreDocs.length); + assertNotNull(elapsedNanos); + assertTrue(elapsedNanos >= 0L); + } + } + } + + @Test + public void testRejectsAnIncompleteRequestBeforeSearching() { + var request = new HashMap(); + request.put(IndexSearcherTimingBridge.QUERY_KEY, new MatchAllDocsQuery()); + request.put(IndexSearcherTimingBridge.TOP_K_KEY, 1); + + IllegalArgumentException error = + assertThrows( + IllegalArgumentException.class, () -> new IndexSearcherTimingBridge().apply(request)); + + assertEquals("searcher must have type IndexSearcher", error.getMessage()); + } +} diff --git a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py new file mode 100644 index 0000000000..ba933e4bcd --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py @@ -0,0 +1,1424 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# + +"""PyLucene lifecycle, index I/O, and fail-closed CAGRA verification.""" + +from __future__ import annotations + +import hashlib +import importlib +import os +import threading +import time +import zipfile +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Callable, Mapping, Sequence + +import numpy as np + +from ._lucene_runtime_config import maven_artifact_version + +CPU_HNSW_CODEC = "Lucene101" +ACCELERATED_HNSW_CODEC = "Lucene101AcceleratedHNSWCodec" +CAGRA_CODEC = "CuVS2510GPUSearchCodec" +MAX_CAGRA_TOP_K = 1024 +REQUIRED_PYLUCENE_VERSION = "10.2.0" + +_ID_FIELD = "id" +_VECTOR_FIELD = "vector" +_MAX_DIMENSIONS = 4096 +_CAGRA_META_EXTENSION = ".vemc" +_CAGRA_META_CODEC_NAME = "Lucene102CuVSVectorsFormatMeta" +_CAGRA_DATA_EXTENSION = ".vcag" +_CAGRA_DATA_CODEC_NAME = "Lucene102CuVSVectorsFormatIndex" +_CAGRA_FORMAT_VERSION = 0 +_FLOAT32_ENCODING_ORDINAL = 1 +_EUCLIDEAN_SIMILARITY_ORDINAL = 0 +_CUVS_JAVA_PACKAGE = "com.nvidia.cuvs" +_CUVS_LUCENE_PACKAGE = f"{_CUVS_JAVA_PACKAGE}.lucene" +_CUVS_JAVA_PACKAGE_DIRECTORY = _CUVS_JAVA_PACKAGE.replace(".", "/") +_CUVS_LUCENE_PACKAGE_DIRECTORY = _CUVS_LUCENE_PACKAGE.replace(".", "/") +_MAVEN_METADATA_DIRECTORY = "META-INF/maven" +_CUVS_JAVA_MAVEN_PROPERTIES = ( + f"{_MAVEN_METADATA_DIRECTORY}/{_CUVS_JAVA_PACKAGE}/" + "cuvs-java/pom.properties" +) +_CUVS_LUCENE_MAVEN_PROPERTIES = ( + f"{_MAVEN_METADATA_DIRECTORY}/{_CUVS_LUCENE_PACKAGE}/" + "cuvs-lucene/pom.properties" +) +_CODEC_SERVICE_DESCRIPTOR = "META-INF/services/org.apache.lucene.codecs.Codec" +_INDEX_SEARCHER_TIMING_BRIDGE = ( + f"{_CUVS_LUCENE_PACKAGE}.IndexSearcherTimingBridge" +) +_SEARCHER_REQUEST_KEY = "searcher" +_QUERY_REQUEST_KEY = "query" +_TOP_K_REQUEST_KEY = "top_k" +_TOP_DOCS_RESPONSE_KEY = "top_docs" +_ELAPSED_NANOS_RESPONSE_KEY = "elapsed_nanos" +DIRECT_PYLUCENE_DISPATCH = "direct_pylucene" +TIMED_BRIDGE_PYLUCENE_DISPATCH = "thin_jar_timing_bridge" + +_JVM_LOCK = threading.Lock() +_INITIALIZED_CLASSPATH: str | None = None +_INITIALIZED_VMARGS: tuple[str, ...] | None = None +_INITIALIZED_ARTIFACT_PROVENANCE: dict[str, str] | None = None +_INITIALIZED_ARTIFACT_TOKENS: dict[str, tuple[int, ...]] | None = None + + +class _CleanupStack: + """Close resources in reverse order without hiding an earlier failure.""" + + def __init__(self) -> None: + self._cleanups: list[tuple[str, Callable[[], None]]] = [] + + def __enter__(self) -> "_CleanupStack": + return self + + def add(self, description: str, cleanup: Callable[[], None]) -> None: + self._cleanups.append((description, cleanup)) + + def __exit__( + self, _kind: Any, error: BaseException | None, _tb: Any + ) -> bool: + pending = error + for description, cleanup in reversed(self._cleanups): + try: + cleanup() + except BaseException as cleanup_error: + if pending is None: + pending = cleanup_error + elif isinstance( + cleanup_error, (KeyboardInterrupt, SystemExit) + ) and isinstance(pending, Exception): + cleanup_error.add_note( + "Resource handling first failed: " + f"{type(pending).__name__}: {pending}" + ) + pending = cleanup_error + elif isinstance(pending, Exception): + pending.add_note( + f"Failed to {description}: " + f"{type(cleanup_error).__name__}: {cleanup_error}" + ) + if pending is not None and pending is not error: + raise pending + return False + + +def _rollback_writer(writer: Any, error: BaseException) -> None: + """Roll back a failed writer without swallowing process-control errors.""" + try: + writer.rollback() + except BaseException as rollback_error: + if isinstance( + rollback_error, (KeyboardInterrupt, SystemExit) + ) and isinstance(error, Exception): + rollback_error.add_note( + f"Lucene writer first failed: {type(error).__name__}: {error}" + ) + raise + error.add_note( + "Failed to roll back Lucene writer: " + f"{type(rollback_error).__name__}: {rollback_error}" + ) + + +def _read_jar(path: Path, label: str) -> tuple[set[str], dict[str, bytes]]: + inspected = { + "META-INF/MANIFEST.MF", + _CUVS_JAVA_MAVEN_PROPERTIES, + _CUVS_LUCENE_MAVEN_PROPERTIES, + _CODEC_SERVICE_DESCRIPTOR, + } + try: + with zipfile.ZipFile(path) as archive: + entries = set(archive.namelist()) + contents = { + name: archive.read(name) + for name in inspected + if name in entries + } + except (OSError, zipfile.BadZipFile) as error: + raise RuntimeError(f"{label} is not a readable JAR: {path}") from error + return entries, contents + + +def _maven_coordinates( + contents: Mapping[str, bytes], descriptor: str +) -> tuple[str, str, str]: + try: + text = contents[descriptor].decode("utf-8") + except (KeyError, UnicodeDecodeError) as error: + raise RuntimeError( + f"Java artifact is missing valid Maven coordinates at {descriptor}" + ) from error + properties = {} + for line in text.splitlines(): + key, separator, value = line.partition("=") + if separator and not key.lstrip().startswith(("#", "!")): + properties[key.strip()] = value.strip() + try: + return ( + properties["groupId"], + properties["artifactId"], + properties["version"], + ) + except KeyError as error: + raise RuntimeError( + f"Incomplete Maven coordinates at {descriptor}" + ) from error + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _artifact_stat_token(path: Path) -> tuple[int, ...]: + stat = path.stat() + return ( + stat.st_dev, + stat.st_ino, + stat.st_size, + stat.st_mtime_ns, + stat.st_ctime_ns, + ) + + +def _verify_artifact_tokens( + tokens: Mapping[str, tuple[int, ...]], + provenance: Mapping[str, str], +) -> None: + expected_hashes = { + provenance.get("cuvs_java_jar_path"): provenance.get( + "cuvs_java_jar_sha256" + ), + provenance.get("cuvs_lucene_jar_path"): provenance.get( + "cuvs_lucene_jar_sha256" + ), + } + for raw_path, expected in tokens.items(): + path = Path(raw_path) + try: + before = _artifact_stat_token(path) + except OSError as error: + raise RuntimeError( + f"Initialized Java artifact is no longer available: {path}" + ) from error + expected_hash = expected_hashes.get(raw_path) + if ( + before != expected + or not isinstance(expected_hash, str) + or _sha256(path) != expected_hash + or _artifact_stat_token(path) != expected + ): + raise RuntimeError( + "A Java artifact changed after the process-wide JVM was " + f"initialized: {path}. Start a new process." + ) + + +def _verify_artifact_stat_tokens( + tokens: Mapping[str, tuple[int, ...]], +) -> None: + """Catch ordinary artifact replacement between operation boundaries.""" + for raw_path, expected in tokens.items(): + path = Path(raw_path) + try: + actual = _artifact_stat_token(path) + except OSError as error: + raise RuntimeError( + f"Initialized Java artifact is no longer available: {path}" + ) from error + if actual != expected: + raise RuntimeError( + "A Java artifact changed after the process-wide JVM was " + f"initialized: {path}. Start a new process." + ) + + +def _validate_artifacts( + java_jar: Path, lucene_jar: Path +) -> tuple[dict[str, str], dict[str, tuple[int, ...]]]: + before = { + str(java_jar): _artifact_stat_token(java_jar), + str(lucene_jar): _artifact_stat_token(lucene_jar), + } + java_entries, java_contents = _read_jar(java_jar, "cuvs_java_jar") + java_required = { + f"{_CUVS_JAVA_PACKAGE_DIRECTORY}/CagraIndex.class", + f"{_CUVS_JAVA_PACKAGE_DIRECTORY}/CuVSResources.class", + "META-INF/versions/22/" + f"{_CUVS_JAVA_PACKAGE_DIRECTORY}/spi/JDKProvider.class", + } + missing = sorted(java_required - java_entries) + if missing: + raise RuntimeError( + "cuvs_java_jar is not the base cuvs-java artifact; missing: " + + ", ".join(missing) + ) + manifest = java_contents.get("META-INF/MANIFEST.MF", b"").decode( + "utf-8", errors="replace" + ) + if "multi-release: true" not in manifest.casefold(): + raise RuntimeError("cuvs_java_jar must declare Multi-Release: true") + if any( + entry.rpartition("/")[2] in {"libcuvs.so", "libcuvs_c.so"} + for entry in java_entries + ): + raise RuntimeError( + "cuvs_java_jar embeds native libraries; use the base JAR, not a " + "native-classifier artifact" + ) + + lucene_entries, lucene_contents = _read_jar(lucene_jar, "cuvs_lucene_jar") + lucene_required = { + f"{_CUVS_LUCENE_PACKAGE_DIRECTORY}/CuVS2510GPUVectorsFormat.class", + f"{_CUVS_LUCENE_PACKAGE_DIRECTORY}/CuVS2510GPUSearchCodec.class", + f"{_CUVS_LUCENE_PACKAGE_DIRECTORY}/IndexSearcherTimingBridge.class", + f"{_CUVS_LUCENE_PACKAGE_DIRECTORY}/" + "Lucene101AcceleratedHNSWCodec.class", + _CODEC_SERVICE_DESCRIPTOR, + } + missing = sorted(lucene_required - lucene_entries) + if missing: + raise RuntimeError( + "cuvs_lucene_jar is not the standard cuvs-lucene artifact; missing: " + + ", ".join(missing) + ) + bundled_lucene = next( + ( + entry + for entry in lucene_entries + if entry.endswith(".class") + and ( + entry.startswith("org/apache/lucene/") + or "/org/apache/lucene/" in entry + ) + ), + None, + ) + if bundled_lucene: + raise RuntimeError( + "cuvs_lucene_jar bundles Lucene classes; use the dependency-thin " + f"artifact. First bundled class: {bundled_lucene}" + ) + providers = { + line.partition("#")[0].strip() + for line in lucene_contents[_CODEC_SERVICE_DESCRIPTOR] + .decode("utf-8") + .splitlines() + if line.partition("#")[0].strip() + } + if f"{_CUVS_LUCENE_PACKAGE}.CuVS2510GPUSearchCodec" not in providers: + raise RuntimeError( + "cuvs_lucene_jar does not advertise CuVS2510GPUSearchCodec" + ) + if ( + f"{_CUVS_LUCENE_PACKAGE}.Lucene101AcceleratedHNSWCodec" + not in providers + ): + raise RuntimeError( + "cuvs_lucene_jar does not advertise Lucene101AcceleratedHNSWCodec" + ) + + java_coordinates = _maven_coordinates( + java_contents, + _CUVS_JAVA_MAVEN_PROPERTIES, + ) + lucene_coordinates = _maven_coordinates( + lucene_contents, + _CUVS_LUCENE_MAVEN_PROPERTIES, + ) + if java_coordinates[:2] != (_CUVS_JAVA_PACKAGE, "cuvs-java"): + raise RuntimeError( + f"Unexpected cuvs-java coordinates: {java_coordinates[:2]}" + ) + if lucene_coordinates[:2] != ( + _CUVS_LUCENE_PACKAGE, + "cuvs-lucene", + ): + raise RuntimeError( + f"Unexpected cuvs-lucene coordinates: {lucene_coordinates[:2]}" + ) + expected_version = maven_artifact_version() + if java_coordinates[2] != lucene_coordinates[2]: + raise RuntimeError( + "cuvs-java and cuvs-lucene JAR versions differ: " + f"{java_coordinates[2]} != {lucene_coordinates[2]}" + ) + if java_coordinates[2] != expected_version: + raise RuntimeError( + "Java artifacts do not match this cuVS Bench release: expected " + f"{expected_version}, found {java_coordinates[2]}" + ) + provenance = { + "cuvs_java_coordinates": ":".join(java_coordinates), + "cuvs_java_jar_path": str(java_jar), + "cuvs_java_jar_sha256": _sha256(java_jar), + "cuvs_lucene_coordinates": ":".join(lucene_coordinates), + "cuvs_lucene_jar_path": str(lucene_jar), + "cuvs_lucene_jar_sha256": _sha256(lucene_jar), + } + after = { + str(java_jar): _artifact_stat_token(java_jar), + str(lucene_jar): _artifact_stat_token(lucene_jar), + } + if before != after: + raise RuntimeError( + "Java artifacts changed while their identities were being validated" + ) + return provenance, after + + +def _load_pylucene() -> Any: + try: + return importlib.import_module("lucene") + except ImportError as error: + raise ImportError( + "The Lucene backend requires the custom PyLucene 10.2.0 runtime, " + "which is not included in cuVS Bench packages. Install and " + "activate the optional Lucene runtime before selecting this " + f"backend. PyLucene import failed: {error}" + ) from error + + +def _classpath( + config: Mapping[str, Any], lucene: Any, *, validate_artifacts: bool +) -> tuple[str, dict[str, str], dict[str, tuple[int, ...]]]: + entries = [] + provenance = {} + artifact_tokens = {} + java_value = config.get("cuvs_java_jar") + lucene_value = config.get("cuvs_lucene_jar") + if bool(java_value) != bool(lucene_value): + raise RuntimeError( + "Both cuvs_java_jar and cuvs_lucene_jar are required together" + ) + if java_value: + java_jar = Path(os.fspath(java_value)).resolve() + lucene_jar = Path(os.fspath(lucene_value)).resolve() + if validate_artifacts: + provenance, artifact_tokens = _validate_artifacts( + java_jar, lucene_jar + ) + entries.extend((str(java_jar), str(lucene_jar))) + entries.append(str(lucene.CLASSPATH)) + return os.pathsep.join(entries), provenance, artifact_tokens + + +def _vmargs(config: Mapping[str, Any]) -> list[str]: + arguments = [ + "--enable-native-access=ALL-UNNAMED", + "--add-modules=jdk.incubator.vector", + ] + if library_path := config.get("java_library_path"): + arguments.append(f"-Djava.library.path={os.fspath(library_path)}") + extra = config.get("jvm_args", ()) + if isinstance(extra, (str, bytes)) or not isinstance(extra, (list, tuple)): + raise TypeError("jvm_args must be a list or tuple of strings") + if not all(isinstance(argument, str) for argument in extra): + raise TypeError("Every jvm_args entry must be a string") + arguments.extend(extra) + return arguments + + +def initialize_pylucene( + config: Mapping[str, Any], +) -> tuple[Any, dict[str, str], dict[str, tuple[int, ...]]]: + """Start PyLucene once with an immutable, validated JVM configuration.""" + global _INITIALIZED_ARTIFACT_PROVENANCE + global _INITIALIZED_ARTIFACT_TOKENS + global _INITIALIZED_CLASSPATH, _INITIALIZED_VMARGS + lucene = _load_pylucene() + actual_version = str(getattr(lucene, "VERSION", "")) + if actual_version != REQUIRED_PYLUCENE_VERSION: + raise RuntimeError( + "PyLucene must match cuvs-lucene's Lucene version: expected " + f"{REQUIRED_PYLUCENE_VERSION}, found {actual_version}. Install " + "and activate a compatible PyLucene runtime before selecting " + "this backend." + ) + vmargs = _vmargs(config) + with _JVM_LOCK: + environment = lucene.getVMEnv() + if environment is None: + classpath, artifact_provenance, artifact_tokens = _classpath( + config, lucene, validate_artifacts=True + ) + environment = lucene.initVM(classpath=classpath, vmargs=vmargs) + environment = environment or lucene.getVMEnv() + if environment is None: + raise RuntimeError("PyLucene did not return a JVM environment") + _INITIALIZED_CLASSPATH = classpath + _INITIALIZED_VMARGS = tuple(vmargs) + _INITIALIZED_ARTIFACT_PROVENANCE = dict(artifact_provenance) + _INITIALIZED_ARTIFACT_TOKENS = dict(artifact_tokens) + else: + classpath, _unused_provenance, _unused_tokens = _classpath( + config, lucene, validate_artifacts=False + ) + if ( + _INITIALIZED_CLASSPATH != classpath + or _INITIALIZED_VMARGS != tuple(vmargs) + ): + raise RuntimeError( + "PyLucene's process-wide JVM is already initialized with a " + "different classpath or JVM arguments. Start a new process and " + "let cuVS Bench initialize PyLucene." + ) + artifact_provenance = dict(_INITIALIZED_ARTIFACT_PROVENANCE or {}) + artifact_tokens = dict(_INITIALIZED_ARTIFACT_TOKENS or {}) + _verify_artifact_tokens(artifact_tokens, artifact_provenance) + environment.attachCurrentThread() + return lucene, artifact_provenance, artifact_tokens + + +@dataclass(frozen=True) +class CagraVerification: + segment_count: int + field_count: int + vector_count: int + dimensions: int + + def metadata(self) -> dict[str, int | str]: + return { + "persisted_index_kind": "gpu_cagra_only", + "segment_count": self.segment_count, + "field_count": self.field_count, + "vector_count": self.vector_count, + "dimensions": self.dimensions, + } + + +@dataclass(frozen=True) +class _RawCagraField: + number: int + encoding: int + similarity: int + dimensions: int + vector_count: int + cagra_offset: int + cagra_length: int + brute_force_offset: int + brute_force_length: int + + +class CagraVerificationError(RuntimeError): + """Raised when persisted files do not contain an all-CAGRA index.""" + + +@dataclass(frozen=True) +class LuceneIndexVerification: + """Observed structure of one committed Lucene vector index.""" + + codec: str + segment_count: int + field_count: int + vector_count: int + dimensions: int + + def metadata(self) -> dict[str, int | str]: + persisted_index_kind = { + CPU_HNSW_CODEC: "cpu_hnsw", + ACCELERATED_HNSW_CODEC: "hnsw", + CAGRA_CODEC: "gpu_cagra", + }[self.codec] + return { + "persisted_index_kind": persisted_index_kind, + "segment_count": self.segment_count, + "field_count": self.field_count, + "vector_count": self.vector_count, + "dimensions": self.dimensions, + } + + +class LuceneIndexVerifier: + """Validate the physical codec, vector schema, counts, and live-doc state.""" + + def __init__(self, runtime: "LuceneRuntime") -> None: + self.runtime = runtime + + def verify( + self, + index_path: Path, + *, + expected_codec: str, + expected_vector_count: int, + expected_dimensions: int, + ) -> LuceneIndexVerification: + runtime = self.runtime + runtime.attach_current_thread() + directory = runtime.FSDirectory.open( + runtime.Paths.get(str(index_path)) + ) + with _CleanupStack() as cleanups: + cleanups.add("close Lucene directory", directory.close) + segment_infos = runtime.SegmentInfos.readLatestCommit(directory) + segments = [ + runtime.SegmentCommitInfo.cast_(raw) for raw in segment_infos + ] + if not segments: + raise RuntimeError("Lucene index has no committed segments") + for segment in segments: + segment_name = str(segment.info.name) + if ( + segment.hasDeletions() + or int(segment.getDelCount()) + or int(segment.getSoftDelCount()) + ): + raise RuntimeError( + "Lucene benchmark index has committed deletions in " + f"{segment_name!r}" + ) + actual_codec = str(segment.info.getCodec().getName()) + if actual_codec != expected_codec: + raise RuntimeError( + f"Segment {segment_name!r} uses {actual_codec}, not " + f"{expected_codec}" + ) + + reader = runtime.DirectoryReader.open(directory) + cleanups.add("close Lucene reader", reader.close) + if int(reader.numDocs()) != int(reader.maxDoc()): + raise RuntimeError("Lucene benchmark index contains deletions") + vector_count = 0 + field_count = 0 + dimensions = set() + for leaf in reader.leaves(): + leaf_reader = leaf.reader() + vector_fields = [] + for raw_info in leaf_reader.getFieldInfos(): + info = runtime.FieldInfo.cast_(raw_info) + if int(info.getVectorDimension()) > 0: + vector_fields.append(info) + if len(vector_fields) != 1: + raise RuntimeError( + "Each Lucene segment must contain exactly one vector " + f"field; found {len(vector_fields)}" + ) + info = vector_fields[0] + if str(info.getName()) != _VECTOR_FIELD: + raise RuntimeError( + f"Unexpected Lucene vector field {info.getName()!r}" + ) + if info.getVectorEncoding() != runtime.VectorEncoding.FLOAT32: + raise RuntimeError("Lucene vector field is not FLOAT32") + if ( + info.getVectorSimilarityFunction() + != runtime.VectorSimilarityFunction.EUCLIDEAN + ): + raise RuntimeError("Lucene vector field is not Euclidean") + values = leaf_reader.getFloatVectorValues(_VECTOR_FIELD) + if values is None: + raise RuntimeError( + "Lucene segment is missing vector values" + ) + segment_vectors = int(values.size()) + if segment_vectors != int(leaf_reader.numDocs()): + raise RuntimeError( + "Lucene segment vector and document counts differ: " + f"{segment_vectors} != {leaf_reader.numDocs()}" + ) + vector_count += segment_vectors + field_count += 1 + dimensions.add(int(info.getVectorDimension())) + + if vector_count != expected_vector_count: + raise RuntimeError( + f"Lucene index has {vector_count} vectors; expected " + f"{expected_vector_count}" + ) + if dimensions != {expected_dimensions}: + raise RuntimeError( + f"Lucene index dimensions are {sorted(dimensions)}; expected " + f"{expected_dimensions}" + ) + return LuceneIndexVerification( + codec=expected_codec, + segment_count=len(segments), + field_count=field_count, + vector_count=vector_count, + dimensions=expected_dimensions, + ) + + +class CagraIndexVerifier: + """Verify that every committed vector field contains CAGRA and no BFI.""" + + def __init__(self, runtime: "LuceneRuntime") -> None: + self.runtime = runtime + + @staticmethod + def _suffix(segment_name: str, metadata_file: str) -> str: + stem = metadata_file[: -len(_CAGRA_META_EXTENSION)] + if stem == segment_name: + return "" + prefix = f"{segment_name}_" + if not stem.startswith(prefix) or stem == prefix: + raise CagraVerificationError( + f"Unexpected CAGRA metadata filename {metadata_file!r} " + f"for segment {segment_name!r}" + ) + return stem[len(prefix) :] + + @staticmethod + def _decode_fields( + metadata_input: Any, metadata_file: str + ) -> list[_RawCagraField]: + fields = [] + numbers = set() + while True: + number = int(metadata_input.readInt()) + if number == -1: + return fields + if number < 0 or number in numbers: + raise CagraVerificationError( + f"Invalid field number {number} in {metadata_file!r}" + ) + numbers.add(number) + fields.append( + _RawCagraField( + number=number, + encoding=int(metadata_input.readInt()), + similarity=int(metadata_input.readInt()), + dimensions=int(metadata_input.readInt()), + vector_count=int(metadata_input.readInt()), + cagra_offset=int(metadata_input.readVLong()), + cagra_length=int(metadata_input.readVLong()), + brute_force_offset=int(metadata_input.readVLong()), + brute_force_length=int(metadata_input.readVLong()), + ) + ) + + @staticmethod + def _validate_field(field: _RawCagraField, metadata_file: str) -> None: + if field.encoding != _FLOAT32_ENCODING_ORDINAL: + raise CagraVerificationError( + f"Non-FLOAT32 vector encoding in {metadata_file!r}" + ) + if field.similarity != _EUCLIDEAN_SIMILARITY_ORDINAL: + raise CagraVerificationError( + f"Non-Euclidean vector similarity in {metadata_file!r}" + ) + if ( + not 1 <= field.dimensions <= _MAX_DIMENSIONS + or field.vector_count < 1 + ): + raise CagraVerificationError( + f"Invalid vector shape in {metadata_file!r}" + ) + if field.brute_force_length != 0: + raise CagraVerificationError( + f"Persisted brute-force fallback for field {field.number} in " + f"{metadata_file!r}" + ) + if field.cagra_length <= 0: + raise CagraVerificationError( + f"No persisted CAGRA index for field {field.number} in " + f"{metadata_file!r}; cuVS may have fallen back" + ) + + def _read_fields( + self, directory: Any, segment: Any, metadata_file: str + ) -> list[_RawCagraField]: + runtime = self.runtime + suffix = self._suffix(str(segment.info.name), metadata_file) + metadata_input = directory.openChecksumInput(metadata_file) + with _CleanupStack() as cleanups: + cleanups.add("close CAGRA metadata", metadata_input.close) + runtime.CodecUtil.checkIndexHeader( + metadata_input, + _CAGRA_META_CODEC_NAME, + _CAGRA_FORMAT_VERSION, + _CAGRA_FORMAT_VERSION, + segment.info.getId(), + suffix, + ) + fields = self._decode_fields(metadata_input, metadata_file) + runtime.CodecUtil.checkFooter(metadata_input) + for field in fields: + self._validate_field(field, metadata_file) + return fields + + def _verify_data( + self, + directory: Any, + segment: Any, + metadata_file: str, + fields: Sequence[_RawCagraField], + ) -> None: + runtime = self.runtime + data_file = ( + metadata_file[: -len(_CAGRA_META_EXTENSION)] + + _CAGRA_DATA_EXTENSION + ) + suffix = self._suffix(str(segment.info.name), metadata_file) + data_input = directory.openInput(data_file, runtime.IOContext.READONCE) + with _CleanupStack() as cleanups: + cleanups.add("close CAGRA data", data_input.close) + runtime.CodecUtil.checkIndexHeader( + data_input, + _CAGRA_DATA_CODEC_NAME, + _CAGRA_FORMAT_VERSION, + _CAGRA_FORMAT_VERSION, + segment.info.getId(), + suffix, + ) + payload_start = int(data_input.getFilePointer()) + payload_end = int(data_input.length()) - int( + runtime.CodecUtil.footerLength() + ) + expected = payload_start + for field in sorted(fields, key=lambda item: item.cagra_offset): + if field.cagra_offset != expected: + raise CagraVerificationError( + f"CAGRA metadata does not exactly cover {data_file!r}" + ) + expected = field.cagra_offset + field.cagra_length + if expected != payload_end: + raise CagraVerificationError( + f"CAGRA metadata does not exactly cover {data_file!r}" + ) + runtime.CodecUtil.checksumEntireFile(data_input) + + def _segment_directory(self, root: Any, segment: Any) -> tuple[Any, bool]: + if not segment.info.getUseCompoundFile(): + return root, False + compound_format = segment.info.getCodec().compoundFormat() + return compound_format.getCompoundReader(root, segment.info), True + + @staticmethod + def _metadata_files( + directory: Any, segment: Any, *, compound: bool + ) -> list[str]: + # A compound reader exposes only one segment's embedded files. The root + # directory does not, so ask SegmentInfo for that segment's files. + names = directory.listAll() if compound else segment.info.files() + return sorted( + str(name) + for name in names + if str(name).endswith(_CAGRA_META_EXTENSION) + ) + + def _verify_segment( + self, root: Any, segment: Any + ) -> tuple[int, int, set[int]]: + runtime = self.runtime + segment_name = str(segment.info.name) + if ( + segment.hasDeletions() + or int(segment.getDelCount()) + or int(segment.getSoftDelCount()) + ): + raise CagraVerificationError( + f"CAGRA benchmark index has committed deletions in {segment_name!r}" + ) + if str(segment.info.getCodec().getName()) != CAGRA_CODEC: + raise CagraVerificationError( + f"Segment {segment_name!r} uses {segment.info.getCodec().getName()}, " + f"not {CAGRA_CODEC}" + ) + directory, compound = self._segment_directory(root, segment) + with _CleanupStack() as cleanups: + if compound: + cleanups.add("close compound directory", directory.close) + metadata_files = self._metadata_files( + directory, segment, compound=compound + ) + if not metadata_files: + raise CagraVerificationError( + f"No CAGRA metadata found for segment {segment_name!r}" + ) + field_infos = ( + segment.info.getCodec() + .fieldInfosFormat() + .read(directory, segment.info, "", runtime.IOContext.READONCE) + ) + verified: dict[int, _RawCagraField] = {} + for metadata_file in metadata_files: + fields = self._read_fields(directory, segment, metadata_file) + self._verify_data(directory, segment, metadata_file, fields) + for field in fields: + if field.number in verified: + raise CagraVerificationError( + f"Duplicate vector field {field.number} in segment " + f"{segment_name!r}" + ) + verified[field.number] = field + + lucene_vector_fields = set() + for raw_info in field_infos: + info = runtime.FieldInfo.cast_(raw_info) + if int(info.getVectorDimension()) <= 0: + continue + lucene_vector_fields.add(int(info.number)) + field = verified.get(int(info.number)) + if field is None: + raise CagraVerificationError( + f"Vector field {info.getName()!r} has no CAGRA metadata" + ) + if str(info.getName()) != _VECTOR_FIELD: + raise CagraVerificationError( + f"Unexpected vector field {info.getName()!r}" + ) + if int(info.getVectorDimension()) != field.dimensions: + raise CagraVerificationError( + f"Dimension mismatch in segment {segment_name!r}" + ) + if info.getVectorEncoding() != runtime.VectorEncoding.FLOAT32: + raise CagraVerificationError("Lucene field is not FLOAT32") + if ( + info.getVectorSimilarityFunction() + != runtime.VectorSimilarityFunction.EUCLIDEAN + ): + raise CagraVerificationError( + "Lucene field is not Euclidean" + ) + if set(verified) != lucene_vector_fields: + raise CagraVerificationError( + f"CAGRA/Lucene vector-field mismatch in {segment_name!r}" + ) + vector_count = sum( + field.vector_count for field in verified.values() + ) + if vector_count != int(segment.info.maxDoc()): + raise CagraVerificationError( + f"Segment {segment_name!r} has {vector_count} vectors for " + f"{segment.info.maxDoc()} documents" + ) + return ( + len(verified), + vector_count, + {field.dimensions for field in verified.values()}, + ) + + def verify( + self, + index_path: Path, + *, + expected_vector_count: int, + expected_dimensions: int, + ) -> CagraVerification: + self.runtime.attach_current_thread() + root = self.runtime.FSDirectory.open( + self.runtime.Paths.get(str(index_path)) + ) + segments = [] + with _CleanupStack() as cleanups: + cleanups.add("close Lucene directory", root.close) + segment_infos = self.runtime.SegmentInfos.readLatestCommit(root) + for raw_segment in segment_infos: + segment = self.runtime.SegmentCommitInfo.cast_(raw_segment) + segments.append(self._verify_segment(root, segment)) + if not segments: + raise CagraVerificationError( + "Lucene index has no committed segments" + ) + field_count = sum(item[0] for item in segments) + vector_count = sum(item[1] for item in segments) + dimensions = set().union(*(item[2] for item in segments)) + if vector_count != expected_vector_count: + raise CagraVerificationError( + f"CAGRA index has {vector_count} vectors; expected " + f"{expected_vector_count}" + ) + if dimensions != {expected_dimensions}: + raise CagraVerificationError( + f"CAGRA index dimensions are {sorted(dimensions)}; expected " + f"{expected_dimensions}" + ) + return CagraVerification( + segment_count=len(segments), + field_count=field_count, + vector_count=vector_count, + dimensions=expected_dimensions, + ) + + +@dataclass(frozen=True) +class SearchHit: + document_id: int + score: float + + +@dataclass(frozen=True) +class RuntimeBuildTiming: + """Nanosecond timings for the Lucene build lifecycle.""" + + directory_open_ns: int + writer_setup_ns: int + document_ingest_ns: int + writer_commit_close_ns: int + post_build_reader_ns: int + directory_close_ns: int + runtime_build_wall_ns: int + + +@dataclass(frozen=True) +class RuntimeBuildResult: + segment_count: int + timing: RuntimeBuildTiming + + +@dataclass(frozen=True) +class QueryTiming: + """Measured boundaries for one query in this backend's serial loop.""" + + query_prepare_ns: int + pylucene_search_dispatch_ns: int + java_index_searcher_search_ns: int | None + result_materialization_ns: int + client_query_ns: int + + +@dataclass(frozen=True) +class RuntimeSearchTiming: + """Nanosecond timings for one search-parameter plan.""" + + directory_open_ns: int + reader_searcher_setup_ns: int + query_corpus_wall_ns: int + reader_close_ns: int + directory_close_ns: int + runtime_plan_wall_ns: int + search_dispatch_kind: str + queries: tuple[QueryTiming, ...] + + +@dataclass(frozen=True) +class RuntimeSearchResult: + hits: list[list[SearchHit]] + timing: RuntimeSearchTiming + document_count: int + dimensions: int + + +class LuceneRuntime: + """Own the generated bindings and the narrow Lucene operations Bench uses.""" + + def __init__(self, lucene: Any): + from java.lang import Class, Integer, Long + from java.nio.file import Paths + from java.util import HashMap, Map + from java.util.function import Function + from org.apache.lucene.codecs import Codec, CodecUtil + from org.apache.lucene.document import ( + Document, + KnnFloatVectorField, + NumericDocValuesField, + ) + from org.apache.lucene.index import ( + DirectoryReader, + FieldInfo, + IndexWriter, + IndexWriterConfig, + SegmentCommitInfo, + SegmentInfos, + VectorEncoding, + VectorSimilarityFunction, + ) + from org.apache.lucene.search import ( + IndexSearcher, + KnnFloatVectorQuery, + TopDocs, + ) + from org.apache.lucene.store import FSDirectory, IOContext + + self.lucene = lucene + self.Class = Class + self.Integer = Integer + self.Long = Long + self.HashMap = HashMap + self.Map = Map + self.Function = Function + self.Paths = Paths + self.Codec = Codec + self.CodecUtil = CodecUtil + self.Document = Document + self.KnnFloatVectorField = KnnFloatVectorField + self.NumericDocValuesField = NumericDocValuesField + self.DirectoryReader = DirectoryReader + self.FieldInfo = FieldInfo + self.IndexWriter = IndexWriter + self.IndexWriterConfig = IndexWriterConfig + self.SegmentCommitInfo = SegmentCommitInfo + self.SegmentInfos = SegmentInfos + self.VectorEncoding = VectorEncoding + self.VectorSimilarityFunction = VectorSimilarityFunction + self.IndexSearcher = IndexSearcher + self.KnnFloatVectorQuery = KnnFloatVectorQuery + self.TopDocs = TopDocs + self.FSDirectory = FSDirectory + self.IOContext = IOContext + self.index_verifier = LuceneIndexVerifier(self) + self.cagra_verifier = CagraIndexVerifier(self) + self._codecs: dict[str, Any] = {} + self.artifact_provenance: dict[str, str] = {} + self._artifact_tokens: dict[str, tuple[int, ...]] = {} + self._java_search_timer: Any | None = None + + @classmethod + def create(cls, config: Mapping[str, Any]) -> "LuceneRuntime": + lucene, artifact_provenance, artifact_tokens = initialize_pylucene( + config + ) + runtime = cls(lucene) + runtime.artifact_provenance = artifact_provenance + runtime._artifact_tokens = artifact_tokens + if artifact_provenance: + runtime._java_search_timer = runtime._load_java_search_timer() + return runtime + + def _load_java_search_timer(self) -> Any: + """Load the thin-JAR timer through JCC's wrapped Function interface.""" + try: + instance = self.Class.forName( + _INDEX_SEARCHER_TIMING_BRIDGE + ).newInstance() + return self.Function.cast_(instance) + except Exception as error: + raise RuntimeError( + f"Could not load or adapt {_INDEX_SEARCHER_TIMING_BRIDGE} " + "through PyLucene/JCC: " + f"{type(error).__name__}: {error}" + ) from error + + @property + def pylucene_version(self) -> str: + return str(self.lucene.VERSION) + + def attach_current_thread(self) -> None: + _verify_artifact_stat_tokens(self._artifact_tokens) + environment = self.lucene.getVMEnv() + if environment is None: + raise RuntimeError("PyLucene JVM is not initialized") + environment.attachCurrentThread() + + def verify_artifacts(self) -> None: + """Verify immutable artifact bytes once at an operation boundary.""" + _verify_artifact_tokens( + self._artifact_tokens, self.artifact_provenance + ) + + def resolve_codec(self, name: str) -> Any: + self.attach_current_thread() + if name in self._codecs: + return self._codecs[name] + available = self.Codec.availableCodecs() + if not available.contains(name): + raise RuntimeError( + f"Lucene codec {name!r} is unavailable. Available codecs: " + + ", ".join(sorted(str(value) for value in available)) + ) + codec = self.Codec.forName(name) + if str(codec.getName()) != name: + raise RuntimeError( + f"Requested codec {name}, resolved {codec.getName()}" + ) + if codec.knnVectorsFormat() is None: + raise RuntimeError(f"{name} did not initialize a vector format") + self._codecs[name] = codec + return codec + + def _java_vector(self, vector: np.ndarray) -> Any: + return self.lucene.JArray("float")(vector.tolist()) + + def _document(self, document_id: int, vector: np.ndarray) -> Any: + document = self.Document() + document.add(self.NumericDocValuesField(_ID_FIELD, document_id)) + document.add( + self.KnnFloatVectorField( + _VECTOR_FIELD, + self._java_vector(vector), + self.VectorSimilarityFunction.EUCLIDEAN, + ) + ) + return document + + def build_index( + self, index_path: Path, vectors: np.ndarray, codec_name: str + ) -> RuntimeBuildResult: + self.attach_current_thread() + runtime_started = time.perf_counter_ns() + directory_open_started = time.perf_counter_ns() + directory = self.FSDirectory.open(self.Paths.get(str(index_path))) + directory_open_ns = time.perf_counter_ns() - directory_open_started + directory_close_ns = 0 + writer_setup_ns = 0 + document_ingest_ns = 0 + writer_commit_close_ns = 0 + post_build_reader_ns = 0 + segment_count = 0 + + def close_directory() -> None: + nonlocal directory_close_ns + started = time.perf_counter_ns() + try: + directory.close() + finally: + directory_close_ns += time.perf_counter_ns() - started + + with _CleanupStack() as cleanups: + cleanups.add("close Lucene directory", close_directory) + writer_setup_started = time.perf_counter_ns() + config = self.IndexWriterConfig() + config.setOpenMode(self.IndexWriterConfig.OpenMode.CREATE) + config.setCodec(self.resolve_codec(codec_name)) + writer = self.IndexWriter(directory, config) + writer_setup_ns = time.perf_counter_ns() - writer_setup_started + try: + ingest_started = time.perf_counter_ns() + for document_id, vector in enumerate(vectors): + writer.addDocument(self._document(document_id, vector)) + document_ingest_ns = time.perf_counter_ns() - ingest_started + commit_started = time.perf_counter_ns() + writer.commit() + writer.close() + writer_commit_close_ns = ( + time.perf_counter_ns() - commit_started + ) + except BaseException as error: + _rollback_writer(writer, error) + raise + post_build_started = time.perf_counter_ns() + reader = self.DirectoryReader.open(directory) + with _CleanupStack() as reader_cleanups: + reader_cleanups.add("close Lucene reader", reader.close) + segment_count = int(reader.leaves().size()) + post_build_reader_ns = time.perf_counter_ns() - post_build_started + return RuntimeBuildResult( + segment_count=segment_count, + timing=RuntimeBuildTiming( + directory_open_ns=directory_open_ns, + writer_setup_ns=writer_setup_ns, + document_ingest_ns=document_ingest_ns, + writer_commit_close_ns=writer_commit_close_ns, + post_build_reader_ns=post_build_reader_ns, + directory_close_ns=directory_close_ns, + runtime_build_wall_ns=time.perf_counter_ns() - runtime_started, + ), + ) + + @staticmethod + def _index_dimensions(reader: Any) -> int: + dimensions = { + int(values.dimension()) + for leaf in reader.leaves() + if (values := leaf.reader().getFloatVectorValues(_VECTOR_FIELD)) + is not None + and values.size() > 0 + } + if len(dimensions) != 1: + raise RuntimeError( + f"Lucene index has invalid vector dimensions: {sorted(dimensions)}" + ) + return dimensions.pop() + + def _search( + self, searcher: Any, query: Any, k: int + ) -> tuple[Any, int, int | None]: + """Search once and return optional in-JVM timing evidence.""" + dispatch_started = time.perf_counter_ns() + if self._java_search_timer is None: + top_docs = searcher.search(query, k) + return ( + top_docs, + time.perf_counter_ns() - dispatch_started, + None, + ) + request = self.HashMap() + request.put(_SEARCHER_REQUEST_KEY, searcher) + request.put(_QUERY_REQUEST_KEY, query) + request.put(_TOP_K_REQUEST_KEY, self.Integer.valueOf(k)) + raw_response = self._java_search_timer.apply(request) + response = self.Map.cast_(raw_response) + top_docs = self.TopDocs.cast_(response.get(_TOP_DOCS_RESPONSE_KEY)) + elapsed = self.Long.cast_( + response.get(_ELAPSED_NANOS_RESPONSE_KEY) + ).longValue() + elapsed_ns = int(elapsed) + if elapsed_ns < 0: + raise RuntimeError( + "The Java IndexSearcher timing bridge returned a negative duration" + ) + return ( + top_docs, + time.perf_counter_ns() - dispatch_started, + elapsed_ns, + ) + + @staticmethod + def _materialize_hits( + reader: Any, score_docs: Sequence[Any] + ) -> list[SearchHit]: + """Read each leaf's forward-only IDs, preserving score rank.""" + ranked_score_docs = sorted( + enumerate(score_docs), key=lambda item: int(item[1].doc) + ) + hits_by_rank: dict[int, SearchHit] = {} + next_hit = 0 + for leaf in reader.leaves(): + leaf_reader = leaf.reader() + doc_base = int(leaf.docBase) + doc_limit = doc_base + int(leaf_reader.maxDoc()) + if ( + next_hit == len(ranked_score_docs) + or int(ranked_score_docs[next_hit][1].doc) >= doc_limit + ): + continue + document_ids = leaf_reader.getNumericDocValues(_ID_FIELD) + if document_ids is None: + raise RuntimeError( + "Lucene index has no numeric dataset IDs; rebuild the " + "index with --force" + ) + while next_hit < len(ranked_score_docs): + rank, score_doc = ranked_score_docs[next_hit] + lucene_doc_id = int(score_doc.doc) + if lucene_doc_id >= doc_limit: + break + if lucene_doc_id < doc_base: + raise RuntimeError( + f"Lucene document {lucene_doc_id} is outside its leaf" + ) + if not document_ids.advanceExact(lucene_doc_id - doc_base): + raise RuntimeError( + f"Lucene document {lucene_doc_id} has no numeric " + "dataset ID; rebuild the index with --force" + ) + hits_by_rank[rank] = SearchHit( + int(document_ids.longValue()), float(score_doc.score) + ) + next_hit += 1 + if next_hit != len(ranked_score_docs): + lucene_doc_id = int(ranked_score_docs[next_hit][1].doc) + raise RuntimeError( + f"Lucene document {lucene_doc_id} is outside the index" + ) + return [hits_by_rank[rank] for rank in range(len(ranked_score_docs))] + + def _search_one( + self, + searcher: Any, + reader: Any, + vector: np.ndarray, + k: int, + candidates: int, + ) -> tuple[list[SearchHit], QueryTiming]: + client_started = time.perf_counter_ns() + prepare_started = time.perf_counter_ns() + query = self.KnnFloatVectorQuery( + _VECTOR_FIELD, self._java_vector(vector), candidates + ) + query_prepare_ns = time.perf_counter_ns() - prepare_started + + top_docs, pylucene_search_dispatch_ns, java_search_ns = self._search( + searcher, query, k + ) + + materialization_started = time.perf_counter_ns() + hits = self._materialize_hits(reader, top_docs.scoreDocs) + result_materialization_ns = ( + time.perf_counter_ns() - materialization_started + ) + return hits, QueryTiming( + query_prepare_ns=query_prepare_ns, + pylucene_search_dispatch_ns=pylucene_search_dispatch_ns, + java_index_searcher_search_ns=java_search_ns, + result_materialization_ns=result_materialization_ns, + client_query_ns=time.perf_counter_ns() - client_started, + ) + + def search_index( + self, + index_path: Path, + queries: np.ndarray, + *, + k: int, + num_candidates: int, + ) -> RuntimeSearchResult: + self.attach_current_thread() + plan_started = time.perf_counter_ns() + directory_open_started = time.perf_counter_ns() + directory = self.FSDirectory.open(self.Paths.get(str(index_path))) + directory_open_ns = time.perf_counter_ns() - directory_open_started + reader_close_ns = 0 + directory_close_ns = 0 + + def close_reader() -> None: + nonlocal reader_close_ns + started = time.perf_counter_ns() + try: + reader.close() + finally: + reader_close_ns += time.perf_counter_ns() - started + + def close_directory() -> None: + nonlocal directory_close_ns + started = time.perf_counter_ns() + try: + directory.close() + finally: + directory_close_ns += time.perf_counter_ns() - started + + with _CleanupStack() as cleanups: + cleanups.add("close Lucene directory", close_directory) + reader_setup_started = time.perf_counter_ns() + reader = self.DirectoryReader.open(directory) + cleanups.add("close Lucene reader", close_reader) + dimensions = self._index_dimensions(reader) + if queries.shape[1] != dimensions: + raise ValueError( + "Query dimensions do not match the index: " + f"{queries.shape[1]} != {dimensions}" + ) + document_count = int(reader.numDocs()) + if document_count < k: + raise ValueError( + f"Lucene index has {document_count} documents, fewer than k={k}" + ) + searcher = self.IndexSearcher(reader) + reader_searcher_setup_ns = ( + time.perf_counter_ns() - reader_setup_started + ) + all_hits: list[list[SearchHit]] = [] + query_timings: list[QueryTiming] = [] + corpus_started = time.perf_counter_ns() + for vector in queries: + hits, query_timing = self._search_one( + searcher, + reader, + vector, + k, + min(num_candidates, document_count), + ) + all_hits.append(hits) + query_timings.append(query_timing) + query_corpus_wall_ns = time.perf_counter_ns() - corpus_started + return RuntimeSearchResult( + hits=all_hits, + timing=RuntimeSearchTiming( + directory_open_ns=directory_open_ns, + reader_searcher_setup_ns=reader_searcher_setup_ns, + query_corpus_wall_ns=query_corpus_wall_ns, + reader_close_ns=reader_close_ns, + directory_close_ns=directory_close_ns, + runtime_plan_wall_ns=time.perf_counter_ns() - plan_started, + search_dispatch_kind=( + TIMED_BRIDGE_PYLUCENE_DISPATCH + if self._java_search_timer is not None + else DIRECT_PYLUCENE_DISPATCH + ), + queries=tuple(query_timings), + ), + document_count=document_count, + dimensions=dimensions, + ) diff --git a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime_config.py b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime_config.py new file mode 100644 index 0000000000..e89cf8ee7d --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime_config.py @@ -0,0 +1,312 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# + +"""Locate the preinstalled artifacts used by the optional Lucene backend.""" + +from __future__ import annotations + +import os +import platform +import sys +from pathlib import Path +from typing import Any, Iterable, Mapping + +_JAVA_JAR_ENV = "CUVS_LUCENE_CUVS_JAVA_JAR" +_LUCENE_JAR_ENV = "CUVS_LUCENE_JAR" +_MAVEN_REPOSITORY_ENV = "MAVEN_LOCAL_REPO" + +_PACKAGE_ROOT = Path(__file__).resolve().parents[1] +_REPOSITORY_ROOT = Path(__file__).resolve().parents[4] +_VERSION_FILE = _PACKAGE_ROOT / "VERSION" +_CUVS_MAVEN_DIRECTORY = Path("com/nvidia/cuvs") +_CUVS_LUCENE_MAVEN_DIRECTORY = _CUVS_MAVEN_DIRECTORY / "lucene" + +_CUDA_TARGET_BY_MACHINE = { + "aarch64": "sbsa-linux", + "x86_64": "x86_64-linux", +} + + +def maven_artifact_version() -> str: + """Return the package version using Maven's non-zero-padded spelling.""" + raw_version = _VERSION_FILE.read_text(encoding="utf-8").strip() + release_version = raw_version.partition("a")[0] + components = release_version.split(".") + if len(components) != 3 or not all(part.isdigit() for part in components): + raise RuntimeError( + f"Cannot derive a Maven artifact version from {raw_version!r}" + ) + return ".".join(str(int(part)) for part in components) + + +def _maven_repository() -> Path: + configured = os.environ.get(_MAVEN_REPOSITORY_ENV) + if configured: + return Path(configured).expanduser() + return Path.home() / ".m2" / "repository" + + +def _artifact_candidates(kind: str) -> tuple[Path, ...]: + version = maven_artifact_version() + repository = _maven_repository() + if kind == "cuvs_java_jar": + return ( + _REPOSITORY_ROOT + / "java" + / "cuvs-java" + / "target" + / f"cuvs-java-{version}.jar", + repository + / _CUVS_MAVEN_DIRECTORY + / "cuvs-java" + / version + / f"cuvs-java-{version}.jar", + ) + if kind == "cuvs_lucene_jar": + return ( + _REPOSITORY_ROOT + / "java" + / "cuvs-lucene" + / "target" + / f"cuvs-lucene-{version}.jar", + repository + / _CUVS_LUCENE_MAVEN_DIRECTORY + / "cuvs-lucene" + / version + / f"cuvs-lucene-{version}.jar", + ) + raise ValueError(f"Unknown Lucene artifact kind: {kind}") + + +def _configured_path(value: Any, key: str) -> Path: + try: + path = Path(os.fspath(value)).expanduser().resolve() + except TypeError as error: + raise TypeError( + f"{key} must be a filesystem path, got {value!r}" + ) from error + if not path.is_file(): + raise FileNotFoundError(f"{key} does not exist: {path}") + return path + + +def _explicit_artifact_pair( + config: Mapping[str, Any], + *, + include_environment: bool = True, +) -> tuple[Path, Path] | None: + configured = ( + config.get("cuvs_java_jar"), + config.get("cuvs_lucene_jar"), + ) + if configured[0] or configured[1]: + if bool(configured[0]) != bool(configured[1]): + raise RuntimeError( + "Lucene backend configuration must provide both cuvs_java_jar " + "and cuvs_lucene_jar or neither" + ) + values = configured + elif include_environment: + environment = ( + os.environ.get(_JAVA_JAR_ENV), + os.environ.get(_LUCENE_JAR_ENV), + ) + if bool(environment[0]) != bool(environment[1]): + raise RuntimeError( + "Lucene artifact environment overrides must provide both " + f"{_JAVA_JAR_ENV} and {_LUCENE_JAR_ENV} or neither" + ) + values = environment + else: + values = (None, None) + if not values[0]: + return None + return ( + _configured_path(values[0], "cuvs_java_jar"), + _configured_path(values[1], "cuvs_lucene_jar"), + ) + + +def _conventional_artifact_pair() -> tuple[Path, Path] | None: + for java_candidate, lucene_candidate in zip( + _artifact_candidates("cuvs_java_jar"), + _artifact_candidates("cuvs_lucene_jar"), + ): + if java_candidate.is_file() and lucene_candidate.is_file(): + return java_candidate.resolve(), lucene_candidate.resolve() + return None + + +def _required_artifact_pair( + config: Mapping[str, Any], +) -> tuple[Path, Path]: + pair = _explicit_artifact_pair(config) or _conventional_artifact_pair() + if pair is not None: + return pair + searched = ", ".join( + f"({java_path}, {lucene_path})" + for java_path, lucene_path in zip( + _artifact_candidates("cuvs_java_jar"), + _artifact_candidates("cuvs_lucene_jar"), + ) + ) + raise RuntimeError( + "cuVS-backed Lucene algorithms require matching standard cuvs-java " + "and thin cuvs-lucene JARs. Build both artifacts or set " + f"{_JAVA_JAR_ENV} and {_LUCENE_JAR_ENV}. Searched: {searched}" + ) + + +def _python_prefixes() -> Iterable[Path]: + prefixes: list[Path] = [] + + def add(path: Path) -> None: + if path not in prefixes: + prefixes.append(path) + + add(Path(sys.prefix)) + add(Path(sys.base_prefix)) + if conda_prefix := os.environ.get("CONDA_PREFIX"): + add(Path(conda_prefix)) + for entry in sys.path: + path = Path(entry) if entry else None + if path is not None and path.name in { + "site-packages", + "dist-packages", + }: + python_dir = path.parent + if python_dir.parent.name == "lib": + add(python_dir.parent.parent) + yield from prefixes + + +def _cuda_library_directories() -> tuple[Path, ...]: + configured = os.environ.get("CUDA_HOME") or os.environ.get("CUDA_PATH") + candidates = [Path("/usr/local/cuda/lib64")] + if configured: + candidates.insert(0, Path(configured) / "lib64") + return tuple(candidates) + + +def _native_library_groups() -> Iterable[tuple[Path, ...]]: + cuda_directories = _cuda_library_directories() + if cuvs_home := os.environ.get("CUVS_HOME"): + build = Path(cuvs_home) / "cpp" / "build" + yield (build / "c", build, *cuda_directories) + source_build = _REPOSITORY_ROOT / "cpp" / "build" + yield (source_build / "c", source_build, *cuda_directories) + cuda_target = _CUDA_TARGET_BY_MACHINE.get(platform.machine()) + for prefix in _python_prefixes(): + target_directories = ( + (prefix / "targets" / cuda_target / "lib",) + if cuda_target is not None + else () + ) + yield ( + prefix / "lib", + *target_directories, + *cuda_directories, + ) + + +def _contains_library(directory: Path, pattern: str) -> bool: + return ( + directory.is_dir() and next(directory.glob(pattern), None) is not None + ) + + +def _library_path_directories(value: str) -> list[Path]: + return [ + Path(component).expanduser().resolve() + for component in value.split(os.pathsep) + if component + ] + + +def _has_unversioned_cuvs_c(directories: Iterable[Path]) -> bool: + return any( + (directory / "libcuvs_c.so").is_file() for directory in directories + ) + + +def _validated_explicit_library_path(value: Any) -> str: + try: + path_value = os.fspath(value) + except TypeError as error: + raise TypeError( + f"java_library_path must be path-like, got {value!r}" + ) from error + directories = _library_path_directories(path_value) + if not _has_unversioned_cuvs_c(directories): + raise RuntimeError( + "java_library_path does not contain the required unversioned " + "libcuvs_c.so" + ) + return os.pathsep.join(str(directory) for directory in directories) + + +def _discover_native_library_path() -> str | None: + if configured := os.environ.get("JAVA_LIBRARY_PATH"): + return _validated_explicit_library_path(configured) + ambient = _library_path_directories(os.environ.get("LD_LIBRARY_PATH", "")) + if _has_unversioned_cuvs_c(ambient): + return os.pathsep.join(str(path) for path in ambient) + for group in _native_library_groups(): + directories = [] + seen = set() + for candidate in group: + resolved = candidate.resolve() + if resolved in seen or not resolved.is_dir(): + continue + if not any( + _contains_library(resolved, pattern) + for pattern in ( + "libcuvs.so*", + "libcuvs_c.so*", + "libcudart.so*", + ) + ): + continue + seen.add(resolved) + directories.append(resolved) + if _has_unversioned_cuvs_c(directories): + combined = list(directories) + combined.extend(path for path in ambient if path not in combined) + return os.pathsep.join(str(path) for path in combined) + return None + + +def resolve_lucene_runtime_config( + config: Mapping[str, Any] | None = None, + *, + requires_cuvs: bool, + include_cuvs: bool = False, +) -> dict[str, Any]: + """Resolve explicit overrides, then matching artifacts from known locations.""" + options = dict(config or {}) + explicit_pair = _explicit_artifact_pair( + options, include_environment=requires_cuvs or include_cuvs + ) + if requires_cuvs: + pair = explicit_pair or _required_artifact_pair(options) + elif include_cuvs: + pair = explicit_pair or _conventional_artifact_pair() + else: + pair = explicit_pair + resolved: dict[str, Any] = {"requires_cuvs": requires_cuvs} + if pair is not None: + resolved["cuvs_java_jar"] = str(pair[0]) + resolved["cuvs_lucene_jar"] = str(pair[1]) + + library_path = options.get("java_library_path") + if library_path is not None and pair is not None: + library_path = _validated_explicit_library_path(library_path) + if library_path is None and pair is not None: + library_path = _discover_native_library_path() + if library_path: + resolved["java_library_path"] = os.fspath(library_path) + if "jvm_args" in options: + resolved["jvm_args"] = options["jvm_args"] + return resolved diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py new file mode 100644 index 0000000000..50287522a9 --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime.py @@ -0,0 +1,603 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# + +"""Tests for immutable JVM setup and Java artifact provenance.""" + +import hashlib +import zipfile +from pathlib import Path +from types import SimpleNamespace + +import numpy as np +import pytest + +from cuvs_bench.backends import _lucene_runtime +from cuvs_bench.backends._lucene_runtime import ( + _CleanupStack, + _load_pylucene, + _rollback_writer, + _validate_artifacts, + CagraIndexVerifier, + initialize_pylucene, + LuceneRuntime, +) +from cuvs_bench.backends._lucene_runtime_config import maven_artifact_version + + +_ARTIFACT_VERSION = maven_artifact_version() + + +def _properties(group: str, artifact: str, version: str) -> bytes: + return ( + f"groupId={group}\nartifactId={artifact}\nversion={version}\n" + ).encode() + + +def _write_artifacts( + directory: Path, + *, + java_version: str = _ARTIFACT_VERSION, + lucene_version: str = _ARTIFACT_VERSION, +) -> tuple[Path, Path]: + java_jar = directory / "cuvs-java.jar" + with zipfile.ZipFile(java_jar, "w") as archive: + archive.writestr("META-INF/MANIFEST.MF", "Multi-Release: true\n") + archive.writestr("com/nvidia/cuvs/CagraIndex.class", b"") + archive.writestr("com/nvidia/cuvs/CuVSResources.class", b"") + archive.writestr( + "META-INF/versions/22/com/nvidia/cuvs/spi/JDKProvider.class", b"" + ) + archive.writestr( + "META-INF/maven/com.nvidia.cuvs/cuvs-java/pom.properties", + _properties("com.nvidia.cuvs", "cuvs-java", java_version), + ) + + lucene_jar = directory / "cuvs-lucene.jar" + with zipfile.ZipFile(lucene_jar, "w") as archive: + archive.writestr( + "com/nvidia/cuvs/lucene/CuVS2510GPUVectorsFormat.class", b"" + ) + archive.writestr( + "com/nvidia/cuvs/lucene/CuVS2510GPUSearchCodec.class", b"" + ) + archive.writestr( + "com/nvidia/cuvs/lucene/IndexSearcherTimingBridge.class", b"" + ) + archive.writestr( + "com/nvidia/cuvs/lucene/Lucene101AcceleratedHNSWCodec.class", b"" + ) + archive.writestr( + "META-INF/services/org.apache.lucene.codecs.Codec", + "com.nvidia.cuvs.lucene.CuVS2510GPUSearchCodec\n" + "com.nvidia.cuvs.lucene.Lucene101AcceleratedHNSWCodec\n", + ) + archive.writestr( + "META-INF/maven/com.nvidia.cuvs.lucene/cuvs-lucene/pom.properties", + _properties( + "com.nvidia.cuvs.lucene", "cuvs-lucene", lucene_version + ), + ) + return java_jar, lucene_jar + + +class _FakeEnvironment: + def attachCurrentThread(self) -> None: + pass + + +class _FakeLucene: + VERSION = "10.2.0" + CLASSPATH = "/pylucene/lucene-core.jar" + + def __init__(self) -> None: + self.environment = None + self.initializations: list[tuple[str, tuple[str, ...]]] = [] + + def getVMEnv(self): + return self.environment + + def initVM(self, *, classpath, vmargs): + self.initializations.append((classpath, tuple(vmargs))) + self.environment = _FakeEnvironment() + return self.environment + + +def _reset_jvm_state(monkeypatch: pytest.MonkeyPatch) -> None: + for name in ( + "_INITIALIZED_CLASSPATH", + "_INITIALIZED_VMARGS", + "_INITIALIZED_ARTIFACT_PROVENANCE", + "_INITIALIZED_ARTIFACT_TOKENS", + ): + monkeypatch.setattr(_lucene_runtime, name, None) + + +def test_java_search_timer_reports_class_loading_failure() -> None: + class MissingBridgeClass: + @staticmethod + def forName(_name: str): + raise RuntimeError("unsupported bridge bytecode") + + runtime = object.__new__(LuceneRuntime) + runtime.Class = MissingBridgeClass + + with pytest.raises(RuntimeError) as failure: + runtime._load_java_search_timer() + + assert str(failure.value) == ( + "Could not load or adapt " + "com.nvidia.cuvs.lucene.IndexSearcherTimingBridge through " + "PyLucene/JCC: RuntimeError: unsupported bridge bytecode" + ) + assert isinstance(failure.value.__cause__, RuntimeError) + + +def test_java_search_timer_reports_jcc_adaptation_failure() -> None: + class LoadedBridgeClass: + @staticmethod + def newInstance(): + return object() + + class LoadableClass: + @staticmethod + def forName(_name: str): + return LoadedBridgeClass() + + class UnsupportedFunction: + @staticmethod + def cast_(_instance): + raise TypeError("Function.cast_ rejected bridge") + + runtime = object.__new__(LuceneRuntime) + runtime.Class = LoadableClass + runtime.Function = UnsupportedFunction + + with pytest.raises(RuntimeError) as failure: + runtime._load_java_search_timer() + + assert "TypeError: Function.cast_ rejected bridge" in str(failure.value) + assert isinstance(failure.value.__cause__, TypeError) + + +def test_float32_vectors_are_converted_to_a_jcc_compatible_sequence() -> None: + received = [] + + class RecordingLucene: + @staticmethod + def JArray(element_type: str): + assert element_type == "float" + + def record(values): + received.append(values) + return values + + return record + + runtime = object.__new__(LuceneRuntime) + runtime.lucene = RecordingLucene() + vector = np.asarray([1.25, -2.5], dtype=np.float32) + + converted = runtime._java_vector(vector) + + assert converted == [1.25, -2.5] + assert received == [[1.25, -2.5]] + assert all(type(value) is float for value in received[0]) + + +def test_numeric_document_ids_preserve_score_order_across_leaves() -> None: + class ForwardOnlyDocumentIds: + def __init__(self, values: dict[int, int]) -> None: + self.values = values + self.current = -1 + self.visited = [] + + def advanceExact(self, document_id: int) -> bool: + assert document_id >= self.current + self.current = document_id + self.visited.append(document_id) + return document_id in self.values + + def longValue(self) -> int: + return self.values[self.current] + + class LeafReader: + def __init__( + self, max_doc: int, document_ids: ForwardOnlyDocumentIds + ) -> None: + self.max_doc = max_doc + self.document_ids = document_ids + + def maxDoc(self) -> int: + return self.max_doc + + def getNumericDocValues(self, field: str): + assert field == "id" + return self.document_ids + + class Leaf: + def __init__(self, doc_base: int, reader: LeafReader) -> None: + self.docBase = doc_base + self._reader = reader + + def reader(self) -> LeafReader: + return self._reader + + class Reader: + def __init__(self, leaves) -> None: + self._leaves = leaves + + def leaves(self): + return self._leaves + + first_ids = ForwardOnlyDocumentIds({2: 102, 5: 105}) + second_ids = ForwardOnlyDocumentIds({2: 108}) + reader = Reader( + [ + Leaf(0, LeafReader(6, first_ids)), + Leaf(6, LeafReader(4, second_ids)), + ] + ) + score_docs = [ + SimpleNamespace(doc=8, score=0.9), + SimpleNamespace(doc=2, score=0.8), + SimpleNamespace(doc=5, score=0.7), + ] + + hits = LuceneRuntime._materialize_hits(reader, score_docs) + + assert first_ids.visited == [2, 5] + assert second_ids.visited == [2] + assert hits == [ + _lucene_runtime.SearchHit(108, 0.9), + _lucene_runtime.SearchHit(102, 0.8), + _lucene_runtime.SearchHit(105, 0.7), + ] + + +def test_missing_numeric_document_id_requests_an_index_rebuild() -> None: + class MissingDocumentIds: + @staticmethod + def advanceExact(_document_id: int) -> bool: + return False + + class LeafReader: + @staticmethod + def maxDoc() -> int: + return 4 + + @staticmethod + def getNumericDocValues(_field: str): + return MissingDocumentIds() + + class Leaf: + docBase = 0 + + @staticmethod + def reader(): + return LeafReader() + + class Reader: + @staticmethod + def leaves(): + return [Leaf()] + + with pytest.raises(RuntimeError, match="rebuild the index with --force"): + LuceneRuntime._materialize_hits( + Reader(), [SimpleNamespace(doc=3, score=1.0)] + ) + + +def test_missing_numeric_document_id_field_requests_an_index_rebuild() -> None: + class LeafReader: + @staticmethod + def maxDoc() -> int: + return 1 + + @staticmethod + def getNumericDocValues(_field: str): + return None + + class Leaf: + docBase = 0 + + @staticmethod + def reader(): + return LeafReader() + + class Reader: + @staticmethod + def leaves(): + return [Leaf()] + + with pytest.raises(RuntimeError, match="rebuild the index with --force"): + LuceneRuntime._materialize_hits( + Reader(), [SimpleNamespace(doc=0, score=1.0)] + ) + + +def test_cagra_verifier_selects_only_the_current_noncompound_segment() -> None: + class RootDirectory: + @staticmethod + def listAll(): + return ("_c.vemc", "_n.vemc") + + class SegmentInfo: + @staticmethod + def files(): + return ("_c.si", "_c.vcag", "_c.vemc") + + class Segment: + info = SegmentInfo() + + assert CagraIndexVerifier._metadata_files( + RootDirectory(), Segment(), compound=False + ) == ["_c.vemc"] + + +def test_cagra_verifier_reads_metadata_inside_a_compound_segment() -> None: + class CompoundDirectory: + @staticmethod + def listAll(): + return ("_c.fnm", "_c.vcag", "_c.vemc") + + class SegmentInfo: + @staticmethod + def files(): + return ("_c.cfe", "_c.cfs", "_c.si") + + class Segment: + info = SegmentInfo() + + assert CagraIndexVerifier._metadata_files( + CompoundDirectory(), Segment(), compound=True + ) == ["_c.vemc"] + + +def test_missing_pylucene_reports_the_required_runtime( + monkeypatch: pytest.MonkeyPatch, +) -> None: + def missing_module(_name: str): + raise ImportError("no lucene module") + + monkeypatch.setattr( + _lucene_runtime.importlib, "import_module", missing_module + ) + + with pytest.raises(ImportError) as error: + _load_pylucene() + + message = str(error.value) + assert "requires the custom PyLucene 10.2.0 runtime" in message + assert "Install and activate the optional Lucene runtime" in message + assert "http" not in message + assert ".md" not in message + assert ".sh" not in message + assert "PyLucene import failed: no lucene module" in message + + +def test_incompatible_pylucene_version_fails_before_vm_initialization( + monkeypatch: pytest.MonkeyPatch, +) -> None: + fake_lucene = _FakeLucene() + fake_lucene.VERSION = "10.1.0" + _reset_jvm_state(monkeypatch) + monkeypatch.setattr(_lucene_runtime, "_load_pylucene", lambda: fake_lucene) + + with pytest.raises(RuntimeError) as error: + initialize_pylucene({}) + + message = str(error.value) + assert "expected 10.2.0, found 10.1.0" in message + assert "Install and activate a compatible PyLucene runtime" in message + assert "http" not in message + assert ".md" not in message + assert ".sh" not in message + assert fake_lucene.initializations == [] + + +def test_artifact_validation_returns_exact_coordinates_paths_and_hashes( + tmp_path: Path, +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + + provenance, tokens = _validate_artifacts(java_jar, lucene_jar) + + assert provenance == { + "cuvs_java_coordinates": ( + f"com.nvidia.cuvs:cuvs-java:{_ARTIFACT_VERSION}" + ), + "cuvs_java_jar_path": str(java_jar), + "cuvs_java_jar_sha256": hashlib.sha256( + java_jar.read_bytes() + ).hexdigest(), + "cuvs_lucene_coordinates": ( + f"com.nvidia.cuvs.lucene:cuvs-lucene:{_ARTIFACT_VERSION}" + ), + "cuvs_lucene_jar_path": str(lucene_jar), + "cuvs_lucene_jar_sha256": hashlib.sha256( + lucene_jar.read_bytes() + ).hexdigest(), + } + assert set(tokens) == {str(java_jar), str(lucene_jar)} + + +def test_artifact_validation_rejects_mismatched_versions( + tmp_path: Path, +) -> None: + java_jar, lucene_jar = _write_artifacts( + tmp_path, lucene_version=f"{_ARTIFACT_VERSION}-mismatch" + ) + + with pytest.raises(RuntimeError, match="JAR versions differ"): + _validate_artifacts(java_jar, lucene_jar) + + +def test_artifact_validation_rejects_a_native_assembled_jar( + tmp_path: Path, +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + with zipfile.ZipFile(java_jar, "a") as archive: + archive.writestr("native/linux-x86_64/libcuvs.so", b"native") + + with pytest.raises(RuntimeError, match="embeds native libraries"): + _validate_artifacts(java_jar, lucene_jar) + + +def test_artifact_validation_rejects_an_unreadable_jar(tmp_path: Path) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + java_jar.write_bytes(b"not a zip archive") + + with pytest.raises(RuntimeError, match="not a readable JAR"): + _validate_artifacts(java_jar, lucene_jar) + + +def test_artifact_validation_rejects_a_lucene_assembled_jar( + tmp_path: Path, +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + with zipfile.ZipFile(lucene_jar, "a") as archive: + archive.writestr("org/apache/lucene/index/IndexReader.class", b"") + + with pytest.raises(RuntimeError, match="bundles Lucene classes"): + _validate_artifacts(java_jar, lucene_jar) + + +@pytest.mark.parametrize("control_error", (KeyboardInterrupt, SystemExit)) +def test_cleanup_process_control_takes_precedence_over_an_ordinary_failure( + control_error: type[BaseException], +) -> None: + def interrupt_cleanup() -> None: + raise control_error("cleanup interrupted") + + with pytest.raises(control_error) as failure: + with _CleanupStack() as cleanups: + cleanups.add("close test resource", interrupt_cleanup) + raise RuntimeError("operation failed") + + assert failure.value.__notes__ == [ + "Resource handling first failed: RuntimeError: operation failed" + ] + + +@pytest.mark.parametrize("control_error", (KeyboardInterrupt, SystemExit)) +def test_writer_rollback_process_control_takes_precedence( + control_error: type[BaseException], +) -> None: + class InterruptingWriter: + @staticmethod + def rollback() -> None: + raise control_error("rollback interrupted") + + with pytest.raises(control_error) as failure: + _rollback_writer(InterruptingWriter(), RuntimeError("write failed")) + + assert failure.value.__notes__ == [ + "Lucene writer first failed: RuntimeError: write failed" + ] + + +def test_reused_jvm_reports_the_initialized_artifact_identity( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + fake_lucene = _FakeLucene() + _reset_jvm_state(monkeypatch) + monkeypatch.setattr(_lucene_runtime, "_load_pylucene", lambda: fake_lucene) + config = { + "cuvs_java_jar": str(java_jar), + "cuvs_lucene_jar": str(lucene_jar), + } + + first = initialize_pylucene(config) + second = initialize_pylucene(config) + + assert len(fake_lucene.initializations) == 1 + assert second[1:] == first[1:] + + +def test_reused_jvm_rejects_a_same_path_artifact_replacement( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + fake_lucene = _FakeLucene() + _reset_jvm_state(monkeypatch) + monkeypatch.setattr(_lucene_runtime, "_load_pylucene", lambda: fake_lucene) + config = { + "cuvs_java_jar": str(java_jar), + "cuvs_lucene_jar": str(lucene_jar), + } + initialize_pylucene(config) + with zipfile.ZipFile(java_jar, "a") as archive: + archive.writestr("replacement", b"changed") + + with pytest.raises(RuntimeError, match="changed after"): + initialize_pylucene(config) + + +def test_reused_jvm_hashes_artifacts_when_stat_tokens_collide( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + fake_lucene = _FakeLucene() + _reset_jvm_state(monkeypatch) + monkeypatch.setattr(_lucene_runtime, "_load_pylucene", lambda: fake_lucene) + config = { + "cuvs_java_jar": str(java_jar), + "cuvs_lucene_jar": str(lucene_jar), + } + initialize_pylucene(config) + original_tokens = dict(_lucene_runtime._INITIALIZED_ARTIFACT_TOKENS or {}) + replacement = bytearray(java_jar.read_bytes()) + replacement[len(replacement) // 2] ^= 1 + java_jar.write_bytes(replacement) + monkeypatch.setattr( + _lucene_runtime, + "_artifact_stat_token", + lambda path: original_tokens[str(path)], + ) + + with pytest.raises(RuntimeError, match="changed after"): + initialize_pylucene(config) + + +def test_artifact_hashing_occurs_once_per_operation_boundary( + monkeypatch: pytest.MonkeyPatch, +) -> None: + runtime = object.__new__(LuceneRuntime) + runtime._artifact_tokens = {} + runtime.artifact_provenance = {} + runtime.lucene = _FakeLucene() + runtime.lucene.environment = _FakeEnvironment() + hash_verifications = [] + monkeypatch.setattr( + _lucene_runtime, + "_verify_artifact_tokens", + lambda _tokens, _provenance: hash_verifications.append(True), + ) + + runtime.verify_artifacts() + runtime.attach_current_thread() + runtime.attach_current_thread() + + assert hash_verifications == [True] + + +def test_reused_jvm_rejects_different_vm_arguments( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _write_artifacts(tmp_path) + fake_lucene = _FakeLucene() + _reset_jvm_state(monkeypatch) + monkeypatch.setattr(_lucene_runtime, "_load_pylucene", lambda: fake_lucene) + config = { + "cuvs_java_jar": str(java_jar), + "cuvs_lucene_jar": str(lucene_jar), + } + initialize_pylucene(config) + + with pytest.raises( + RuntimeError, match="different classpath or JVM arguments" + ): + initialize_pylucene({**config, "jvm_args": ["-Xmx1g"]}) diff --git a/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime_config.py b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime_config.py new file mode 100644 index 0000000000..2fd258c312 --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/tests/test_lucene_runtime_config.py @@ -0,0 +1,309 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# + +"""Tests for locating the optional Lucene runtime artifact pair.""" + +import os +from pathlib import Path + +import pytest + +from cuvs_bench.backends import _lucene_runtime_config +from cuvs_bench.backends._lucene_runtime_config import ( + resolve_lucene_runtime_config, +) + + +_ARTIFACT_ENVIRONMENT = ( + "CUVS_LUCENE_CUVS_JAVA_JAR", + "CUVS_LUCENE_JAR", +) + + +def _empty_jar_pair(directory: Path, prefix: str = "") -> tuple[Path, Path]: + java_jar = directory / f"{prefix}cuvs-java.jar" + lucene_jar = directory / f"{prefix}cuvs-lucene.jar" + java_jar.touch() + lucene_jar.touch() + return java_jar, lucene_jar + + +def _native_directory(directory: Path) -> Path: + native = directory / "native" + native.mkdir(exist_ok=True) + (native / "libcuvs_c.so").touch() + return native + + +def test_explicit_artifact_pair_is_resolved_together(tmp_path: Path) -> None: + java_jar, lucene_jar = _empty_jar_pair(tmp_path) + native = _native_directory(tmp_path) + + resolved = resolve_lucene_runtime_config( + { + "cuvs_java_jar": java_jar, + "cuvs_lucene_jar": lucene_jar, + "java_library_path": native, + "jvm_args": ["-Xmx1g"], + }, + requires_cuvs=True, + ) + + assert resolved == { + "requires_cuvs": True, + "cuvs_java_jar": str(java_jar.resolve()), + "cuvs_lucene_jar": str(lucene_jar.resolve()), + "java_library_path": str(native.resolve()), + "jvm_args": ["-Xmx1g"], + } + + +@pytest.mark.parametrize( + "configured_key", + ("cuvs_java_jar", "cuvs_lucene_jar"), +) +def test_partial_artifact_override_names_both_required_settings( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, configured_key: str +) -> None: + for variable in _ARTIFACT_ENVIRONMENT: + monkeypatch.delenv(variable, raising=False) + artifact = tmp_path / "one.jar" + artifact.touch() + + with pytest.raises(RuntimeError) as error: + resolve_lucene_runtime_config( + {configured_key: artifact}, requires_cuvs=True + ) + + assert "both cuvs_java_jar and cuvs_lucene_jar" in str(error.value) + + +def test_environment_artifact_pair_is_used_when_config_has_no_override( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _empty_jar_pair(tmp_path) + native = _native_directory(tmp_path) + monkeypatch.setenv("CUVS_LUCENE_CUVS_JAVA_JAR", str(java_jar)) + monkeypatch.setenv("CUVS_LUCENE_JAR", str(lucene_jar)) + + resolved = resolve_lucene_runtime_config( + {"java_library_path": str(native)}, requires_cuvs=True + ) + + assert resolved["cuvs_java_jar"] == str(java_jar.resolve()) + assert resolved["cuvs_lucene_jar"] == str(lucene_jar.resolve()) + assert resolved["java_library_path"] == str(native.resolve()) + + +@pytest.mark.parametrize("configured_environment", ("neither", "java", "both")) +def test_cpu_only_resolution_ignores_ambient_cuvs_artifacts( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + configured_environment: str, +) -> None: + for variable in _ARTIFACT_ENVIRONMENT: + monkeypatch.delenv(variable, raising=False) + java_jar, lucene_jar = _empty_jar_pair(tmp_path) + if configured_environment in {"java", "both"}: + monkeypatch.setenv("CUVS_LUCENE_CUVS_JAVA_JAR", str(java_jar)) + if configured_environment == "both": + monkeypatch.setenv("CUVS_LUCENE_JAR", str(lucene_jar)) + + resolved = resolve_lucene_runtime_config( + requires_cuvs=False, include_cuvs=False + ) + + assert resolved == {"requires_cuvs": False} + + +def test_cpu_only_resolution_still_rejects_partial_explicit_artifacts( + tmp_path: Path, +) -> None: + java_jar, _lucene_jar = _empty_jar_pair(tmp_path) + + with pytest.raises(RuntimeError, match="both cuvs_java_jar"): + resolve_lucene_runtime_config( + {"cuvs_java_jar": java_jar}, + requires_cuvs=False, + include_cuvs=False, + ) + + +def test_backend_config_artifacts_take_precedence_over_environment_pair( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + configured_java, configured_lucene = _empty_jar_pair( + tmp_path, "configured-" + ) + environment_java, environment_lucene = _empty_jar_pair( + tmp_path, "environment-" + ) + native = _native_directory(tmp_path) + monkeypatch.setenv("CUVS_LUCENE_CUVS_JAVA_JAR", str(environment_java)) + monkeypatch.setenv("CUVS_LUCENE_JAR", str(environment_lucene)) + + resolved = resolve_lucene_runtime_config( + { + "cuvs_java_jar": configured_java, + "cuvs_lucene_jar": configured_lucene, + "java_library_path": str(native), + }, + requires_cuvs=True, + ) + + assert resolved["cuvs_java_jar"] == str(configured_java.resolve()) + assert resolved["cuvs_lucene_jar"] == str(configured_lucene.resolve()) + + +def test_complete_backend_pair_ignores_an_incomplete_environment_override( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + configured_java, configured_lucene = _empty_jar_pair(tmp_path) + native = _native_directory(tmp_path) + unrelated_environment_jar = tmp_path / "environment-java.jar" + unrelated_environment_jar.touch() + monkeypatch.setenv( + "CUVS_LUCENE_CUVS_JAVA_JAR", str(unrelated_environment_jar) + ) + monkeypatch.delenv("CUVS_LUCENE_JAR", raising=False) + + resolved = resolve_lucene_runtime_config( + { + "cuvs_java_jar": configured_java, + "cuvs_lucene_jar": configured_lucene, + "java_library_path": native, + }, + requires_cuvs=True, + ) + + assert resolved["cuvs_java_jar"] == str(configured_java.resolve()) + assert resolved["cuvs_lucene_jar"] == str(configured_lucene.resolve()) + + +def test_artifacts_from_configuration_and_environment_cannot_form_a_pair( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + configured_lucene = tmp_path / "configured-lucene.jar" + environment_java = tmp_path / "environment-java.jar" + configured_lucene.touch() + environment_java.touch() + monkeypatch.setenv("CUVS_LUCENE_CUVS_JAVA_JAR", str(environment_java)) + + with pytest.raises( + RuntimeError, match="both cuvs_java_jar and cuvs_lucene_jar" + ): + resolve_lucene_runtime_config( + {"cuvs_lucene_jar": configured_lucene}, requires_cuvs=True + ) + + +def test_explicit_native_path_must_contain_unversioned_cuvs_c( + tmp_path: Path, +) -> None: + java_jar, lucene_jar = _empty_jar_pair(tmp_path) + empty_native = tmp_path / "empty-native" + empty_native.mkdir() + + with pytest.raises(RuntimeError, match="unversioned libcuvs_c.so"): + resolve_lucene_runtime_config( + { + "cuvs_java_jar": java_jar, + "cuvs_lucene_jar": lucene_jar, + "java_library_path": empty_native, + }, + requires_cuvs=True, + ) + + +def test_cuda_only_ld_library_path_is_augmented_with_discovered_cuvs( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + java_jar, lucene_jar = _empty_jar_pair(tmp_path) + cuvs_native = _native_directory(tmp_path) + cuda_native = tmp_path / "cuda" + cuda_native.mkdir() + (cuda_native / "libcudart.so").touch() + monkeypatch.setenv("LD_LIBRARY_PATH", str(cuda_native)) + monkeypatch.delenv("JAVA_LIBRARY_PATH", raising=False) + monkeypatch.setattr( + _lucene_runtime_config, + "_native_library_groups", + lambda: iter(((cuvs_native,),)), + ) + + resolved = resolve_lucene_runtime_config( + {"cuvs_java_jar": java_jar, "cuvs_lucene_jar": lucene_jar}, + requires_cuvs=True, + ) + + assert resolved["java_library_path"].split(":") == [ + str(cuvs_native.resolve()), + str(cuda_native.resolve()), + ] + + +@pytest.mark.parametrize( + ("machine", "cuda_target"), + ( + pytest.param("x86_64", "x86_64-linux", id="x86-64"), + pytest.param("aarch64", "sbsa-linux", id="arm64"), + ), +) +def test_native_library_discovery_uses_the_host_cuda_target( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + machine: str, + cuda_target: str, +) -> None: + prefix = tmp_path / "python-prefix" + cuvs_directory = prefix / "lib" + cuda_directory = prefix / "targets" / cuda_target / "lib" + cuvs_directory.mkdir(parents=True) + cuda_directory.mkdir(parents=True) + (cuvs_directory / "libcuvs_c.so").touch() + (cuda_directory / "libcudart.so").touch() + monkeypatch.delenv("CUVS_HOME", raising=False) + monkeypatch.delenv("JAVA_LIBRARY_PATH", raising=False) + monkeypatch.delenv("LD_LIBRARY_PATH", raising=False) + monkeypatch.setattr( + _lucene_runtime_config.platform, "machine", lambda: machine + ) + monkeypatch.setattr( + _lucene_runtime_config, "_REPOSITORY_ROOT", tmp_path / "repository" + ) + monkeypatch.setattr( + _lucene_runtime_config, "_python_prefixes", lambda: (prefix,) + ) + monkeypatch.setattr( + _lucene_runtime_config, "_cuda_library_directories", tuple + ) + + discovered = _lucene_runtime_config._discover_native_library_path() + + assert discovered == os.pathsep.join( + (str(cuvs_directory.resolve()), str(cuda_directory.resolve())) + ) + + +def test_unknown_host_architecture_does_not_assume_an_x86_cuda_target( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + prefix = tmp_path / "python-prefix" + monkeypatch.setattr( + _lucene_runtime_config.platform, "machine", lambda: "riscv64" + ) + monkeypatch.setattr( + _lucene_runtime_config, "_REPOSITORY_ROOT", tmp_path / "repository" + ) + monkeypatch.setattr( + _lucene_runtime_config, "_python_prefixes", lambda: (prefix,) + ) + monkeypatch.setattr( + _lucene_runtime_config, "_cuda_library_directories", tuple + ) + + groups = tuple(_lucene_runtime_config._native_library_groups()) + + assert groups[-1] == (prefix / "lib",) diff --git a/python/cuvs_bench/cuvs_bench/tests/test_pylucene_builder.py b/python/cuvs_bench/cuvs_bench/tests/test_pylucene_builder.py new file mode 100644 index 0000000000..aee7bed70a --- /dev/null +++ b/python/cuvs_bench/cuvs_bench/tests/test_pylucene_builder.py @@ -0,0 +1,76 @@ +# +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Tests for the pinned PyLucene source builder.""" + +import os +import subprocess +from pathlib import Path + + +_REPOSITORY_ROOT = Path(__file__).parents[4] +_RECIPE_DIRECTORY = _REPOSITORY_ROOT / "conda" / "recipes" / "cuvs-bench" +_BUILD_HELPER = _RECIPE_DIRECTORY / "build_pylucene_10_2.sh" +_COMPATIBILITY_PATCH = _RECIPE_DIRECTORY / "pylucene-10.2.0.patch" + + +def test_source_builder_is_self_describing() -> None: + completed = subprocess.run( + ["bash", str(_BUILD_HELPER), "--help"], + check=False, + capture_output=True, + text=True, + ) + + assert os.access(_BUILD_HELPER, os.X_OK) + assert _COMPATIBILITY_PATCH.is_file() + assert completed.returncode == 0, completed.stderr + assert "Build an isolated PyLucene 10.2.0 environment" in completed.stdout + + +def test_source_builder_rejects_unsafe_build_roots(tmp_path: Path) -> None: + for unsafe_character in (" ", ":", ";", "$", "`", "&", "#", "|"): + build_root = tmp_path / f"unsafe{unsafe_character}root" + completed = subprocess.run( + [ + "bash", + str(_BUILD_HELPER), + "--build-root", + str(build_root), + "--prepare-only", + ], + check=False, + capture_output=True, + text=True, + ) + + assert completed.returncode != 0 + assert "--build-root may contain only" in completed.stderr + assert not build_root.exists() + + +def test_source_builder_rejects_unsafe_resolved_build_root( + tmp_path: Path, +) -> None: + unsafe_target = tmp_path / "unsafe;target" + unsafe_target.mkdir() + build_root = tmp_path / "safe-link" + build_root.symlink_to(unsafe_target, target_is_directory=True) + + completed = subprocess.run( + [ + "bash", + str(_BUILD_HELPER), + "--build-root", + str(build_root), + "--prepare-only", + ], + check=False, + capture_output=True, + text=True, + ) + + assert completed.returncode != 0 + assert "--build-root may contain only" in completed.stderr + assert not (unsafe_target / ".build.lock").exists() From 1f6460d76596a7313e5dc54babf4b123d81c60f9 Mon Sep 17 00:00:00 2001 From: nvzm123 Date: Tue, 6 Oct 2026 21:12:56 +0000 Subject: [PATCH 2/2] Document CAGRA verifier format ownership --- python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py index ba933e4bcd..429472fedb 100644 --- a/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py +++ b/python/cuvs_bench/cuvs_bench/backends/_lucene_runtime.py @@ -30,6 +30,11 @@ _ID_FIELD = "id" _VECTOR_FIELD = "vector" _MAX_DIMENSIONS = 4096 +# These private-format constants power an independent verifier. Keep them +# synchronized with +# CuVS2510GPUVectorsFormat.VERSION_CURRENT, +# CuVS2510GPUVectorsWriter.writeMeta(), and +# CuVS2510GPUVectorsReader.FieldEntry.readEntry(). _CAGRA_META_EXTENSION = ".vemc" _CAGRA_META_CODEC_NAME = "Lucene102CuVSVectorsFormatMeta" _CAGRA_DATA_EXTENSION = ".vcag"