diff --git a/.github/workflows/unit-tests-arm64.yaml b/.github/workflows/unit-tests-arm64.yaml new file mode 100644 index 000000000..a54844fc6 --- /dev/null +++ b/.github/workflows/unit-tests-arm64.yaml @@ -0,0 +1,137 @@ +name: Unit Test CI — ARM64 (NEON / SVE) + +on: + workflow_dispatch: + pull_request: + push: + branches: + - main + paths: + - .github/workflows/unit-tests-arm64.yaml + - '**.java' + - '**/pom.xml' + +jobs: + build-arm64: + concurrency: + group: arm64-${{ matrix.max_isa }}-${{ matrix.jdk }} + cancel-in-progress: false + strategy: + matrix: + jdk: [ 24 ] + # Three ISA tiers in ascending capability order, mirroring avx512f/avx2/sse42. + # GitHub-hosted ubuntu-24.04-arm is a Neoverse-N1 (Graviton 2): NEON only, no SVE. + # The sve/sve2 matrix entries still exercise the JVECTOR_MAX_ISA cap path and + # compile all three ISA variants; the native kernel tests that require actual SVE + # hardware are gated on the runtime feature check below. + max_isa: [ neon, sve, sve2 ] + runs-on: ubuntu-24.04-arm + steps: + - name: Report ARM64 ISA capabilities + id: cpu-features + run: | + # Parse the "Features" line from /proc/cpuinfo — the kernel only exposes a + # token here when the OS has set up context-switch support for it, so this is + # the same authority as getauxval(AT_HWCAP / AT_HWCAP2). + # "asimd" is the NEON token; "sve"/"sve2"/"sveaes" appear on Graviton 3/4. + flags="$(grep '^Features' /proc/cpuinfo | head -1 | cut -d: -f2)" + has_neon=false; has_sve=false; has_sve2=false + [[ " $flags " == *" asimd "* ]] && has_neon=true + [[ " $flags " == *" sve "* ]] && has_sve=true + [[ " $flags " == *" sve2 "* ]] && has_sve2=true + printf "NEON=%s SVE=%s SVE2=%s\n" "$has_neon" "$has_sve" "$has_sve2" + if [[ "$has_neon" != "true" ]]; then + echo "ERROR: NEON (asimd) not found in /proc/cpuinfo — not a valid AArch64 runner" + exit 2 + fi + # Expose as step outputs for conditional steps below. + echo "has_neon=$has_neon" >> "$GITHUB_OUTPUT" + echo "has_sve=$has_sve" >> "$GITHUB_OUTPUT" + echo "has_sve2=$has_sve2" >> "$GITHUB_OUTPUT" + + - name: Set up GCC + run: | + sudo apt install -y gcc g++ + + - name: Install Meson, Ninja, and GTest + run: | + sudo apt update && sudo apt install -y meson ninja-build pkg-config libgtest-dev + + - uses: actions/checkout@v4 + + - name: Initialize Git Submodules + run: git submodule update --init + + - name: Build test_simd_kernels (native C++) + # Meson detects aarch64 and compiles all three ISA variants (neon/sve/sve2) + # regardless of what the host CPU supports at runtime. + working-directory: jvector-native/src/main/native + run: | + meson setup build --wipe + ninja -C build test_simd_kernels + + - name: Run test_simd_kernels — no ISA cap (auto-detect, neon job) + if: matrix.max_isa == 'neon' + working-directory: jvector-native/src/main/native + run: ./build/test_simd_kernels + + - name: Run test_simd_kernels — capped at neon (sve job, host may lack SVE) + if: matrix.max_isa == 'sve' + working-directory: jvector-native/src/main/native + env: + JVECTOR_MAX_ISA: neon + run: ./build/test_simd_kernels + + - name: Run test_simd_kernels — no ISA cap on SVE hardware (sve job) + if: matrix.max_isa == 'sve' && steps.cpu-features.outputs.has_sve == 'true' + working-directory: jvector-native/src/main/native + run: ./build/test_simd_kernels + + - name: Run test_simd_kernels — capped at neon (sve2 job baseline check) + if: matrix.max_isa == 'sve2' + working-directory: jvector-native/src/main/native + env: + JVECTOR_MAX_ISA: neon + run: ./build/test_simd_kernels + + - name: Run test_simd_kernels — no ISA cap on SVE2 hardware (sve2 job) + if: matrix.max_isa == 'sve2' && steps.cpu-features.outputs.has_sve2 == 'true' + working-directory: jvector-native/src/main/native + run: ./build/test_simd_kernels + + - name: Set up JDK ${{ matrix.jdk }} + uses: actions/setup-java@v3 + with: + java-version: ${{ matrix.jdk }} + distribution: temurin + cache: maven + + - name: Verify native-access vector support (JDK ${{ matrix.jdk }}) + env: + JVECTOR_MAX_ISA: ${{ matrix.max_isa }} + run: >- + mvn -B -Punix-amd64-profile -pl jvector-tests -am test + -DTest_RequireSpecificVectorizationProvider=NativeVectorizationProvider + -Dsurefire.failIfNoSpecifiedTests=false + -Dtest=TestVectorizationProvider + + - name: Test full suite with native vectorization (JDK ${{ matrix.jdk }}) + env: + JVECTOR_MAX_ISA: ${{ matrix.max_isa }} + run: >- + mvn -B -Punix-amd64-profile test + -DTest_RequireSpecificVectorizationProvider=NativeVectorizationProvider + + - name: Test Summary for (ARM64/max:${{ matrix.max_isa }},JDK${{ matrix.jdk }}) + if: always() + uses: test-summary/action@v2 + with: + paths: | + **/target/surefire-reports/TEST-*.xml + + - name: Upload Surefire Test Results + uses: actions/upload-artifact@v4 + if: always() + with: + name: surefire-results--arm64-${{ matrix.max_isa }}-${{ matrix.jdk }} + path: "**/target/surefire-reports/**" diff --git a/README.md b/README.md index 4d376007b..3d7bb2546 100644 --- a/README.md +++ b/README.md @@ -77,8 +77,8 @@ git clone --recurse-submodules ### Building native libraries The native SIMD library (`libjvector.so`) is built with [Meson](https://mesonbuild.com/) + [Ninja](https://ninja-build.org/) -and requires **g++ 11+**. The entry-point script is -`jvector-native/src/main/native/build_native_lib.sh`. Run it from that directory: +and requires **g++ 11+**. Supported platforms: **Linux x86-64** (SSE4.2, AVX2, AVX-512) and **Linux AArch64** (NEON, SVE, SVE2). +The entry-point script is `jvector-native/src/main/native/build_native_lib.sh`. Run it from that directory. ```bash cd jvector-native/src/main/native diff --git a/docs/release notes/4.0.2/723.performance.md b/docs/release notes/4.0.2/723.performance.md new file mode 100644 index 000000000..66e6851dd --- /dev/null +++ b/docs/release notes/4.0.2/723.performance.md @@ -0,0 +1,91 @@ +### Native SIMD Acceleration on AArch64 (NEON, SVE, SVE2) + +**Description** + +This PR extends JVector's native Highway SIMD backend to **AArch64 (64-bit ARM)**, bringing +the same native acceleration that x86-64 users already have to ARM-based servers and +development machines (e.g. AWS Graviton, Apple Silicon under Linux, Ampere Altra). + +AArch64 support was previously absent from the native layer — ARM hosts silently fell back to +the Panama Vector API path. The native binary now ships three AArch64 ISA tiers, all compiled +as vector-length-agnostic (scalable) code: + +| Tier | Highway target | Notes | +|---|---|---| +| NEON | `HWY_NEON` | Baseline; all AArch64 CPUs | +| SVE | `HWY_SVE` | Scalable Vector Extension | +| SVE2 | `HWY_SVE2` | SVE2 + `i8mm` + `bf16` | + +SVE and SVE2 are compiled as VL-agnostic (scalable) code, so the same binary runs correctly +on any SVE/SVE2-capable CPU regardless of its physical vector length. NEON operates on fixed +128-bit vectors. At JVM startup, the best available tier is selected automatically via CPU +feature detection — identical to the existing CPUID-based dispatch on x86-64. + +The following changes were made: + +- **`NativeVectorizationProvider`** — the hard-coded `x86_64` architecture check is extended + to include `aarch64`, so the provider is offered on ARM Linux hosts. +- **`jvector_arch.h`** — new header introducing `JV_ARCH_X86_64` / `JV_ARCH_AARCH64` + preprocessor macros; all arch-specific code is guarded with `#if`/`#elif`/`#endif`. +- **`jvector_cpu_features.h`** — AArch64 feature probing (NEON, SVE, SVE2 presence and + vector-length queries) added alongside the existing x86 CPUID path. +- **Meson build** — three AArch64 ISA variant targets added; SVE/SVE2 targets skip gracefully + on Clang < 22 or non-Linux hosts where SVE is unavailable. +- **Cross-compilation** — Meson cross-files and updated READMEs provided for building the + AArch64 native library from an x86-64 host. +- **CI** — a dedicated AArch64 GitHub Actions runner added to the matrix so native builds and + kernel correctness are verified on real ARM hardware on every PR. + +**Purpose / Impact** + +Benchmarked on 1M-scale datasets on both AWS Graviton 3 and Graviton 4, Native SIMD (Highway) +consistently outperforms Panama SIMD on AArch64: + +- Up to **40% lower search latency** at the same recall level +- Up to **29% faster index construction** + +- All kernels already ported to Highway for x86-64 (FP32 similarity, PQ, NVQ, element-wise + arithmetic) benefit immediately on AArch64 — no separate ARM implementation was required. +- On AArch64 Linux with native vectorization enabled, the Highway backend is used exclusively + for all similarity and quantization operations; the Panama Vector API path is bypassed. +- On platforms where the native library cannot be loaded, JVector falls back to the Panama or + pure-Java provider transparently, as before. + +**How to Enable** + +Both `libjvector.so` variants — x86-64 and AArch64 — are built and bundled into the release +JAR. At JVM startup, `NativeVectorizationProvider` detects the current architecture and loads +the matching native library automatically. The same flags work on both architectures: + +```bash +java --enable-native-access=ALL-UNNAMED \ + -Djvector.experimental.enable_native_vectorization=true \ + -jar your-app.jar +``` + +To cap the ISA tier for debugging or benchmarking: + +```bash +JVECTOR_MAX_ISA=neon java --enable-native-access=ALL-UNNAMED \ + -Djvector.experimental.enable_native_vectorization=true \ + -jar your-app.jar +``` + +**Building releases:** + +To produce a release JAR with native libraries for all supported +architectures (x86-64 and AArch64), the build must be run with the `-Dnative.crossarch` +Maven property. Without this flag, only the native library for the current host architecture +is built and bundled. + +```bash +mvn package -Dnative.crossarch +``` + + +**Notes** + +- SVE/SVE2 compilation requires Clang ≥ 22; older compilers silently skip those targets and + fall back to NEON. +- There is no change to the on-disk index format; AArch64 and x86-64 indexes are fully + interchangeable. diff --git a/jvector-native/pom.xml b/jvector-native/pom.xml index ac663fbdc..ab0df815a 100644 --- a/jvector-native/pom.xml +++ b/jvector-native/pom.xml @@ -14,8 +14,22 @@ release true + + false + + + + ${project.build.directory}/meson-build + + libjvector-x86_64.so + libjvector-aarch64.so + + + org.apache.maven.plugins @@ -89,12 +103,24 @@ release + + + native.crossarch + + + native.crossarch + + + + true + + + unix-amd64-profile - unix - amd64 + Linux @@ -157,6 +183,7 @@ build_native_lib.sh ${native.buildtype} + ${native.crossarch} false ${project.basedir}/src/main/native/src/ diff --git a/jvector-native/src/main/java/io/github/jbellis/jvector/vector/NativeVectorizationProvider.java b/jvector-native/src/main/java/io/github/jbellis/jvector/vector/NativeVectorizationProvider.java index f73f237a6..213e53f9a 100644 --- a/jvector-native/src/main/java/io/github/jbellis/jvector/vector/NativeVectorizationProvider.java +++ b/jvector-native/src/main/java/io/github/jbellis/jvector/vector/NativeVectorizationProvider.java @@ -31,8 +31,8 @@ public class NativeVectorizationProvider extends VectorizationProvider { public NativeVectorizationProvider() { var arch = System.getProperty("os.arch", ""); - if (!arch.equals("amd64") && !arch.equals("x86_64")) { - throw new UnsupportedOperationException("Native SIMD operations are only supported on x86_64."); + if (!arch.equals("amd64") && !arch.equals("x86_64") && !arch.equals("aarch64")) { + throw new UnsupportedOperationException("Native SIMD operations are only supported on x86_64 and aarch64."); } var libraryLoaded = LibraryLoader.loadJvector(); if (!libraryLoaded) { diff --git a/jvector-native/src/main/java/io/github/jbellis/jvector/vector/cnative/LibraryLoader.java b/jvector-native/src/main/java/io/github/jbellis/jvector/vector/cnative/LibraryLoader.java index 80d7b2d94..763eddf22 100644 --- a/jvector-native/src/main/java/io/github/jbellis/jvector/vector/cnative/LibraryLoader.java +++ b/jvector-native/src/main/java/io/github/jbellis/jvector/vector/cnative/LibraryLoader.java @@ -22,9 +22,32 @@ /** * This class is used to load supporting native libraries. First, it tries to load the library from the system path. * If that fails, it tries to load the library from the classpath (using the usual copying to a tmp directory route). + *

+ * Two resource names are bundled in the jar: + *

    + *
  • {@code /libjvector-x86_64.so} — built natively for x86_64
  • + *
  • {@code /libjvector-aarch64.so} — cross-compiled for aarch64
  • + *
+ * At runtime the correct file is chosen based on {@code os.arch}. */ public class LibraryLoader { private LibraryLoader() {} + + /** + * Returns the classpath resource name for the native library appropriate for the + * current CPU architecture, or {@code null} when the architecture is not supported. + */ + static String resourceNameForArch() { + String arch = System.getProperty("os.arch", ""); + if (arch.equals("aarch64") || arch.equals("arm64")) { + return "/libjvector-aarch64.so"; + } + if (arch.equals("amd64") || arch.equals("x86_64")) { + return "/libjvector-x86_64.so"; + } + return null; + } + public static boolean loadJvector() { try { System.loadLibrary("jvector"); @@ -35,9 +58,14 @@ public static boolean loadJvector() { try { // reinventing the wheel instead of picking up deps, so we'll just use the classloader to load the library // as a resource and then copy it to a tmp directory and load it from there - String libName = System.mapLibraryName("jvector"); - File tmpLibFile = File.createTempFile(libName.substring(0, libName.lastIndexOf('.')), libName.substring(libName.lastIndexOf('.'))); - try (var in = LibraryLoader.class.getResourceAsStream("/" + libName); + String resourceName = resourceNameForArch(); + if (resourceName == null) { + return false; // unsupported architecture + } + String baseName = resourceName.substring(1, resourceName.lastIndexOf('.')); // e.g. "libjvector-aarch64" + String ext = resourceName.substring(resourceName.lastIndexOf('.')); // e.g. ".so" + File tmpLibFile = File.createTempFile(baseName, ext); + try (var in = LibraryLoader.class.getResourceAsStream(resourceName); var out = Files.newOutputStream(tmpLibFile.toPath())) { if (in != null) { in.transferTo(out); @@ -54,4 +82,4 @@ public static boolean loadJvector() { return false; } -} \ No newline at end of file +} diff --git a/jvector-native/src/main/native/README.md b/jvector-native/src/main/native/README.md index 54f5ef441..91e3060ce 100644 --- a/jvector-native/src/main/native/README.md +++ b/jvector-native/src/main/native/README.md @@ -16,16 +16,13 @@ limitations under the License. # JVector Native SIMD Library -This directory contains the C++ source for `libjvector.so`, the native SIMD -backend that accelerates vector operations in JVector via the Java Foreign -Function & Memory (FFM) API. +This directory contains the C++ source for `libjvector-x86_64.so` and +`libjvector-aarch64.so`, the native SIMD backends that accelerate vector +operations in JVector via the Java Foreign Function & Memory (FFM) API. > **Platform support:** Currently enabled on **Linux x86-64** (SSE4.2, AVX2, -> and AVX-512). Windows and macOS are not yet supported. Support for **ARM** -> (NEON and SVE) is planned for the near future; the -> [Google Highway](https://github.com/google/highway) library used for SIMD -> portability already targets both AArch64 targets, which will make the -> extension straightforward. +> AVX-512) and **Linux AArch64** (NEON, SVE, SVE2). Windows and macOS are not +> yet supported. --- @@ -40,8 +37,10 @@ jvector_simd_kernel_list.h - X-Macro table: single source of truth for all ke jvector_cpu_features.h - CPUID/XGETBV-based CPU feature detection assert_hwy_targets.h - Compile-time assertions that the expected HWY target is active meson.build - Build description +aarch64-cross.ini - Meson cross-compile file: build aarch64 from an x86_64 host +x86_64-cross.ini - Meson cross-compile file: build x86_64 from an aarch64 host jextract_generate_bindings.sh - Generate Java bindings from native headers using jextract -build_native_lib.sh - Build the native library +build_native_lib.sh - Build the native library (native arch + optional cross-compile) third_party/highway/ - Google Highway header-only library (git submodule) ``` @@ -53,31 +52,46 @@ third_party/highway/ - Google Highway header-only library (git submodul | Tool | Minimum version | Notes | |------|----------------|-------| -| g++ / clang++ | GCC 11+ | Must support `-march=skylake-avx512` | +| g++ (x86-64) | GCC 11+ | Must support `-march=skylake-avx512` | +| g++ (AArch64) | GCC 11+ | All three tiers (NEON, SVE, SVE2) built | +| clang++ (x86-64) | Clang any recent | Must support `-march=skylake-avx512` (Clang 7+) | +| clang++ (AArch64) | Clang any recent | NEON tier always built; SVE/SVE2 require **Clang ≥ 22** — older Clang produces a NEON-only build with a warning | | [Meson](https://mesonbuild.com/) | 0.55 | `pip install meson` | | [Ninja](https://ninja-build.org/) | any | `sudo apt install ninja-build` | | Git submodules | — | `git submodule update --init` (needed once) | +> **AArch64 compiler note:** On Clang < 22, the SVE and SVE2 tiers are +> automatically skipped at configure time (`meson setup` prints a warning). +> The resulting library contains only the NEON tier and falls back to it at +> runtime even on SVE2-capable hardware (e.g. Graviton 4). Upgrade to +> Clang ≥ 22 or use GCC to get full SVE/SVE2 acceleration. + ### Build steps Run the build script from this directory: ```bash -bash build_native_lib.sh [buildtype] +bash build_native_lib.sh [buildtype] [crossarch] ``` -The `buildtype` parameter is optional and defaults to `release`. Valid values are: -- `release` (default) - Optimized build with no debug symbols -- `debug` - Unoptimized build with debug symbols (`-g -O0`) -- `debugoptimized` - Optimized build with debug symbols (`-g -O2`) +Parameters (all optional): + +| Parameter | Default | Description | +|-----------|---------|-------------| +| `buildtype` | `release` | `release`, `debug`, or `debugoptimized` | +| `crossarch` | `false` | Set to `true` to also cross-compile for the other arch | The script: -1. Verifies prerequisites (g++, meson, ninja, Highway submodule). -2. Runs `meson setup ../../../target/meson-build --wipe --buildtype=` then `meson compile`. -3. Copies the versioned `.so` to `../resources/libjvector.so` where the Java - `LibraryLoader` expects it. +1. Detects the host arch via `uname -m` (x86_64 or aarch64). +2. Verifies prerequisites (g++, meson, ninja, Highway submodule). +3. Always builds the **native arch** library into `target/meson-build-/`. +4. If `crossarch=true`, also cross-compiles the **other arch** library into + `target/meson-build-/` using the bundled cross-compile file. +5. Copies each built `.so` to `src/main/resources/libjvector-.so`. + Debug symbols are stripped from the resource copy and saved separately as + `target/meson-build-/libjvector-.so.debug`. -To install all required dependencies automatically (g++, meson, ninja) and then build, pass `--auto-install-deps`: +To install all required dependencies automatically and then build, pass `--auto-install-deps` as the first argument: ```bash bash build_native_lib.sh --auto-install-deps @@ -87,40 +101,107 @@ This is the easiest way to get started on a fresh Ubuntu machine. For other distributions the script will print an error indicating which install commands need to be added. +### Cross-compilation + +Both `.so` files can be produced on a single machine without access to the +target hardware. + +#### Build aarch64 from an x86_64 host + +Install the cross-toolchain: + +```bash +sudo apt-get install -y g++-aarch64-linux-gnu +``` + +Build both libraries: + +```bash +bash build_native_lib.sh release true +# produces: src/main/resources/libjvector-x86_64.so +# src/main/resources/libjvector-aarch64.so +``` + +#### Build x86_64 from an aarch64 host + +Install the cross-toolchain: + +```bash +sudo apt-get install -y g++-x86-64-linux-gnu +``` + +Build both libraries: + +```bash +bash build_native_lib.sh release true +# produces: src/main/resources/libjvector-aarch64.so +# src/main/resources/libjvector-x86_64.so +``` + +The cross-compile file used for each direction lives in this directory: + +| File | Direction | +|------|-----------| +| [`aarch64-cross.ini`](../aarch64-cross.ini) | x86_64 host → aarch64 target | +| [`x86_64-cross.ini`](../x86_64-cross.ini) | aarch64 host → x86_64 target | + +Meson reads the cross file (via `--cross-file`) and sets +`host_machine.cpu_family` to the target arch, which causes `meson.build` to +select the correct ISA variant list (x86 tiers vs AArch64 tiers) automatically. + ### Building with Maven From the project root, you can build the native module using Maven: -**Release build (default):** +**Native arch only — release (default):** ```bash mvn clean install ``` -**Debug build:** +**Native arch only — debug:** ```bash mvn clean install -Dnative.debug ``` -**Debug-optimized build:** +**Native arch only — debug-optimized:** ```bash mvn clean install -Dnative.debugoptimized ``` -The Maven build automatically invokes the `build_native_lib.sh` script with the -appropriate buildtype parameter. The available profiles are: -- `release` (default) - Optimized build with no debug symbols -- `debug` - Unoptimized build with debug symbols (`-g -O0`) -- `debugoptimized` - Optimized build with debug symbols (`-g -O2`) +**Both architectures (native + cross-compiled):** +```bash +mvn clean install -Dnative.crossarch +``` + +**Both architectures, debug build:** +```bash +mvn clean install -Dnative.crossarch -Dnative.debug +``` + +The `-Dnative.crossarch` flag activates the `native.crossarch` Maven profile, +which passes `crossarch=true` as the second argument to `build_native_lib.sh`. +Without it, only the library for the host architecture is built. ### Manual meson build +**Native build:** ```bash cd jvector-native/src/main/native -meson setup ../../../target/meson-build --wipe --buildtype=release -meson compile -C ../../../target/meson-build +meson setup ../../../target/meson-build-x86_64 . --wipe --buildtype=release +meson compile -C ../../../target/meson-build-x86_64 +``` + +**Cross-compile aarch64 from x86_64:** +```bash +cd jvector-native/src/main/native +meson setup ../../../target/meson-build-aarch64 . --wipe \ + --cross-file aarch64-cross.ini --buildtype=release +meson compile -C ../../../target/meson-build-aarch64 ``` -The output is `target/meson-build/libjvector.so.` (relative to the project root). +The versioned output (e.g. `libjvector.so.0.1.0`) is the real file; copy and +rename it to `src/main/resources/libjvector-.so` to make it available to +`LibraryLoader`. ### Generating Java bindings for native code @@ -271,25 +352,39 @@ JVECTOR_MAX_ISA=sse42 ../../../target/meson-build/bench_simd_kernels ## How it is integrated into JVector +**x86-64:** ``` Java caller └─ NativeVectorUtilSupport (jvector-native/.../vector/) └─ NativeSimdOps (jvector-native/.../vector/cnative/ — FFM glue, generated by jextract) └─ libjvector.so (this library, loaded at runtime by LibraryLoader) └─ jvector_simd.cpp — dispatches to the best ISA vtable - ├─ AVX3_SPR::* (compiled with -march=skylake-avx512 -mavx512fp16 …, Sapphire Rapids) - ├─ AVX3_DL::* (compiled with -march=skylake-avx512 -mavx512vnni …, Ice Lake) + ├─ AVX3_SPR::* (compiled with -march=sapphirerapids) + ├─ AVX3_DL::* (compiled with -march=icelake-server) ├─ AVX3::* (compiled with -march=skylake-avx512) ├─ AVX2::* (compiled with -march=haswell) └─ SSE42::* (compiled with -msse4.2, scalar fallback) ``` +**AArch64:** +``` +Java caller + └─ NativeVectorUtilSupport (jvector-native/.../vector/) + └─ NativeSimdOps (jvector-native/.../vector/cnative/ — FFM glue, generated by jextract) + └─ libjvector.so (this library, loaded at runtime by LibraryLoader) + └─ jvector_simd.cpp — dispatches to the best ISA vtable + ├─ SVE2::* (compiled with -march=armv9-a+sve2, scalable VL) + ├─ SVE::* (compiled with -march=armv8.4-a+sve, scalable VL) + └─ NEON::* (compiled with -march=armv8-a+crypto, baseline) +``` + ### Load sequence 1. `NativeVectorizationProvider` calls `LibraryLoader.loadJvector()` at startup. 2. `LibraryLoader` first tries `System.loadLibrary("jvector")` (picks up a - system-installed `.so`), then falls back to extracting `libjvector.so` from - the JAR's resources and loading it from a temp file. + system-installed `.so`), then falls back to reading `os.arch` to select the + correct resource name (`libjvector-x86_64.so` or `libjvector-aarch64.so`), + extracting it from the JAR, and loading it from a temp file. 3. On first call into `NativeSimdOps`, the FFM `SymbolLookup` resolves each exported symbol directly against the loaded library. @@ -297,9 +392,10 @@ Java caller Dispatch happens **once** at C++ static-init time (before `main()`): -1. `populate_cpu_features()` issues CPUID / XGETBV and fills a feature array. -2. `dispatch_kernels()` checks the feature array in descending capability order - (`AVX3` ⊃ `AVX2` ⊃ `SSE42`) and returns a copy of the matching `KernelVTable`. +1. `populate_cpu_features()` issues CPUID / XGETBV (x86) or `getauxval` (AArch64) and fills a feature array. +2. `dispatch_kernels()` checks the feature array in descending capability order and returns a copy of the matching `KernelVTable`: + - x86-64: `AVX3_SPR` ⊃ `AVX3_DL` ⊃ `AVX3` ⊃ `AVX2` ⊃ `SSE42` + - AArch64: `SVE2` ⊃ `SVE` ⊃ `NEON` 3. All public API functions are one-liner wrappers that call through `kernels.`. @@ -308,15 +404,23 @@ Dispatch happens **once** at C++ static-init time (before `main()`): Set the `JVECTOR_MAX_ISA` environment variable before starting the JVM to cap the selected ISA without recompiling: +**x86-64:** ```bash JVECTOR_MAX_ISA=avx3_spr java ... # use Sapphire-Rapids FP16 tier JVECTOR_MAX_ISA=avx3_dl java ... # use Ice Lake tier -JVECTOR_MAX_ISA=avx3 java ... # use AVX-512 even if a higher tier is available -JVECTOR_MAX_ISA=avx2 java ... # use AVX2 even on an AVX-512 machine -JVECTOR_MAX_ISA=sse42 java ... # force scalar/SSE4.2 fallback +JVECTOR_MAX_ISA=avx3 java ... # use AVX-512 even if a higher tier is available +JVECTOR_MAX_ISA=avx2 java ... # use AVX2 even on an AVX-512 machine +JVECTOR_MAX_ISA=sse42 java ... # force scalar/SSE4.2 fallback +``` + +**AArch64:** +```bash +JVECTOR_MAX_ISA=sve2 java ... # use SVE2 tier (Graviton 4 / Neoverse V2/N2) +JVECTOR_MAX_ISA=sve java ... # use SVE tier (Graviton 3 / Neoverse V1) +JVECTOR_MAX_ISA=neon java ... # force NEON baseline ``` -Accepted values (case-sensitive): `avx3_spr`, `avx3_dl`, `avx3`, `avx2`, `sse42`. +Accepted values (case-sensitive): `avx3_spr`, `avx3_dl`, `avx3`, `avx2`, `sse42` (x86-64); `sve2`, `sve`, `neon` (AArch64). An unrecognised value is silently ignored and full CPU detection is used. --- diff --git a/jvector-native/src/main/native/aarch64-cross.ini b/jvector-native/src/main/native/aarch64-cross.ini new file mode 100644 index 000000000..d9b213b9b --- /dev/null +++ b/jvector-native/src/main/native/aarch64-cross.ini @@ -0,0 +1,36 @@ +# Copyright DataStax, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Meson cross-compilation file for building libjvector for aarch64 on an x86_64 host. +# +# Prerequisites (Ubuntu/Debian): +# sudo apt-get install -y gcc-aarch64-linux-gnu g++-aarch64-linux-gnu +# +# The cross file is passed to meson via: +# meson setup --cross-file /aarch64-cross.ini ... + +[binaries] +c = 'aarch64-linux-gnu-gcc' +cpp = 'aarch64-linux-gnu-g++' +ar = 'aarch64-linux-gnu-ar' +strip = 'aarch64-linux-gnu-strip' +# pkg-config wrapper that searches the sysroot; fall back to plain pkg-config +# if the multiarch wrapper is not installed. +pkg-config = ['aarch64-linux-gnu-pkg-config'] + +[host_machine] +system = 'linux' +cpu_family = 'aarch64' +cpu = 'aarch64' +endian = 'little' diff --git a/jvector-native/src/main/native/meson.build b/jvector-native/src/main/native/meson.build index 42fec1ada..ec85d86e4 100644 --- a/jvector-native/src/main/native/meson.build +++ b/jvector-native/src/main/native/meson.build @@ -25,32 +25,91 @@ hwy_inc = include_directories('third_party/highway') # Each ISA variant: name, JV_ISA namespace, and extra compiler flags. # These all compile jvector_simd_kernels.cpp (the generic kernel file). -isa_variants = [ - { - 'name' : 'avx3', - 'namespace': 'AVX3', - 'args' : ['-march=skylake-avx512', - '-DHWY_COMPILE_ONLY_STATIC', - '-DJV_REQUIRE_HWY_AVX3'], - }, - { - 'name' : 'avx2', - 'namespace': 'AVX2', - 'args' : ['-march=haswell', - '-maes', - '-DHWY_COMPILE_ONLY_STATIC', - '-DJV_REQUIRE_HWY_AVX2'], - }, - { - 'name' : 'sse42', - 'namespace': 'SSE42', - 'args' : ['-msse4.2', '-mpclmul', '-maes', - '-DHWY_COMPILE_ONLY_STATIC', - '-DJV_REQUIRE_HWY_SCALAR'], - }, -] +cpu = host_machine.cpu_family() + +if cpu == 'x86_64' + isa_variants = [ + { + 'name' : 'avx3', + 'namespace': 'AVX3', + 'args' : ['-march=skylake-avx512', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_AVX3'], + }, + { + 'name' : 'avx2', + 'namespace': 'AVX2', + 'args' : ['-march=haswell', + '-maes', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_AVX2'], + }, + { + 'name' : 'sse42', + 'namespace': 'SSE42', + 'args' : ['-msse4.2', '-mpclmul', '-maes', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_SCALAR'], + }, + ] +elif cpu == 'aarch64' + # Three tiers in ascending capability order. SVE and SVE2 use scalable + # Highway targets (HWY_HAVE_SCALABLE=1): Lanes() is a runtime value and + # all loop counters are vector-length agnostic. The fixed-width fast-paths + # in calculate_partial_sums_* are gated behind #if !HWY_HAVE_SCALABLE. + # + # NEON — baseline AArch64. Highway target: HWY_NEON. + # All Graviton generations and all Apple Silicon. + # + # SVE — Graviton 3 / Neoverse V1 and later. Highway target: HWY_SVE. + # No -msve-vector-bits flag; Highway queries the VL at runtime. + # Requires Clang >= 22 for reliable SVE codegen via Highway. + # + # SVE2 — Graviton 4 / Neoverse V2/N2 and later. Highway target: HWY_SVE2. + # No -msve-vector-bits flag; Highway queries the VL at runtime. + # Requires Clang >= 22 for reliable SVE2 codegen via Highway. + isa_variants = [ + { + 'name' : 'neon', + 'namespace': 'NEON', + 'args' : ['-march=armv8-a+crypto', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_NEON'], + }, + ] + + cpp = meson.get_compiler('cpp') + # SVE and SVE2 require Clang >= 22 for correct Highway SVE codegen. + # GCC does not have this restriction; only gate on Clang. + sve_supported = cpp.get_id() != 'clang' or cpp.version().version_compare('>=22') + + if sve_supported + isa_variants += [ + { + 'name' : 'sve', + 'namespace': 'SVE', + 'args' : ['-march=armv8.4-a+sve', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_SVE'], + }, + { + 'name' : 'sve2', + 'namespace': 'SVE2', + 'args' : ['-march=armv9-a+sve2', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_SVE2'], + }, + ] + else + warning('Clang ' + cpp.version() + ' detected — SVE and SVE2 tiers require Clang >= 22 and will not be built. Only the NEON tier will be available.') + endif +else + error('Unsupported CPU family: ' + cpu + '. Supported: x86_64, aarch64.') +endif isa_libs = [] +# Extra -D flags forwarded to jvector_simd.cpp to record which tiers were built. +dispatch_defines = [] foreach isa : isa_variants lib = static_library( 'simdKernels_' + isa['name'], @@ -59,38 +118,41 @@ foreach isa : isa_variants cpp_args : isa['args'] + ['-DJV_ISA=' + isa['namespace'], '-fvisibility=hidden'] ) isa_libs += lib + dispatch_defines += '-DJV_HAS_' + isa['namespace'].to_upper() + '=1' endforeach -# AVX3_DL (Ice Lake) tier: -march=icelake-server enables the full ICX feature -# set (AVX3 + VNNI, VBMI, VBMI2, IFMA, BITALG, VPOPCNTDQ, GFNI, VAES, VPCLMULQDQ). -avx3_dl_lib = static_library( - 'simdKernels_avx3_dl', - sources : 'src/jvector_avx3_dl_kernels.cpp', - include_directories: hwy_inc, - cpp_args : [ - '-march=icelake-server', - '-DHWY_COMPILE_ONLY_STATIC', - '-DJV_REQUIRE_HWY_AVX3_DL', - '-fvisibility=hidden', - ], -) -isa_libs += avx3_dl_lib - -# AVX3_SPR (Sapphire Rapids) tier: -march=sapphirerapids adds AVX512FP16 and -# AVX512BF16 on top of the icelake-server feature set. -# Requires GCC >= 12 or Clang >= 14. -avx3_spr_lib = static_library( - 'simdKernels_avx3_spr', - sources : 'src/jvector_avx3_spr_kernels.cpp', - include_directories: hwy_inc, - cpp_args : [ - '-march=sapphirerapids', - '-DHWY_COMPILE_ONLY_STATIC', - '-DJV_REQUIRE_HWY_AVX3_SPR', - '-fvisibility=hidden', - ], -) -isa_libs += avx3_spr_lib +if cpu == 'x86_64' + # AVX3_DL (Ice Lake) tier: -march=icelake-server enables the full ICX feature + # set (AVX3 + VNNI, VBMI, VBMI2, IFMA, BITALG, VPOPCNTDQ, GFNI, VAES, VPCLMULQDQ). + avx3_dl_lib = static_library( + 'simdKernels_avx3_dl', + sources : 'src/jvector_avx3_dl_kernels.cpp', + include_directories: hwy_inc, + cpp_args : [ + '-march=icelake-server', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_AVX3_DL', + '-fvisibility=hidden', + ], + ) + isa_libs += avx3_dl_lib + + # AVX3_SPR (Sapphire Rapids) tier: -march=sapphirerapids adds AVX512FP16 and + # AVX512BF16 on top of the icelake-server feature set. + # Requires GCC >= 12 or Clang >= 14. + avx3_spr_lib = static_library( + 'simdKernels_avx3_spr', + sources : 'src/jvector_avx3_spr_kernels.cpp', + include_directories: hwy_inc, + cpp_args : [ + '-march=sapphirerapids', + '-DHWY_COMPILE_ONLY_STATIC', + '-DJV_REQUIRE_HWY_AVX3_SPR', + '-fvisibility=hidden', + ], + ) + isa_libs += avx3_spr_lib +endif # vectorUtil provides a single runtime-dispatch entry point. # link_whole pulls every ISA object into the shared library so that no @@ -103,7 +165,7 @@ vectorutil_lib = shared_library( sources : ['src/jvector_simd.cpp', 'third_party/highway/hwy/abort.cc'], include_directories: [include_directories('src'), hwy_inc], - cpp_args : ['-DJVECTOR_BUILD', '-fvisibility=hidden'], + cpp_args : ['-DJVECTOR_BUILD', '-fvisibility=hidden'] + dispatch_defines, link_whole : isa_libs, version : meson.project_version(), install : true, @@ -126,13 +188,18 @@ vectorutil_dep = declare_dependency( gtest_dep = dependency('gtest_main', required: false) if gtest_dep.found() + # Select the architecture-specific CPU features test. + cpu_features_test_src = cpu == 'x86_64' \ + ? 'tests/test_x86_cpu_features.cpp' \ + : 'tests/test_aarch64_cpu_features.cpp' + simd_kernels_test = executable( 'test_simd_kernels', sources : [ 'tests/test_helpers.cpp', 'tests/test_similarity.cpp', 'tests/test_elementwise.cpp', - 'tests/test_cpu_features.cpp', + cpu_features_test_src, ], dependencies: [vectorutil_dep, gtest_dep], ) diff --git a/jvector-native/src/main/native/src/assert_hwy_targets.h b/jvector-native/src/main/native/src/assert_hwy_targets.h index b1f4b6347..1c51bedbd 100644 --- a/jvector-native/src/main/native/src/assert_hwy_targets.h +++ b/jvector-native/src/main/native/src/assert_hwy_targets.h @@ -14,22 +14,49 @@ * limitations under the License. */ -#if defined(__x86_64__) || defined(_M_X64) +#include "jvector_arch.h" + +#if JV_ARCH_X86_64 + #if defined(JV_REQUIRE_HWY_AVX3_SPR) -#if HWY_STATIC_TARGET != HWY_AVX3_SPR -#error "Highway did not select HWY_AVX3_SPR for the Sapphire Rapids build. Check compiler flags, compiler support, and Highway blocklists." -#endif +# if HWY_STATIC_TARGET != HWY_AVX3_SPR +# error "Highway did not select HWY_AVX3_SPR for the Sapphire Rapids build. Check compiler flags, compiler support, and Highway blocklists." +# endif #elif defined(JV_REQUIRE_HWY_AVX3_DL) -#if HWY_STATIC_TARGET != HWY_AVX3_DL -#error "Highway did not select HWY_AVX3_DL for the Ice Lake build. Check compiler flags, compiler support, and Highway blocklists." -#endif +# if HWY_STATIC_TARGET != HWY_AVX3_DL +# error "Highway did not select HWY_AVX3_DL for the Ice Lake build. Check compiler flags, compiler support, and Highway blocklists." +# endif #elif defined(JV_REQUIRE_HWY_AVX3) -#if HWY_STATIC_TARGET != HWY_AVX3 -#error "Highway did not select HWY_AVX3 for the AVX-512 build. Check compiler flags, compiler support, and Highway blocklists." -#endif +# if HWY_STATIC_TARGET != HWY_AVX3 +# error "Highway did not select HWY_AVX3 for the AVX-512 build. Check compiler flags, compiler support, and Highway blocklists." +# endif #elif defined(JV_REQUIRE_HWY_AVX2) -#if HWY_STATIC_TARGET != HWY_AVX2 -#error "Highway did not select HWY_AVX2 for the AVX2 build. Check compiler flags, compiler support, and Highway blocklists." -#endif -#endif // -#endif // __X86_64__ +# if HWY_STATIC_TARGET != HWY_AVX2 +# error "Highway did not select HWY_AVX2 for the AVX2 build. Check compiler flags, compiler support, and Highway blocklists." +# endif +#endif // JV_REQUIRE_HWY_* + +#elif JV_ARCH_AARCH64 + +// Compiler flags per tier and the Highway target each must produce: +// neon: -march=armv8-a+crypto → HWY_NEON +// (no BF16/dotprod/I8MM features, so never HWY_NEON_BF16) +// sve: -march=armv8.4-a+sve → HWY_SVE (scalable, VL-agnostic) +// (Graviton 3 / Neoverse V1 and later) +// sve2: -march=armv9-a+sve2 → HWY_SVE2 (scalable, VL-agnostic) +// (Graviton 4 / Neoverse V2/N2 and later) +#if defined(JV_REQUIRE_HWY_SVE2) +# if HWY_STATIC_TARGET != HWY_SVE2 +# error "Highway did not select HWY_SVE2 for the SVE2 build. Check compiler flags (-march=armv9-a+sve2) and Highway blocklists." +# endif +#elif defined(JV_REQUIRE_HWY_SVE) +# if HWY_STATIC_TARGET != HWY_SVE +# error "Highway did not select HWY_SVE for the SVE build. Check compiler flags (-march=armv8.4-a+sve) and Highway blocklists." +# endif +#elif defined(JV_REQUIRE_HWY_NEON) +# if HWY_STATIC_TARGET != HWY_NEON +# error "Highway did not select HWY_NEON for the NEON build. Check compiler flags (-march=armv8-a+crypto) and Highway blocklists." +# endif +#endif // JV_REQUIRE_HWY_* + +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 diff --git a/jvector-native/src/main/native/src/build_native_lib.sh b/jvector-native/src/main/native/src/build_native_lib.sh index 05b750823..060b1b7a0 100755 --- a/jvector-native/src/main/native/src/build_native_lib.sh +++ b/jvector-native/src/main/native/src/build_native_lib.sh @@ -30,8 +30,10 @@ NATIVE_DIR="${REPO_ROOT}/jvector-native/src/main/native" MODULE_ROOT="${REPO_ROOT}/jvector-native" HIGHWAY_DIR="${NATIVE_DIR}/third_party/highway" -BUILD_DIR="${MODULE_ROOT}/target/meson-build" -RESOURCES_DIR="${MODULE_ROOT}/src/main/resources" +# Drop the .so files into target/meson-build/ so that mvn clean removes them +# along with every other build artefact, and Maven's resource plugin picks them +# up from there instead of from src/main/resources/. +RESOURCES_DIR="${MODULE_ROOT}/target/meson-build" if [ "$1" == "--auto-install-deps" ] ; then AUTO_INSTALL_DEPS=true ; shift ; fi printf "AUTO_INSTALL_DEPS=%s\n" "${AUTO_INSTALL_DEPS}" @@ -45,11 +47,16 @@ if [ "$BUILDTYPE" != "release" ] && [ "$BUILDTYPE" != "debug" ] && [ "$BUILDTYPE fi printf "BUILDTYPE=%s\n" "${BUILDTYPE}" -mkdir -p "${RESOURCES_DIR}" +# Accept crossarch flag (default: false). +# When true, builds both the native arch library AND cross-compiles for the other arch. +# On x86_64: builds libjvector-x86_64.so + cross-compiles libjvector-aarch64.so +# On aarch64: builds libjvector-aarch64.so + cross-compiles libjvector-x86_64.so +CROSSARCH="${2:-false}" +printf "CROSSARCH=%s\n" "${CROSSARCH}" -# compile jvector_simd_check.cpp as x86-64 -# compile jvector_simd.cpp as skylake-avx512 -# produce one shared library +mkdir -p "${RESOURCES_DIR}" +# target/meson-build/ is under target/, which IS cleaned by mvn clean. +# src/main/resources/ is intentionally NOT written to any more. # Check that the Google Highway submodule has been initialised if [ ! -f "${HIGHWAY_DIR}/hwy/highway.h" ]; then @@ -83,33 +90,92 @@ require_cmd() { fi } -require_cmd g++ "sudo apt-get install -y g++" require_cmd meson "sudo apt-get install -y meson" require_cmd ninja "sudo apt-get install -y ninja-build" -# Check g++ version -CURRENT_GPP_VERSION=$(g++ -dumpversion) +# --------------------------------------------------------------------------- +# Detect the host CPU architecture. +# --------------------------------------------------------------------------- +HOST_ARCH="$(uname -m)" # e.g. x86_64 or aarch64 +printf "HOST_ARCH=%s\n" "${HOST_ARCH}" -# Check if the current GCC version is greater than or equal to the minimum required version -if [ "$(printf '%s\n' "$MIN_GCC_VERSION" "$CURRENT_GPP_VERSION" | sort -V | head -n1)" != "$MIN_GCC_VERSION" ]; then - echo "WARNING: g++ version $CURRENT_GPP_VERSION is too old. Please upgrade to g++ $MIN_GCC_VERSION or newer." - exit 1 -fi +# --------------------------------------------------------------------------- +# build_native +# Compiles the library for using either a native build (when arch == +# HOST_ARCH) or a cross-compile via the bundled cross file. +# Output: RESOURCES_DIR/libjvector-.so +# --------------------------------------------------------------------------- +build_native() { + local ARCH="$1" + local BUILD_DIR="${MODULE_ROOT}/target/meson-build-${ARCH}" + local OUT_SO="${RESOURCES_DIR}/libjvector-${ARCH}.so" + + rm -rf "${OUT_SO}" + + if [ "${ARCH}" == "${HOST_ARCH}" ]; then + # ---- Native build ------------------------------------------------------ + require_cmd g++ "sudo apt-get install -y g++" + + CURRENT_GPP_VERSION=$(g++ -dumpversion) + if [ "$(printf '%s\n' "$MIN_GCC_VERSION" "$CURRENT_GPP_VERSION" | sort -V | head -n1)" != "$MIN_GCC_VERSION" ]; then + echo "WARNING: g++ version $CURRENT_GPP_VERSION is too old. Please upgrade to g++ $MIN_GCC_VERSION or newer." + exit 1 + fi + + meson setup "${BUILD_DIR}" "${NATIVE_DIR}" \ + --wipe \ + --buildtype="${BUILDTYPE}" + else + # ---- Cross-compile build ----------------------------------------------- + # Only x86_64 <-> aarch64 is supported. + if [ "${ARCH}" == "aarch64" ]; then + CROSS_TOOLCHAIN="aarch64-linux-gnu-g++" + CROSS_INSTALL="sudo apt-get install -y g++-aarch64-linux-gnu" + CROSS_FILE="${NATIVE_DIR}/aarch64-cross.ini" + elif [ "${ARCH}" == "x86_64" ]; then + CROSS_TOOLCHAIN="x86_64-linux-gnu-g++" + CROSS_INSTALL="sudo apt-get install -y g++-x86-64-linux-gnu" + CROSS_FILE="${NATIVE_DIR}/x86_64-cross.ini" + else + echo "ERROR: No cross-compile support for arch '${ARCH}'." ; exit 1 + fi -rm -rf "${RESOURCES_DIR}/libjvector.so" + require_cmd "${CROSS_TOOLCHAIN}" "${CROSS_INSTALL}" -# Configure (--wipe resets any stale configuration) then compile -meson setup "${BUILD_DIR}" "${NATIVE_DIR}" \ - --wipe \ - --buildtype="${BUILDTYPE}" + meson setup "${BUILD_DIR}" "${NATIVE_DIR}" \ + --wipe \ + --cross-file "${CROSS_FILE}" \ + --buildtype="${BUILDTYPE}" + fi -meson compile -C "${BUILD_DIR}" + meson compile -C "${BUILD_DIR}" -# The versioned .so (e.g. libjvector.so.0.1.0) is the real file; symlinks point to it. -# Copy it to src/main/resources/ so Maven packages it into the jar for LibraryLoader. -SOFILE=$(find "${BUILD_DIR}" -maxdepth 1 -name 'libjvector.so.*' -type f | head -1) -if [ -z "${SOFILE}" ]; then - echo "ERROR: libjvector.so not found in ${BUILD_DIR} after build." + # The versioned .so (e.g. libjvector.so.0.1.0) is the real file; symlinks point to it. + SOFILE=$(find "${BUILD_DIR}" -maxdepth 1 -name 'libjvector.so.*' -type f | head -1) + if [ -z "${SOFILE}" ]; then + echo "ERROR: libjvector.so not found in ${BUILD_DIR} after ${ARCH} build." exit 1 + fi + cp "${SOFILE}" "${OUT_SO}" + printf "Built: %s\n" "${OUT_SO}" +} + +# --------------------------------------------------------------------------- +# Determine which archs to build. +# --------------------------------------------------------------------------- +if [ "${HOST_ARCH}" == "x86_64" ]; then + OTHER_ARCH="aarch64" +elif [ "${HOST_ARCH}" == "aarch64" ]; then + OTHER_ARCH="x86_64" +else + echo "ERROR: Unsupported host architecture '${HOST_ARCH}'. Supported: x86_64, aarch64." + exit 1 +fi + +# Always build the native arch. +build_native "${HOST_ARCH}" + +# Optionally cross-compile the other arch. +if [ "${CROSSARCH}" == "true" ]; then + build_native "${OTHER_ARCH}" fi -cp "${SOFILE}" "${RESOURCES_DIR}/libjvector.so" diff --git a/jvector-native/src/main/native/src/jvector_arch.h b/jvector-native/src/main/native/src/jvector_arch.h new file mode 100644 index 000000000..21a4a1b78 --- /dev/null +++ b/jvector-native/src/main/native/src/jvector_arch.h @@ -0,0 +1,45 @@ +/* + * Copyright DataStax, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// jvector_arch.h — canonical architecture detection macros for jvector-native. +// +// Use JV_ARCH_X86_64 / JV_ARCH_AARCH64 throughout the codebase to guard +// architecture-specific code instead of scattering raw compiler predefined +// macros (__x86_64__, __aarch64__, etc.). This keeps the guards readable and +// makes it trivial to add a new architecture in one place. +// +// Exactly one of these will be defined to 1 on a supported build; the other +// will be defined to 0. Unsupported architectures define neither to 1 so that +// #if JV_ARCH_X86_64 / #if JV_ARCH_AARCH64 simply evaluate false. + +#ifndef JVECTOR_ARCH_H +#define JVECTOR_ARCH_H + +// ---- x86-64 ----------------------------------------------------------------- +#if defined(__x86_64__) || defined(_M_X64) +# define JV_ARCH_X86_64 1 +# define JV_ARCH_AARCH64 0 +// ---- AArch64 ---------------------------------------------------------------- +#elif defined(__aarch64__) || defined(_M_ARM64) +# define JV_ARCH_X86_64 0 +# define JV_ARCH_AARCH64 1 +// ---- Unsupported ------------------------------------------------------------ +#else +# define JV_ARCH_X86_64 0 +# define JV_ARCH_AARCH64 0 +#endif + +#endif // JVECTOR_ARCH_H diff --git a/jvector-native/src/main/native/src/jvector_cpu_features.h b/jvector-native/src/main/native/src/jvector_cpu_features.h index 27ef3f13e..708042000 100644 --- a/jvector-native/src/main/native/src/jvector_cpu_features.h +++ b/jvector-native/src/main/native/src/jvector_cpu_features.h @@ -19,18 +19,42 @@ #include #include +#include "jvector_arch.h" -#if defined(_MSC_VER) -#include -#elif defined(__GNUC__) || defined(__clang__) -#include +#if JV_ARCH_X86_64 +# if defined(_MSC_VER) +# include +# elif defined(__GNUC__) || defined(__clang__) +# include +# endif +#elif JV_ARCH_AARCH64 +# if defined(__APPLE__) +# include +# else +# include +# include + // Older sysroots may not define these; provide fallbacks matching + // the pattern in highway/hwy/targets.cc. +# ifndef HWCAP_SVE +# define HWCAP_SVE (1 << 22) +# endif +# ifndef HWCAP2_SVE2 +# define HWCAP2_SVE2 (1 << 1) +# endif +# ifndef HWCAP2_SVEAES +# define HWCAP2_SVEAES (1 << 2) +# endif +# endif #endif // Features needed by the ISA dispatch table. Extend as new targets are added. // -// ICX = Intel Ice Lake-SP (Xeon Scalable 3rd Gen) -// SPR = Intel Sapphire Rapids (Xeon Scalable 4th Gen) +// x86-64: +// ICX = Intel Ice Lake-SP (Xeon Scalable 3rd Gen) +// SPR = Intel Sapphire Rapids (Xeon Scalable 4th Gen) +// AArch64 tier flags start at 200 to stay well clear of the x86 entries. enum class CpuFeature : uint32_t { +#if JV_ARCH_X86_64 // ---- Base AVX2 / AVX-512 foundation (all SKUs) ---------------------- AVX2 = 0, AVX512F = 1, @@ -59,19 +83,31 @@ enum class CpuFeature : uint32_t { AVX3_DL = 101, // AVX3 + VNNI + VBMI + VBMI2 + IFMA + BITALG + VPOPCNTDQ // + GFNI + VAES + VPCLMULQDQ (Ice Lake) AVX3_SPR = 102, // AVX3_DL + AVX512_FP16 (Sapphire Rapids) +#elif JV_ARCH_AARCH64 + // ---- AArch64 ISA tier flags ----------------------------------------- + // Numbered from 200 to stay clear of the x86 entries above. + NEON = 200, // baseline AArch64 NEON + AES (AT_HWCAP: HWCAP_AES, or always + // true on Apple where all CPUs support AES) + SVE = 201, // Scalable Vector Extension (AT_HWCAP: HWCAP_SVE). + // Never set on Apple Silicon (no SVE through M4/A18). + SVE2 = 202, // SVE2 + SVE2-AES (AT_HWCAP2: HWCAP2_SVE2 | HWCAP2_SVEAES). + // Never set on Apple Silicon. +#endif COUNT }; -// Populate `features` by issuing CPUID and XGETBV. -// All entries are false on non-x86 architectures. +// Populate `features` by probing CPU capabilities: +// x86-64: CPUID + XGETBV +// AArch64: getauxval(AT_HWCAP/AT_HWCAP2) on Linux; sysctlbyname on macOS +// All entries default to false; only the flags for the current architecture +// are ever set to true. inline void populate_cpu_features(std::array(CpuFeature::COUNT)> &features) noexcept { features.fill(false); -#if defined(__i386__) || defined(__x86_64__) || defined(_M_IX86) \ - || defined(_M_X64) +#if JV_ARCH_X86_64 // Portable CPUID: GCC/Clang use ; MSVC uses . auto run_cpuid = [](uint32_t leaf, @@ -192,7 +228,40 @@ populate_cpu_features(std::array(CpuFeature::COUNT)> features[static_cast(CpuFeature::AVX3_SPR)] = f(CpuFeature::AVX3_DL) && f(CpuFeature::AVX512_FP16); -#endif // x86 / x86_64 +#elif JV_ARCH_AARCH64 + +#if defined(__APPLE__) + // macOS: use sysctlbyname for capability queries. + // NEON (with AES) — present on every shipping Apple Silicon through M4. + { + int val = 0; size_t len = sizeof(val); + if (sysctlbyname("hw.optional.arm.FEAT_AES", &val, &len, nullptr, 0) == 0 && val) + features[static_cast(CpuFeature::NEON)] = true; + } + // SVE and SVE2 are never available on Apple Silicon; leave them false. + +#else // Linux AArch64 + { + const unsigned long hw = getauxval(AT_HWCAP); + + // NEON: all AArch64 CPUs have NEON; require AES too (matches HWY_NEON). +#if defined(HWCAP_AES) + if (hw & HWCAP_AES) + features[static_cast(CpuFeature::NEON)] = true; +#endif + + // SVE: Graviton 3 / Neoverse V1 and later. + if (hw & HWCAP_SVE) + features[static_cast(CpuFeature::SVE)] = true; + + // SVE2: requires SVE2 *and* SVE2-AES (matches HWY_SVE2 requirement). + const unsigned long hw2 = getauxval(AT_HWCAP2); + if ((hw2 & (HWCAP2_SVE2 | HWCAP2_SVEAES)) == (HWCAP2_SVE2 | HWCAP2_SVEAES)) + features[static_cast(CpuFeature::SVE2)] = true; + } +#endif // __APPLE__ + +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 } #endif // CPU_FEATURES_H diff --git a/jvector-native/src/main/native/src/jvector_simd.cpp b/jvector-native/src/main/native/src/jvector_simd.cpp index 8bf8ebef5..f12746e64 100644 --- a/jvector-native/src/main/native/src/jvector_simd.cpp +++ b/jvector-native/src/main/native/src/jvector_simd.cpp @@ -15,12 +15,16 @@ */ // Runtime SIMD dispatch: selects the best available ISA tier at startup. -// Tiers in descending capability order: AVX3_SPR, AVX3_DL, AVX3, AVX2, SSE42. -// SSE42 is the baseline and is assumed always available without a CPUID check. +// x86-64 tiers (descending): AVX3_SPR, AVX3_DL, AVX3, AVX2, SSE42. +// SSE42 is the x86-64 baseline — assumed always available, no CPUID check. +// AArch64 tiers (descending): SVE2 (HWY_SVE2), SVE (HWY_SVE), NEON. +// SVE/SVE2 are scalable targets (HWY_HAVE_SCALABLE=1): Lanes() is runtime. +// NEON is the AArch64 baseline — always available on any aarch64 CPU. // Function pointers are resolved once at static-init time; each public call is // a single indirect branch. #include "jvector_simd.h" -#include "jvector_simd_kernels.h" // AVX3_SPR::, AVX3_DL::, AVX3::, AVX2::, SSE42:: kernel declarations +#include "jvector_arch.h" // JV_ARCH_X86_64, JV_ARCH_AARCH64 +#include "jvector_simd_kernels.h" // per-arch namespace declarations #include "jvector_cpu_features.h" // populate_cpu_features(), CpuFeature enum #include @@ -36,29 +40,46 @@ namespace { // comparisons (e.g. max_isa > MaxIsa::AVX2) correctly gate higher tiers. // Unset (INT_MAX) means "no override; use best available CPU capability" // and must be greater than every named tier so that all guards pass. +#if JV_ARCH_X86_64 enum class MaxIsa { SSE42 = 0, AVX2 = 1, AVX3 = 2, AVX3_DL = 3, AVX3_SPR = 4, Unset = INT_MAX }; static_assert( - (int)MaxIsa::SSE42 < (int)MaxIsa::AVX2 - && (int)MaxIsa::AVX2 < (int)MaxIsa::AVX3 - && (int)MaxIsa::AVX3 < (int)MaxIsa::AVX3_DL + (int)MaxIsa::SSE42 < (int)MaxIsa::AVX2 + && (int)MaxIsa::AVX2 < (int)MaxIsa::AVX3 + && (int)MaxIsa::AVX3 < (int)MaxIsa::AVX3_DL && (int)MaxIsa::AVX3_DL < (int)MaxIsa::AVX3_SPR && (int)MaxIsa::AVX3_SPR < (int)MaxIsa::Unset, "MaxIsa values must be in strict ascending capability order with Unset at the top"); +#elif JV_ARCH_AARCH64 +enum class MaxIsa { NEON = 0, SVE = 1, SVE2 = 2, + Unset = INT_MAX }; +static_assert( + (int)MaxIsa::NEON < (int)MaxIsa::SVE + && (int)MaxIsa::SVE < (int)MaxIsa::SVE2 + && (int)MaxIsa::SVE2 < (int)MaxIsa::Unset, + "MaxIsa values must be in strict ascending capability order with Unset at the top"); +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 // Reads the JVECTOR_MAX_ISA environment variable and maps it to a MaxIsa // value. This lets callers cap the ISA at runtime without recompiling — // useful for benchmarking or working around CPU errata. -// Accepted values (case-sensitive): "avx3", "avx2", "sse42". +// x86-64 values (case-sensitive): "avx3_spr", "avx3_dl", "avx3", "avx2", "sse42". +// AArch64 values (case-sensitive): "sve2", "sve", "neon". static MaxIsa read_max_isa() noexcept { const char *val = std::getenv("JVECTOR_MAX_ISA"); if (!val) return MaxIsa::Unset; +#if JV_ARCH_X86_64 if (std::strcmp(val, "avx3_spr") == 0) return MaxIsa::AVX3_SPR; if (std::strcmp(val, "avx3_dl") == 0) return MaxIsa::AVX3_DL; - if (std::strcmp(val, "avx3") == 0) return MaxIsa::AVX3; - if (std::strcmp(val, "avx2") == 0) return MaxIsa::AVX2; - if (std::strcmp(val, "sse42") == 0) return MaxIsa::SSE42; + if (std::strcmp(val, "avx3") == 0) return MaxIsa::AVX3; + if (std::strcmp(val, "avx2") == 0) return MaxIsa::AVX2; + if (std::strcmp(val, "sse42") == 0) return MaxIsa::SSE42; +#elif JV_ARCH_AARCH64 + if (std::strcmp(val, "sve2") == 0) return MaxIsa::SVE2; + if (std::strcmp(val, "sve") == 0) return MaxIsa::SVE; + if (std::strcmp(val, "neon") == 0) return MaxIsa::NEON; +#endif return MaxIsa::Unset; // unrecognised value: ignore and use CPU detection } @@ -74,7 +95,11 @@ struct KernelVTable { }; // One pre-filled vtable per ISA. These are constant data; no heap allocation. -// Auto-generated from jvector_simd_kernel_list.h +// Auto-generated from jvector_simd_kernel_list.h. +// Guarded by JV_ARCH_* so that only the vtables for the current build +// architecture are instantiated (avoiding references to non-existent symbols). + +#if JV_ARCH_X86_64 #define KERNEL_ENTRY(ret_type, name, params, names) AVX3::name, static const KernelVTable AVX3_vtable = { @@ -110,6 +135,32 @@ static const KernelVTable SSE42_vtable = { }; #undef KERNEL_ENTRY +#elif JV_ARCH_AARCH64 + +#define KERNEL_ENTRY(ret_type, name, params, names) NEON::name, +static const KernelVTable NEON_vtable = { + JVECTOR_SIMD_KERNEL_LIST +}; +#undef KERNEL_ENTRY + +#if JV_HAS_SVE +#define KERNEL_ENTRY(ret_type, name, params, names) SVE::name, +static const KernelVTable SVE_vtable = { + JVECTOR_SIMD_KERNEL_LIST +}; +#undef KERNEL_ENTRY +#endif // JV_HAS_SVE + +#if JV_HAS_SVE2 +#define KERNEL_ENTRY(ret_type, name, params, names) SVE2::name, +static const KernelVTable SVE2_vtable = { + JVECTOR_SIMD_KERNEL_LIST +}; +#undef KERNEL_ENTRY +#endif // JV_HAS_SVE2 + +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 + // Bundles the chosen vtable and the tier that was selected together so that // both can be initialised atomically from a single dispatch call. struct DispatchResult { @@ -126,26 +177,28 @@ static DispatchResult dispatch_kernels() noexcept // Check whether the caller has capped the ISA via the environment variable. const MaxIsa max_isa = read_max_isa(); - // Populate a boolean feature array by issuing CPUID and reading XCR0. + // Populate a boolean feature array via CPUID (x86) or getauxval/sysctl (ARM). std::array(CpuFeature::COUNT)> features; populate_cpu_features(features); - auto has = [&](CpuFeature f) noexcept { + // [[maybe_unused]]: on AArch64 NEON-only builds (no SVE/SVE2) the lambda is + // never called since both JV_HAS_SVE and JV_HAS_SVE2 blocks are compiled out. + auto has [[maybe_unused]] = [&](CpuFeature f) noexcept { return features[static_cast(f)]; }; - // Select the highest tier the CPU supports and the cap allows. - // max_isa > MaxIsa::X means "user has not capped at X or below". - // Adding a new tier above AVX3_SPR only requires one new if at the top. - // Capture the recognised env-var string (or nullptr) for later retrieval. const char *env_str = nullptr; - if (max_isa == MaxIsa::AVX3_SPR) env_str = "avx3_spr"; - else if (max_isa == MaxIsa::AVX3_DL) env_str = "avx3_dl"; - else if (max_isa == MaxIsa::AVX3) env_str = "avx3"; - else if (max_isa == MaxIsa::AVX2) env_str = "avx2"; - else if (max_isa == MaxIsa::SSE42) env_str = "sse42"; +#if JV_ARCH_X86_64 + if (max_isa == MaxIsa::AVX3_SPR) env_str = "avx3_spr"; + else if (max_isa == MaxIsa::AVX3_DL) env_str = "avx3_dl"; + else if (max_isa == MaxIsa::AVX3) env_str = "avx3"; + else if (max_isa == MaxIsa::AVX2) env_str = "avx2"; + else if (max_isa == MaxIsa::SSE42) env_str = "sse42"; + // Select the highest tier the CPU supports and the cap allows. + // max_isa > MaxIsa::X means "user has not capped at X or below". + // Adding a new tier above AVX3_SPR only requires one new if at the top. if (max_isa > MaxIsa::AVX3_DL && has(CpuFeature::AVX3_SPR)) return { AVX3_SPR_vtable, MaxIsa::AVX3_SPR, env_str }; if (max_isa > MaxIsa::AVX3 && has(CpuFeature::AVX3_DL)) @@ -154,8 +207,35 @@ static DispatchResult dispatch_kernels() noexcept return { AVX3_vtable, MaxIsa::AVX3, env_str }; if (max_isa > MaxIsa::SSE42 && has(CpuFeature::AVX2)) return { AVX2_vtable, MaxIsa::AVX2, env_str }; - // SSE42 is the baseline — assumed always present, no CPUID check needed. + // SSE42 is the x86-64 baseline — assumed always present, no CPUID check. return { SSE42_vtable, MaxIsa::SSE42, env_str }; +#elif JV_ARCH_AARCH64 +#if JV_HAS_SVE2 + if (max_isa == MaxIsa::SVE2) env_str = "sve2"; + else +#endif +#if JV_HAS_SVE + if (max_isa == MaxIsa::SVE) env_str = "sve"; + else +#endif + if (max_isa == MaxIsa::NEON) env_str = "neon"; + + // SVE2 and SVE are not available on Apple Silicon (up to and including M4). + // populate_cpu_features() will return false for those on HWY_OS_APPLE. + // The JV_HAS_SVE / JV_HAS_SVE2 guards also handle the case where those tiers + // were not compiled in (e.g. Clang < 22), ensuring we never reference a + // vtable or a CpuFeature that does not exist in this build. +#if JV_HAS_SVE2 + if (max_isa > MaxIsa::SVE && has(CpuFeature::SVE2)) + return { SVE2_vtable, MaxIsa::SVE2, env_str }; +#endif +#if JV_HAS_SVE + if (max_isa > MaxIsa::NEON && has(CpuFeature::SVE)) + return { SVE_vtable, MaxIsa::SVE, env_str }; +#endif + // NEON is the AArch64 baseline — always available, no auxval check needed. + return { NEON_vtable, MaxIsa::NEON, env_str }; +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 } // Both are initialised once at static-init time from a single dispatch call. @@ -190,11 +270,18 @@ JVECTOR_SIMD_KERNEL_LIST const char *jvector_simd_get_active_isa() { switch (active_isa) { +#if JV_ARCH_X86_64 case MaxIsa::AVX3_SPR: return "avx3_spr"; case MaxIsa::AVX3_DL: return "avx3_dl"; case MaxIsa::AVX3: return "avx3"; case MaxIsa::AVX2: return "avx2"; - default: return "sse42"; + case MaxIsa::SSE42: return "sse42"; +#elif JV_ARCH_AARCH64 + case MaxIsa::SVE2: return "sve2"; + case MaxIsa::SVE: return "sve"; + case MaxIsa::NEON: return "neon"; +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 + default: return "unknown"; } } diff --git a/jvector-native/src/main/native/src/jvector_simd_kernels.cpp b/jvector-native/src/main/native/src/jvector_simd_kernels.cpp index f4e8c2453..875182af9 100644 --- a/jvector-native/src/main/native/src/jvector_simd_kernels.cpp +++ b/jvector-native/src/main/native/src/jvector_simd_kernels.cpp @@ -175,6 +175,45 @@ namespace hn = hwy::HWY_NAMESPACE; +#if !HWY_HAVE_SCALABLE +// The two helpers below rely on MaxLanes being a compile-time constant and on +// fixed-stride Combine/Half arithmetic. They are written for fixed vector +// widths (x86, NEON) and must not be instantiated on scalable targets (SVE/SVE2) +// where MaxLanes is a loose upper bound unrelated to the runtime VL. +// Their only call sites are also inside #if !HWY_HAVE_SCALABLE blocks. + +// Loads 4 floats from ptr and broadcasts them to fill the full vector D. +// Uses LoadU + Combine instead of hn::LoadDup128 to avoid a GCC 14 internal +// compiler error (ICE in convert_move/expr.cc:301) triggered by hn::LoadDup128 +// (which emits ld1rq) on SVE targets. +// MaxLanes(D) == 4 → plain LoadU (ptr holds exactly one full vector) +// MaxLanes(D) == 8 → load 4-lane half, Combine to duplicate into both halves +// MaxLanes(D) == 16 → load 4-lane half-of-half, Combine twice (4→8→16 lanes) +// hn::Quarter does not exist in this Highway version; two applications +// of hn::Half<> are used instead. +template +HWY_INLINE hn::Vec BroadcastDup128(D d, const float *HWY_RESTRICT ptr) +{ + static_assert(hn::MaxLanes(d) <= 16, + "BroadcastDup128 is not implemented for ISAs wider than 512-bit"); + if constexpr (hn::MaxLanes(d) > 8) { + // 16-lane (AVX-512): half-of-half = 4 lanes; combine up twice. + const hn::Half> dq; + const auto quarter = hn::LoadU(dq, ptr); + const hn::Half dh; + const auto half = hn::Combine(dh, quarter, quarter); + return hn::Combine(d, half, half); + } else if constexpr (hn::MaxLanes(d) > 4) { + // MaxLanes == 8 (AVX2): load 4-lane half, combine to full. + const hn::Half dh; + const auto half = hn::LoadU(dh, ptr); + return hn::Combine(d, half, half); + } else { + // MaxLanes == 4 (SSE4, NEON): ptr holds exactly one full vector. + return hn::LoadU(d, ptr); + } +} + // Loads 8 floats from ptr and broadcasts them to fill the full vector D. // On ISAs where D is exactly 8 lanes (e.g. AVX2) this is a plain LoadU. // On wider ISAs (e.g. AVX-512, 16 lanes) the 8 floats are loaded into the @@ -197,6 +236,7 @@ HWY_INLINE hn::Vec LoadDup256(D d, const float *HWY_RESTRICT ptr) return hn::LoadU(d, ptr); } } +#endif // !HWY_HAVE_SCALABLE // ============================================================================= // Base Fp32 kernels // ============================================================================= @@ -239,10 +279,10 @@ HWY_INLINE float L2SquareDistanceImpl(Tag tag, const float *a, const float *b, s auto acc2 = hn::Zero(tag), acc3 = hn::Zero(tag); size_t ii = 0; for (; ii + 4 * lanes <= size; ii += 4 * lanes) { - auto d0 = hn::LoadU(tag, a + ii + 0*lanes) - hn::LoadU(tag, b + ii + 0*lanes); - auto d1 = hn::LoadU(tag, a + ii + 1*lanes) - hn::LoadU(tag, b + ii + 1*lanes); - auto d2 = hn::LoadU(tag, a + ii + 2*lanes) - hn::LoadU(tag, b + ii + 2*lanes); - auto d3 = hn::LoadU(tag, a + ii + 3*lanes) - hn::LoadU(tag, b + ii + 3*lanes); + auto d0 = hn::Sub(hn::LoadU(tag, a + ii + 0*lanes), hn::LoadU(tag, b + ii + 0*lanes)); + auto d1 = hn::Sub(hn::LoadU(tag, a + ii + 1*lanes), hn::LoadU(tag, b + ii + 1*lanes)); + auto d2 = hn::Sub(hn::LoadU(tag, a + ii + 2*lanes), hn::LoadU(tag, b + ii + 2*lanes)); + auto d3 = hn::Sub(hn::LoadU(tag, a + ii + 3*lanes), hn::LoadU(tag, b + ii + 3*lanes)); acc0 = hn::MulAdd(d0, d0, acc0); acc1 = hn::MulAdd(d1, d1, acc1); acc2 = hn::MulAdd(d2, d2, acc2); @@ -250,11 +290,11 @@ HWY_INLINE float L2SquareDistanceImpl(Tag tag, const float *a, const float *b, s } auto acc = hn::Add(hn::Add(acc0, acc1), hn::Add(acc2, acc3)); for (; ii + lanes <= size; ii += lanes) { - auto d = hn::LoadU(tag, a + ii) - hn::LoadU(tag, b + ii); + auto d = hn::Sub(hn::LoadU(tag, a + ii), hn::LoadU(tag, b + ii)); acc = hn::MulAdd(d, d, acc); } if (ii < size) { - auto d = hn::LoadN(tag, a + ii, size - ii) - hn::LoadN(tag, b + ii, size - ii); + auto d = hn::Sub(hn::LoadN(tag, a + ii, size - ii), hn::LoadN(tag, b + ii, size - ii)); acc = hn::MulAdd(d, d, acc); } return hn::ReduceSum(tag, acc); @@ -368,7 +408,15 @@ HWY_FLATTEN float euclidean_f32( // #pragma GCC unroll 4: unroll by 4 to hide the 4-cycle FMA latency and keep // both AVX-512 FMA ports saturated across independent load–op–store chains. // -__attribute__((optimize("rename-registers"))) +// Clang does not support the GCC "rename-registers" optimisation attribute and +// warns on unknown attributes, so we suppress it on Clang builds. +#ifdef __clang__ +# define JV_RENAME_REGS +#else +# define JV_RENAME_REGS __attribute__((optimize("rename-registers"))) +#endif + +JV_RENAME_REGS HWY_FLATTEN void add_in_place_f32(float *HWY_RESTRICT v1, const float *HWY_RESTRICT v2, size_t length) @@ -390,7 +438,7 @@ HWY_FLATTEN void add_in_place_f32(float *HWY_RESTRICT v1, } } -__attribute__((optimize("rename-registers"))) +JV_RENAME_REGS HWY_FLATTEN void add_scalar_in_place_f32(float *HWY_RESTRICT v1, float value, size_t length) @@ -411,7 +459,7 @@ HWY_FLATTEN void add_scalar_in_place_f32(float *HWY_RESTRICT v1, } } -__attribute__((optimize("rename-registers"))) +JV_RENAME_REGS HWY_FLATTEN void sub_in_place_f32(float *HWY_RESTRICT v1, const float *HWY_RESTRICT v2, size_t length) @@ -433,7 +481,7 @@ HWY_FLATTEN void sub_in_place_f32(float *HWY_RESTRICT v1, } } -__attribute__((optimize("rename-registers"))) +JV_RENAME_REGS HWY_FLATTEN void sub_scalar_in_place_f32(float *HWY_RESTRICT v1, float value, size_t length) @@ -454,7 +502,7 @@ HWY_FLATTEN void sub_scalar_in_place_f32(float *HWY_RESTRICT v1, } } -__attribute__((optimize("rename-registers"))) +JV_RENAME_REGS HWY_FLATTEN float max_f32(const float *HWY_RESTRICT v, size_t length) { hn::ScalableTag d; @@ -472,7 +520,7 @@ HWY_FLATTEN float max_f32(const float *HWY_RESTRICT v, size_t length) return result; } -__attribute__((optimize("rename-registers"))) +JV_RENAME_REGS HWY_FLATTEN void min_in_place_f32(float *HWY_RESTRICT v1, const float *HWY_RESTRICT v2, size_t length) @@ -549,22 +597,24 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, float *HWY_RESTRICT partialSums) { int codebookBase = codebookIndex * clusterCount; + int ii = 0; + +#if !HWY_HAVE_SCALABLE + // Fixed-width ISAs (x86, NEON): MaxLanes is a compile-time constant so + // centroids_per_iter and the horizontal-reduction shuffles are valid. using FloatTag = hn::ScalableTag; FloatTag tag; - constexpr size_t kLanes = hn::MaxLanes(tag); - alignas(64) float tmp[kLanes]; - int ii = 0; + constexpr size_t kMaxLanes = hn::MaxLanes(tag); + alignas(64) float tmp[kMaxLanes]; - if constexpr (kLanes >= 2) { + if constexpr (kMaxLanes >= 2) { if (size == 2) { float qtmp[4] = {query[queryOffset], query[queryOffset + 1], query[queryOffset], query[queryOffset + 1]}; - hn::Vec queryVec = hn::LoadDup128(tag, qtmp); - - constexpr size_t kBlock = 2; - constexpr int centroids_per_iter = kLanes / kBlock; + hn::Vec queryVec = BroadcastDup128(tag, qtmp); + constexpr int centroids_per_iter = static_cast(kMaxLanes / 2); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -573,7 +623,7 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, hn::Vec score = partial_sum_score( centroidVec, queryVec); hn::Vec swapped = hn::Shuffle2301(score); - hn::Vec sum = score + swapped; + hn::Vec sum = hn::Add(score, swapped); hn::StoreU(sum, tag, tmp); #pragma GCC unroll 8 for (int jj = 0; jj < centroids_per_iter; ++jj) { @@ -582,11 +632,11 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, } } } - if constexpr (kLanes >= 4) { + if constexpr (kMaxLanes >= 4) { if (size == 4) { - constexpr int centroids_per_iter = static_cast(kLanes / 4); + constexpr int centroids_per_iter = static_cast(kMaxLanes / 4); hn::Vec queryVec - = hn::LoadDup128(tag, query + queryOffset); + = BroadcastDup128(tag, query + queryOffset); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -606,10 +656,10 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, } } } - if constexpr (kLanes >= 8) { + if constexpr (kMaxLanes >= 8) { if (size == 8) { hn::Vec queryVec = LoadDup256(tag, query + queryOffset); - constexpr int centroids_per_iter = static_cast(kLanes / 8); + constexpr int centroids_per_iter = static_cast(kMaxLanes / 8); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -631,8 +681,7 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, } } } - if constexpr (kLanes == 16) { - // Don't have to worry about making this work on 1024-bit lanes just yet + if constexpr (kMaxLanes == 16) { if (size == 16) { const hn::Vec queryVec = hn::LoadU(tag, query + queryOffset); @@ -645,6 +694,8 @@ HWY_INLINE void calculate_partial_sums_f32(const float *HWY_RESTRICT codebook, } } } +#endif // !HWY_HAVE_SCALABLE + for (; ii < clusterCount; ii++) { partialSums[codebookBase + ii] = distance_func
( codebook, ii * size, query, queryOffset, size); @@ -930,14 +981,17 @@ HWY_FLATTEN void calculate_partial_sums_self_magnitude_f32( const int codebookBase = codebookIndex * clusterCount; using FloatTag = hn::ScalableTag; FloatTag tag; - constexpr size_t kLanes = hn::MaxLanes(tag); - alignas(64) float tmp[kLanes]; int ii = 0; - if constexpr (kLanes >= 2) { +#if !HWY_HAVE_SCALABLE + // Fixed-width ISAs (x86, NEON): MaxLanes is a compile-time constant so + // centroids_per_iter and the horizontal-reduction shuffles are valid. + constexpr size_t kMaxLanes = hn::MaxLanes(tag); + alignas(64) float tmp[kMaxLanes]; + + if constexpr (kMaxLanes >= 2) { if (size == 2) { - constexpr size_t kBlock = 2; - constexpr int centroids_per_iter = kLanes / kBlock; + constexpr int centroids_per_iter = static_cast(kMaxLanes / 2); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -954,9 +1008,9 @@ HWY_FLATTEN void calculate_partial_sums_self_magnitude_f32( } } } - if constexpr (kLanes >= 4) { + if constexpr (kMaxLanes >= 4) { if (size == 4) { - constexpr int centroids_per_iter = static_cast(kLanes / 4); + constexpr int centroids_per_iter = static_cast(kMaxLanes / 4); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -975,9 +1029,9 @@ HWY_FLATTEN void calculate_partial_sums_self_magnitude_f32( } } } - if constexpr (kLanes >= 8) { + if constexpr (kMaxLanes >= 8) { if (size == 8) { - constexpr int centroids_per_iter = static_cast(kLanes / 8); + constexpr int centroids_per_iter = static_cast(kMaxLanes / 8); for (; ii + centroids_per_iter <= clusterCount; ii += centroids_per_iter) { @@ -998,8 +1052,7 @@ HWY_FLATTEN void calculate_partial_sums_self_magnitude_f32( } } } - if constexpr (kLanes == 16) { - // AVX-512 only: one full register holds exactly one size==16 centroid. + if constexpr (kMaxLanes == 16) { if (size == 16) { for (; ii < clusterCount; ++ii) { const hn::Vec cv @@ -1009,12 +1062,15 @@ HWY_FLATTEN void calculate_partial_sums_self_magnitude_f32( } } } +#endif // !HWY_HAVE_SCALABLE + // General fallback: one centroid at a time, vector-accumulate then reduce. + const size_t lanes = hn::Lanes(tag); for (; ii < clusterCount; ii++) { const float *cptr = codebook + ii * size; auto accVec = hn::Zero(tag); size_t j = 0; - for (; j + kLanes <= size; j += kLanes) { + for (; j + lanes <= size; j += lanes) { const auto v = hn::LoadU(tag, cptr + j); accVec = hn::MulAdd(v, v, accVec); } @@ -1156,8 +1212,9 @@ HWY_FLATTEN void nvq_quantize_8bit(const float *HWY_RESTRICT vector, using Int32Tag = hn::RebindToSigned; FloatTag d_f; Int32Tag d_i; - constexpr size_t kLanes = hn::MaxLanes(d_f); - alignas(64) int32_t tmp[kLanes]; + constexpr size_t kMaxLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); + alignas(64) int32_t tmp[kMaxLanes]; float delta = maxValue - minValue; float scaledAlpha = alpha / delta; @@ -1206,7 +1263,7 @@ HWY_FLATTEN float nvq_loss(const float *HWY_RESTRICT vector, using Int32Tag = hn::RebindToSigned; FloatTag d_f; Int32Tag d_i; - constexpr size_t kLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); int constant = (1 << nBits) - 1; float delta = maxValue - minValue; @@ -1264,7 +1321,7 @@ HWY_FLATTEN float nvq_uniform_loss(const float *HWY_RESTRICT vector, using Int32Tag = hn::RebindToSigned; FloatTag d_f; Int32Tag d_i; - constexpr size_t kLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); float constant = (float)((1 << nBits) - 1); float delta = maxValue - minValue; @@ -1373,7 +1430,7 @@ HWY_FLATTEN float nvq_square_l2_distance_8bit(const float *HWY_RESTRICT vecto Uint8x4Tag d_b; Uint16Tag d_u16; Uint8Tag d_u8; - constexpr size_t kLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); float delta = maxValue - minValue; float scaledAlpha = alpha / delta; @@ -1449,7 +1506,7 @@ HWY_FLATTEN float nvq_dot_product_8bit(const float *HWY_RESTRICT vector, Uint8x4Tag d_b; Uint16Tag d_u16; Uint8Tag d_u8; - constexpr size_t kLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); float delta = maxValue - minValue; float scaledAlpha = alpha / delta; @@ -1572,7 +1629,7 @@ HWY_FLATTEN int64_t nvq_cosine_8bit_packed(const float *HWY_RESTRICT vector, Uint8x4Tag d_b; Uint16Tag d_u16; Uint8Tag d_u8; - constexpr size_t kLanes = hn::MaxLanes(d_f); + const size_t kLanes = hn::Lanes(d_f); float delta = maxValue - minValue; float scaledAlpha = alpha / delta; diff --git a/jvector-native/src/main/native/src/jvector_simd_kernels.h b/jvector-native/src/main/native/src/jvector_simd_kernels.h index 3261b20ce..2d80724c0 100644 --- a/jvector-native/src/main/native/src/jvector_simd_kernels.h +++ b/jvector-native/src/main/native/src/jvector_simd_kernels.h @@ -14,19 +14,22 @@ * limitations under the License. */ -// Header file for the SIMD kernels -// Kernel declarations are auto-generated from jvector_simd_kernel_list.h +// Header file for the SIMD kernels. +// Kernel declarations are auto-generated from jvector_simd_kernel_list.h. +// Architecture guards use the canonical macros from jvector_arch.h so that +// only the namespaces that exist for the current build target are declared. #ifndef SIMD_KERNELS_H #define SIMD_KERNELS_H #include #include +#include "jvector_arch.h" -// Macro to declare a kernel function signature from the kernel list +// Macro to declare a kernel function signature from the kernel list. #define KERNEL_ENTRY(ret_type, name, params, names) \ ret_type name params; -// Generate namespace declarations for each ISA +// Declares all kernel signatures inside a namespace named ISA. #define DECLARE_SIMD_KERNELS(ISA) \ namespace ISA { \ JVECTOR_SIMD_KERNEL_LIST \ @@ -34,11 +37,19 @@ #include "jvector_simd_kernel_list.h" +#if JV_ARCH_X86_64 +// x86-64 ISA namespaces (SSE4.2 baseline → AVX2 → AVX3 → Ice Lake → Sapphire Rapids) DECLARE_SIMD_KERNELS(AVX3_SPR) DECLARE_SIMD_KERNELS(AVX3_DL) DECLARE_SIMD_KERNELS(AVX3) DECLARE_SIMD_KERNELS(AVX2) DECLARE_SIMD_KERNELS(SSE42) +#elif JV_ARCH_AARCH64 +// AArch64 ISA namespaces (NEON baseline → SVE → SVE2) +DECLARE_SIMD_KERNELS(NEON) +DECLARE_SIMD_KERNELS(SVE) +DECLARE_SIMD_KERNELS(SVE2) +#endif // JV_ARCH_X86_64 / JV_ARCH_AARCH64 #undef KERNEL_ENTRY diff --git a/jvector-native/src/main/native/tests/test_aarch64_cpu_features.cpp b/jvector-native/src/main/native/tests/test_aarch64_cpu_features.cpp new file mode 100644 index 000000000..7fe50eb90 --- /dev/null +++ b/jvector-native/src/main/native/tests/test_aarch64_cpu_features.cpp @@ -0,0 +1,189 @@ +/* + * Copyright DataStax, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// Validates that the native dispatcher selects the correct AArch64 ISA tier. +// +// Ground truth on Linux: /proc/cpuinfo "Features" line — the kernel only +// exposes a feature token when the OS has set up the necessary context-switch +// support, so this is the same authority as getauxval(AT_HWCAP). +// Relevant tokens: "aes" (→ NEON), "sve" (→ SVE), "sve2" + "sveaes" (→ SVE2). +// +// Ground truth on macOS: sysctlbyname("hw.optional.arm.FEAT_AES") for NEON. +// SVE and SVE2 are never available on any Apple Silicon through M4/A18. + +#include "test_helpers.h" + +#include +#include +#include +#include +#include + +#if defined(__APPLE__) +# include +#endif + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +// Tier names in ascending capability order for AArch64. +static const std::vector kIsaTiers = { "neon", "sve", "sve2" }; + +static int tier_index(const std::string& name) +{ + auto it = std::find(kIsaTiers.begin(), kIsaTiers.end(), name); + return (it == kIsaTiers.end()) ? -1 : static_cast(it - kIsaTiers.begin()); +} + +// Compute the expected ISA tier from the parsed feature set and any cap. +// Mirrors the logic in populate_cpu_features() / dispatch_kernels(). +static std::string expected_isa(const std::unordered_set& f, + const std::string& cap) +{ + std::string best; + // SVE2: requires both "sve2" and "sveaes" (matches HWCAP2_SVE2 | HWCAP2_SVEAES). + if (f.count("sve2") && f.count("sveaes")) best = "sve2"; + else if (f.count("sve")) best = "sve"; + else best = "neon"; + + if (!cap.empty() && tier_index(cap) < tier_index(best)) + return cap; + return best; +} + +// On macOS, detect NEON capability via sysctlbyname (no /proc/cpuinfo). +// SVE/SVE2 are never present on Apple Silicon, so "neon" is always the result. +#if defined(__APPLE__) +static std::string detect_host_isa_apple() +{ + int val = 0; size_t len = sizeof(val); + if (sysctlbyname("hw.optional.arm.FEAT_AES", &val, &len, nullptr, 0) == 0 && val) + return "neon"; + return "neon"; // baseline — every AArch64 Apple CPU has NEON +} +#endif + +// --------------------------------------------------------------------------- +// Fixture +// --------------------------------------------------------------------------- + +class AArch64CpuFeaturesTest : public ::testing::Test +{ +protected: + static void SetUpTestSuite() + { + const char* active = jvector_simd_get_active_isa(); + const char* cap_c = jvector_simd_get_max_isa_env(); + s_active = active ? active : ""; + s_cap = cap_c ? cap_c : ""; + +#if defined(__APPLE__) + s_host_isa = detect_host_isa_apple(); + s_available = true; + std::printf("[ CPU ] active_isa=%s JVECTOR_MAX_ISA=%s host_isa=%s (macOS sysctl)\n", + s_active.c_str(), + s_cap.empty() ? "(unset)" : s_cap.c_str(), + s_host_isa.c_str()); +#else + s_cpuinfo = parse_cpuinfo_line("Features"); + s_available = !s_cpuinfo.empty(); + + if (s_available) { + s_host_isa = expected_isa(s_cpuinfo, ""); // uncapped hardware capability + // Collect feature string for diagnostics. + s_feature_str = std::accumulate( + s_cpuinfo.begin(), s_cpuinfo.end(), std::string{}, + [](const std::string& a, const std::string& b) { + return a.empty() ? b : a + " " + b; + }); + } + std::printf("[ CPU ] active_isa=%s JVECTOR_MAX_ISA=%s host_isa=%s " + "cpuinfo_features=%zu\n", + s_active.c_str(), + s_cap.empty() ? "(unset)" : s_cap.c_str(), + s_host_isa.c_str(), + s_cpuinfo.size()); +#endif + } + + static std::string s_active; + static std::string s_cap; + static std::string s_host_isa; + static bool s_available; + // Linux only: + static std::unordered_set s_cpuinfo; + static std::string s_feature_str; +}; + +std::string AArch64CpuFeaturesTest::s_active; +std::string AArch64CpuFeaturesTest::s_cap; +std::string AArch64CpuFeaturesTest::s_host_isa; +bool AArch64CpuFeaturesTest::s_available = false; +std::unordered_set AArch64CpuFeaturesTest::s_cpuinfo; +std::string AArch64CpuFeaturesTest::s_feature_str; + +#define SKIP_IF_UNAVAILABLE() \ + do { if (!s_available) GTEST_SKIP() << "/proc/cpuinfo unavailable"; } while (0) + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +// The active ISA must be one of the three valid AArch64 tier names. +TEST_F(AArch64CpuFeaturesTest, ActiveIsaIsValidTier) +{ + EXPECT_GE(tier_index(s_active), 0) + << "active_isa '" << s_active << "' is not a valid AArch64 tier " + << "(expected one of: neon, sve, sve2)"; +} + +// NEON is always available on any AArch64 CPU. +TEST_F(AArch64CpuFeaturesTest, NeonAlwaysAvailable) +{ + SKIP_IF_UNAVAILABLE(); + EXPECT_GE(tier_index(s_host_isa), tier_index("neon")) + << "Expected at least NEON on any AArch64 CPU"; +} + +// The dispatcher must not select a tier higher than the hardware supports. +TEST_F(AArch64CpuFeaturesTest, DispatcherDoesNotExceedHardware) +{ + SKIP_IF_UNAVAILABLE(); + EXPECT_LE(tier_index(s_active), tier_index(s_host_isa)) + << "Dispatcher selected '" << s_active + << "' but host only supports up to '" << s_host_isa << "'" + << "\nCPU features: " << s_feature_str; +} + +// End-to-end: the tier the dispatcher chose must match what /proc/cpuinfo implies. +TEST_F(AArch64CpuFeaturesTest, DispatcherMatchesCpuInfo) +{ + SKIP_IF_UNAVAILABLE(); + std::string exp = expected_isa(s_cpuinfo, s_cap); + EXPECT_EQ(s_active, exp) + << "Dispatcher chose '" << s_active + << "' but /proc/cpuinfo implies '" << exp << "'." + << "\nCPU features: " << s_feature_str; +} + +// When capped to "neon", the dispatcher must select the baseline tier. +TEST_F(AArch64CpuFeaturesTest, NeonCapForcesFallback) +{ + if (s_cap != "neon") GTEST_SKIP() << "JVECTOR_MAX_ISA != neon; skipping"; + EXPECT_EQ(s_active, "neon") + << "Expected 'neon' when JVECTOR_MAX_ISA=neon, got: " << s_active; +} diff --git a/jvector-native/src/main/native/tests/test_elementwise.cpp b/jvector-native/src/main/native/tests/test_elementwise.cpp index 04c836461..82dab16f4 100644 --- a/jvector-native/src/main/native/tests/test_elementwise.cpp +++ b/jvector-native/src/main/native/tests/test_elementwise.cpp @@ -191,3 +191,26 @@ INSTANTIATE_TEST_SUITE_P( [](const ::testing::TestParamInfo& info) { return info.param.description; }); + +// --------------------------------------------------------------------------- +// ISA-tier sanity test: active ISA must be a recognised tier name +// --------------------------------------------------------------------------- + +TEST(IsaDispatch, ActiveIsaIsKnownTier) +{ + const char* active = jvector_simd_get_active_isa(); + ASSERT_NE(active, nullptr) << "jvector_simd_get_active_isa() returned null"; + +#if defined(__aarch64__) + static const char* kOrder[] = {"neon", "sve", "sve2"}; + constexpr int kOrderLen = 3; +#else + static const char* kOrder[] = {"sse42", "avx2", "avx3", "avx3_dl", "avx3_spr"}; + constexpr int kOrderLen = 5; +#endif + bool found = false; + for (int i = 0; i < kOrderLen; ++i) + if (std::strcmp(kOrder[i], active) == 0) { found = true; break; } + + EXPECT_TRUE(found) << "Active ISA '" << active << "' is not a recognised tier"; +} diff --git a/jvector-native/src/main/native/tests/test_helpers.cpp b/jvector-native/src/main/native/tests/test_helpers.cpp index 957e42947..654e86734 100644 --- a/jvector-native/src/main/native/tests/test_helpers.cpp +++ b/jvector-native/src/main/native/tests/test_helpers.cpp @@ -16,6 +16,9 @@ #include "test_helpers.h" +#include +#include + // --------------------------------------------------------------------------- // Global test environment — prints the active ISA once for the whole binary. // Registered via AddGlobalTestEnvironment at static-init time so it fires @@ -75,6 +78,25 @@ const std::vector kKernelTestParams = { {255, "large_odd_tail_15"}, }; +std::unordered_set parse_cpuinfo_line(const std::string& key) +{ + std::unordered_set tokens; + std::ifstream f("/proc/cpuinfo"); + if (!f.is_open()) return tokens; + + std::string line; + while (std::getline(f, line)) { + if (line.rfind(key, 0) != 0) continue; + auto colon = line.find(':'); + if (colon == std::string::npos) continue; + std::istringstream iss(line.substr(colon + 1)); + std::string token; + while (iss >> token) tokens.insert(token); + break; + } + return tokens; +} + std::vector make_vec(size_t n, float seed) { std::vector v(n); diff --git a/jvector-native/src/main/native/tests/test_helpers.h b/jvector-native/src/main/native/tests/test_helpers.h index a48ea47cf..8127cd7b3 100644 --- a/jvector-native/src/main/native/tests/test_helpers.h +++ b/jvector-native/src/main/native/tests/test_helpers.h @@ -25,10 +25,26 @@ #include #include #include +#include #include #include "jvector_simd.h" +// --------------------------------------------------------------------------- +// /proc/cpuinfo helpers. +// +// parse_cpuinfo_line(key) reads /proc/cpuinfo, finds the first line that +// starts with `key`, and returns all whitespace-separated tokens after the +// colon as a set. Returns an empty set when the file is unavailable +// (e.g. macOS) or the key is not found. +// +// Usage: +// x86-64: parse_cpuinfo_line("flags") → {"avx2", "avx512f", ...} +// AArch64: parse_cpuinfo_line("Features") → {"aes", "sve", "sve2", ...} +// --------------------------------------------------------------------------- + +std::unordered_set parse_cpuinfo_line(const std::string& key); + // --------------------------------------------------------------------------- // Deterministic test vectors. // make_vec(n, seed) produces n floats with a mix of signs and magnitudes diff --git a/jvector-native/src/main/native/tests/test_similarity.cpp b/jvector-native/src/main/native/tests/test_similarity.cpp index c08947e4f..bcde00b0b 100644 --- a/jvector-native/src/main/native/tests/test_similarity.cpp +++ b/jvector-native/src/main/native/tests/test_similarity.cpp @@ -246,9 +246,15 @@ TEST(IsaDispatch, MaxIsaEnvHonoured) } // Tiers ordered by capability (ascending index = lower capability). +#if defined(__aarch64__) + static const char* kOrder[] = {"neon", "sve", "sve2"}; + constexpr int kOrderLen = 3; +#else static const char* kOrder[] = {"sse42", "avx2", "avx3", "avx3_dl", "avx3_spr"}; + constexpr int kOrderLen = 5; +#endif auto tier_idx = [](const char* name) -> int { - for (int i = 0; i < 5; ++i) + for (int i = 0; i < kOrderLen; ++i) if (std::strcmp(kOrder[i], name) == 0) return i; return -1; }; diff --git a/jvector-native/src/main/native/tests/test_cpu_features.cpp b/jvector-native/src/main/native/tests/test_x86_cpu_features.cpp similarity index 85% rename from jvector-native/src/main/native/tests/test_cpu_features.cpp rename to jvector-native/src/main/native/tests/test_x86_cpu_features.cpp index a921660b6..995fd451e 100644 --- a/jvector-native/src/main/native/tests/test_cpu_features.cpp +++ b/jvector-native/src/main/native/tests/test_x86_cpu_features.cpp @@ -14,52 +14,21 @@ * limitations under the License. */ -// Validates that the native dispatcher selects the ISA tier that matches the -// CPU capabilities reported in /proc/cpuinfo, respecting any JVECTOR_MAX_ISA cap. +// Validates that the native dispatcher selects the correct x86-64 ISA tier, +// using /proc/cpuinfo as ground truth. Mirrors DispatcherCpuFlagsTest.java. // -// Logic mirrors DispatcherCpuFlagsTest.java and the C implementation in -// jvector_cpu_features.h / jvector_simd.cpp exactly. -// -// /proc/cpuinfo is the authoritative ground-truth: the kernel only exposes a -// flag when the OS context-switch support (XCR0) is also in place, so checking -// it is equivalent to checking CPUID + XCR0 together. +// /proc/cpuinfo is authoritative: the kernel only exposes a flag when the OS +// context-switch support (XCR0) is also in place, so checking it is equivalent +// to checking CPUID + XCR0 together. #include "test_helpers.h" #include -#include -#include #include -#include #include #include #include -// --------------------------------------------------------------------------- -// /proc/cpuinfo helpers — mirrors DispatcherCpuFlagsTest.java -// --------------------------------------------------------------------------- - -// Parse the flags line from the first processor entry in /proc/cpuinfo. -// Returns an empty set if unavailable (non-Linux, non-x86, or unreadable). -static std::unordered_set parse_cpuinfo_flags() -{ - std::unordered_set flags; - std::ifstream f("/proc/cpuinfo"); - if (!f.is_open()) return flags; - - std::string line; - while (std::getline(f, line)) { - if (line.rfind("flags", 0) != 0) continue; - auto colon = line.find(':'); - if (colon == std::string::npos) continue; - std::istringstream iss(line.substr(colon + 1)); - std::string token; - while (iss >> token) flags.insert(token); - break; - } - return flags; -} - // Tier names in ascending capability order — index is ordinal (mirrors Java). static const std::vector kIsaTiers = { "sse42", "avx2", "avx3", "avx3_dl", "avx3_spr" @@ -128,7 +97,7 @@ class CpuFeaturesTest : public ::testing::Test protected: static void SetUpTestSuite() { - s_flags = parse_cpuinfo_flags(); + s_flags = parse_cpuinfo_line("flags"); const char* active = jvector_simd_get_active_isa(); const char* cap_c = jvector_simd_get_max_isa_env(); diff --git a/jvector-native/src/main/native/x86_64-cross.ini b/jvector-native/src/main/native/x86_64-cross.ini new file mode 100644 index 000000000..f51769fd5 --- /dev/null +++ b/jvector-native/src/main/native/x86_64-cross.ini @@ -0,0 +1,34 @@ +# Copyright DataStax, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Meson cross-compilation file for building libjvector for x86_64 on an aarch64 host. +# +# Prerequisites (Ubuntu/Debian): +# sudo apt-get install -y gcc-x86-64-linux-gnu g++-x86-64-linux-gnu +# +# The cross file is passed to meson via: +# meson setup --cross-file /x86_64-cross.ini ... + +[binaries] +c = 'x86_64-linux-gnu-gcc' +cpp = 'x86_64-linux-gnu-g++' +ar = 'x86_64-linux-gnu-ar' +strip = 'x86_64-linux-gnu-strip' +pkg-config = ['x86_64-linux-gnu-pkg-config'] + +[host_machine] +system = 'linux' +cpu_family = 'x86_64' +cpu = 'x86_64' +endian = 'little' diff --git a/jvector-native/verify_native_lib.sh b/jvector-native/verify_native_lib.sh index 47dd245c0..ecf98369b 100755 --- a/jvector-native/verify_native_lib.sh +++ b/jvector-native/verify_native_lib.sh @@ -14,11 +14,17 @@ # See the License for the specific language governing permissions and # limitations under the License. -# Sanity-checks the libjvector.so that actually ships inside the built -# jvector-native jar (not the loose copy in src/main/resources): correct -# architecture, resolvable dynamic dependencies, and presence of the expected -# exported symbols. Non-fatal (warns and skips) for any check whose tool isn't -# available on the current OS (e.g. readelf/ldd on macOS). +# Sanity-checks both libjvector-x86_64.so and libjvector-aarch64.so that ship +# inside the built jvector-native jar (not the loose copies in +# src/main/resources): correct architecture, resolvable dynamic dependencies, +# and presence of the expected exported symbols. Both libraries MUST be present +# and pass all checks — this script is used in the release process where the +# jar is built with -Dnative.crossarch. Non-fatal (warns and skips) for any +# check whose tool isn't available on the current OS (e.g. readelf/ldd on +# macOS). +# +# Usage: verify_native_lib.sh [] +# Defaults: jar auto-detected from target/. set -euo pipefail @@ -35,83 +41,106 @@ if [ -z "${JAR}" ] || [ ! -f "${JAR}" ]; then exit 1 fi -LIBNAME="libjvector.so" +REQUIRED_SYMBOLS=( + jvector_simd_get_active_isa + jvector_simd_get_max_isa_env + dot_product_f32 + cosine_f32 + euclidean_f32 +) + WORKDIR=$(mktemp -d) trap 'rm -rf "${WORKDIR}"' EXIT -if ! unzip -p "${JAR}" "${LIBNAME}" > "${WORKDIR}/${LIBNAME}" 2>/dev/null || [ ! -s "${WORKDIR}/${LIBNAME}" ]; then - echo "ERROR: ${LIBNAME} not found (or empty) inside ${JAR}." >&2 - echo " This is expected if ${JAR} was not built on a unix/amd64 host: the native" >&2 - echo " library is only compiled and packaged there (see unix-amd64-profile in" >&2 - echo " jvector-native/pom.xml). On e.g. Apple Silicon (arm64) Macs, the native" >&2 - echo " build is skipped entirely, so the jar never contains ${LIBNAME}." >&2 - exit 1 -fi +OVERALL_FAIL=0 +VERIFIED=0 -LIB="${WORKDIR}/${LIBNAME}" -FAIL=0 +for ARCH in x86_64 aarch64; do + LIBNAME="libjvector-${ARCH}.so" -printf '\n== %s (from %s) ==\n' "${LIBNAME}" "$(basename "${JAR}")" + printf '\n== %s (from %s) ==\n' "${LIBNAME}" "$(basename "${JAR}")" -echo "-- file --" -file "${LIB}" - -echo "-- architecture (readelf -h) --" -if command -v readelf &>/dev/null; then - MACHINE_LINE=$(readelf -h "${LIB}" | grep -i 'Machine:') - echo "${MACHINE_LINE}" - if [[ "${MACHINE_LINE}" != *"X86-64"* ]]; then - echo "ERROR: expected an x86-64 shared object, got: ${MACHINE_LINE}" >&2 - FAIL=1 + # Both libraries are required; missing either one is a hard failure. + if ! unzip -p "${JAR}" "${LIBNAME}" > "${WORKDIR}/${LIBNAME}" 2>/dev/null \ + || [ ! -s "${WORKDIR}/${LIBNAME}" ]; then + rm -f "${WORKDIR}/${LIBNAME}" + echo "ERROR: ${LIBNAME} not found (or empty) inside ${JAR}." >&2 + echo " Release jars must be built with -Dnative.crossarch to bundle both libraries." >&2 + OVERALL_FAIL=1 + continue fi -else - echo "WARNING: readelf not available on this OS, skipping architecture check (rely on 'file' output above)." >&2 -fi -echo "-- dynamic dependencies (ldd) --" -if command -v ldd &>/dev/null; then - LDD_OUT=$(ldd "${LIB}" 2>&1 || true) - echo "${LDD_OUT}" - if echo "${LDD_OUT}" | grep -qi "not found"; then - echo "ERROR: unresolved shared library dependencies detected above." >&2 - FAIL=1 - fi -else - echo "WARNING: ldd not available on this OS (e.g. macOS); use 'otool -L ${LIBNAME}' manually if needed." >&2 -fi + LIB="${WORKDIR}/${LIBNAME}" + FAIL=0 -# Note that an exhaustive check of exported symbols is not strictly necessary, because the -# Java code will load but fail at runtime if any of the expected symbols are missing. But -# this check is a useful sanity-check to catch any accidental changes to the native code that -# would break the Java code, and to catch any accidental changes to the build process that -# would result in a library that doesn't export the expected symbols. -echo "-- exported symbols (nm) --" -REQUIRED_SYMBOLS=( - jvector_simd_get_active_isa - jvector_simd_get_max_isa_env - dot_product_f32 - cosine_f32 - euclidean_f32 -) -if command -v nm &>/dev/null; then - SYMBOLS=$(nm -D "${LIB}" 2>/dev/null || nm "${LIB}" 2>/dev/null || true) - for sym in "${REQUIRED_SYMBOLS[@]}"; do - # Mach-O (macOS) nm output underscore-prefixes C symbols; ELF (Linux) does not. - if echo "${SYMBOLS}" | grep -qE "[[:space:]]_?${sym}\$"; then - echo " OK ${sym}" + echo "-- file --" + file "${LIB}" + + echo "-- architecture (readelf -h) --" + if command -v readelf &>/dev/null; then + MACHINE_LINE=$(readelf -h "${LIB}" | grep -i 'Machine:') + echo "${MACHINE_LINE}" + if [ "${ARCH}" = "aarch64" ]; then + EXPECTED_MACHINE="AArch64" else - echo " MISSING ${sym}" >&2 + EXPECTED_MACHINE="X86-64" + fi + if [[ "${MACHINE_LINE}" != *"${EXPECTED_MACHINE}"* ]]; then + echo "ERROR: expected a ${EXPECTED_MACHINE} shared object, got: ${MACHINE_LINE}" >&2 FAIL=1 fi - done -else - echo "WARNING: nm not available, skipping exported-symbol check." >&2 -fi + else + echo "WARNING: readelf not available on this OS, skipping architecture check (rely on 'file' output above)." >&2 + fi + + echo "-- dynamic dependencies (ldd) --" + if command -v ldd &>/dev/null; then + LDD_OUT=$(ldd "${LIB}" 2>&1 || true) + echo "${LDD_OUT}" + if echo "${LDD_OUT}" | grep -qi "not found"; then + echo "ERROR: unresolved shared library dependencies detected above." >&2 + FAIL=1 + fi + else + echo "WARNING: ldd not available on this OS (e.g. macOS); use 'otool -L ${LIBNAME}' manually if needed." >&2 + fi + + # Note that an exhaustive check of exported symbols is not strictly necessary, because the + # Java code will load but fail at runtime if any of the expected symbols are missing. But + # this check is a useful sanity-check to catch any accidental changes to the native code that + # would break the Java code, and to catch any accidental changes to the build process that + # would result in a library that doesn't export the expected symbols. + echo "-- exported symbols (nm) --" + if command -v nm &>/dev/null; then + SYMBOLS=$(nm -D "${LIB}" 2>/dev/null || nm "${LIB}" 2>/dev/null || true) + for sym in "${REQUIRED_SYMBOLS[@]}"; do + # Mach-O (macOS) nm output underscore-prefixes C symbols; ELF (Linux) does not. + if echo "${SYMBOLS}" | grep -qE "[[:space:]]_?${sym}\$"; then + echo " OK ${sym}" + else + echo " MISSING ${sym}" >&2 + FAIL=1 + fi + done + else + echo "WARNING: nm not available, skipping exported-symbol check." >&2 + fi + + echo + if [ "${FAIL}" -ne 0 ]; then + echo "${LIBNAME} verification FAILED" >&2 + OVERALL_FAIL=1 + else + echo "${LIBNAME} verification passed" + VERIFIED=$((VERIFIED + 1)) + fi +done +EXPECTED=2 echo -if [ "${FAIL}" -ne 0 ]; then - echo "libjvector.so verification FAILED" >&2 +if [ "${OVERALL_FAIL}" -ne 0 ] || [ "${VERIFIED}" -lt "${EXPECTED}" ]; then + echo "Release verification FAILED: expected both libjvector-x86_64.so and libjvector-aarch64.so to be present and pass all checks." >&2 exit 1 fi -echo "libjvector.so verification passed" +echo "All verified libraries passed" diff --git a/rat-excludes.txt b/rat-excludes.txt index 4d0eb0740..dbc696f0a 100644 --- a/rat-excludes.txt +++ b/rat-excludes.txt @@ -3,6 +3,7 @@ CONTRIBUTIONS.md .github/workflows/checklist_comment_on_new_pr.yml .github/workflows/pr_checklist.md .github/workflows/unit-tests.yaml +.github/workflows/unit-tests-arm64.yaml .github/workflows/generate-changelog.yaml .github/workflows/generate-release-notes.yml package.json