Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
23 commits
Select commit Hold shift + click to select a range
05e2efe
native: add ARM64 (NEON/SVE/SVE2) ISA tiers to meson.build
r-devulap Aug 20, 2026
ed9c97d
native: introduce jvector_arch.h and guard all arch-specific code wit…
r-devulap Aug 20, 2026
c002a11
native: guard arch-specific code with #if JV_ARCH_X86_64 / #elif JV_A…
r-devulap Aug 20, 2026
d0e4e89
native: add AArch64 CPU feature detection to jvector_cpu_features.h
r-devulap Aug 20, 2026
a54bb5f
native: allow aarch64 in NativeVectorizationProvider architecture check
r-devulap Aug 20, 2026
7962fb7
tests: add AArch64 ISA tier coverage (Sub-Task 7)
r-devulap Aug 20, 2026
5815aa0
aarch64: target fixed-width HWY_SVE_256 and HWY_SVE2_128
r-devulap Aug 20, 2026
9678d8b
aarch64/sve2: add +i8mm+bf16 to march flag for HWY_SVE2_128
r-devulap Aug 20, 2026
96fc0a6
kernels: replace vector operator+/- with hn::Add/Sub for SVE portability
r-devulap Aug 20, 2026
d6ad7ff
kernels: add BroadcastDup128 to work around GCC 14 ICE with ld1rq+SVE…
r-devulap Aug 20, 2026
d8178f6
run native module on any linux
r-devulap Aug 21, 2026
9e9da2b
Add CI runner for arm64
r-devulap Aug 27, 2026
e50a33e
arm64: build SVE/SVE2 as scalable (VL-agnostic) targets
r-devulap Sep 1, 2026
200f726
docs: update READMEs for AArch64 SVE/SVE2 support
r-devulap Sep 1, 2026
5225643
native: skip SVE/SVE2 on Clang < 22, fix macOS build crash
r-devulap Sep 9, 2026
a9a8504
native: fix clang warnings for unused variable and unknown attribute
r-devulap Sep 9, 2026
1d4fd1a
Add aarch64 cross-compilation support
r-devulap Sep 10, 2026
589d257
Add license headers to cross.ini files; update README for cross-compi…
r-devulap Sep 10, 2026
068f286
Add release notes for aarch64 native PR
r-devulap Sep 25, 2026
7b8b6fb
Require both x86_64 and aarch64 .so in release verification
r-devulap Sep 28, 2026
e572c77
Rename kLanes to kMaxLanes when using hn::MaxLanes
r-devulap Sep 28, 2026
4966d4a
update release notes
r-devulap Sep 28, 2026
a4fb28a
build: write .so files to target/meson-build/ instead of src/main/res…
r-devulap Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 137 additions & 0 deletions .github/workflows/unit-tests-arm64.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,137 @@
name: Unit Test CI — ARM64 (NEON / SVE)

on:
workflow_dispatch:
pull_request:
push:
branches:
- main
paths:
- .github/workflows/unit-tests-arm64.yaml
- '**.java'
- '**/pom.xml'

jobs:
build-arm64:
concurrency:
group: arm64-${{ matrix.max_isa }}-${{ matrix.jdk }}
cancel-in-progress: false
strategy:
matrix:
jdk: [ 24 ]
# Three ISA tiers in ascending capability order, mirroring avx512f/avx2/sse42.
# GitHub-hosted ubuntu-24.04-arm is a Neoverse-N1 (Graviton 2): NEON only, no SVE.
# The sve/sve2 matrix entries still exercise the JVECTOR_MAX_ISA cap path and
# compile all three ISA variants; the native kernel tests that require actual SVE
# hardware are gated on the runtime feature check below.
max_isa: [ neon, sve, sve2 ]
runs-on: ubuntu-24.04-arm
steps:
- name: Report ARM64 ISA capabilities
id: cpu-features
run: |
# Parse the "Features" line from /proc/cpuinfo — the kernel only exposes a
# token here when the OS has set up context-switch support for it, so this is
# the same authority as getauxval(AT_HWCAP / AT_HWCAP2).
# "asimd" is the NEON token; "sve"/"sve2"/"sveaes" appear on Graviton 3/4.
flags="$(grep '^Features' /proc/cpuinfo | head -1 | cut -d: -f2)"
has_neon=false; has_sve=false; has_sve2=false
[[ " $flags " == *" asimd "* ]] && has_neon=true
[[ " $flags " == *" sve "* ]] && has_sve=true
[[ " $flags " == *" sve2 "* ]] && has_sve2=true
printf "NEON=%s SVE=%s SVE2=%s\n" "$has_neon" "$has_sve" "$has_sve2"
if [[ "$has_neon" != "true" ]]; then
echo "ERROR: NEON (asimd) not found in /proc/cpuinfo — not a valid AArch64 runner"
exit 2
fi
# Expose as step outputs for conditional steps below.
echo "has_neon=$has_neon" >> "$GITHUB_OUTPUT"
echo "has_sve=$has_sve" >> "$GITHUB_OUTPUT"
echo "has_sve2=$has_sve2" >> "$GITHUB_OUTPUT"

- name: Set up GCC
run: |
sudo apt install -y gcc g++

- name: Install Meson, Ninja, and GTest
run: |
sudo apt update && sudo apt install -y meson ninja-build pkg-config libgtest-dev

- uses: actions/checkout@v4

- name: Initialize Git Submodules
run: git submodule update --init

- name: Build test_simd_kernels (native C++)
# Meson detects aarch64 and compiles all three ISA variants (neon/sve/sve2)
# regardless of what the host CPU supports at runtime.
working-directory: jvector-native/src/main/native
run: |
meson setup build --wipe
ninja -C build test_simd_kernels

- name: Run test_simd_kernels — no ISA cap (auto-detect, neon job)
if: matrix.max_isa == 'neon'
working-directory: jvector-native/src/main/native
run: ./build/test_simd_kernels

- name: Run test_simd_kernels — capped at neon (sve job, host may lack SVE)
if: matrix.max_isa == 'sve'
working-directory: jvector-native/src/main/native
env:
JVECTOR_MAX_ISA: neon
run: ./build/test_simd_kernels

- name: Run test_simd_kernels — no ISA cap on SVE hardware (sve job)
if: matrix.max_isa == 'sve' && steps.cpu-features.outputs.has_sve == 'true'
working-directory: jvector-native/src/main/native
run: ./build/test_simd_kernels

- name: Run test_simd_kernels — capped at neon (sve2 job baseline check)
if: matrix.max_isa == 'sve2'
working-directory: jvector-native/src/main/native
env:
JVECTOR_MAX_ISA: neon
run: ./build/test_simd_kernels

- name: Run test_simd_kernels — no ISA cap on SVE2 hardware (sve2 job)
if: matrix.max_isa == 'sve2' && steps.cpu-features.outputs.has_sve2 == 'true'
working-directory: jvector-native/src/main/native
run: ./build/test_simd_kernels

- name: Set up JDK ${{ matrix.jdk }}
uses: actions/setup-java@v3
with:
java-version: ${{ matrix.jdk }}
distribution: temurin
cache: maven

- name: Verify native-access vector support (JDK ${{ matrix.jdk }})
env:
JVECTOR_MAX_ISA: ${{ matrix.max_isa }}
run: >-
mvn -B -Punix-amd64-profile -pl jvector-tests -am test
-DTest_RequireSpecificVectorizationProvider=NativeVectorizationProvider
-Dsurefire.failIfNoSpecifiedTests=false
-Dtest=TestVectorizationProvider

- name: Test full suite with native vectorization (JDK ${{ matrix.jdk }})
env:
JVECTOR_MAX_ISA: ${{ matrix.max_isa }}
run: >-
mvn -B -Punix-amd64-profile test
-DTest_RequireSpecificVectorizationProvider=NativeVectorizationProvider

- name: Test Summary for (ARM64/max:${{ matrix.max_isa }},JDK${{ matrix.jdk }})
if: always()
uses: test-summary/action@v2
with:
paths: |
**/target/surefire-reports/TEST-*.xml

- name: Upload Surefire Test Results
uses: actions/upload-artifact@v4
if: always()
with:
name: surefire-results--arm64-${{ matrix.max_isa }}-${{ matrix.jdk }}
path: "**/target/surefire-reports/**"
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -77,8 +77,8 @@ git clone --recurse-submodules <repo-url>
### Building native libraries

The native SIMD library (`libjvector.so`) is built with [Meson](https://mesonbuild.com/) + [Ninja](https://ninja-build.org/)
and requires **g++ 11+**. The entry-point script is
`jvector-native/src/main/native/build_native_lib.sh`. Run it from that directory:
and requires **g++ 11+**. Supported platforms: **Linux x86-64** (SSE4.2, AVX2, AVX-512) and **Linux AArch64** (NEON, SVE, SVE2).
The entry-point script is `jvector-native/src/main/native/build_native_lib.sh`. Run it from that directory.

```bash
cd jvector-native/src/main/native
Expand Down
91 changes: 91 additions & 0 deletions docs/release notes/4.0.2/723.performance.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
### Native SIMD Acceleration on AArch64 (NEON, SVE, SVE2)

**Description**

This PR extends JVector's native Highway SIMD backend to **AArch64 (64-bit ARM)**, bringing
the same native acceleration that x86-64 users already have to ARM-based servers and
development machines (e.g. AWS Graviton, Apple Silicon under Linux, Ampere Altra).

AArch64 support was previously absent from the native layer — ARM hosts silently fell back to
the Panama Vector API path. The native binary now ships three AArch64 ISA tiers, all compiled
as vector-length-agnostic (scalable) code:

| Tier | Highway target | Notes |
|---|---|---|
| NEON | `HWY_NEON` | Baseline; all AArch64 CPUs |
| SVE | `HWY_SVE` | Scalable Vector Extension |
| SVE2 | `HWY_SVE2` | SVE2 + `i8mm` + `bf16` |

SVE and SVE2 are compiled as VL-agnostic (scalable) code, so the same binary runs correctly
on any SVE/SVE2-capable CPU regardless of its physical vector length. NEON operates on fixed
128-bit vectors. At JVM startup, the best available tier is selected automatically via CPU
feature detection — identical to the existing CPUID-based dispatch on x86-64.

The following changes were made:

- **`NativeVectorizationProvider`** — the hard-coded `x86_64` architecture check is extended
to include `aarch64`, so the provider is offered on ARM Linux hosts.
- **`jvector_arch.h`** — new header introducing `JV_ARCH_X86_64` / `JV_ARCH_AARCH64`
preprocessor macros; all arch-specific code is guarded with `#if`/`#elif`/`#endif`.
- **`jvector_cpu_features.h`** — AArch64 feature probing (NEON, SVE, SVE2 presence and
vector-length queries) added alongside the existing x86 CPUID path.
- **Meson build** — three AArch64 ISA variant targets added; SVE/SVE2 targets skip gracefully
on Clang < 22 or non-Linux hosts where SVE is unavailable.
- **Cross-compilation** — Meson cross-files and updated READMEs provided for building the
AArch64 native library from an x86-64 host.
- **CI** — a dedicated AArch64 GitHub Actions runner added to the matrix so native builds and
kernel correctness are verified on real ARM hardware on every PR.

**Purpose / Impact**

Benchmarked on 1M-scale datasets on both AWS Graviton 3 and Graviton 4, Native SIMD (Highway)
consistently outperforms Panama SIMD on AArch64:

- Up to **40% lower search latency** at the same recall level
- Up to **29% faster index construction**

- All kernels already ported to Highway for x86-64 (FP32 similarity, PQ, NVQ, element-wise
arithmetic) benefit immediately on AArch64 — no separate ARM implementation was required.
- On AArch64 Linux with native vectorization enabled, the Highway backend is used exclusively
for all similarity and quantization operations; the Panama Vector API path is bypassed.
- On platforms where the native library cannot be loaded, JVector falls back to the Panama or
pure-Java provider transparently, as before.

**How to Enable**

Both `libjvector.so` variants — x86-64 and AArch64 — are built and bundled into the release
JAR. At JVM startup, `NativeVectorizationProvider` detects the current architecture and loads
the matching native library automatically. The same flags work on both architectures:
Comment thread
r-devulap marked this conversation as resolved.

```bash
java --enable-native-access=ALL-UNNAMED \
-Djvector.experimental.enable_native_vectorization=true \
-jar your-app.jar
```

To cap the ISA tier for debugging or benchmarking:

```bash
JVECTOR_MAX_ISA=neon java --enable-native-access=ALL-UNNAMED \
-Djvector.experimental.enable_native_vectorization=true \
-jar your-app.jar
```

**Building releases:**

To produce a release JAR with native libraries for all supported
architectures (x86-64 and AArch64), the build must be run with the `-Dnative.crossarch`
Maven property. Without this flag, only the native library for the current host architecture
is built and bundled.

```bash
mvn package -Dnative.crossarch

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Note that this requires installing additional packages, at least in some cases. On my test system I had to sudo apt-get install -y g++-aarch64-linux-gnu for this to work.

Is there any way we can simply add this as part of the standard mvn verify that is run to prepare a release without requiring additional command line switches?

@r-devulap r-devulap Oct 1, 2026 •

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

We could pass --auto-install-deps as default to the build script, that way it auto installs when compiling.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

sorry, I was unclear and mixed 2 things in my comment

  1. package dependencies - not a big deal, this happens. I just thought it was worth documenting that if you are on x86 you need to install g++-aarch64-linux-gnu and presumably there is a similar alternative for compiling on ARM. I don't think this requires anything more than documentation
  2. With regards to passing -Dnative.crossarch I'm wondering if this can be automatically added to release prep. Currently the release instructions list the last step to push the release as mvn -Prelease clean deploy, which builds the release and then pushes it to the portal. What I'd like to see is this release profile have the -Dnative.crossarch automatically applied at that step rather than having to pass it on the command line. I think all you'd need to do is add a release profile to the pom in jvector-native, e.g.
<profile>
      <id>release</id>
      <properties>
          <native.crossarch>true</native.crossarch>
      </properties>
  </profile>

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Clarification: -Dnative.crossarch would have to be applied at the mvn verify step as well, since that's what we run before running the verification script. The approach is the same, the release guide would just need to be updated to say run mvn -Prelease verify instead of mvn verify.

```


**Notes**

- SVE/SVE2 compilation requires Clang ≥ 22; older compilers silently skip those targets and
fall back to NEON.
- There is no change to the on-disk index format; AArch64 and x86-64 indexes are fully
interchangeable.
31 changes: 29 additions & 2 deletions jvector-native/pom.xml
Original file line number Diff line number Diff line change
Expand Up @@ -14,8 +14,22 @@
<properties>
<native.buildtype>release</native.buildtype>
<native.jextract.skip>true</native.jextract.skip>
<!-- Set to true via -Dnative.crossarch to also cross-compile for the other arch -->
<native.crossarch>false</native.crossarch>
</properties>
<build>
<!-- Pick up the .so files from target/meson-build/ so that mvn clean
removes stale libraries along with every other build artefact.
The build script no longer writes into src/main/resources/. -->
<resources>
<resource>
<directory>${project.build.directory}/meson-build</directory>
<includes>
<include>libjvector-x86_64.so</include>
<include>libjvector-aarch64.so</include>
</includes>
</resource>
</resources>
<plugins>
<plugin>
<groupId>org.apache.maven.plugins</groupId>
Expand Down Expand Up @@ -89,12 +103,24 @@
<native.buildtype>release</native.buildtype>
</properties>
</profile>
<!-- Activate with -Dnative.crossarch to also cross-compile for the other arch -->
<profile>
<id>native.crossarch</id>
<activation>
<property>
<name>native.crossarch</name>
</property>
</activation>
<properties>
<native.crossarch>true</native.crossarch>
</properties>
</profile>

<profile>
<id>unix-amd64-profile</id>
<activation>
<os>
<family>unix</family>
<arch>amd64</arch>
<name>Linux</name>
</os>
</activation>
<build>
Expand Down Expand Up @@ -157,6 +183,7 @@
<arguments>
<argument>build_native_lib.sh</argument>
<argument>${native.buildtype}</argument>
<argument>${native.crossarch}</argument>
</arguments>
<skip>false</skip>
<workingDirectory>${project.basedir}/src/main/native/src/</workingDirectory>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -31,8 +31,8 @@ public class NativeVectorizationProvider extends VectorizationProvider {

public NativeVectorizationProvider() {
var arch = System.getProperty("os.arch", "");
if (!arch.equals("amd64") && !arch.equals("x86_64")) {
throw new UnsupportedOperationException("Native SIMD operations are only supported on x86_64.");
if (!arch.equals("amd64") && !arch.equals("x86_64") && !arch.equals("aarch64")) {
throw new UnsupportedOperationException("Native SIMD operations are only supported on x86_64 and aarch64.");
}
var libraryLoaded = LibraryLoader.loadJvector();
if (!libraryLoaded) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -22,9 +22,32 @@
/**
* This class is used to load supporting native libraries. First, it tries to load the library from the system path.
* If that fails, it tries to load the library from the classpath (using the usual copying to a tmp directory route).
* <p>
* Two resource names are bundled in the jar:
* <ul>
* <li>{@code /libjvector-x86_64.so} — built natively for x86_64</li>
* <li>{@code /libjvector-aarch64.so} — cross-compiled for aarch64</li>
* </ul>
* At runtime the correct file is chosen based on {@code os.arch}.
*/
public class LibraryLoader {
private LibraryLoader() {}

/**
* Returns the classpath resource name for the native library appropriate for the
* current CPU architecture, or {@code null} when the architecture is not supported.
*/
static String resourceNameForArch() {
String arch = System.getProperty("os.arch", "");
if (arch.equals("aarch64") || arch.equals("arm64")) {
return "/libjvector-aarch64.so";
}
if (arch.equals("amd64") || arch.equals("x86_64")) {
return "/libjvector-x86_64.so";
}
return null;
}

public static boolean loadJvector() {
try {
System.loadLibrary("jvector");
Expand All @@ -35,9 +58,14 @@ public static boolean loadJvector() {
try {
// reinventing the wheel instead of picking up deps, so we'll just use the classloader to load the library
// as a resource and then copy it to a tmp directory and load it from there
String libName = System.mapLibraryName("jvector");
File tmpLibFile = File.createTempFile(libName.substring(0, libName.lastIndexOf('.')), libName.substring(libName.lastIndexOf('.')));
try (var in = LibraryLoader.class.getResourceAsStream("/" + libName);
String resourceName = resourceNameForArch();
if (resourceName == null) {
return false; // unsupported architecture
}
String baseName = resourceName.substring(1, resourceName.lastIndexOf('.')); // e.g. "libjvector-aarch64"
String ext = resourceName.substring(resourceName.lastIndexOf('.')); // e.g. ".so"
File tmpLibFile = File.createTempFile(baseName, ext);
try (var in = LibraryLoader.class.getResourceAsStream(resourceName);
var out = Files.newOutputStream(tmpLibFile.toPath())) {
if (in != null) {
in.transferTo(out);
Expand All @@ -54,4 +82,4 @@ public static boolean loadJvector() {
return false;
}

}
}
Loading
Loading