Skip to content

Commit 85d7be0

Browse files
committed
simd masking ops: address-only vocabulary, survival conditions, key-run distinct fold
The five data-indexed primitives no longer describe themselves in foreign-key / join / table terms: parameters are `index` and `table`, docs speak of addresses and populations. `mask_gather_u32` and `mask_scatter_or_u32` carry their survival conditions — a gather reads only resident state into a tile-local output; a scatter writes only the demanded sink or the accumulator of the fold whose scalar leaves. New: `masked_key_run_count_u32` + `KeyRunCarry` — on a key-clustered lane, the distinct count over selected elements as a run fold with a two-word carry and no population-sized set. Documented as over-counting on an unclustered lane; parity 0xD50 threads the carry across uneven tiles against a seen-set reference. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG
1 parent b4bda4c commit 85d7be0

4 files changed

Lines changed: 304 additions & 66 deletions

File tree

‎.claude/blackboard.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3276,3 +3276,5 @@ Loose ends: the general strided path still gathers scalar (correct — at row
32763276
strides ≥ a cache line a hardware gather buys nothing, per the doc); a
32773277
`stride_bytes == 8` twin for `u64` lanes does not exist yet because no caller
32783278
compares `u64` lanes.
3279+
3280+
2026-09-21 (materialisation ruling): the five data-indexed primitives are re-documented in ADDRESS terms only — `index`/`table` parameters, no foreign-key / join / semijoin / table-name vocabulary; `mask_gather_u32` and `mask_scatter_or_u32` now carry their SURVIVAL CONDITIONS (gather: source must be resident state, output tile-local; scatter: destination must be the demanded sink or the accumulator of the fold whose scalar leaves). Sixth arm of the 0xDxx group: `masked_key_run_count_u32(keys, mask_words, &mut KeyRunCarry)` — on a key-clustered lane the distinct count over selected elements as a two-word-carry run fold, no population-sized set; parity `0xD50` threads the carry across uneven tiles against a seen-set reference. Not exact on an unclustered lane by construction (documented; unit test pins the over-count).

‎crates/simd-masking-parity/src/lib.rs‎

Lines changed: 38 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -25,8 +25,8 @@
2525
//! permutation/scatter family (`mask_gather_u32`/`mask_scatter_or_u32`/
2626
//! `masked_group_sum_i32`/`masked_group_sum_i32_via`, for
2727
//! lance-graph-mask-risc's Gather/ScatterOr/GroupSum verbs); `0xD3x` the
28-
//! fk-indirected `masked_group_sum_i32_via` (two-hop zero-fallback); `0xD4x`
29-
//! `eq_u32_via_to_mask` (the same fk lane, packed as a predicate rather than
28+
//! index-addressed `masked_group_sum_i32_via` (two-hop zero-fallback); `0xD4x`
29+
//! `eq_u32_via_to_mask` (the same index lane, packed as a predicate rather than
3030
//! folded into a sum). `main.rs` (native / qemu) and
3131
//! `selfcheck()` (the wasm cdylib export, driven by `run.mjs`) both call
3232
//! [`run`].
@@ -39,10 +39,10 @@ use ndarray::simd::{
3939
lt_u8_to_mask, mask_all, mask_and, mask_and_assign, mask_andnot, mask_andnot_assign, mask_any, mask_gather_u32,
4040
mask_not, mask_not_assign, mask_or, mask_or_assign, mask_scatter_or_u32, mask_set_range, mask_shift_morton,
4141
mask_ternlog, mask_ternlog_assign, mask_xor, mask_xor_assign, masked_group_sum_i32, masked_group_sum_i32_via,
42-
masked_max_i32, masked_min_i32, masked_strided_group_sum, masked_sum_i32, ne_i32_to_mask, ne_i32_to_mask_under,
43-
ne_u32_to_mask, ne_u32_to_mask_under, ne_u64_to_mask, ne_u8_to_mask, ternary_match_strided_to_mask,
44-
ternary_match_u32_to_mask, ternary_match_u32_to_mask_under, ternary_match_u64_to_mask,
45-
ternary_match_u64_to_mask_under, ternlog, I32x16, MortonDir, U32x16, U64x8,
42+
masked_key_run_count_u32, masked_max_i32, masked_min_i32, masked_strided_group_sum, masked_sum_i32, ne_i32_to_mask,
43+
ne_i32_to_mask_under, ne_u32_to_mask, ne_u32_to_mask_under, ne_u64_to_mask, ne_u8_to_mask,
44+
ternary_match_strided_to_mask, ternary_match_u32_to_mask, ternary_match_u32_to_mask_under,
45+
ternary_match_u64_to_mask, ternary_match_u64_to_mask_under, ternlog, I32x16, KeyRunCarry, MortonDir, U32x16, U64x8,
4646
};
4747

4848
/// Number of check groups [`run`] executes (for the log line only).
@@ -1335,7 +1335,7 @@ fn check_gather_scatter_group() -> Result<(), u32> {
13351335
}
13361336

13371337
// ── eq_u32_via_to_mask ────────────────────────────────────────────
1338-
// A predicate evaluated through the same fk lane `masked_group_sum_i32_via`
1338+
// A predicate evaluated through the same index lane `masked_group_sum_i32_via`
13391339
// uses for its key, but packed into a bitmask rather than folded into a
13401340
// sum: `fk[i] < foreign.len() && foreign[fk[i]] == v`.
13411341
let foreign_len = 9usize;
@@ -1360,6 +1360,37 @@ fn check_gather_scatter_group() -> Result<(), u32> {
13601360
if via_mask != want_via_mask {
13611361
return Err(0xD40);
13621362
}
1363+
1364+
// ── masked_key_run_count_u32 ─────────────────────────────────────
1365+
// A key-clustered lane (sorted with repeats), folded in uneven tiles
1366+
// with the carry threaded through; the reference is a plain
1367+
// seen-set over the selected elements — the population-sized state
1368+
// the fold replaces on a clustered lane.
1369+
let mut keys: Vec<u32> = (0..n).map(|_| (rng.next() % 23) as u32).collect();
1370+
keys.sort_unstable();
1371+
let sel: Vec<bool> = (0..n).map(|_| rng.next().is_multiple_of(3)).collect();
1372+
let sel_bits = reference_mask(n, nw, |i| sel[i]);
1373+
let mut carry = KeyRunCarry::default();
1374+
let mut got = 0usize;
1375+
let mut start = 0usize;
1376+
let tile = 37usize;
1377+
while start < n {
1378+
let end = (start + tile).min(n);
1379+
let mut tile_bits = vec![0u64; (end - start).div_ceil(64).max(1)];
1380+
for i in start..end {
1381+
if sel[i] {
1382+
tile_bits[(i - start) / 64] |= 1 << ((i - start) % 64);
1383+
}
1384+
}
1385+
got += masked_key_run_count_u32(&keys[start..end], &tile_bits, &mut carry);
1386+
start = end;
1387+
}
1388+
got += carry.finish();
1389+
let want: std::collections::BTreeSet<u32> = (0..n).filter(|&i| sel[i]).map(|i| keys[i]).collect();
1390+
if got != want.len() {
1391+
return Err(0xD50);
1392+
}
1393+
let _ = sel_bits;
13631394
}
13641395
Ok(())
13651396
}

‎src/simd.rs‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -828,6 +828,7 @@ pub use crate::simd_masking_ops::{
828828
mask_xor_assign,
829829
masked_group_sum_i32,
830830
masked_group_sum_i32_via,
831+
masked_key_run_count_u32,
831832
masked_max_i32,
832833
masked_min_i32,
833834
masked_strided_group_sum,
@@ -843,6 +844,7 @@ pub use crate::simd_masking_ops::{
843844
ternary_match_u32_to_mask_under,
844845
ternary_match_u64_to_mask,
845846
ternary_match_u64_to_mask_under,
847+
KeyRunCarry,
846848
MortonDir,
847849
};
848850
// The popcount that closes the loop on the masks above: `mask_count` in ABI

0 commit comments

Comments
 (0)