From ca5d23d73d1d77c6399ae705e21f02d17e699064 Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Fri, 2 Oct 2026 10:51:19 +0500 Subject: [PATCH] test(sync): observe receiver admission after cancellation --- crates/rds-sync/tests/sync_e2e.rs | 81 ++++++++++++++++++- .../rds-sync-release-fence-20261002.md | 34 ++++++++ 2 files changed, 112 insertions(+), 3 deletions(-) create mode 100644 docs/reports/rds-sync-release-fence-20261002.md diff --git a/crates/rds-sync/tests/sync_e2e.rs b/crates/rds-sync/tests/sync_e2e.rs index df05293..02c9dd0 100644 --- a/crates/rds-sync/tests/sync_e2e.rs +++ b/crates/rds-sync/tests/sync_e2e.rs @@ -98,7 +98,16 @@ fn count_parts(state_dir: &Path) -> usize { .map(|e| e.path().join("parts")) .map(|p| { std::fs::read_dir(&p) - .map(|d| d.flatten().filter(|f| f.path().is_file()).count()) + .map(|d| { + d.flatten() + .filter(|f| { + f.path().is_file() + && f.file_name().to_str().is_some_and(|name| { + name.len() == 64 && name.bytes().all(|b| b.is_ascii_hexdigit()) + }) + }) + .count() + }) .unwrap_or(0) }) .sum() @@ -123,6 +132,35 @@ async fn push( send_file(&conn, path, send, recv).await } +/// Client-task cancellation does not join a receiver's running filesystem +/// operation. Wait for its actual exclusive journal admission before the one +/// resumed transfer; only a typed lock-contention error is transient here. +async fn wait_receive_journal( + dir: &Path, + rel: &str, + manifest: &Manifest, +) -> rds_sync::journal::Journal { + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let (dir, rel, manifest) = (dir.to_owned(), rel.to_owned(), manifest.clone()); + let opened = tokio::task::spawn_blocking(move || { + rds_sync::journal::Journal::open(&dir, &rel, &manifest) + }) + .await + .unwrap(); + match opened { + Ok(journal) => break journal, + Err(rds_sync::SyncError::Io(error)) + if error.kind() == std::io::ErrorKind::WouldBlock => {} + Err(error) => panic!("unexpected receive admission error: {error}"), + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("canceled receive retained its journal lock beyond the cleanup bound") +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn transfer_completes_byte_identical() { let (_s, c_ep, target, _task, server_dir) = pair().await; @@ -354,8 +392,16 @@ async fn kill_mid_transfer_resumes_identical() { tokio::time::sleep(Duration::from_millis(10)).await; } attempt.abort(); - // Let the server observe the drop before reconnecting. - tokio::time::sleep(Duration::from_millis(150)).await; + let _ = attempt.await; + assert!(landed >= 3, "the kill must follow verified part progress"); + // A fixed sleep is not a receiver cleanup fence. Observe the exact root's + // lock and retain its verified progress before making one resumed push. + let journal = wait_receive_journal(&server_dir, "killme.bin", &manifest_of(&data)).await; + assert!( + journal.have_set().len() >= 3, + "canceled parts did not survive" + ); + drop(journal); let stats = push(&c_ep, target.clone(), &src).await.unwrap(); assert_eq!(std::fs::read(server_dir.join("killme.bin")).unwrap(), data); @@ -367,6 +413,35 @@ async fn kill_mid_transfer_resumes_identical() { ); } +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn receive_admission_waits_for_the_actual_lock_release() { + use rds_sync::journal::Journal; + let dir = scratch("release-fence"); + let manifest = manifest_of(b"retained receiver work"); + let held = Journal::open(&dir, "data.bin", &manifest).unwrap(); + // Reproduce the old assumption: 150ms elapsed, but an executing receiver + // still owns the real lock. This must remain a refusal, not a forced release. + tokio::time::sleep(Duration::from_millis(150)).await; + assert!(matches!(Journal::open(&dir, "data.bin", &manifest), + Err(rds_sync::SyncError::Io(error)) if error.kind() == std::io::ErrorKind::WouldBlock)); + let path = dir.clone(); + let waiting = + tokio::spawn(async move { wait_receive_journal(&path, "data.bin", &manifest).await }); + tokio::time::sleep(Duration::from_millis(50)).await; + assert!( + !waiting.is_finished(), + "admission bypassed the receiver lock" + ); + drop(held); + let journal = tokio::time::timeout(Duration::from_secs(1), waiting) + .await + .unwrap() + .unwrap(); + assert_eq!(journal.need(), [0]); + drop(journal); + std::fs::remove_dir_all(dir).unwrap(); +} + /// G6: kill at randomized progress points until done — byte-identical /// every time. 20 iterations of a small file keep it fast. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] diff --git a/docs/reports/rds-sync-release-fence-20261002.md b/docs/reports/rds-sync-release-fence-20261002.md new file mode 100644 index 0000000..97ae5ce --- /dev/null +++ b/docs/reports/rds-sync-release-fence-20261002.md @@ -0,0 +1,34 @@ +# Sync cancellation fixture: observe receiver admission + +Scope: repair an independent W2.5/W8 fixture assumption exposed by the native +viewer PR's full Ubuntu CI; no runtime or wire change. + +The original [Ubuntu failure](https://github.com/NDDev-OpenNetwork/remote-device-sync/actions/runs/36969086539/job/110719084734) +is retained in [issue79](https://github.com/NDDev-OpenNetwork/remote-device-sync/issues/79): +`kill_mid_transfer_resumes_identical` reopened a transfer after a fixed 150 ms +sleep and failed with `offer refused: cannot open transfer journal`. +Canceling the sender does not join an executing receiver filesystem operation; +the exclusive journal lock correctly remains held until that operation ends. + +The fixture now joins the aborted sender, counts only hash-named part files, +requires actual part progress, and probes the receiver's journal admission under +one three-second bound. Only `SyncError::Io(WouldBlock)` is retried; all other +errors fail immediately. The admitted journal verifies retained chunks before +the single resumed network transfer. Byte identity and fewer-than-total fetched +chunks remain required. Accepted transfers are not retried by this helper. + +A real-lock fixture retains a Journal past 150 ms and confirms typed refusal. +The bounded helper remains pending while the lock is held and admits after the +owner releases it. This reproduces the invalid timing assumption with an actual +filesystem lock, without pretending to cancel a running syscall. + +The corrected macOS sync E2E lane passed 15 tests with zero failures or ignores; +strict all-target/all-feature sync lint and formatting passed. Linux and full +CI results accompany the repairing PR. These fixtures do not close physical +power-loss, native desktop latency/quality or sustained stability acceptance. + +```sh +cargo test --locked -p rds-sync --test sync_e2e +cargo clippy --locked -p rds-sync --all-targets --all-features -- -D warnings +cargo fmt --check +```