From c6db2bc49da6260a763c64de7d7b982b87cac408 Mon Sep 17 00:00:00 2001 From: Hauser Date: Tue, 22 Sep 2026 23:55:09 +0800 Subject: [PATCH 01/22] fix(bench): reap benchmark groups after wrapper failure Terminate the owned process group even when its leader was already reaped. Cover surviving descendants with FIFO EOF and preserve failures when the group no longer exists. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- scripts/bench-abba.py | 16 ++++++------ scripts/test_bench_abba.py | 51 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 60 insertions(+), 7 deletions(-) diff --git a/scripts/bench-abba.py b/scripts/bench-abba.py index 11e787b..d99d272 100644 --- a/scripts/bench-abba.py +++ b/scripts/bench-abba.py @@ -113,13 +113,15 @@ def execute(command, env, stdout, stderr, timeout, unit): if code != 0: raise RuntimeError(f"benchmark failed with exit code {code}; inspect the leg stderr") except BaseException: - if process.poll() is None: - try: - os.killpg(process.pid, signal.SIGKILL) - except ProcessLookupError: - pass - finally: - process.wait() + # The time wrapper can exit before its benchmark child. The owned + # process group can therefore still need cleanup after wait/poll + # has reaped its leader. + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + finally: + process.wait() raise diff --git a/scripts/test_bench_abba.py b/scripts/test_bench_abba.py index 5339ae5..9295728 100644 --- a/scripts/test_bench_abba.py +++ b/scripts/test_bench_abba.py @@ -3,6 +3,8 @@ import importlib.util import os from pathlib import Path +import select +import signal import sys import tempfile import unittest @@ -78,6 +80,55 @@ def test_timeout_terminates_the_owned_process_group(self): with self.assertRaises(ProcessLookupError): os.kill(pid, 0) + def test_failed_leader_does_not_leave_its_child_running(self): + # The child holds a FIFO writer open before allowing the leader to exit. + # EOF proves it exited without depending on orphan/zombie reaping timing. + child = """ +import os, sys, time +print(os.getpid(), flush=True) +ready_read, ready_write = os.pipe() +if os.fork() == 0: + os.close(ready_read) + writer = os.open(sys.argv[1], os.O_WRONLY) + os.write(writer, b"ready") + os.write(ready_write, b"ready") + time.sleep(30) + os._exit(0) +os.close(ready_write) +os.read(ready_read, 5) +os._exit(7) +""" + with tempfile.TemporaryDirectory() as directory, patch.object(BENCH, "environment_guard"): + out = Path(directory) / "out" + err = Path(directory) / "err" + fifo = Path(directory) / "child-lifetime" + os.mkfifo(fifo) + reader = os.open(fifo, os.O_RDONLY | os.O_NONBLOCK) + try: + with self.assertRaisesRegex(RuntimeError, "exit code 7"): + BENCH.execute([sys.executable, "-c", child, str(fifo)], os.environ.copy(), out, err, 5, None) + self.assertEqual(os.read(reader, 5), b"ready") + readable, _, _ = select.select([reader], [], [], 5) + self.assertEqual(readable, [reader], "benchmark descendant still holds the FIFO open") + self.assertEqual(os.read(reader, 1), b"") + finally: + os.close(reader) + # Also clean up when this regression is run against broken code. + if out.exists() and out.read_text().strip(): + try: + os.killpg(int(out.read_text().strip()), signal.SIGKILL) + except ProcessLookupError: + pass + + def test_missing_process_group_preserves_the_benchmark_failure(self): + with tempfile.TemporaryDirectory() as directory, patch.object(BENCH, "environment_guard"), \ + patch.object(BENCH.os, "killpg", side_effect=ProcessLookupError) as killpg: + out = Path(directory) / "out" + err = Path(directory) / "err" + with self.assertRaisesRegex(RuntimeError, "exit code 7"): + BENCH.execute([sys.executable, "-c", "raise SystemExit(7)"], os.environ.copy(), out, err, 5, None) + killpg.assert_called_once() + if __name__ == "__main__": unittest.main() From 9bebf07830efd0a36e9212ec577d5db21c6df065 Mon Sep 17 00:00:00 2001 From: Hauser Date: Tue, 22 Sep 2026 23:54:07 +0800 Subject: [PATCH 02/22] fix(driver): recover empty-SQ completions and preserve direct-read errors Advance kernel overflow/task work even without new SQEs and retry remaining submission work after reaping. Return metadata errors rather than reporting an unconfirmed O_DIRECT tail as EOF. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/fault-recovery.md | 39 +++++++++ src/driver.rs | 94 +++++++++++---------- src/driver_fault_recovery_tests.rs | 127 +++++++++++++++++++++++++++++ 3 files changed, 216 insertions(+), 44 deletions(-) create mode 100644 docs/fault-recovery.md create mode 100644 src/driver_fault_recovery_tests.rs diff --git a/docs/fault-recovery.md b/docs/fault-recovery.md new file mode 100644 index 0000000..879d4d9 --- /dev/null +++ b/docs/fault-recovery.md @@ -0,0 +1,39 @@ +# Completion progress and direct-read integrity + +The driver skips `io_uring_enter` only when the submission queue is empty and +neither `IORING_SQ_CQ_OVERFLOW` nor `IORING_SQ_TASKRUN` requests kernel work. +The io-uring 0.7.15 submitter supplies `GETEVENTS` for these flags. After reaping, +the driver attempts submission again so newly freed CQ space can receive the +kernel's NODROP overflow entries even when no new reads arrive. + +Each driver turn makes at most two submission attempts. Partial submission, +zero progress, EINTR and EBUSY do not start an unbounded retry loop. The next +event or heartbeat drives further progress. Pending buffers, file descriptors +and admission permits retain their existing final-CQE ownership rules. + +A non-block-aligned O_DIRECT short read that does not cover the logical range +cannot be resumed from its unaligned endpoint. The driver checks file metadata: +confirmed EOF returns the initialized logical prefix; a mid-file endpoint +returns an error; metadata failure propagates its original I/O error. A metadata +failure is not evidence of EOF. Concurrent file mutation retains ordinary read +semantics, without providing snapshot isolation. + +## Validation + +On Linux, run: + +```sh +cargo test --all-features --lib fault_recovery_tests -- --nocapture +``` + +The metadata-result tests inject a completed prefix and metadata outcomes into +the production decision helper. They do not simulate a kernel short read or a +real `fstat` failure. The NOP overflow test creates a two-entry CQ, overflows it, +drains it, and verifies the empty-SQ submission path recovers the third CQE. It +uses actual kernel CQEs but no user buffers. The backlog test covers SQ-capacity +backpressure, not forced partial acceptance by `io_uring_enter`. + +Kernel tests print `SKIP` when setup returns an expected restriction error; +such a run is not evidence of overflow recovery. Run on an unrestricted Linux +host for kernel coverage. Advanced taskrun ring modes remain disabled by default +and are not enabled or kernel-validated by these tests. diff --git a/src/driver.rs b/src/driver.rs index 4072c3f..c420584 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -838,6 +838,8 @@ impl UringDriver { /// just a (correct but pointless) buffered read of the superset range. /// Invalid alignment, range, or reserved-offset inputs are returned through /// the awaited result as `io::ErrorKind::InvalidInput`. + /// A non-aligned short read that does not cover the requested range succeeds + /// only when metadata confirms EOF; a failed metadata lookup returns an error. pub fn read_at_direct(&self, file: Arc, offset: u64, len: usize, align: usize) -> ReadHandle { self.submit(file, offset, len, align, false) } @@ -1347,13 +1349,15 @@ impl Drop for DriverState { } } -/// Best-effort file length via `fstat` on the driver thread, used to tell a -/// genuine O_DIRECT tail short read from a non-block-multiple short read that -/// happened mid-file on a stacked filesystem (rustfs/backlog#1168). `None` when -/// the stat fails, in which case the caller keeps the conservative EOF -/// assumption rather than risk a wrong error or an unbounded resubmit loop. -fn file_len(file: &File) -> Option { - file.metadata().ok().map(|m| m.len()) +/// Finish a non-aligned O_DIRECT short read only after confirming the file tail. +/// Keeping the metadata result explicit lets tests inject errors without closing +/// a live fd or changing process-wide environment state (rustfs/backlog#2647). +fn finish_direct_short_read(p: &mut Pending, file_len: io::Result) -> io::Result> { + let file_len = file_len?; + if p.offset + (p.nread as u64) < file_len { + return Err(io::Error::other("io_uring O_DIRECT: non-block-aligned short read before EOF")); + } + Ok(deliver(p)) } /// Hand the caller exactly the logical range `[head, head + want)` of the read @@ -1410,10 +1414,25 @@ fn flush_backlog(ring: &mut IoUring, backlog: &mut VecDeque Option> { + let needs_enter = { + let sq = ring.submission(); + !sq.is_empty() || sq.cq_overflow() || sq.taskrun() + }; + needs_enter.then(|| ring.submit()) +} + +#[cfg(test)] +#[path = "driver_fault_recovery_tests.rs"] +mod fault_recovery_tests; + /// Flush the backlog into the SQ and submit it, with submit-error classification /// (rustfs/backlog#1162). The single submit path for the whole loop: called once -/// after intake and once more after reap when resubmits were queued. Skips the -/// `io_uring_enter` syscall on an empty SQ (rustfs/backlog#1169). EINTR/EBUSY are +/// after intake and once more after reap. Skips the `io_uring_enter` syscall only +/// when both submissions and kernel completion work are absent. EINTR/EBUSY are /// transient; any other errno is counted and, after a bounded run, transitions /// the shard to shutdown so callers fall back to the std backend. fn submit_ring( @@ -1425,19 +1444,19 @@ fn submit_ring( queued_cancels: &mut HashSet, ) { flush_backlog(&mut state.ring, &mut state.backlog); - if state.ring.submission().is_empty() { + let Some(result) = submit_if_needed(&mut state.ring) else { return; - } - match state.ring.submit() { + }; + match result { Ok(_) => *consecutive_submit_errors = 0, // CQ-overflow backpressure (EBUSY) and signal interruption (EINTR) are // transient — retry next turn without counting them (C5, backlog#1056). Err(e) if matches!(e.raw_os_error(), Some(libc::EBUSY) | Some(libc::EINTR)) => *consecutive_submit_errors = 0, Err(e) => { - // The queued SQEs were not accepted, so their CQEs never arrive. A - // brief run may be transient (EAGAIN); a persistent one (e.g. EPERM - // from a seccomp/LSM policy applied after startup) must not be retried - // forever in silence. + // Submission or kernel completion progress failed. Do not infer + // per-op acceptance or reclaim buffers from this syscall error. + // A brief run may be transient (EAGAIN); a persistent one (e.g. + // EPERM from a later seccomp policy) must not retry forever silently. stats.submit_errors.fetch_add(1, Ordering::SeqCst); *consecutive_submit_errors += 1; if !*submit_error_logged { @@ -1751,20 +1770,8 @@ fn drive( // mid-file, and assuming EOF there would silently truncate // the delivered range. Disambiguate with the actual file // length instead of inferring it (rustfs/backlog#1168). - match file_len(&p.file) { - // Genuine tail: at or past EOF — deliver what we read. - Some(len) if p.offset + p.nread as u64 >= len => ReapStep::Finish(Ok(deliver(p))), - // Mid-file non-multiple: an O_DIRECT read cannot resubmit - // from a non-block-aligned offset, so surface an error - // rather than truncate. The integration falls back to - // the std backend for this read, preserving correctness. - Some(_) => ReapStep::Finish(Err(io::Error::other( - "io_uring O_DIRECT: non-block-aligned short read before EOF", - ))), - // fstat failed: keep the conservative EOF assumption - // rather than risk a wrong error or an infinite loop. - None => ReapStep::Finish(Ok(deliver(p))), - } + let file_len = p.file.metadata().map(|metadata| metadata.len()); + ReapStep::Finish(finish_direct_short_read(p, file_len)) } else { // Positioned short read, not EOF, block-aligned: resubmit // the remainder into the read region. The buffer stays @@ -1810,20 +1817,19 @@ fn drive( } } - // A short-read resubmit queued during reap must reach the kernel in THIS - // turn, not wait out the next heartbeat (rustfs/backlog#1163). Reap runs - // after the submit above, so re-run the single submit path when reap left - // work in the backlog; an idle turn leaves it empty and skips the call. - if !state.backlog.is_empty() { - submit_ring( - &mut state, - &stats, - &mut consecutive_submit_errors, - &mut submit_error_logged, - &mut shutting_down, - &mut queued_cancels, - ); - } + // Reap freed CQ space: flush kernel overflow/task work even with an + // empty SQ/backlog. Also retry SQEs left by partial submission and submit + // short-read continuations this turn. Two bounded attempts per loop + // preserve heartbeat pacing on EBUSY/EINTR/zero-progress submission; + // fully idle calls skip the syscall (rustfs/backlog#2647). + submit_ring( + &mut state, + &stats, + &mut consecutive_submit_errors, + &mut submit_error_logged, + &mut shutting_down, + &mut queued_cancels, + ); // Monitor CQ overflow. With NODROP (asserted at probe) overflowed CQEs // are BUFFERED in the kernel overflow list and flushed on the next enter, diff --git a/src/driver_fault_recovery_tests.rs b/src/driver_fault_recovery_tests.rs new file mode 100644 index 0000000..f9d9560 --- /dev/null +++ b/src/driver_fault_recovery_tests.rs @@ -0,0 +1,127 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +fn direct_prefix() -> Pending { + Pending { + #[cfg(feature = "diagnostics")] + timing: None, + // Simulated completed CQE: no kernel ever references this allocation. + buf: vec![99, 10, 11, 12, 13, 14, 88, 88, 88], + file: Arc::new(File::open("/dev/null").unwrap()), + done: None, + offset: 4096, + nread: 5, + _permit: Arc::new(Semaphore::new(1)).try_acquire_owned().unwrap(), + pad: 1, + head: 2, + want: 6, + region_len: 8, + align: 4, + transient_retries: 0, + } +} + +#[test] +fn direct_metadata_failure_preserves_errno_without_delivering_prefix() { + let mut pending = direct_prefix(); + let before = pending.buf.clone(); + let error = finish_direct_short_read(&mut pending, Err(io::Error::from_raw_os_error(libc::EIO))).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EIO)); + assert_eq!(pending.buf, before, "metadata failure must not turn into prefix delivery"); +} + +#[test] +fn direct_mid_file_short_read_errors_without_delivering_prefix() { + let mut pending = direct_prefix(); + let error = finish_direct_short_read(&mut pending, Ok(8192)).unwrap_err(); + assert_eq!(error.kind(), io::ErrorKind::Other); + assert!(!pending.buf.is_empty()); +} + +#[test] +fn direct_confirmed_tail_returns_only_initialized_logical_bytes() { + let mut pending = direct_prefix(); + assert_eq!(finish_direct_short_read(&mut pending, Ok(4101)).unwrap(), [12, 13, 14]); +} + +#[test] +fn direct_concurrent_truncation_preserves_completed_prefix() { + let mut pending = direct_prefix(); + // A truncate after the CQE does not erase bytes the read already completed. + assert_eq!(finish_direct_short_read(&mut pending, Ok(4096)).unwrap(), [12, 13, 14]); +} + +#[test] +fn direct_eof_before_logical_start_returns_empty() { + let mut pending = direct_prefix(); + pending.nread = 1; + assert!(finish_direct_short_read(&mut pending, Ok(4097)).unwrap().is_empty()); +} + +fn small_ring() -> Option { + match IoUring::builder().setup_cqsize(2).build(2) { + Ok(ring) => { + assert!(ring.params().is_feature_nodrop(), "overflow recovery requires NODROP"); + Some(ring) + } + Err(error) => { + let failure = ProbeFailure::Setup(error); + assert!(failure.is_expected_restriction(), "unexpected io_uring setup failure: {failure}"); + eprintln!("SKIP fault recovery kernel test: io_uring restricted ({failure})"); + None + } + } +} + +fn push_nop(ring: &mut IoUring, id: u64) { + // SAFETY: NOP has no pointers or external resources, including on unwind. + unsafe { ring.submission().push(&opcode::Nop::new().build().user_data(id)) }.unwrap(); +} + +#[test] +fn idle_ring_skips_enter() { + let Some(mut ring) = small_ring() else { return }; + assert!(submit_if_needed(&mut ring).is_none()); +} + +#[test] +fn empty_sq_flushes_real_kernel_overflow_after_reap() { + let Some(mut ring) = small_ring() else { return }; + assert_eq!(ring.completion().capacity(), 2); + push_nop(&mut ring, 1); + push_nop(&mut ring, 2); + assert_eq!(ring.submit_and_wait(2).unwrap(), 2); + // Leave the CQ full, then complete a third NOP into the NODROP overflow list. + push_nop(&mut ring, 3); + assert_eq!(ring.submit().unwrap(), 1); + assert!(ring.submission().is_empty()); + assert!(ring.submission().cq_overflow(), "test must exercise real kernel overflow"); + let first: Vec<_> = ring.completion().map(|cqe| (cqe.user_data(), cqe.result())).collect(); + assert_eq!(first, [(1, 0), (2, 0)]); + + // Regression: the old empty-SQ early return skipped exactly this enter. + assert_eq!(submit_if_needed(&mut ring).expect("overflow must enter").unwrap(), 0); + let recovered: Vec<_> = ring.completion().map(|cqe| (cqe.user_data(), cqe.result())).collect(); + assert_eq!(recovered, [(3, 0)]); + assert!(submit_if_needed(&mut ring).is_none()); +} + +#[test] +fn sq_capacity_leaves_backlog_owned_until_later_submission() { + let Some(mut ring) = small_ring() else { return }; + let mut backlog: VecDeque<_> = (1..=5).map(|id| opcode::Nop::new().build().user_data(id)).collect(); + let mut completed = Vec::new(); + for remaining in [3, 1, 0] { + flush_backlog(&mut ring, &mut backlog); + assert_eq!(backlog.len(), remaining); + submit_if_needed(&mut ring).expect("queued NOPs must submit").unwrap(); + completed.extend(ring.completion().map(|cqe| { + assert_eq!(cqe.result(), 0); + cqe.user_data() + })); + } + assert_eq!(completed, [1, 2, 3, 4, 5]); + assert!(submit_if_needed(&mut ring).is_none()); +} From 536398d2c91842d24a7c4eca6528b76253502527 Mon Sep 17 00:00:00 2001 From: Hauser Date: Tue, 22 Sep 2026 23:55:52 +0800 Subject: [PATCH 03/22] feat(admission): bound logical reads and in-flight buffer bytes Add opt-in shared byte admission with aligned allocation accounting, cancellation-safe permit ownership, and shutdown wakeup coverage. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- README.md | 27 ++++++++ src/admission_tests.rs | 150 +++++++++++++++++++++++++++++++++++++++++ src/driver.rs | 150 ++++++++++++++++++++++++++++++++++++----- src/lib.rs | 2 +- tests/admission.rs | 120 +++++++++++++++++++++++++++++++++ 5 files changed, 433 insertions(+), 16 deletions(-) create mode 100644 src/admission_tests.rs create mode 100644 tests/admission.rs diff --git a/README.md b/README.md index d30fc77..ed1f44c 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,33 @@ assert_eq!(snapshot.delivered + snapshot.orphan_reclaimed, snapshot.submitted); - `read_at_direct(file, offset, len, align)` — the same for an `O_DIRECT` fd; `offset`/`len` need not be aligned (the driver reads a block-aligned superset and returns exactly the requested range). - `read_current(file, len)` — `read(2)` semantics from the current position, for pipes and other non-seekable fds (a short read is a valid final result). - `probe_and_start_sharded(entries, shards)` — several independent rings per disk (each ring caps at one core's memory bandwidth for cache-hit reads); `probe_and_start(entries)` equals `..._sharded(entries, 1)`. +- `probe_and_start_with_limits(entries, shards, ReadLimits { max_read_len, max_in_flight_bytes })` — optional logical read-size and driver-wide read-buffer limits. Both fields default to `None`, preserving existing constructor behavior. + +### Read allocation admission + +With `max_in_flight_bytes: Some(budget)`, all shards share one byte budget. +Buffered reads reserve `len` bytes; direct reads reserve the block-aligned +superset length plus `align - 1` bytes of allocation padding, including for +zero-length direct reads. A request whose allocation exceeds the entire budget, +or whose logical length exceeds `max_read_len`, returns `InvalidInput` before +allocation. A byte budget of zero or above `tokio::sync::Semaphore::MAX_PERMITS` +is rejected at construction. `max_read_len: Some(0)` allows only zero-length reads. + +Admission acquires the shard's count permit before its byte permits. Saturated +handles wait asynchronously, holding no read buffer; a byte waiter may hold a +count permit, and Tokio's fair byte semaphore can put small reads behind a large +waiter. Dropping a waiting handle returns all partial reservations. After enqueue, +both permits travel with the read until its terminal CQE, even if its caller is +canceled. Short-read retries retain the same reservation. A leaked read retains +its charge. Shutdown closes the byte semaphore immediately; any shard-thread exit +also closes it for the whole driver, conservatively rejecting further byte +admission and waking byte waiters even on a bounded-drain escape. + +This limits reserved driver read-buffer allocation bytes, **not process RSS**. +It excludes queued handle/FD metadata, allocator overhead, result copies and +completed `Vec` results retained in channels or by callers. The caller must bound +its task fan-out and result queue separately. Completion releases admission even +when the returned result remains alive. ## API contract diff --git a/src/admission_tests.rs b/src/admission_tests.rs new file mode 100644 index 0000000..acd05ee --- /dev/null +++ b/src/admission_tests.rs @@ -0,0 +1,150 @@ +//! Deterministic admission tests; no kernel ring is needed. +use super::*; +use std::task::Waker; + +fn poll_once(future: &mut AcquireFut) -> Poll> { + future.as_mut().poll(&mut Context::from_waker(Waker::noop())) +} + +#[test] +fn byte_saturation_returns_partial_count_reservation_on_drop() { + let count = Arc::new(Semaphore::new(2)); + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8); + assert!(poll_once(&mut waiting).is_pending()); + assert_eq!(count.available_permits(), 0); + drop(waiting); + assert_eq!(count.available_permits(), 1); + assert_eq!(bytes.available_permits(), 0); + drop(first); + assert_eq!(bytes.available_permits(), 8); +} + +#[test] +fn dropping_count_waiter_does_not_reserve_bytes() { + let count = Arc::new(Semaphore::new(1)); + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&count, Some(&bytes), 4).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 4); + assert!(poll_once(&mut waiting).is_pending()); + drop(waiting); + assert_eq!(bytes.available_permits(), 4); + drop(first); + assert_eq!(count.available_permits(), 1); +} + +#[test] +fn shard_exit_closes_bytes_and_releases_waiting_count_permit() { + let count = Arc::new(Semaphore::new(2)); + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8); + assert!(poll_once(&mut waiting).is_pending()); + drop(CloseByteAdmission(Some(bytes.clone()))); + assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); + assert_eq!(count.available_permits(), 1); + assert_eq!(bytes.available_permits(), 0); + drop(first); +} + +#[test] +fn permits_retained_by_leaked_operation_remain_charged_after_close() { + let count = Arc::new(Semaphore::new(1)); + let bytes = Arc::new(Semaphore::new(8)); + let leaked = try_read_permits(&count, Some(&bytes), 8).unwrap(); + std::mem::forget(leaked); + drop(CloseByteAdmission(Some(bytes.clone()))); + assert_eq!(bytes.available_permits(), 0); + assert_eq!(count.available_permits(), 0); +} + +#[test] +fn shared_byte_budget_serializes_reads_from_distinct_shards() { + let counts = [Arc::new(Semaphore::new(1)), Arc::new(Semaphore::new(1))]; + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&counts[0], Some(&bytes), 8).unwrap(); + assert!(matches!(try_read_permits(&counts[1], Some(&bytes), 1), Err(TryAcquireError::NoPermits))); + assert_eq!(counts[1].available_permits(), 1); + let mut waiting = acquire_read_permits(counts[1].clone(), Some(bytes.clone()), 8); + assert!(poll_once(&mut waiting).is_pending()); + drop(first); + let Poll::Ready(Ok(second)) = poll_once(&mut waiting) else { panic!("budget was not released") }; + drop(second); + assert_eq!(bytes.available_permits(), 8); +} + +fn fake_driver(limits: ReadLimits) -> (UringDriver, mpsc::Receiver) { + let (tx, rx) = mpsc::channel(); + ( + UringDriver { + limits, + byte_sem: limits.max_in_flight_bytes.map(|bytes| Arc::new(Semaphore::new(bytes))), + shards: vec![Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem: Arc::new(Semaphore::new(8)), + wake_efd: Arc::new(EventFd::new().unwrap()), + }], + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + rx, + ) +} + +#[tokio::test] +async fn logical_limit_rejects_before_message_or_allocation() { + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: Some(4), + max_in_flight_bytes: Some(8192), + }); + let err = driver + .read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 5) + .await + .unwrap_err(); + assert_eq!(err.kind(), io::ErrorKind::InvalidInput); + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); +} + +#[tokio::test] +async fn direct_budget_includes_superset_padding_and_zero_length_geometry() { + for (offset, len, charge) in [(1, 4096, 12287), (0, 0, 4095), (1, 0, 8191)] { + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(charge - 1), + }); + let file = Arc::new(File::open("/dev/zero").unwrap()); + let err = driver.read_at_direct(file.clone(), offset, len, 4096).await.unwrap_err(); + assert_eq!(err.kind(), io::ErrorKind::InvalidInput); + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); + + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(charge), + }); + let handle = driver.read_at_direct(file, offset, len, 4096).without_cancel_on_drop(); + let msg = rx.try_recv().unwrap(); + assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), 0); + drop(handle); + assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), 0, "caller drop released bytes"); + drop(msg); + assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), charge); + } +} + +#[test] +fn invalid_budget_is_rejected_before_kernel_probe() { + for budget in [0, Semaphore::MAX_PERMITS + 1] { + let result = UringDriver::probe_and_start_with_limits( + 8, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(budget), + }, + ); + assert!(matches!(result, Err(ProbeFailure::Setup(err)) if err.kind() == io::ErrorKind::InvalidInput)); + } +} diff --git a/src/driver.rs b/src/driver.rs index c420584..4838e83 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -28,6 +28,10 @@ use std::time::{Duration, Instant}; use io_uring::{IoUring, opcode, types}; +#[cfg(test)] +#[path = "admission_tests.rs"] +mod admission_tests; + #[cfg(feature = "diagnostics")] use crate::diagnostics::{Diagnostics, DiagnosticsSnapshot, Trace}; @@ -262,7 +266,7 @@ impl ProbeFailure { // resident memory and reopening the memory-DoS surface. // // That rule is now enforced by the type system rather than by a manual -// `release()` call: the `OwnedSemaphorePermit` travels with `Msg::Read` into the +// `release()` call: the count and optional byte permits travel with `Msg::Read` into the // `Pending` entry and is dropped exactly when the entry is removed at the final // CQE. A short-read resubmit keeps the entry — and thus the permit. // @@ -272,7 +276,70 @@ impl ProbeFailure { // returned `ReadHandle`, which awaits it on its first poll and submits then. /// Boxed `Semaphore::acquire_owned` future held by a saturated `ReadHandle`. -type AcquireFut = Pin> + Send>>; +type AcquireFut = Pin> + Send>>; + +/// Optional resource limits shared by all shards of one driver. +/// +/// Defaults preserve the existing count-only admission policy. Limits cover +/// driver-owned read allocations, not queued handle metadata, allocator overhead, +/// completion copies, or returned results retained by the caller. They are not +/// a whole-process RSS bound. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct ReadLimits { + /// Maximum logical length of one read. `None` retains the kernel read cap. + /// A request exceeding this limit returns `InvalidInput` without allocation. + pub max_read_len: Option, + /// Maximum sum of reserved read-buffer bytes across all shards. + /// + /// Direct reads charge their aligned superset plus `align - 1` allocation + /// padding. Reservations start before allocation and end at terminal read + /// completion, including canceled operations; leaked operations stay charged. + /// Requests larger than this budget return `InvalidInput`, never wait. + /// Zero and values above Tokio's `Semaphore::MAX_PERMITS` are invalid. + /// Shutdown or any shard exit closes byte admission for the entire driver + /// when enabled, waking byte waiters even if buffers must be leaked. + pub max_in_flight_bytes: Option, +} + +struct ReadPermits { + _count: OwnedSemaphorePermit, + _bytes: Option, +} + +fn try_read_permits(count: &Arc, bytes: Option<&Arc>, charge: u32) -> Result { + let count = Arc::clone(count).try_acquire_owned()?; + let bytes = bytes.map(|sem| Arc::clone(sem).try_acquire_many_owned(charge)).transpose()?; + Ok(ReadPermits { + _count: count, + _bytes: bytes, + }) +} + +fn acquire_read_permits(count: Arc, bytes: Option>, charge: u32) -> AcquireFut { + Box::pin(async move { + // Every admission takes count before bytes. Pending reads need neither + // resource to complete, so there is no inverse acquisition cycle. + let count = count.acquire_owned().await?; + let bytes = match bytes { + Some(sem) => Some(sem.acquire_many_owned(charge).await?), + None => None, + }; + Ok(ReadPermits { + _count: count, + _bytes: bytes, + }) + }) +} + +struct CloseByteAdmission(Option>); + +impl Drop for CloseByteAdmission { + fn drop(&mut self) { + if let Some(bytes) = &self.0 { + bytes.close(); + } + } +} #[derive(Default)] struct DriverStats { @@ -342,7 +409,7 @@ enum Msg { /// released only when the pending entry is dropped at the final CQE /// (rustfs/backlog#1060/#1102). If the driver rejects the op (shutting /// down) the permit is dropped with the message — released immediately. - permit: OwnedSemaphorePermit, + permit: ReadPermits, /// Block size the read must be aligned to. `1` means a normal buffered /// read; `> 1` means the file was opened `O_DIRECT` and the driver must /// read the block-aligned superset range into a block-aligned buffer @@ -401,7 +468,7 @@ struct Pending { offset: u64, /// Bytes already read into the read region (`buf[pad..]`). nread: usize, - _permit: OwnedSemaphorePermit, + _permit: ReadPermits, /// Offset inside `buf` where the block-aligned read region starts. pad: usize, /// Bytes of the read region that precede the caller's logical range. @@ -601,8 +668,8 @@ impl Drop for ReadHandle { fn drop(&mut self) { // The buffer is deliberately NOT touched here: the driver owns it // until the CQE. All we may do is ask the kernel to hurry up. A handle - // dropped before it was submitted (Inert / WaitingPermit) has no buffer, - // no permit and no SQE, so there is nothing to cancel. + // dropped before it was submitted (Inert / WaitingPermit) has no buffer + // and no SQE. A waiting handle releases any partial reservation by drop. if let HandleState::Submitted { wake } = &self.state && !self.finished && self.cancel_on_drop @@ -672,6 +739,8 @@ impl Drop for Shard { /// kernel. Construct it through [`UringDriver::probe_and_start`] so restricted /// environments can fall back to a blocking backend before serving traffic. pub struct UringDriver { + limits: ReadLimits, + byte_sem: Option>, /// One or more independent rings. A cache-hit buffered read completes inline /// inside `io_uring_enter`, so the thread driving a ring performs that /// read's memcpy — which caps a single-ring driver at one core's memory @@ -714,6 +783,26 @@ impl UringDriver { /// later shard fails to start, the ones already running are shut down and /// joined before the error is returned. pub fn probe_and_start_sharded(entries: u32, shards: usize) -> Result { + Self::probe_and_start_with_limits(entries, shards, ReadLimits::default()) + } + + /// Start one or more shards with opt-in logical-size and read-allocation limits. + /// + /// Ring setup and probing follow [`Self::probe_and_start_sharded`]. Invalid + /// limits return [`ProbeFailure::Setup`] with `InvalidInput` before probing. + /// Saturated admission waits asynchronously and fairly on Tokio semaphores; + /// dropping a waiting handle returns any partial reservation. + pub fn probe_and_start_with_limits(entries: u32, shards: usize, limits: ReadLimits) -> Result { + if limits + .max_in_flight_bytes + .is_some_and(|bytes| bytes == 0 || bytes > Semaphore::MAX_PERMITS) + { + return Err(ProbeFailure::Setup(io::Error::new( + io::ErrorKind::InvalidInput, + "byte budget must be between 1 and Semaphore::MAX_PERMITS", + ))); + } + let byte_sem = limits.max_in_flight_bytes.map(|bytes| Arc::new(Semaphore::new(bytes))); let mut started = Vec::with_capacity(shards.max(1)); for i in 0..shards.max(1) { // Probe only the first shard (rustfs/backlog#1165): the probe read @@ -722,9 +811,11 @@ impl UringDriver { // verify NODROP — this avoids `shards - 1` extra O_TMPFILE // create+write+read round-trips per disk on every start and renew. // `?` drops `started`, whose `Shard::drop` joins each running thread. - started.push(Self::start_shard(entries, i == 0)?); + started.push(Self::start_shard(entries, i == 0, byte_sem.clone())?); } Ok(Self { + limits, + byte_sem, shards: started, next_id: AtomicU64::new(1), rr: AtomicUsize::new(0), @@ -739,7 +830,7 @@ impl UringDriver { &self.shards[self.rr.fetch_add(1, Ordering::Relaxed) % n] } - fn start_shard(entries: u32, probe: bool) -> Result { + fn start_shard(entries: u32, probe: bool, byte_sem: Option>) -> Result { let mut ring = IoUring::new(entries).map_err(ProbeFailure::Setup)?; // Require the NODROP feature (kernel >= 5.5). Without it, CQ overflow // silently drops CQEs, stranding pending entries forever and hanging @@ -799,7 +890,10 @@ impl UringDriver { // (moved into the closure) drop cleanly with no SQE in flight. let handle = std::thread::Builder::new() .name("uring-spike-driver".into()) - .spawn(move || drive(ring, rx, thread_stats, thread_sem, cq_efd, thread_wake)) + .spawn(move || { + let _close_bytes = CloseByteAdmission(byte_sem); + drive(ring, rx, thread_stats, thread_sem, cq_efd, thread_wake); + }) .map_err(ProbeFailure::Setup)?; Ok(Shard { @@ -909,10 +1003,10 @@ impl UringDriver { // rustfs/backlog#1057). Failing fast here also removes the caller- // controlled `vec![0u8; len]` capacity-overflow panic that made the // unwind-UAF (rustfs/backlog#1054) reachable. P2 must chunk instead. - if len > MAX_READ_LEN { + if len > MAX_READ_LEN || self.limits.max_read_len.is_some_and(|max| len > max) { let _ = done.send(Err(io::Error::new( io::ErrorKind::InvalidInput, - "read length exceeds MAX_RW_COUNT (2 GiB - 4 KiB); caller must chunk", + "read length exceeds MAX_RW_COUNT or configured logical read limit; caller must chunk", ))); return ReadHandle { id, @@ -934,7 +1028,7 @@ impl UringDriver { // submit (rustfs/backlog#1102, #1166). Pre-empting it here also makes // every resubmit's `next_off < kernel_offset + region_len` provably // <= i64::MAX. `align == 1` (buffered) always passes the alignment part. - match aligned_geometry(offset, len, align) { + let allocation_bytes = match aligned_geometry(offset, len, align) { // CURRENT_POSITION (stream) reads use no positional offset — the // kernel reads from the current file position — so the i64::MAX end // check does not apply to them (their sentinel offset would overflow @@ -944,7 +1038,10 @@ impl UringDriver { && (allow_current_position && offset == CURRENT_POSITION || kernel_offset .checked_add(region_len as u64) - .is_some_and(|end| end <= i64::MAX as u64)) => {} + .is_some_and(|end| end <= i64::MAX as u64)) => + { + region_len + align - 1 + } _ => { let _ = done.send(Err(io::Error::new( io::ErrorKind::InvalidInput, @@ -961,13 +1058,30 @@ impl UringDriver { state: HandleState::Inert, }; } + }; + + if self.limits.max_in_flight_bytes.is_some_and(|max| allocation_bytes > max) { + let _ = done.send(Err(io::Error::new(io::ErrorKind::InvalidInput, "aligned allocation exceeds byte budget"))); + return ReadHandle { + id, + #[cfg(feature = "diagnostics")] + timing, + rx, + tx: shard.tx.clone(), + finished: false, + cancel_on_drop: false, + state: HandleState::Inert, + }; } + // Both region and alignment are capped at MAX_READ_LEN, so their sum + // fits u32, even on 32-bit Linux. Charge the exact Vec allocation length. + let charge = allocation_bytes as u32; // Take a backpressure permit BEFORE the op reaches the driver; it is // released only when the pending entry is dropped at the CQE (C10, // rustfs/backlog#1060). Acquisition never blocks the caller's thread // (rustfs/backlog#1102). - match Arc::clone(&shard.sem).try_acquire_owned() { + match try_read_permits(&shard.sem, self.byte_sem.as_ref(), charge) { // Fast path: a permit was free, so submit eagerly — no allocation, // no await, and the op is in flight the moment `submit` returns, // exactly as with the previous blocking implementation. @@ -1033,7 +1147,7 @@ impl UringDriver { finished: false, cancel_on_drop: true, state: HandleState::WaitingPermit { - acquire: Box::pin(Arc::clone(&shard.sem).acquire_owned()), + acquire: acquire_read_permits(Arc::clone(&shard.sem), self.byte_sem.clone(), charge), file, offset, len, @@ -1117,6 +1231,9 @@ impl UringDriver { /// Shards are asked to stop first and joined afterwards, so their bounded /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. pub fn shutdown(mut self) -> StatsSnapshot { + if let Some(bytes) = &self.byte_sem { + bytes.close(); + } for shard in &self.shards { let _ = shard.tx.send(Msg::Shutdown); shard.wake_efd.signal(); @@ -1142,6 +1259,9 @@ impl UringDriver { impl Drop for UringDriver { fn drop(&mut self) { + if let Some(bytes) = &self.byte_sem { + bytes.close(); + } // Ask every shard to stop before joining any of them, so their bounded // drains overlap. Dropping the `Vec` would instead run each // `Shard::drop` in turn, serializing up to `shards * DRAIN_TIMEOUT` on a diff --git a/src/lib.rs b/src/lib.rs index d75e314..406f8fb 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -55,4 +55,4 @@ mod diagnostics; pub use diagnostics::{DIAGNOSTICS_SAMPLE_INTERVAL, DiagnosticsSnapshot, LatencyHistogram}; #[cfg(target_os = "linux")] -pub use driver::{ProbeFailure, ReadHandle, StatsSnapshot, UringDriver}; +pub use driver::{ProbeFailure, ReadHandle, ReadLimits, StatsSnapshot, UringDriver}; diff --git a/tests/admission.rs b/tests/admission.rs new file mode 100644 index 0000000..bd9d42d --- /dev/null +++ b/tests/admission.rs @@ -0,0 +1,120 @@ +//! Native Linux resource-admission tests (rustfs/backlog#2647). +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io::Write; +use std::os::fd::FromRawFd; +use std::sync::Arc; +use std::task::Poll; +use std::time::Duration; + +use rustfs_uring::{ReadLimits, UringDriver}; + +fn driver_or_skip() -> Option { + match UringDriver::probe_and_start_with_limits( + 8, + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) { + Ok(driver) => Some(driver), + Err(err) => { + assert!(err.is_expected_restriction(), "unexpected probe failure: {err}"); + eprintln!("SKIP admission: restricted environment ({err})"); + None + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid output array; successful pipe creates two owned descriptors. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0); + // SAFETY: each fresh descriptor transfers ownership exactly once. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("read did not enter the driver"); +} + +async fn assert_pending(future: &mut rustfs_uring::ReadHandle) { + std::future::poll_fn(|cx| { + assert!(std::pin::Pin::new(&mut *future).poll(cx).is_pending()); + Poll::Ready(()) + }) + .await; +} + +#[tokio::test] +async fn orphan_holds_shared_bytes_until_its_real_read_completes() { + let Some(driver) = driver_or_skip() else { return }; + let (read, mut write) = pipe(); + let first = driver.read_current(read, 8).without_cancel_on_drop(); + wait_submitted(&driver, 1).await; + let mut next = driver.read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 8); + assert_pending(&mut next).await; + drop(first); + assert_pending(&mut next).await; + assert_eq!(driver.stats().submitted, 1); + write.write_all(b"12345678").unwrap(); + assert_eq!(tokio::time::timeout(Duration::from_secs(2), next).await.unwrap().unwrap(), vec![0; 8]); + let stats = driver.shutdown(); + assert_eq!(stats.orphan_reclaimed, 1); + assert_eq!(stats.submitted, 2); + assert_eq!(stats.in_flight, 0); +} + +#[tokio::test] +async fn dropping_a_byte_waiter_does_not_submit_or_strand_capacity() { + let Some(driver) = driver_or_skip() else { return }; + let (read, mut write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let file = Arc::new(File::open("/dev/zero").unwrap()); + let mut canceled = driver.read_at(file.clone(), 0, 8); + assert_pending(&mut canceled).await; + drop(canceled); + write.write_all(b"12345678").unwrap(); + assert_eq!(first.await.unwrap(), b"12345678"); + let result = tokio::time::timeout(Duration::from_secs(2), driver.read_at(file, 0, 8)) + .await + .unwrap() + .unwrap(); + assert_eq!(result, vec![0; 8]); + assert_eq!(driver.shutdown().submitted, 2); +} + +#[tokio::test] +async fn shutdown_wakes_byte_waiters_while_canceling_pending_read() { + let Some(driver) = driver_or_skip() else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let mut waiting = driver.read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 8); + assert_pending(&mut waiting).await; + driver.shutdown(); + assert!(tokio::time::timeout(Duration::from_secs(2), waiting).await.unwrap().is_err()); + assert!(tokio::time::timeout(Duration::from_secs(2), first).await.unwrap().is_err()); +} + +#[tokio::test] +async fn completed_results_do_not_hold_driver_byte_admission() { + let Some(driver) = driver_or_skip() else { return }; + let file = Arc::new(File::open("/dev/zero").unwrap()); + let retained = driver.read_at(file.clone(), 0, 8).await.unwrap(); + let next = tokio::time::timeout(Duration::from_secs(2), driver.read_at(file, 0, 8)) + .await + .unwrap() + .unwrap(); + assert_eq!(retained, next); + assert_eq!(driver.shutdown().submitted, 2); +} From 63fd05e27a96d25ffdccf69811a0f5b454afd19d Mon Sep 17 00:00:00 2001 From: Hauser Date: Tue, 22 Sep 2026 23:54:49 +0800 Subject: [PATCH 04/22] perf(driver): avoid cancelled read continuations and orphan copies Retry interrupted eventfd transfers and retain bounded wake diagnostics. Stop positioned continuation only after a read CQE confirms safe reclamation, preserving stream and no-cancel semantics. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/cancellation-efficiency.md | 46 ++++ src/driver.rs | 387 ++++++++++++++++++++++++----- src/driver_fault_recovery_tests.rs | 26 +- 3 files changed, 382 insertions(+), 77 deletions(-) create mode 100644 docs/cancellation-efficiency.md diff --git a/docs/cancellation-efficiency.md b/docs/cancellation-efficiency.md new file mode 100644 index 0000000..576e7a2 --- /dev/null +++ b/docs/cancellation-efficiency.md @@ -0,0 +1,46 @@ +# Cancellation and eventfd efficiency + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647), step 3 (partial). + +## Implemented scope + +- [x] Retry interrupted eventfd signal/drain syscalls. Treat `EAGAIN` as an + already-pending wake on write, or an already-empty counter on read. Log an + unexpected error at most once per eventfd; the existing heartbeat remains. +- [x] Drain one successful counter read per turn. A concurrent later signal + stays readable; intake and completion processing still run after draining. +- [x] Record explicit drop-cancel intent separately from receiver closure. + After the read CQE arrives, stop positioned short-read and transient-error + continuations on explicit cancellation or shutdown with `ECANCELED`. +- [x] Preserve successful complete reads, EOF results, and `read_current` + short-read semantics when they race with cancellation. +- [x] Preserve `without_cancel_on_drop`: closing the receiver alone does not + stop positioned continuation. At the terminal completion, a closed receiver + skips result memmove/truncation/materialization, including direct-I/O padding. +- [x] Keep fd, buffer, and permit ownership until read completion and pending + removal. AsyncCancel CQEs still only update cancel statistics. +- [ ] Intake/reap budgets and wake coalescing (separate liveness work). +- [ ] Native Linux execution of the new tests and the full regression suite. +- [ ] Isolated before/after benchmark evidence; no measured speedup is claimed. + +## Review and validation + +The completion decision now lives in `reap_read` so deterministic tests can +provide short reads and transient errnos without regular-file timing races. +Tests cover explicit cancellation, shutdown, abandoned receivers without +cancellation, stream semantics, successful cancel races, direct result ranges, +unchanged orphan allocations, and resource ownership at terminal decisions. +Eventfd tests cover interrupted retries, saturated counters, empty drains, and +unexpected errors. They do not depend on io_uring availability or accept a skip. + +Local `cargo fmt --all --check`, `git diff --check`, and Linux-target +`cargo check` / `cargo clippy` with all targets and features passed. Cross-checks +compile the Linux-only tests but do not execute them. Native execution remains +an integration gate for this batch. + +Review confirms that no buffer is reclaimed at a cancel CQE or merely because +the receiver closed. The optimization only observes cancellation at a read +completion, with no live SQE writing into that buffer. A receiver closing after +the final `is_closed` check may still incur a copy; this is an accepted race, +not an ownership or correctness failure. Current direct-I/O `fstat` handling +is unchanged and is tracked by the separate integrity work. diff --git a/src/driver.rs b/src/driver.rs index 4838e83..ae9615d 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -20,7 +20,7 @@ use std::os::fd::{AsRawFd, FromRawFd}; use std::os::unix::ffi::OsStrExt; use std::pin::Pin; use std::sync::Arc; -use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; use std::sync::mpsc::{self, TryRecvError}; use std::task::{Context, Poll}; use std::thread::JoinHandle; @@ -122,6 +122,7 @@ const MAX_TRANSIENT_RETRIES: u32 = 16; /// together they replace the spike's 200 µs busy-poll. struct EventFd { fd: std::os::fd::RawFd, + error_logged: AtomicBool, } impl EventFd { @@ -131,7 +132,10 @@ impl EventFd { if fd < 0 { return Err(io::Error::last_os_error()); } - Ok(Self { fd }) + Ok(Self { + fd, + error_logged: AtomicBool::new(false), + }) } fn as_raw(&self) -> std::os::fd::RawFd { @@ -142,18 +146,53 @@ impl EventFd { /// already readable, which is all a wakeup needs. fn signal(&self) { let v: u64 = 1; - // SAFETY: writing 8 bytes from a valid u64 to an eventfd we own. - unsafe { - libc::write(self.fd, (&v as *const u64).cast(), 8); - } + self.transfer("signal", || { + // SAFETY: writing 8 bytes from a valid u64 to an eventfd we own. + unsafe { libc::write(self.fd, (&v as *const u64).cast(), 8) } + }); } /// Reset the counter. EFD_NONBLOCK guarantees this never blocks; a single - /// successful read drains the whole counter, the next returns EAGAIN. + /// successful read drains the whole counter. A later concurrent signal is + /// left readable for the next turn; intake and CQ are checked after drain. fn drain(&self) { let mut v: u64 = 0; - // SAFETY: reading 8 bytes into a valid u64 from an eventfd we own. - while unsafe { libc::read(self.fd, (&mut v as *mut u64).cast(), 8) } == 8 {} + self.transfer("drain", || { + // SAFETY: reading 8 bytes into a valid u64 from an eventfd we own. + unsafe { libc::read(self.fd, (&mut v as *mut u64).cast(), 8) } + }); + } + + fn transfer(&self, operation: &'static str, mut syscall: impl FnMut() -> isize) { + let result = eventfd_transfer(|| { + let count = syscall(); + if count < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(count as usize) + } + }); + if let Err(error) = result + && !self.error_logged.swap(true, Ordering::Relaxed) + { + // At most one warning per fd, including shared producer wake fds. + // The heartbeat still checks queued work if a wake syscall fails. + tracing::warn!(operation, %error, "uring driver: eventfd operation failed; heartbeat remains active"); + } + } +} + +/// An eventfd transfer is all-or-nothing. Retry interrupted syscalls; EAGAIN +/// means a signal is already pending (write) or no signal remains (read). +fn eventfd_transfer(mut syscall: impl FnMut() -> io::Result) -> io::Result<()> { + loop { + match syscall() { + Ok(8) => return Ok(()), + Ok(_) => return Err(io::Error::other("eventfd transferred an unexpected byte count")), + Err(error) if error.raw_os_error() == Some(libc::EINTR) => continue, + Err(error) if error.raw_os_error() == Some(libc::EAGAIN) => return Ok(()), + Err(error) => return Err(error), + } } } @@ -483,6 +522,9 @@ struct Pending { /// progress, bounded by `MAX_TRANSIENT_RETRIES` so a storm cannot spin the /// driver thread (rustfs/backlog#1166). Reset whenever a read makes progress. transient_retries: u32, + /// Explicit cancel intent, independent of receiver closure: opting out of + /// drop-cancel must still finish positioned reads after abandoning results. + cancel_requested: bool, } impl Pending { @@ -1490,6 +1532,11 @@ fn finish_direct_short_read(p: &mut Pending, file_len: io::Result) -> io::R /// clamped to `nread`, so the zero-filled remainder of the buffer stays hidden /// (content hygiene, C12 / rustfs/backlog#1062). fn deliver(p: &mut Pending) -> Vec { + if p.done.as_ref().is_none_or(oneshot::Sender::is_closed) { + // Only called after the terminal read CQE. Keep the allocation in the + // entry for normal reclamation, avoiding an orphan's O_DIRECT memmove. + return Vec::new(); + } let avail = p.nread.saturating_sub(p.head).min(p.want); let start = p.pad + p.head; // The buffered path (`align == 1`) has `pad == 0` and `head == 0`, so the @@ -1510,6 +1557,52 @@ enum ReapStep { Resubmit(io_uring::squeue::Entry), } +/// Decide what follows a READ completion, never an AsyncCancel completion. +/// Cancellation only suppresses continuation: an already-complete successful +/// read still wins the race, and streams keep their read(2) short-read result. +fn reap_read(p: &mut Pending, id: u64, res: i32, shutting_down: bool) -> ReapStep { + let stop_continuation = p.cancel_requested || shutting_down; + if res < 0 { + let err = -res; + let transient = err == libc::EINTR || err == libc::EAGAIN; + if transient && p.offset != CURRENT_POSITION && p.nread < p.region_len { + if stop_continuation { + return ReapStep::Finish(Err(io::Error::from_raw_os_error(libc::ECANCELED))); + } + if p.transient_retries < MAX_TRANSIENT_RETRIES { + p.transient_retries += 1; + return ReapStep::Resubmit(p.read_sqe(id)); + } + } + return ReapStep::Finish(Err(io::Error::from_raw_os_error(err))); + } + if res == 0 { + return ReapStep::Finish(Ok(deliver(p))); + } + p.nread += res as usize; + // Progress resets the transient retry budget (rustfs/backlog#1166). + p.transient_retries = 0; + let is_stream = p.offset == CURRENT_POSITION; + let covered = p.nread >= p.head + p.want; + if is_stream || covered || p.nread >= p.region_len { + return ReapStep::Finish(Ok(deliver(p))); + } + if stop_continuation { + // This read SQE has completed, so reclamation is safe now. Do not start + // another positioned read for an explicitly cancelled logical request. + // Never report its incomplete prefix as successful whole-range output. + return ReapStep::Finish(Err(io::Error::from_raw_os_error(libc::ECANCELED))); + } + if p.align > 1 && !p.nread.is_multiple_of(p.align) { + // Disambiguate a genuine direct-I/O tail from a non-aligned short read + // before EOF; that offset cannot be resubmitted (rustfs/backlog#1168). + let file_len = p.file.metadata().map(|metadata| metadata.len()); + ReapStep::Finish(finish_direct_short_read(p, file_len)) + } else { + ReapStep::Resubmit(p.read_sqe(id)) + } +} + /// Queue at most one `AsyncCancel` per op (rustfs/backlog#1167): a drop-cancel /// followed by a shutdown, or the submit-error shutdown, must not enqueue a /// second cancel for the same id. The set is bounded by the pending table @@ -1752,6 +1845,7 @@ fn drive( region_len, align, transient_retries: 0, + cancel_requested: false, }, ); let sqe = state.pending.get(&id).expect("just inserted").read_sqe(id); @@ -1764,7 +1858,8 @@ fn drive( } } Msg::Cancel { id } => { - if state.pending.contains_key(&id) { + if let Some(pending) = state.pending.get_mut(&id) { + pending.cancel_requested = true; queue_cancel(&mut state.backlog, &mut queued_cancels, id); } } @@ -1843,63 +1938,7 @@ fn drive( // the borrow ends (finish removes it; resubmit re-queues an SQE). let step = { let p = state.pending.get_mut(&ud).expect("checked above"); - if res < 0 { - let err = -res; - // C7 three-class contract (rustfs/backlog#1166): a transient - // errno (EINTR/EAGAIN) must be retried, not surfaced as the - // read's final result — surfacing it would also discard the - // already-read prefix of a resubmit. Bounded per logical read - // so a storm cannot spin the driver thread. Streams - // (CURRENT_POSITION) cannot resubmit positionally; ECANCELED - // and every other errno terminate the logical read. - let transient = err == libc::EINTR || err == libc::EAGAIN; - if transient - && p.offset != CURRENT_POSITION - && p.nread < p.region_len - && p.transient_retries < MAX_TRANSIENT_RETRIES - { - p.transient_retries += 1; - ReapStep::Resubmit(p.read_sqe(ud)) - } else { - // Error (incl. ECANCELED, or a transient errno past its - // retry budget) terminates the logical read. - ReapStep::Finish(Err(io::Error::from_raw_os_error(err))) - } - } else if res == 0 { - // Real EOF: deliver whatever of the logical range was read. - ReapStep::Finish(Ok(deliver(p))) - } else { - p.nread += res as usize; - // Progress resets the transient-retry budget (rustfs/backlog#1166). - p.transient_retries = 0; - // Only POSITIONED reads (read_at / read_at_direct, whole-range - // pread contract) resubmit a short read. CURRENT_POSITION - // reads (read_current on pipes/streams) follow read(2) - // semantics: a short read is a valid final result and must be - // delivered as-is — resubmitting would block forever waiting - // for stream data that may never come. - let is_stream = p.offset == CURRENT_POSITION; - let covered = p.nread >= p.head + p.want; - if is_stream || covered || p.nread >= p.region_len { - ReapStep::Finish(Ok(deliver(p))) - } else if p.align > 1 && !p.nread.is_multiple_of(p.align) { - // O_DIRECT non-block-multiple short read below the covered - // range. The kernel returns block multiples EXCEPT at the - // file tail — but a stacked filesystem (NFS/FUSE, or a - // signal-split direct I/O) can legally return a non-multiple - // mid-file, and assuming EOF there would silently truncate - // the delivered range. Disambiguate with the actual file - // length instead of inferring it (rustfs/backlog#1168). - let file_len = p.file.metadata().map(|metadata| metadata.len()); - ReapStep::Finish(finish_direct_short_read(p, file_len)) - } else { - // Positioned short read, not EOF, block-aligned: resubmit - // the remainder into the read region. The buffer stays - // owned by the driver and in_flight is unchanged — one - // logical op. - ReapStep::Resubmit(p.read_sqe(ud)) - } - } + reap_read(p, ud, res, shutting_down) }; #[cfg(feature = "diagnostics")] @@ -2019,3 +2058,217 @@ fn drive( // the next CQE, message, or heartbeat (backlog#1102). } } + +#[cfg(test)] +mod cancellation_efficiency_tests { + use super::*; + + // No real I/O is submitted: completion decisions get deterministic short + // reads / errno results, so a naturally completed file cannot mask a retry. + fn pending() -> (Pending, oneshot::Receiver>>, Arc) { + let sem = Arc::new(Semaphore::new(1)); + let (done, rx) = oneshot::channel(); + let p = Pending { + #[cfg(feature = "diagnostics")] + timing: None, + buf: (0..16).collect(), + file: Arc::new(File::open("/dev/null").expect("open fixture fd")), + done: Some(done), + offset: 0, + nread: 0, + _permit: ReadPermits { + _count: Arc::clone(&sem).try_acquire_owned().expect("fixture permit"), + _bytes: None, + }, + pad: 0, + head: 0, + want: 16, + region_len: 16, + align: 1, + transient_retries: 0, + cancel_requested: false, + }; + (p, rx, sem) + } + + fn assert_cancelled(step: ReapStep) { + match step { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::ECANCELED)), + _ => panic!("cancelled positioned continuation must finish with ECANCELED"), + } + } + + #[test] + fn explicit_cancel_stops_short_read_without_releasing_resources_early() { + let (mut p, _rx, sem) = pending(); + let file = Arc::downgrade(&p.file); + let buffer = p.buf.as_ptr(); + p.cancel_requested = true; + assert_eq!(sem.available_permits(), 0); + assert_cancelled(reap_read(&mut p, 1, 4, false)); + assert_eq!(p.nread, 4); + assert_eq!(p.buf.as_ptr(), buffer); + assert_eq!(sem.available_permits(), 0); + assert!(file.upgrade().is_some()); + drop(p); + assert_eq!(sem.available_permits(), 1); + assert!(file.upgrade().is_none()); + } + + #[test] + fn explicit_cancel_stops_both_transient_errno_retries() { + for errno in [libc::EINTR, libc::EAGAIN] { + let (mut p, _rx, _sem) = pending(); + p.cancel_requested = true; + assert_cancelled(reap_read(&mut p, 1, -errno, false)); + assert_eq!(p.transient_retries, 0); + } + } + + #[test] + fn shutdown_stops_short_read_and_transient_continuations() { + for result in [4, -libc::EINTR, -libc::EAGAIN] { + let (mut p, _rx, _sem) = pending(); + assert_cancelled(reap_read(&mut p, 1, result, true)); + } + } + + #[test] + fn closed_receiver_without_cancel_still_continues_positioned_reads() { + for result in [4, -libc::EINTR, -libc::EAGAIN] { + let (mut p, rx, sem) = pending(); + drop(rx); + assert!(matches!(reap_read(&mut p, 1, result, false), ReapStep::Resubmit(_))); + assert_eq!(sem.available_permits(), 0); + } + } + + #[test] + fn current_position_short_read_survives_cancel_and_shutdown_race() { + let (mut p, _rx, _sem) = pending(); + p.offset = CURRENT_POSITION; + p.cancel_requested = true; + match reap_read(&mut p, 1, 4, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [0, 1, 2, 3]), + _ => panic!("stream short read must keep its read(2) result"), + } + } + + #[test] + fn successful_complete_read_wins_cancel_race() { + let (mut p, _rx, _sem) = pending(); + p.cancel_requested = true; + match reap_read(&mut p, 1, 16, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, (0..16).collect::>()), + _ => panic!("complete read must retain success"), + } + } + + #[test] + fn eof_after_prefix_preserves_success_during_shutdown() { + let (mut p, _rx, _sem) = pending(); + p.nread = 4; + p.cancel_requested = true; + match reap_read(&mut p, 1, 0, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [0, 1, 2, 3]), + _ => panic!("observed EOF must preserve the completed prefix"), + } + } + + #[test] + fn transient_retry_keeps_prefix_and_exhausts_budget() { + let (mut p, _rx, sem) = pending(); + p.nread = 4; + for _ in 0..MAX_TRANSIENT_RETRIES { + assert!(matches!(reap_read(&mut p, 1, -libc::EAGAIN, false), ReapStep::Resubmit(_))); + assert_eq!(p.nread, 4); + assert_eq!(sem.available_permits(), 0); + } + match reap_read(&mut p, 1, -libc::EAGAIN, false) { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::EAGAIN)), + _ => panic!("transient retry budget must remain bounded"), + } + } + + #[test] + fn current_position_transient_error_never_retries() { + let (mut p, _rx, _sem) = pending(); + p.offset = CURRENT_POSITION; + p.cancel_requested = true; + match reap_read(&mut p, 1, -libc::EINTR, true) { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::EINTR)), + _ => panic!("stream errors preserve read(2) semantics"), + } + } + + #[test] + fn closed_receiver_skips_direct_result_copy_and_materialization() { + let (mut p, rx, _sem) = pending(); + p.pad = 2; + p.head = 3; + p.want = 4; + p.region_len = 8; + p.align = 4; + let before = p.buf.clone(); + let ptr = p.buf.as_ptr(); + drop(rx); + match reap_read(&mut p, 1, 8, false) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes.capacity(), 0), + _ => panic!("terminal direct read must finish"), + } + assert_eq!(p.buf, before, "orphan result must not be memmoved or truncated"); + assert_eq!(p.buf.as_ptr(), ptr); + } + + #[test] + fn live_receiver_gets_exact_direct_result_range() { + let (mut p, _rx, _sem) = pending(); + p.pad = 2; + p.head = 3; + p.want = 4; + p.region_len = 8; + p.align = 4; + match reap_read(&mut p, 1, 8, false) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [5, 6, 7, 8]), + _ => panic!("direct result must contain exactly the logical range"), + } + } + + #[test] + fn eventfd_retries_interruption_until_transfer_or_would_block() { + for terminal in [Ok(8), Err(io::Error::from_raw_os_error(libc::EAGAIN))] { + let mut calls = [ + Err(io::Error::from_raw_os_error(libc::EINTR)), + Err(io::Error::from_raw_os_error(libc::EINTR)), + terminal, + ] + .into_iter(); + eventfd_transfer(|| calls.next().expect("unexpected retry")).expect("transfer or readiness satisfied"); + assert!(calls.next().is_none()); + } + } + + #[test] + fn eventfd_propagates_unexpected_error_and_short_transfer() { + let error = eventfd_transfer(|| Err(io::Error::from_raw_os_error(libc::EBADF))).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EBADF)); + assert!(eventfd_transfer(|| Ok(4)).is_err()); + } + + #[test] + fn saturated_eventfd_signal_keeps_readiness_and_drain_clears_it() { + let event = EventFd::new().expect("eventfd"); + let value = u64::MAX - 1; + // SAFETY: event owns the fd; value is an initialized eight-byte counter. + assert_eq!(unsafe { libc::write(event.as_raw(), (&value as *const u64).cast(), 8) }, 8); + event.signal(); // EAGAIN: the already readable saturated fd is sufficient. + assert!(!event.error_logged.load(Ordering::Relaxed)); + event.drain(); + let mut read = 0_u64; + // SAFETY: event owns the fd; read is a valid eight-byte output buffer. + assert_eq!(unsafe { libc::read(event.as_raw(), (&mut read as *mut u64).cast(), 8) }, -1); + assert_eq!(io::Error::last_os_error().raw_os_error(), Some(libc::EAGAIN)); + event.drain(); // Empty EAGAIN is normal too. + assert!(!event.error_logged.load(Ordering::Relaxed)); + } +} diff --git a/src/driver_fault_recovery_tests.rs b/src/driver_fault_recovery_tests.rs index f9d9560..e938f04 100644 --- a/src/driver_fault_recovery_tests.rs +++ b/src/driver_fault_recovery_tests.rs @@ -3,29 +3,35 @@ use super::*; -fn direct_prefix() -> Pending { - Pending { +fn direct_prefix() -> (Pending, oneshot::Receiver>>) { + let (done, receiver) = oneshot::channel(); + let pending = Pending { #[cfg(feature = "diagnostics")] timing: None, // Simulated completed CQE: no kernel ever references this allocation. buf: vec![99, 10, 11, 12, 13, 14, 88, 88, 88], file: Arc::new(File::open("/dev/null").unwrap()), - done: None, + done: Some(done), offset: 4096, nread: 5, - _permit: Arc::new(Semaphore::new(1)).try_acquire_owned().unwrap(), + _permit: ReadPermits { + _count: Arc::new(Semaphore::new(1)).try_acquire_owned().unwrap(), + _bytes: None, + }, pad: 1, head: 2, want: 6, region_len: 8, align: 4, transient_retries: 0, - } + cancel_requested: false, + }; + (pending, receiver) } #[test] fn direct_metadata_failure_preserves_errno_without_delivering_prefix() { - let mut pending = direct_prefix(); + let (mut pending, _receiver) = direct_prefix(); let before = pending.buf.clone(); let error = finish_direct_short_read(&mut pending, Err(io::Error::from_raw_os_error(libc::EIO))).unwrap_err(); assert_eq!(error.raw_os_error(), Some(libc::EIO)); @@ -34,7 +40,7 @@ fn direct_metadata_failure_preserves_errno_without_delivering_prefix() { #[test] fn direct_mid_file_short_read_errors_without_delivering_prefix() { - let mut pending = direct_prefix(); + let (mut pending, _receiver) = direct_prefix(); let error = finish_direct_short_read(&mut pending, Ok(8192)).unwrap_err(); assert_eq!(error.kind(), io::ErrorKind::Other); assert!(!pending.buf.is_empty()); @@ -42,20 +48,20 @@ fn direct_mid_file_short_read_errors_without_delivering_prefix() { #[test] fn direct_confirmed_tail_returns_only_initialized_logical_bytes() { - let mut pending = direct_prefix(); + let (mut pending, _receiver) = direct_prefix(); assert_eq!(finish_direct_short_read(&mut pending, Ok(4101)).unwrap(), [12, 13, 14]); } #[test] fn direct_concurrent_truncation_preserves_completed_prefix() { - let mut pending = direct_prefix(); + let (mut pending, _receiver) = direct_prefix(); // A truncate after the CQE does not erase bytes the read already completed. assert_eq!(finish_direct_short_read(&mut pending, Ok(4096)).unwrap(), [12, 13, 14]); } #[test] fn direct_eof_before_logical_start_returns_empty() { - let mut pending = direct_prefix(); + let (mut pending, _receiver) = direct_prefix(); pending.nread = 1; assert!(finish_direct_short_read(&mut pending, Ok(4097)).unwrap().is_empty()); } From 8ad360800e69a72873280edf1af41061956fac65 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 00:50:05 +0800 Subject: [PATCH 05/22] fix(admission): close count waiters on global byte shutdown Register shard count semaphores in shared admission state, synchronize registration with terminal close, and invoke semaphore wakeups outside the registry lock. Cover count-stage closure, late/concurrent registration, and partial FIFO byte cancellation. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- README.md | 9 ++- src/admission_tests.rs | 124 ++++++++++++++++++++++++++++++++++++++--- src/driver.rs | 94 +++++++++++++++++++++++++------ 3 files changed, 197 insertions(+), 30 deletions(-) diff --git a/README.md b/README.md index ed1f44c..1021670 100644 --- a/README.md +++ b/README.md @@ -65,9 +65,12 @@ count permit, and Tokio's fair byte semaphore can put small reads behind a large waiter. Dropping a waiting handle returns all partial reservations. After enqueue, both permits travel with the read until its terminal CQE, even if its caller is canceled. Short-read retries retain the same reservation. A leaked read retains -its charge. Shutdown closes the byte semaphore immediately; any shard-thread exit -also closes it for the whole driver, conservatively rejecting further byte -admission and waking byte waiters even on a bounded-drain escape. +its charge. Shutdown or any shard-thread exit closes both the shared byte +semaphore and every shard's count semaphore when byte limits are enabled. +This rejects further admission and wakes waiters at either acquisition stage, +even when another shard has a hung read or takes a bounded-drain escape. +Registration and terminal closure are synchronized at startup/shutdown; ordinary +read admission does not take a registry lock. This limits reserved driver read-buffer allocation bytes, **not process RSS**. It excludes queued handle/FD metadata, allocator overhead, result copies and diff --git a/src/admission_tests.rs b/src/admission_tests.rs index acd05ee..47a14f2 100644 --- a/src/admission_tests.rs +++ b/src/admission_tests.rs @@ -37,11 +37,13 @@ fn dropping_count_waiter_does_not_reserve_bytes() { #[test] fn shard_exit_closes_bytes_and_releases_waiting_count_permit() { let count = Arc::new(Semaphore::new(2)); - let bytes = Arc::new(Semaphore::new(8)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let bytes = admission.bytes.clone(); let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8); assert!(poll_once(&mut waiting).is_pending()); - drop(CloseByteAdmission(Some(bytes.clone()))); + drop(CloseByteAdmission(Some(admission))); assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); assert_eq!(count.available_permits(), 1); assert_eq!(bytes.available_permits(), 0); @@ -49,12 +51,107 @@ fn shard_exit_closes_bytes_and_releases_waiting_count_permit() { } #[test] -fn permits_retained_by_leaked_operation_remain_charged_after_close() { +fn global_close_wakes_count_stage_waiter_without_releasing_hung_read() { + struct LockCheckingWake { + admission: Arc, + wakes: AtomicUsize, + } + impl std::task::Wake for LockCheckingWake { + fn wake(self: Arc) { + self.wake_by_ref(); + } + + fn wake_by_ref(self: &Arc) { + assert!(self.admission.registry.try_lock().is_ok(), "registry lock held across task wake"); + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + let count = Arc::new(Semaphore::new(1)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let hung = try_read_permits(&count, Some(&admission.bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8); + let wake = Arc::new(LockCheckingWake { + admission: admission.clone(), + wakes: AtomicUsize::new(0), + }); + let waker = Waker::from(wake.clone()); + assert!(waiting.as_mut().poll(&mut Context::from_waker(&waker)).is_pending()); + + // Model a different shard exiting while this shard's read stays hung. + drop(CloseByteAdmission(Some(admission.clone()))); + assert!(wake.wakes.load(Ordering::SeqCst) > 0, "count waiter was not woken"); + assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); + assert!(matches!( + try_read_permits(&count, Some(&admission.bytes), 8), + Err(TryAcquireError::Closed) + )); + let mut new_waiter = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8); + assert!(matches!(poll_once(&mut new_waiter), Poll::Ready(Err(_)))); + assert_eq!(admission.bytes.available_permits(), 0, "hung read must remain charged"); + drop(hung); +} + +#[test] +fn late_registration_is_closed_after_terminal_admission_close() { + let admission = ByteAdmission::new(8); + admission.close(); + let count = Arc::new(Semaphore::new(1)); + admission.register(&count); + assert!(count.is_closed()); + assert!(admission.registry.lock().unwrap().counts.is_empty()); +} + +#[test] +fn concurrent_registration_cannot_escape_terminal_close() { + for _ in 0..16 { + let admission = ByteAdmission::new(8); + let count = Arc::new(Semaphore::new(1)); + let barrier = std::sync::Barrier::new(2); + std::thread::scope(|scope| { + scope.spawn(|| { + barrier.wait(); + admission.register(&count); + }); + barrier.wait(); + admission.close(); + }); + assert!(count.is_closed()); + assert!(admission.bytes.is_closed()); + } +} + +#[test] +fn canceling_partial_fifo_byte_reservation_unblocks_next_waiter() { + let count = Arc::new(Semaphore::new(4)); let bytes = Arc::new(Semaphore::new(8)); + let held = try_read_permits(&count, Some(&bytes), 4).unwrap(); + let mut large = acquire_read_permits(count.clone(), Some(bytes.clone()), 6); + assert!(poll_once(&mut large).is_pending()); + assert_eq!(bytes.available_permits(), 0, "large FIFO waiter should reserve the available four bytes"); + let mut small = acquire_read_permits(count.clone(), Some(bytes.clone()), 2); + assert!(poll_once(&mut small).is_pending()); + drop(large); + let Poll::Ready(Ok(small)) = poll_once(&mut small) else { + panic!("canceling partial reservation did not advance FIFO") + }; + assert_eq!(bytes.available_permits(), 2); + drop(held); + drop(small); + assert_eq!(bytes.available_permits(), 8); + assert_eq!(count.available_permits(), 4); +} + +#[test] +fn permits_retained_by_leaked_operation_remain_charged_after_close() { + let count = Arc::new(Semaphore::new(1)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let bytes = admission.bytes.clone(); let leaked = try_read_permits(&count, Some(&bytes), 8).unwrap(); std::mem::forget(leaked); - drop(CloseByteAdmission(Some(bytes.clone()))); + drop(CloseByteAdmission(Some(admission))); assert_eq!(bytes.available_permits(), 0); assert_eq!(count.available_permits(), 0); } @@ -76,15 +173,20 @@ fn shared_byte_budget_serializes_reads_from_distinct_shards() { fn fake_driver(limits: ReadLimits) -> (UringDriver, mpsc::Receiver) { let (tx, rx) = mpsc::channel(); + let count = Arc::new(Semaphore::new(8)); + let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + if let Some(admission) = &byte_admission { + admission.register(&count); + } ( UringDriver { limits, - byte_sem: limits.max_in_flight_bytes.map(|bytes| Arc::new(Semaphore::new(bytes))), + byte_admission, shards: vec![Shard { tx, handle: None, stats: Arc::new(DriverStats::default()), - sem: Arc::new(Semaphore::new(8)), + sem: count, wake_efd: Arc::new(EventFd::new().unwrap()), }], next_id: AtomicU64::new(1), @@ -126,11 +228,15 @@ async fn direct_budget_includes_superset_padding_and_zero_length_geometry() { }); let handle = driver.read_at_direct(file, offset, len, 4096).without_cancel_on_drop(); let msg = rx.try_recv().unwrap(); - assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), 0); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 0); drop(handle); - assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), 0, "caller drop released bytes"); + assert_eq!( + driver.byte_admission.as_ref().unwrap().bytes.available_permits(), + 0, + "caller drop released bytes" + ); drop(msg); - assert_eq!(driver.byte_sem.as_ref().unwrap().available_permits(), charge); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), charge); } } diff --git a/src/driver.rs b/src/driver.rs index ae9615d..67dea67 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -19,9 +19,9 @@ use std::io::Write as _; use std::os::fd::{AsRawFd, FromRawFd}; use std::os::unix::ffi::OsStrExt; use std::pin::Pin; -use std::sync::Arc; use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; use std::sync::mpsc::{self, TryRecvError}; +use std::sync::{Arc, Mutex}; use std::task::{Context, Poll}; use std::thread::JoinHandle; use std::time::{Duration, Instant}; @@ -335,8 +335,8 @@ pub struct ReadLimits { /// completion, including canceled operations; leaked operations stay charged. /// Requests larger than this budget return `InvalidInput`, never wait. /// Zero and values above Tokio's `Semaphore::MAX_PERMITS` are invalid. - /// Shutdown or any shard exit closes byte admission for the entire driver - /// when enabled, waking byte waiters even if buffers must be leaked. + /// Shutdown or any shard exit closes count and byte admission for the entire + /// driver when enabled, waking all waiters even if buffers must be leaked. pub max_in_flight_bytes: Option, } @@ -370,12 +370,63 @@ fn acquire_read_permits(count: Arc, bytes: Option>, ch }) } -struct CloseByteAdmission(Option>); +/// Registration and terminal closure happen only at startup/shutdown, never on +/// read admission. The shared lock orders registration against shard failure. +struct ByteAdmission { + bytes: Arc, + registry: Mutex, +} + +#[derive(Default)] +struct AdmissionRegistry { + closed: bool, + counts: Vec>, +} + +impl ByteAdmission { + fn new(bytes: usize) -> Self { + Self { + bytes: Arc::new(Semaphore::new(bytes)), + registry: Mutex::new(AdmissionRegistry::default()), + } + } + + fn register(&self, count: &Arc) { + let close = { + let mut registry = self.registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if registry.closed { + true + } else { + registry.counts.push(Arc::clone(count)); + false + } + }; + // Semaphore::close may run arbitrary task wakers. Never hold the + // registry lock across it, including registration after terminal close. + if close { + count.close(); + } + } + + fn close(&self) { + let counts = { + let mut registry = self.registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + registry.closed = true; + std::mem::take(&mut registry.counts) + }; + self.bytes.close(); + for count in counts { + count.close(); + } + } +} + +struct CloseByteAdmission(Option>); impl Drop for CloseByteAdmission { fn drop(&mut self) { - if let Some(bytes) = &self.0 { - bytes.close(); + if let Some(admission) = &self.0 { + admission.close(); } } } @@ -782,7 +833,7 @@ impl Drop for Shard { /// environments can fall back to a blocking backend before serving traffic. pub struct UringDriver { limits: ReadLimits, - byte_sem: Option>, + byte_admission: Option>, /// One or more independent rings. A cache-hit buffered read completes inline /// inside `io_uring_enter`, so the thread driving a ring performs that /// read's memcpy — which caps a single-ring driver at one core's memory @@ -844,7 +895,7 @@ impl UringDriver { "byte budget must be between 1 and Semaphore::MAX_PERMITS", ))); } - let byte_sem = limits.max_in_flight_bytes.map(|bytes| Arc::new(Semaphore::new(bytes))); + let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); let mut started = Vec::with_capacity(shards.max(1)); for i in 0..shards.max(1) { // Probe only the first shard (rustfs/backlog#1165): the probe read @@ -853,11 +904,11 @@ impl UringDriver { // verify NODROP — this avoids `shards - 1` extra O_TMPFILE // create+write+read round-trips per disk on every start and renew. // `?` drops `started`, whose `Shard::drop` joins each running thread. - started.push(Self::start_shard(entries, i == 0, byte_sem.clone())?); + started.push(Self::start_shard(entries, i == 0, byte_admission.clone())?); } Ok(Self { limits, - byte_sem, + byte_admission, shards: started, next_id: AtomicU64::new(1), rr: AtomicUsize::new(0), @@ -872,7 +923,7 @@ impl UringDriver { &self.shards[self.rr.fetch_add(1, Ordering::Relaxed) % n] } - fn start_shard(entries: u32, probe: bool, byte_sem: Option>) -> Result { + fn start_shard(entries: u32, probe: bool, byte_admission: Option>) -> Result { let mut ring = IoUring::new(entries).map_err(ProbeFailure::Setup)?; // Require the NODROP feature (kernel >= 5.5). Without it, CQ overflow // silently drops CQEs, stranding pending entries forever and hanging @@ -915,6 +966,9 @@ impl UringDriver { // Cap in-flight at the SQ depth (entries), which is < CQ capacity // (2*entries), so CQ overflow is structurally unreachable (C5/C10). let sem = Arc::new(Semaphore::new(entries as usize)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } let thread_sem = Arc::clone(&sem); // Deterministic spawn-failure seam (rustfs/backlog#1164): exercise the // degrade-not-panic path without a real cgroup pids-limit. Never present @@ -933,7 +987,7 @@ impl UringDriver { let handle = std::thread::Builder::new() .name("uring-spike-driver".into()) .spawn(move || { - let _close_bytes = CloseByteAdmission(byte_sem); + let _close_bytes = CloseByteAdmission(byte_admission); drive(ring, rx, thread_stats, thread_sem, cq_efd, thread_wake); }) .map_err(ProbeFailure::Setup)?; @@ -1123,7 +1177,7 @@ impl UringDriver { // released only when the pending entry is dropped at the CQE (C10, // rustfs/backlog#1060). Acquisition never blocks the caller's thread // (rustfs/backlog#1102). - match try_read_permits(&shard.sem, self.byte_sem.as_ref(), charge) { + match try_read_permits(&shard.sem, self.byte_admission.as_ref().map(|admission| &admission.bytes), charge) { // Fast path: a permit was free, so submit eagerly — no allocation, // no await, and the op is in flight the moment `submit` returns, // exactly as with the previous blocking implementation. @@ -1189,7 +1243,11 @@ impl UringDriver { finished: false, cancel_on_drop: true, state: HandleState::WaitingPermit { - acquire: acquire_read_permits(Arc::clone(&shard.sem), self.byte_sem.clone(), charge), + acquire: acquire_read_permits( + Arc::clone(&shard.sem), + self.byte_admission.as_ref().map(|admission| Arc::clone(&admission.bytes)), + charge, + ), file, offset, len, @@ -1273,8 +1331,8 @@ impl UringDriver { /// Shards are asked to stop first and joined afterwards, so their bounded /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. pub fn shutdown(mut self) -> StatsSnapshot { - if let Some(bytes) = &self.byte_sem { - bytes.close(); + if let Some(admission) = &self.byte_admission { + admission.close(); } for shard in &self.shards { let _ = shard.tx.send(Msg::Shutdown); @@ -1301,8 +1359,8 @@ impl UringDriver { impl Drop for UringDriver { fn drop(&mut self) { - if let Some(bytes) = &self.byte_sem { - bytes.close(); + if let Some(admission) = &self.byte_admission { + admission.close(); } // Ask every shard to stop before joining any of them, so their bounded // drains overlap. Dropping the `Vec` would instead run each From 46c746fc323842e5230011ffcf8c2ad6ca0fabe0 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 00:55:02 +0800 Subject: [PATCH 06/22] perf(driver): bound intake and completion work per turn Alternate bounded message, allocation, and completion batches while remembering pending local work after eventfd drains. Continue partial submissions only after positive progress to retain idle efficiency and avoid zero-progress retry spins. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/driver-fairness.md | 46 +++++++++++++ src/driver.rs | 79 ++++++++++++++++++---- src/driver_loop_budget_tests.rs | 113 ++++++++++++++++++++++++++++++++ 3 files changed, 226 insertions(+), 12 deletions(-) create mode 100644 docs/driver-fairness.md create mode 100644 src/driver_loop_budget_tests.rs diff --git a/docs/driver-fairness.md b/docs/driver-fairness.md new file mode 100644 index 0000000..caa7ea7 --- /dev/null +++ b/docs/driver-fairness.md @@ -0,0 +1,46 @@ +# Driver turn fairness + +Each driver turn processes at most 64 input messages and 64 completion entries. +Intake also ends once the turn has allocated 8 MiB of read buffers, including +alignment padding. These private constants bound batches; they are not claimed +to be optimal throughput settings. Public configuration and ring flags remain +unchanged. + +An individual accepted read may exceed 8 MiB: it is allocated once and then +intake yields to submission and reap. Consequently the allocation threshold may +be exceeded by one allocation. These limits do not preempt a single allocation, +direct-read copy, metadata lookup, syscall, or shutdown's cancellation sweep. +Request-count and optional byte admission limits continue to govern ownership. + +The driver alternates bounded intake, submission, bounded reap, and a second +submission. This prevents a continuously replenished input or completion queue +from starving the other phase or shutdown-deadline checks. Messages remain FIFO; +this does not move cancellation or shutdown ahead of previously queued reads. + +Hitting either phase budget remembers that work may remain and starts the next +turn without waiting for another eventfd edge. This matters because eventfd has +already been drained even when input/CQ entries remain. A ready CQ also prevents +waiting. A partial successful submission continues immediately while queued SQ +work remains; failed or zero-progress submissions alone do not prevent waiting. +The existing active/idle heartbeats and two bounded submission attempts remain. +An exact budget boundary with no actual remaining work costs one empty turn, +after which normal idle waiting resumes. + +No wakeup coalescing or async-only eventfd registration is introduced. Messages +arriving after the driver's queue check retain their eventfd notification; work +left because of a budget retains the explicit continuation flag. Buffers, FDs +and permits retain the existing final-CQE ownership rules. + +## Validation + +```sh +cargo test --all-features --lib loop_budget_tests -- --nocapture +``` + +Tests exercise a real eventfd plus an input queue whose wake signal is drained +before the intake limit, allocation fairness including an oversized read, +completion budget continuation, idle boundaries and submission retry decisions. +Submission-error/partial-submit decisions and CQ batches are deterministic +models, not injected kernel failures. Native read/cancellation/fault suites +remain necessary integration coverage. Throughput and tail-latency changes +require isolated before/after measurements and are currently unmeasured. diff --git a/src/driver.rs b/src/driver.rs index 67dea67..08cc86f 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -116,6 +116,41 @@ const MAX_CONSECUTIVE_SUBMIT_ERRORS: u32 = 128; /// pathological storm cannot spin the driver thread (rustfs/backlog#1166). const MAX_TRANSIENT_RETRIES: u32 = 16; +// Bound each phase so intake/allocation cannot indefinitely defer reap, and a +// CQ burst cannot indefinitely defer cancel/shutdown intake. These are fairness +// limits, not tuned throughput settings (rustfs/backlog#2647). +const TURN_MESSAGES: usize = 64; +const TURN_COMPLETIONS: usize = 64; +const TURN_ALLOCATION_BYTES: usize = 8 * 1024 * 1024; + +#[derive(Default)] +struct TurnBudget { + messages: usize, + completions: usize, + allocation_bytes: usize, +} + +impl TurnBudget { + fn can_take_message(&self) -> bool { + self.messages < TURN_MESSAGES && self.allocation_bytes < TURN_ALLOCATION_BYTES + } + + fn can_reap(&self) -> bool { + self.completions < TURN_COMPLETIONS + } + + fn continue_without_wait(&self, ready_cqe: bool, queued_submission: bool, submitted: usize) -> bool { + // A hit budget can leave work behind after its eventfd edge was drained. + // One extra empty turn at an exact boundary is harmless. Merely having + // unaccepted SQEs must not spin after EBUSY, EINTR, errors or Ok(0). + !self.can_take_message() || !self.can_reap() || ready_cqe || (queued_submission && submitted > 0) + } +} + +#[cfg(test)] +#[path = "driver_loop_budget_tests.rs"] +mod loop_budget_tests; + /// Owned `eventfd(2)` used to wake the driver loop (backlog#1102): one is /// registered with the ring so the kernel signals it on every CQE, the other is /// signaled by `submit`/shutdown so a new message wakes the loop immediately — @@ -1706,6 +1741,7 @@ mod fault_recovery_tests; /// when both submissions and kernel completion work are absent. EINTR/EBUSY are /// transient; any other errno is counted and, after a bounded run, transitions /// the shard to shutdown so callers fall back to the std backend. +/// Returns the accepted SQE count, or zero when idle or submission failed. fn submit_ring( state: &mut DriverState, stats: &DriverStats, @@ -1713,13 +1749,16 @@ fn submit_ring( submit_error_logged: &mut bool, shutting_down: &mut bool, queued_cancels: &mut HashSet, -) { +) -> usize { flush_backlog(&mut state.ring, &mut state.backlog); let Some(result) = submit_if_needed(&mut state.ring) else { - return; + return 0; }; match result { - Ok(_) => *consecutive_submit_errors = 0, + Ok(submitted) => { + *consecutive_submit_errors = 0; + return submitted; + } // CQ-overflow backpressure (EBUSY) and signal interruption (EINTR) are // transient — retry next turn without counting them (C5, backlog#1056). Err(e) if matches!(e.raw_os_error(), Some(libc::EBUSY) | Some(libc::EINTR)) => *consecutive_submit_errors = 0, @@ -1747,6 +1786,7 @@ fn submit_ring( } } } + 0 } fn drive( @@ -1793,6 +1833,7 @@ fn drive( #[cfg(feature = "fault-injection")] let fault_stuck_drain = std::env::var_os("RUSTFS_URING_FAULT_STUCK_DRAIN").is_some(); + let mut continue_without_wait = false; loop { // Block until a CQE is ready (the ring's registered eventfd), a new // message arrives (the wakeup eventfd), or the heartbeat elapses — @@ -1810,13 +1851,16 @@ fn drive( } else { IDLE_HEARTBEAT }; - wait_for_events(&cq_efd, &wake_efd, heartbeat); + if !continue_without_wait { + wait_for_events(&cq_efd, &wake_efd, heartbeat); + } cq_efd.drain(); wake_efd.drain(); - // 1. Intake: drain all queued messages (the wait above did the blocking, - // so this is purely non-blocking). - loop { + let mut budget = TurnBudget::default(); + // 1. Bounded intake. An individual large accepted read still makes + // progress; its allocation ends this phase rather than deferring forever. + while budget.can_take_message() { let msg = match rx.try_recv() { Ok(m) => m, Err(TryRecvError::Empty) => break, @@ -1825,6 +1869,7 @@ fn drive( break; } }; + budget.messages += 1; match msg { Msg::Read { id, @@ -1868,6 +1913,7 @@ fn drive( } }; let buf = vec![0u8; cap]; + budget.allocation_bytes = budget.allocation_bytes.saturating_add(cap); let pad = buf.as_ptr().align_offset(align); // Runtime guard (not a debug-only assert): if the allocator // ever returned a block `align_offset` cannot satisfy, refuse @@ -1943,7 +1989,7 @@ fn drive( // 2. Flush the backlog into the SQ and submit it (the single submit path; // see `submit_ring`). - submit_ring( + let submitted_before_reap = submit_ring( &mut state, &stats, &mut consecutive_submit_errors, @@ -1955,7 +2001,12 @@ fn drive( // 3. Reap. A Pending entry (and thus its buffer) is dropped ONLY when // the logical read finishes; a short read is resubmitted for the // remainder and the entry stays put (C9, rustfs/backlog#1058). - while let Some(cqe) = state.ring.completion().next() { + while budget.can_reap() { + let Some(cqe) = state.ring.completion().next() else { + break; + }; + // Count every consumed CQE, including cancels and test-only drops. + budget.completions += 1; let ud = cqe.user_data(); if ud & CANCEL_BIT != 0 { // Result of the AsyncCancel op itself; the read's own CQE @@ -2039,7 +2090,7 @@ fn drive( // short-read continuations this turn. Two bounded attempts per loop // preserve heartbeat pacing on EBUSY/EINTR/zero-progress submission; // fully idle calls skip the syscall (rustfs/backlog#2647). - submit_ring( + let submitted_after_reap = submit_ring( &mut state, &stats, &mut consecutive_submit_errors, @@ -2112,8 +2163,12 @@ fn drive( return; } } - // No pacing sleep: `wait_for_events` at the top of the loop blocks until - // the next CQE, message, or heartbeat (backlog#1102). + let ready_cqe = !state.ring.completion().is_empty(); + continue_without_wait = budget.continue_without_wait( + ready_cqe, + !state.backlog.is_empty() || !state.ring.submission().is_empty(), + submitted_before_reap.saturating_add(submitted_after_reap), + ); } } diff --git a/src/driver_loop_budget_tests.rs b/src/driver_loop_budget_tests.rs new file mode 100644 index 0000000..0c0f94a --- /dev/null +++ b/src/driver_loop_budget_tests.rs @@ -0,0 +1,113 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +fn readable(event: &EventFd) -> bool { + let mut fd = libc::pollfd { + fd: event.as_raw(), + events: libc::POLLIN, + revents: 0, + }; + // SAFETY: initialized pollfd, zero timeout, no pointer retained by kernel. + unsafe { libc::poll(&mut fd, 1, 0) > 0 } +} + +#[test] +fn drained_eventfd_does_not_strand_messages_beyond_intake_budget() { + let wake = EventFd::new().unwrap(); + let (tx, rx) = mpsc::channel(); + for id in 0..TURN_MESSAGES + 1 { + tx.send(id).unwrap(); + wake.signal(); + } + wake.drain(); + assert!(!readable(&wake)); + + let mut first = TurnBudget::default(); + while first.can_take_message() { + rx.try_recv().unwrap(); + first.messages += 1; + } + assert!(first.continue_without_wait(false, false, 0)); + // No additional producer or wake signal: the next turn consumes the tail. + assert_eq!(rx.try_recv().unwrap(), TURN_MESSAGES); + let second = TurnBudget { + messages: 1, + ..TurnBudget::default() + }; + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); + assert!(!second.continue_without_wait(false, false, 0)); +} + +#[test] +fn exact_message_boundary_costs_one_empty_turn_then_sleeps() { + let full = TurnBudget { + messages: TURN_MESSAGES, + ..TurnBudget::default() + }; + assert!(full.continue_without_wait(false, false, 0)); + assert!(!TurnBudget::default().continue_without_wait(false, false, 0)); +} + +#[test] +fn oversized_allocation_makes_progress_then_yields_to_reap() { + let mut budget = TurnBudget::default(); + assert!(budget.can_take_message()); + budget.messages += 1; + // Record one already-accepted read, larger than the fairness threshold. + // It is not rejected or repeatedly deferred by the turn budget. + budget.allocation_bytes = TURN_ALLOCATION_BYTES * 2; + assert!(!budget.can_take_message()); + assert!(budget.can_reap()); + assert!(budget.continue_without_wait(false, false, 0)); + assert!(TurnBudget::default().can_take_message()); +} + +#[test] +fn allocation_volume_yields_before_message_limit() { + let mut budget = TurnBudget::default(); + while budget.can_take_message() { + budget.messages += 1; + budget.allocation_bytes += TURN_ALLOCATION_BYTES / 4; + } + assert_eq!(budget.messages, 4); + assert!(budget.continue_without_wait(false, false, 0)); +} + +#[test] +fn completion_burst_yields_to_intake_and_continues_without_a_new_edge() { + let mut cq: VecDeque<_> = (0..TURN_COMPLETIONS + 1).collect(); + let mut budget = TurnBudget::default(); + while budget.can_reap() { + cq.pop_front().unwrap(); + budget.completions += 1; + } + assert_eq!(cq.len(), 1); + assert!(budget.continue_without_wait(true, false, 0)); + // A new turn starts with intake, so shutdown/cancel can be observed before + // the next chunk of completions rather than waiting for CQ to empty. + assert!(TurnBudget::default().can_take_message()); +} + +#[test] +fn failed_or_zero_progress_submissions_cannot_spin_from_queued_sqes_alone() { + let budget = TurnBudget::default(); + // submit_ring maps EBUSY/EINTR/other errors and Ok(0) to zero; each must + // return to the eventfd/heartbeat wait when there is no other ready work. + assert!(!budget.continue_without_wait(false, true, 0)); + // A partial positive submission allows a bounded immediate follow-up; if + // that next turn makes no progress, it waits again. + assert!(budget.continue_without_wait(false, true, 1)); + assert!(!budget.continue_without_wait(false, true, 0)); +} + +#[test] +fn completion_readiness_wins_even_when_last_submit_failed() { + assert!(TurnBudget::default().continue_without_wait(true, true, 0)); +} + +#[test] +fn submission_progress_without_remaining_work_does_not_busy_poll() { + assert!(!TurnBudget::default().continue_without_wait(false, false, 2)); +} From a65fb8a262b067a4949b98851b90f48397fa3d19 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 01:06:19 +0800 Subject: [PATCH 07/22] feat(uring): add opt-in capacity-aware shard admission Try actual count permits across healthy shards for positioned reads, retain round-robin by default and for streams, and keep the final owner fixed for waits and cancellation. Preserve shared-byte closure and account diagnostics only to the final shard. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- README.md | 28 +++++ docs/benchmarking.md | 11 +- src/admission_tests.rs | 1 + src/diagnostics.rs | 10 +- src/driver.rs | 110 +++++++++++++++++-- src/lib.rs | 2 +- src/shard_policy_tests.rs | 215 ++++++++++++++++++++++++++++++++++++++ tests/shard_policy.rs | 92 ++++++++++++++++ 8 files changed, 458 insertions(+), 11 deletions(-) create mode 100644 src/shard_policy_tests.rs create mode 100644 tests/shard_policy.rs diff --git a/README.md b/README.md index 1021670..a74b489 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,34 @@ assert_eq!(snapshot.delivered + snapshot.orphan_reclaimed, snapshot.submitted); - `read_current(file, len)` — `read(2)` semantics from the current position, for pipes and other non-seekable fds (a short read is a valid final result). - `probe_and_start_sharded(entries, shards)` — several independent rings per disk (each ring caps at one core's memory bandwidth for cache-hit reads); `probe_and_start(entries)` equals `..._sharded(entries, 1)`. - `probe_and_start_with_limits(entries, shards, ReadLimits { max_read_len, max_in_flight_bytes })` — optional logical read-size and driver-wide read-buffer limits. Both fields default to `None`, preserving existing constructor behavior. +- `with_shard_policy(ShardPolicy::CapacityAware)` — opt-in capacity-aware routing for positioned reads. Constructors keep `ShardPolicy::RoundRobin` by default. + +### Shard selection + +Round-robin binds each read to the next shard, even if that shard is busy or +closed. Capacity-aware selection starts at the same cursor and tries each shard's +count permit at most once, skipping closed count semaphores. It uses actual +permit acquisition, not a free-capacity snapshot. If all healthy shards are busy, +the handle waits on the first healthy candidate using Tokio's fair semaphore. +The wait is local to that shard; it does not rebalance when another shard frees. + +A shared byte-budget shortage waits on the first shard whose count permit was +available, returning that temporary count reservation before constructing the +waiter. A closed shared byte budget rejects admission globally. Once a read is +accepted or deferred, its read, wakeup, retry, and cancel retain the same owning +shard. `read_current` always keeps the original round-robin behavior; concurrent +stream reads still require caller serialization when ordering matters. + +Enable the policy explicitly on the constructed driver before sharing it: + +```rust,ignore +let driver = UringDriver::probe_and_start_sharded(128, 4)? + .with_shard_policy(rustfs_uring::ShardPolicy::CapacityAware); +``` + +This policy has additional admission work under contention. Throughput, CPU cost, +and tail-latency acceptance remain pending target-hardware measurements; it is +not enabled by default. ### Read allocation admission diff --git a/docs/benchmarking.md b/docs/benchmarking.md index 6373202..7119567 100644 --- a/docs/benchmarking.md +++ b/docs/benchmarking.md @@ -85,6 +85,15 @@ sample toward shard zero. Invalid requests can consume a sampling position without recording stages. Deterministic sampling is diagnostic, not an unbiased estimate for every possible periodic workload. +With opt-in `ShardPolicy::CapacityAware`, positioned reads choose their owner +before consuming that shard's sample sequence. Rejected geometry and unsuccessful +candidate attempts do not consume sample positions; closed admission records no +sample. For valid capacity-aware positioned reads, feature-on builds take one +timestamp before selection even for unsampled reads, so sampled admission covers +selection and permit acquisition (but excludes preceding geometry validation). +Round-robin and stream sampling retain their original behavior. Measure this +additional diagnostics cost with the selected routing policy. + `UringDriver::diagnostics()` aggregates shards; `UringDriver::shard_diagnostics()` preserves shard identity. Each stage has a count, total nanoseconds, and 64 log2 nanosecond buckets. Bucket zero covers @@ -117,7 +126,7 @@ even on std strategies (which do not use the instrumented driver). The default build compiles out timing fields, clock reads, histogram storage and sample allocations. Enabled builds add a per-shard atomic sampling counter per -handle and an Arc/timestamps/histogram updates for sampled operations. Measure +eligible handle and an Arc/timestamps/histogram updates for sampled operations. Measure that overhead on target hardware with separate feature-off/feature-on artifacts; do not assume it is free. Runtime schedule-latency and blocking-pool metrics must still be correlated in the application, which owns the Tokio runtime. diff --git a/src/admission_tests.rs b/src/admission_tests.rs index 47a14f2..4abd3c4 100644 --- a/src/admission_tests.rs +++ b/src/admission_tests.rs @@ -181,6 +181,7 @@ fn fake_driver(limits: ReadLimits) -> (UringDriver, mpsc::Receiver) { ( UringDriver { limits, + shard_policy: ShardPolicy::default(), byte_admission, shards: vec![Shard { tx, diff --git a/src/diagnostics.rs b/src/diagnostics.rs index 8c82834..b478e05 100644 --- a/src/diagnostics.rs +++ b/src/diagnostics.rs @@ -179,12 +179,20 @@ pub(crate) struct Trace { impl Trace { pub(crate) fn sample(diagnostics: &Arc) -> Option> { + Self::sample_with_start(diagnostics, Instant::now) + } + + pub(crate) fn sample_since(diagnostics: &Arc, started: Instant) -> Option> { + Self::sample_with_start(diagnostics, || started) + } + + fn sample_with_start(diagnostics: &Arc, start: impl FnOnce() -> Instant) -> Option> { // A global id modulo 64 would sample only shard zero for power-of-two // round-robin sharding. Each shard therefore owns its sampling sequence. let sequence = diagnostics.requests.fetch_add(1, Ordering::Relaxed); sequence.is_multiple_of(DIAGNOSTICS_SAMPLE_INTERVAL).then(|| { Arc::new(Self { - start: Instant::now(), + start: start(), diagnostics: Arc::clone(diagnostics), queued: AtomicU64::new(0), entered: AtomicU64::new(0), diff --git a/src/driver.rs b/src/driver.rs index 08cc86f..61157b8 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -32,6 +32,10 @@ use io_uring::{IoUring, opcode, types}; #[path = "admission_tests.rs"] mod admission_tests; +#[cfg(test)] +#[path = "shard_policy_tests.rs"] +mod shard_policy_tests; + #[cfg(feature = "diagnostics")] use crate::diagnostics::{Diagnostics, DiagnosticsSnapshot, Trace}; @@ -375,6 +379,20 @@ pub struct ReadLimits { pub max_in_flight_bytes: Option, } +/// Shard selection for positioned reads. Stream reads retain round-robin routing. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum ShardPolicy { + /// Bind to the next shard even when its admission is full or closed. + #[default] + RoundRobin, + /// Starting at the round-robin cursor, try each shard's count permit once, + /// skipping closed shards. If every healthy shard is full, wait fairly on + /// the first healthy shard. A shared byte-budget shortage waits on the first + /// shard with count capacity; a closed byte budget terminates admission. + /// Accepted and waiting reads never migrate to another shard. + CapacityAware, +} + struct ReadPermits { _count: OwnedSemaphorePermit, _bytes: Option, @@ -868,6 +886,7 @@ impl Drop for Shard { /// environments can fall back to a blocking backend before serving traffic. pub struct UringDriver { limits: ReadLimits, + shard_policy: ShardPolicy, byte_admission: Option>, /// One or more independent rings. A cache-hit buffered read completes inline /// inside `io_uring_enter`, so the thread driving a ring performs that @@ -943,6 +962,7 @@ impl UringDriver { } Ok(Self { limits, + shard_policy: ShardPolicy::default(), byte_admission, shards: started, next_id: AtomicU64::new(1), @@ -950,9 +970,69 @@ impl UringDriver { }) } + /// Set the policy for future positioned reads, independently of resource limits. + /// + /// Existing handles retain their owning shard. [`Self::read_current`] keeps + /// round-robin behavior under either policy; callers must still serialize + /// stream reads themselves when ordering matters. Capacity-aware selection + /// is opt-in and has not established a throughput or latency improvement. + #[must_use] + pub fn with_shard_policy(mut self, policy: ShardPolicy) -> Self { + self.shard_policy = policy; + self + } + + fn select_read_shard( + &self, + start: usize, + charge: u32, + capacity_aware: bool, + ) -> (&Shard, Result) { + let first = &self.shards[start]; + let bytes = self.byte_admission.as_ref().map(|admission| &admission.bytes); + if !capacity_aware { + return (first, try_read_permits(&first.sem, bytes, charge)); + } + if bytes.is_some_and(|sem| sem.is_closed()) { + return (first, Err(TryAcquireError::Closed)); + } + let mut waiting = None; + for index in (start..self.shards.len()).chain(0..start) { + let shard = &self.shards[index]; + match Arc::clone(&shard.sem).try_acquire_owned() { + Ok(count) => { + // A shared byte shortage cannot be solved on another shard. + // map drops count on either byte error, before async waiting. + let permits = bytes + .map(|sem| Arc::clone(sem).try_acquire_many_owned(charge)) + .transpose() + .map(|bytes| ReadPermits { + _count: count, + _bytes: bytes, + }); + return (shard, permits); + } + Err(TryAcquireError::NoPermits) => { + waiting.get_or_insert(shard); + } + Err(TryAcquireError::Closed) => {} + } + } + // Closure racing with the scan is terminal, even if an earlier count + // attempt saw NoPermits. Later closure wakes the deferred acquire future. + if bytes.is_some_and(|sem| sem.is_closed()) { + return (first, Err(TryAcquireError::Closed)); + } + match waiting { + Some(shard) => (shard, Err(TryAcquireError::NoPermits)), + None => (first, Err(TryAcquireError::Closed)), + } + } + /// Pick the shard for the next op. Round-robin spreads the inline-completion /// memcpy across driver threads; correctness does not depend on the choice, /// because the handle remembers which shard took the op. + #[cfg(feature = "fault-injection")] fn shard(&self) -> &Shard { let n = self.shards.len(); &self.shards[self.rr.fetch_add(1, Ordering::Relaxed) % n] @@ -1074,14 +1154,13 @@ impl UringDriver { assert_eq!(id & CANCEL_BIT, 0, "op id overflowed into the cancel bit"); let (done, rx) = oneshot::channel(); - // Bind the op to one shard for its whole life: the permit, the message, - // the wake, and any later cancel all go to this ring. The handle holds - // clones of that shard's `tx`/`wake_efd`, so nothing can route a cancel - // to a ring whose pending table does not hold the op. The rejection paths - // below return an `Inert` handle that never sends, but still need a `tx`. - let shard = self.shard(); + // Rejected requests retain the legacy round-robin cursor behavior. + // Valid capacity-aware reads choose their final owner after validation. + let start = self.rr.fetch_add(1, Ordering::Relaxed) % self.shards.len(); + let shard = &self.shards[start]; + let capacity_aware = self.shard_policy == ShardPolicy::CapacityAware && !allow_current_position; #[cfg(feature = "diagnostics")] - let timing = Trace::sample(&shard.stats.diagnostics); + let timing = (!capacity_aware).then(|| Trace::sample(&shard.stats.diagnostics)).flatten(); // `CURRENT_POSITION` is an internal sentinel used only by // `read_current`; accepting it through a positioned API would silently @@ -1208,11 +1287,26 @@ impl UringDriver { // fits u32, even on 32-bit Linux. Charge the exact Vec allocation length. let charge = allocation_bytes as u32; + #[cfg(feature = "diagnostics")] + let selection_started = capacity_aware.then(Instant::now); + let (shard, permits) = self.select_read_shard(start, charge, capacity_aware); + // Sample only the final owner, never the unsuccessful candidates. Default + // routing keeps its original sample sequence, including invalid requests. + #[cfg(feature = "diagnostics")] + let timing = match selection_started { + Some(started) if !matches!(&permits, Err(TryAcquireError::Closed)) => { + Trace::sample_since(&shard.stats.diagnostics, started) + } + Some(_) => None, + None => timing, + }; + // Take a backpressure permit BEFORE the op reaches the driver; it is // released only when the pending entry is dropped at the CQE (C10, // rustfs/backlog#1060). Acquisition never blocks the caller's thread // (rustfs/backlog#1102). - match try_read_permits(&shard.sem, self.byte_admission.as_ref().map(|admission| &admission.bytes), charge) { + // This owner is fixed for admission, message send, wake, and cancellation. + match permits { // Fast path: a permit was free, so submit eagerly — no allocation, // no await, and the op is in flight the moment `submit` returns, // exactly as with the previous blocking implementation. diff --git a/src/lib.rs b/src/lib.rs index 406f8fb..d714619 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -55,4 +55,4 @@ mod diagnostics; pub use diagnostics::{DIAGNOSTICS_SAMPLE_INTERVAL, DiagnosticsSnapshot, LatencyHistogram}; #[cfg(target_os = "linux")] -pub use driver::{ProbeFailure, ReadHandle, ReadLimits, StatsSnapshot, UringDriver}; +pub use driver::{ProbeFailure, ReadHandle, ReadLimits, ShardPolicy, StatsSnapshot, UringDriver}; diff --git a/src/shard_policy_tests.rs b/src/shard_policy_tests.rs new file mode 100644 index 0000000..689981c --- /dev/null +++ b/src/shard_policy_tests.rs @@ -0,0 +1,215 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +// No ring or driver thread: queued messages retain real admission permits, and +// tests observe the exact destination of reads and cancels without I/O races. +fn driver(limits: ReadLimits) -> (UringDriver, Vec>) { + let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + let mut receivers = Vec::new(); + let shards = (0..2) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(1)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().unwrap()), + } + }) + .collect(); + ( + UringDriver { + limits, + shard_policy: ShardPolicy::default(), + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) +} + +fn file() -> Arc { + Arc::new(File::open("/dev/zero").unwrap()) +} + +fn poll(handle: &mut ReadHandle) -> Poll>> { + Pin::new(handle).poll(&mut Context::from_waker(std::task::Waker::noop())) +} + +#[test] +fn capacity_aware_read_and_cancel_use_the_acquired_shard() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = driver.read_at(file(), 0, 8); + let id = handle.id; + let HandleState::Submitted { wake } = &handle.state else { panic!("free shard was not selected") }; + assert!(Arc::ptr_eq(wake, &driver.shards[1].wake_efd)); + let message = receivers[1].try_recv().unwrap(); + assert!(matches!(&message, Msg::Read { id: actual, .. } if *actual == id)); + drop(handle); + assert!(matches!(receivers[1].try_recv(), Ok(Msg::Cancel { id: actual }) if actual == id)); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(driver.shards[1].sem.available_permits(), 0, "cancel must not release accepted permits"); + drop(message); + drop(held); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); +} + +#[test] +fn default_and_stream_routing_still_wait_on_the_round_robin_shard() { + for stream in [false, true] { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = if stream { + driver.with_shard_policy(ShardPolicy::CapacityAware) + } else { + driver + }; + let _held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = if stream { + driver.read_current(file(), 8) + } else { + driver.read_at(file(), 0, 8) + }; + let HandleState::WaitingPermit { wake, .. } = &handle.state else { panic!("legacy route changed") }; + assert!(Arc::ptr_eq(wake, &driver.shards[0].wake_efd)); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); + assert_eq!(driver.shards[1].sem.available_permits(), 1); + drop(handle); + } +} + +#[test] +fn capacity_aware_skips_closed_shards_but_default_does_not() { + for policy in [ShardPolicy::RoundRobin, ShardPolicy::CapacityAware] { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(policy); + driver.shards[0].sem.close(); + let mut handle = driver.read_at(file(), 0, 8); + if policy == ShardPolicy::CapacityAware { + assert!(matches!(receivers[1].try_recv(), Ok(Msg::Read { .. }))); + } else { + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); + } + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + } +} + +#[test] +fn capacity_aware_rejects_when_every_shard_is_closed() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + for shard in &driver.shards { + shard.sem.close(); + } + let mut handle = driver.read_at(file(), 0, 8); + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[test] +fn capacity_aware_deferred_owner_remains_fixed_when_another_shard_frees() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let first = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let second = Arc::clone(&driver.shards[1].sem).try_acquire_owned().unwrap(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(poll(&mut handle).is_pending()); + drop(second); + assert!(poll(&mut handle).is_pending()); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); + drop(first); + assert!(poll(&mut handle).is_pending()); // submitted, awaiting its CQE + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { .. }))); + drop(handle); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Cancel { .. }))); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); +} + +#[test] +fn capacity_aware_deferred_waiters_keep_semaphore_queue_order() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let _other = Arc::clone(&driver.shards[1].sem).try_acquire_owned().unwrap(); + let mut first = driver.read_at(file(), 0, 8); + driver.rr.store(0, Ordering::Relaxed); + let mut second = driver.read_at(file(), 8, 8); + assert!(poll(&mut first).is_pending()); + assert!(poll(&mut second).is_pending()); + drop(held); + assert!(poll(&mut second).is_pending(), "later waiter must not overtake the first"); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert!(poll(&mut first).is_pending()); + let message = receivers[0].try_recv().unwrap(); + assert!(matches!(&message, Msg::Read { id, .. } if *id == first.id)); + assert!(poll(&mut second).is_pending()); + drop(message); + assert!(poll(&mut second).is_pending()); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { id, .. }) if id == second.id)); +} + +#[test] +fn byte_shortage_releases_failed_count_reservation_and_waiter_drop_releases_partial_admission() { + let (driver, receivers) = driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(8), + }); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let admission = driver.byte_admission.as_ref().unwrap(); + let bytes = Arc::clone(&admission.bytes).try_acquire_many_owned(8).unwrap(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert!(poll(&mut handle).is_pending()); + assert_eq!(driver.shards[0].sem.available_permits(), 0); + assert_eq!(driver.shards[1].sem.available_permits(), 1); + drop(handle); + drop(bytes); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert_eq!(admission.bytes.available_permits(), 8); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[test] +fn closed_byte_admission_is_terminal_even_with_available_count_permits() { + let (driver, receivers) = driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(8), + }); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + // Close just bytes to exercise the race before registry count closure. + driver.byte_admission.as_ref().unwrap().bytes.close(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[cfg(feature = "diagnostics")] +#[test] +fn capacity_aware_diagnostics_sample_only_the_final_owner() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = driver.read_at(file(), 0, 8); // rr 0, accepted on 1 + assert_eq!(driver.shard_diagnostics()[0].admission.count, 0); + assert_eq!(driver.shard_diagnostics()[1].admission.count, 1); + drop(receivers[1].try_recv().unwrap()); + drop(handle); + drop(held); + // The failed candidate did not consume shard 0's first sample position. + driver.rr.store(0, Ordering::Relaxed); + let _handle = driver.read_at(file(), 0, 8); + assert_eq!(driver.shard_diagnostics()[0].admission.count, 1); +} diff --git a/tests/shard_policy.rs b/tests/shard_policy.rs new file mode 100644 index 0000000..732cce8 --- /dev/null +++ b/tests/shard_policy.rs @@ -0,0 +1,92 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::os::fd::FromRawFd; +use std::sync::Arc; +use std::time::Duration; + +use rustfs_uring::{ShardPolicy, UringDriver}; + +async fn exercise_busy_shard(policy: ShardPolicy) { + let driver = match UringDriver::probe_and_start_sharded(1, 2) { + Ok(driver) => driver.with_shard_policy(policy), + Err(error) => { + assert!(error.is_expected_restriction(), "unexpected setup error: {error}"); + eprintln!("SKIP shard_policy {policy:?}: {error}"); + return; + } + }; + let mut fds = [0; 2]; + // SAFETY: the array holds two fds; each successful fd is owned by one File. + assert_eq!(unsafe { libc::pipe(fds.as_mut_ptr()) }, 0); + let pipe_read = Arc::new(unsafe { File::from_raw_fd(fds[0]) }); + let pipe_write = unsafe { File::from_raw_fd(fds[1]) }; + let held = driver.read_current(pipe_read, 8); // shard 0 stays blocked + let path = std::env::temp_dir().join(format!("uring-shard-policy-{policy:?}-{}.bin", std::process::id())); + std::fs::write(&path, [42; 8]).unwrap(); + let file = Arc::new(File::open(&path).unwrap()); + // Advance the cursor over shard 1 without occupying its permit. Waiting for + // stats.in_flight after a real read would race that read's permit destructor. + assert_eq!( + driver.read_at(Arc::clone(&file), u64::MAX, 8).await.unwrap_err().kind(), + std::io::ErrorKind::InvalidInput + ); + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().in_flight != 1 { + tokio::task::yield_now().await; + } + }) + .await + .expect("only the blocked pipe should remain"); + + let mut read = driver.read_at(file, 0, 8); // round-robin starts at busy shard 0 + match policy { + ShardPolicy::CapacityAware => { + assert_eq!( + tokio::time::timeout(Duration::from_secs(2), &mut read) + .await + .expect("must use the idle shard") + .unwrap(), + [42; 8] + ); + // The pipe writer remains open and empty throughout this read. + drop(held); + } + ShardPolicy::RoundRobin => { + assert!(tokio::time::timeout(Duration::from_millis(100), &mut read).await.is_err()); + drop(held); + assert_eq!( + tokio::time::timeout(Duration::from_secs(2), read) + .await + .expect("cancellation should release the original shard") + .unwrap(), + [42; 8] + ); + } + } + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().in_flight != 0 { + tokio::task::yield_now().await; + } + }) + .await + .expect("reads should drain after cancellation"); + let snapshot = driver.shutdown(); + assert_eq!(snapshot.submitted, 2); + assert_eq!(snapshot.delivered, 1); + assert_eq!(snapshot.orphan_reclaimed, 1); + drop(pipe_write); + std::fs::remove_file(path).unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn capacity_aware_bypasses_a_shard_saturated_by_a_blocked_pipe() { + exercise_busy_shard(ShardPolicy::CapacityAware).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn round_robin_still_waits_for_its_original_busy_shard() { + exercise_busy_shard(ShardPolicy::RoundRobin).await; +} From bb20f6048e2f54c751bc3f56f1120511f5f2b5c3 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 01:12:15 +0800 Subject: [PATCH 08/22] test(admission): cover byte-budget bounded-drain bailout Exercise a real pending pipe read with count-stage and byte-stage waiters under the dropped-CQE fault seam. Assert bounded bailout, retained Pending ownership, no waiter submission, and errors for all stranded handles. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- tests/fault_injection.rs | 75 +++++++++++++++++++++++++++++++++++++++- 1 file changed, 74 insertions(+), 1 deletion(-) diff --git a/tests/fault_injection.rs b/tests/fault_injection.rs index 3500d2c..aaea7fa 100644 --- a/tests/fault_injection.rs +++ b/tests/fault_injection.rs @@ -32,7 +32,7 @@ use std::os::unix::process::ExitStatusExt; use std::sync::Arc; use std::time::{Duration, Instant}; -use rustfs_uring::UringDriver; +use rustfs_uring::{ReadLimits, UringDriver}; /// A pipe whose read side blocks until the write side is written or closed — the /// only portable way to hold an op provably in flight. @@ -247,6 +247,79 @@ fn stranded_handle_errors_after_bounded_drain_bailout() { drop(pipe_write); } +/// #2647: exercise the real Pending ownership path with both admission stages +/// waiting. This seam discards a real EOF/cancel CQE; it models a lost completion +/// for bounded-drain testing, not an actual hung kernel read or storage device. +#[test] +fn byte_budget_bailout_leaks_pending_and_errors_both_admission_waiters() { + // SAFETY: this test binary is run with --test-threads=1. Configure the seam + // before driver creation; remove it only after every driver thread joins. + unsafe { + std::env::set_var("RUSTFS_URING_FAULT_STUCK_DRAIN", "1"); + std::env::set_var("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400"); + } + let driver = match UringDriver::probe_and_start_with_limits( + 1, + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) { + Ok(driver) => driver, + Err(error) => { + clear_stuck_env(); + assert!(error.is_expected_restriction(), "unexpected probe failure: {error}"); + eprintln!("SKIP byte_budget_bailout: restricted environment ({error})"); + return; + } + }; + + let (pipe_read, pipe_write) = os_pipe(); + let leaked_file = Arc::downgrade(&pipe_read); + // Round-robin request 0 holds shard 0's sole count permit and all bytes. + let stranded = driver.read_current(Arc::clone(&pipe_read), 8).without_cancel_on_drop(); + assert!(wait_until(Duration::from_secs(2), || driver.stats().in_flight == 1)); + let waiting_file = Arc::new(File::open("/dev/zero").unwrap()); + // Request 1 acquires shard 1's count but waits on shared bytes; request 2 + // waits on shard 0's count before it can reach the byte semaphore. + let mut byte_waiter = driver.read_at(Arc::clone(&waiting_file), 0, 8); + let mut count_waiter = driver.read_at(waiting_file, 0, 8); + let mut cx = std::task::Context::from_waker(std::task::Waker::noop()); + assert!(std::pin::Pin::new(&mut byte_waiter).poll(&mut cx).is_pending()); + assert!(std::pin::Pin::new(&mut count_waiter).poll(&mut cx).is_pending()); + assert_eq!(driver.stats().submitted, 1, "waiting reads must not allocate or submit"); + + // EOF makes the real pipe read finite. The seam deliberately suppresses its + // completion bookkeeping so the Pending and its reservations must be leaked. + drop(pipe_write); + drop(pipe_read); + let start = Instant::now(); + let stats = driver.shutdown(); + let elapsed = start.elapsed(); + clear_stuck_env(); + assert!(elapsed >= Duration::from_millis(300), "bailout path was not taken: {elapsed:?}"); + assert!(elapsed < Duration::from_secs(3), "bounded drain exceeded its deadline: {elapsed:?}"); + assert_eq!(stats.in_flight, 1, "Pending reservation must remain retained after bailout"); + assert_eq!(stats.submitted, 1, "neither admission waiter may reach the driver"); + assert_eq!(stats.delivered, 0, "discarded EOF CQE must not report successful delivery"); + + let rt = tokio::runtime::Builder::new_current_thread().enable_time().build().unwrap(); + rt.block_on(async { + for (name, handle) in [ + ("stranded", stranded), + ("byte-stage", byte_waiter), + ("count-stage", count_waiter), + ] { + let result = tokio::time::timeout(Duration::from_secs(2), handle).await; + assert!(matches!(result, Ok(Err(_))), "{name} handle must fail after bailout, got {result:?}"); + } + }); + // All caller references and handles are gone. This sole remaining strong + // reference belongs to the leaked Pending (whose ReadPermits stay with it). + assert_eq!(leaked_file.strong_count(), 1, "Pending's owned file must survive the bailout"); +} + fn clear_stuck_env() { // SAFETY: `--test-threads=1` teardown; no concurrent environment readers. unsafe { From 782b6ea65ac35e0d33b60ae8b5b175596c351b58 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 01:44:02 +0800 Subject: [PATCH 09/22] test(driver): cover submit failures and suppress repeated overflow warnings Extract submit-result classification for deterministic progress, reset, shutdown and ownership checks without injecting real kernel failures. Warn only on changed nonzero CQ overflow counters and explicitly handle counter wrap. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/fault-recovery.md | 15 +++ src/driver.rs | 45 ++++++-- src/driver_fault_recovery_tests.rs | 32 ++++++ src/driver_submit_result_tests.rs | 158 +++++++++++++++++++++++++++++ 4 files changed, 244 insertions(+), 6 deletions(-) create mode 100644 src/driver_submit_result_tests.rs diff --git a/docs/fault-recovery.md b/docs/fault-recovery.md index 879d4d9..5b8266c 100644 --- a/docs/fault-recovery.md +++ b/docs/fault-recovery.md @@ -11,6 +11,12 @@ zero progress, EINTR and EBUSY do not start an unbounded retry loop. The next event or heartbeat drives further progress. Pending buffers, file descriptors and admission permits retain their existing final-CQE ownership rules. +CQ-overflow warnings are emitted when the observed kernel counter changes to a +nonzero value, not on every turn while the cumulative counter remains nonzero. +The snapshot continues to report the kernel counter, including its u32 wrap: +a wrap to zero updates the snapshot silently, and a later nonzero value warns +again. This prevents repeated log output without changing completion recovery. + A non-block-aligned O_DIRECT short read that does not cover the logical range cannot be resumed from its unaligned endpoint. The driver checks file metadata: confirmed EOF returns the initialized logical prefix; a mid-file endpoint @@ -24,6 +30,7 @@ On Linux, run: ```sh cargo test --all-features --lib fault_recovery_tests -- --nocapture +cargo test --all-features --lib submit_result_tests -- --nocapture ``` The metadata-result tests inject a completed prefix and metadata outcomes into @@ -33,6 +40,14 @@ drains it, and verifies the empty-SQ submission path recovers the third CQE. It uses actual kernel CQEs but no user buffers. The backlog test covers SQ-capacity backpressure, not forced partial acceptance by `io_uring_enter`. +Submit-result tests inject `io::Result` into the production classification +helper: partial positive acceptance, zero progress, EINTR/EBUSY and persistent +nontransient errors. They verify counter reset, the exact shutdown threshold, +one shutdown transition, deduplicated cancels and retained pending buffer/FD/ +count-permit/byte-permit ownership. They also check the loop's retry decision. +This models syscall return handling without inducing a partial return or failure +from a real kernel; it is not native kernel partial-submission evidence. + Kernel tests print `SKIP` when setup returns an expected restriction error; such a run is not evidence of overflow recovery. Run on an unrestricted Linux host for kernel coverage. Advanced taskrun ring modes remain disabled by default diff --git a/src/driver.rs b/src/driver.rs index 61157b8..bb8cb7c 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -1848,6 +1848,26 @@ fn submit_ring( let Some(result) = submit_if_needed(&mut state.ring) else { return 0; }; + handle_submit_result(result, stats, consecutive_submit_errors, submit_error_logged, shutting_down, || { + let ids: Vec = state.pending.keys().copied().collect(); + for id in ids { + queue_cancel(&mut state.backlog, queued_cancels, id); + } + }) +} + +/// Classify a completed submit syscall without inferring which buffers the +/// kernel owns. The callback only queues cancellation on the first shutdown +/// transition; it never reclaims pending reads. Tests inject syscall results at +/// this boundary, not into a real kernel ring. +fn handle_submit_result( + result: io::Result, + stats: &DriverStats, + consecutive_submit_errors: &mut u32, + submit_error_logged: &mut bool, + shutting_down: &mut bool, + on_shutdown: impl FnOnce(), +) -> usize { match result { Ok(submitted) => { *consecutive_submit_errors = 0; @@ -1873,16 +1893,29 @@ fn submit_ring( "uring driver: consecutive submit failures; shutting down so callers fall back to the std backend" ); *shutting_down = true; - let ids: Vec = state.pending.keys().copied().collect(); - for id in ids { - queue_cancel(&mut state.backlog, queued_cancels, id); - } + on_shutdown(); } } } 0 } +#[cfg(test)] +#[path = "driver_submit_result_tests.rs"] +mod submit_result_tests; + +/// Publish changes to the kernel's cumulative u32 counter. Warn only for a new +/// nonzero observation; a wrap to zero updates the snapshot and rearms logging +/// for the next increment without emitting a misleading zero-overflow warning. +fn update_cq_overflow(stats: &DriverStats, previous: &mut u32, observed: u32) -> bool { + if observed == *previous { + return false; + } + *previous = observed; + stats.cq_overflow.store(u64::from(observed), Ordering::SeqCst); + observed != 0 +} + fn drive( ring: IoUring, rx: mpsc::Receiver, @@ -1902,6 +1935,7 @@ fn drive( // the persistent-submit-failure escape hatch (rustfs/backlog#1162). let mut consecutive_submit_errors: u32 = 0; let mut submit_error_logged = false; + let mut previous_cq_overflow = 0; // Ids with an AsyncCancel already queued, so a drop-cancel followed by a // shutdown (or vice versa) does not enqueue a second cancel for the same op — // keeping total completions <= 2*entries and CQ overflow unreachable @@ -2200,8 +2234,7 @@ fn drive( // `entries` and cancels are deduped (at most one per op), keeping total // completions <= 2*entries, so this should stay 0 in practice. let overflow = state.ring.completion().overflow(); - if overflow != 0 { - stats.cq_overflow.store(overflow as u64, Ordering::SeqCst); + if update_cq_overflow(&stats, &mut previous_cq_overflow, overflow) { tracing::warn!( overflow, "uring driver: CQ overflow; CQEs buffered (NODROP), not lost — backpressure warning" diff --git a/src/driver_fault_recovery_tests.rs b/src/driver_fault_recovery_tests.rs index e938f04..ff3123d 100644 --- a/src/driver_fault_recovery_tests.rs +++ b/src/driver_fault_recovery_tests.rs @@ -3,6 +3,38 @@ use super::*; +#[test] +fn unchanged_cumulative_overflow_warns_only_once() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(!update_cq_overflow(&stats, &mut previous, 0)); + assert!(update_cq_overflow(&stats, &mut previous, 7)); + assert!(!update_cq_overflow(&stats, &mut previous, 7)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 7); + assert!(update_cq_overflow(&stats, &mut previous, 8)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 8); +} + +#[test] +fn overflow_counter_wrap_to_zero_updates_snapshot_and_rearms_warning() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(update_cq_overflow(&stats, &mut previous, u32::MAX)); + assert!(!update_cq_overflow(&stats, &mut previous, 0)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 0); + assert!(update_cq_overflow(&stats, &mut previous, 1)); + assert!(!update_cq_overflow(&stats, &mut previous, 1)); +} + +#[test] +fn overflow_counter_wrap_to_nonzero_warns_for_new_observation() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(update_cq_overflow(&stats, &mut previous, u32::MAX)); + assert!(update_cq_overflow(&stats, &mut previous, 2)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 2); +} + fn direct_prefix() -> (Pending, oneshot::Receiver>>) { let (done, receiver) = oneshot::channel(); let pending = Pending { diff --git a/src/driver_submit_result_tests.rs b/src/driver_submit_result_tests.rs new file mode 100644 index 0000000..71de099 --- /dev/null +++ b/src/driver_submit_result_tests.rs @@ -0,0 +1,158 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +#[derive(Default)] +struct ResultHarness { + stats: DriverStats, + consecutive_errors: u32, + logged: bool, + shutting_down: bool, + shutdown_calls: usize, + pending: HashMap, + backlog: VecDeque, + queued_cancels: HashSet, +} + +impl ResultHarness { + fn apply(&mut self, result: io::Result) -> usize { + handle_submit_result( + result, + &self.stats, + &mut self.consecutive_errors, + &mut self.logged, + &mut self.shutting_down, + || { + self.shutdown_calls += 1; + for id in self.pending.keys() { + queue_cancel(&mut self.backlog, &mut self.queued_cancels, *id); + } + }, + ) + } + + fn fail(&mut self, errno: i32) -> usize { + self.apply(Err(io::Error::from_raw_os_error(errno))) + } +} + +fn pending_read(file: Arc, count: &Arc, bytes: &Arc) -> Pending { + Pending { + #[cfg(feature = "diagnostics")] + timing: None, + buf: vec![42; 16], + file, + done: None, + offset: 0, + nread: 0, + _permit: ReadPermits { + _count: Arc::clone(count).try_acquire_owned().unwrap(), + _bytes: Some(Arc::clone(bytes).try_acquire_many_owned(16).unwrap()), + }, + pad: 0, + head: 0, + want: 16, + region_len: 16, + align: 1, + transient_retries: 0, + cancel_requested: false, + } +} + +#[test] +fn partial_positive_acceptance_resets_errors_and_allows_bounded_retry() { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.apply(Ok(2)); // Model 2 accepted out of a larger SQ. + assert_eq!(submitted, 2); + assert_eq!(harness.consecutive_errors, 0); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); + assert!(TurnBudget::default().continue_without_wait(false, true, submitted)); + assert_eq!(harness.shutdown_calls, 0); +} + +#[test] +fn zero_acceptance_resets_error_streak_but_waits_with_remaining_sqes() { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.apply(Ok(0)); + assert_eq!(harness.consecutive_errors, 0); + assert!(!TurnBudget::default().continue_without_wait(false, true, submitted)); + assert_eq!(harness.shutdown_calls, 0); +} + +#[test] +fn interrupted_or_busy_submission_resets_streak_without_counting_or_spinning() { + for errno in [libc::EINTR, libc::EBUSY] { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.fail(errno); + assert_eq!(harness.consecutive_errors, 0); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); + assert!(!TurnBudget::default().continue_without_wait(false, true, submitted)); + assert!(!harness.shutting_down); + } +} + +#[test] +fn persistent_failures_shutdown_once_and_keep_every_pending_resource() { + let mut harness = ResultHarness::default(); + let file = Arc::new(File::open("/dev/null").unwrap()); + let count = Arc::new(Semaphore::new(2)); + let bytes = Arc::new(Semaphore::new(32)); + harness.pending.insert(1, pending_read(Arc::clone(&file), &count, &bytes)); + harness.pending.insert(2, pending_read(Arc::clone(&file), &count, &bytes)); + let ptrs: Vec<_> = (1..=2).map(|id| harness.pending[&id].buf.as_ptr()).collect(); + // Model an already queued drop-cancel; shutdown must not duplicate it. + queue_cancel(&mut harness.backlog, &mut harness.queued_cancels, 1); + + for attempt in 1..=MAX_CONSECUTIVE_SUBMIT_ERRORS + 2 { + assert_eq!(harness.fail(libc::EPERM), 0); + assert_eq!(harness.consecutive_errors, attempt); + assert_eq!(harness.shutting_down, attempt >= MAX_CONSECUTIVE_SUBMIT_ERRORS); + assert_eq!(harness.shutdown_calls, usize::from(attempt >= MAX_CONSECUTIVE_SUBMIT_ERRORS)); + assert_eq!(harness.pending.len(), 2); + for id in 1..=2 { + assert_eq!(harness.pending[&id].buf.as_ptr(), ptrs[(id - 1) as usize]); + assert_eq!(harness.pending[&id].buf, [42; 16]); + } + assert_eq!(count.available_permits(), 0); + assert_eq!(bytes.available_permits(), 0); + assert_eq!(Arc::strong_count(&file), 3); + } + assert!(harness.logged); + assert_eq!(harness.backlog.len(), 2); + assert_eq!(harness.queued_cancels, HashSet::from([1, 2])); + assert_eq!( + harness.stats.submit_errors.load(Ordering::SeqCst), + u64::from(MAX_CONSECUTIVE_SUBMIT_ERRORS + 2) + ); + // No SQEs were submitted to a kernel in this model, so dropping is safe. + drop(harness); + assert_eq!(count.available_permits(), 2); + assert_eq!(bytes.available_permits(), 32); + assert_eq!(Arc::strong_count(&file), 1); +} + +#[test] +fn success_and_transient_errors_break_the_consecutive_shutdown_threshold() { + for reset in [Ok(1), Ok(0), Err(libc::EINTR), Err(libc::EBUSY)] { + let mut harness = ResultHarness::default(); + for _ in 1..MAX_CONSECUTIVE_SUBMIT_ERRORS { + harness.fail(libc::EIO); + } + harness.apply(reset.map_err(io::Error::from_raw_os_error)); + harness.fail(libc::EIO); + assert_eq!(harness.consecutive_errors, 1); + assert_eq!(harness.shutdown_calls, 0); + } +} + +#[test] +fn eagain_is_counted_as_a_bounded_submit_failure() { + let mut harness = ResultHarness::default(); + assert_eq!(harness.fail(libc::EAGAIN), 0); + assert_eq!(harness.consecutive_errors, 1); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); +} From 7e2f1aa6a007d3cde5cf11e8c7b51d7fee44ba4d Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 02:15:57 +0800 Subject: [PATCH 10/22] feat(driver): add bounded buffered read batches with shared notifications Reuse normal read validation, ownership and admission while emitting one eager wakeup per final owning shard. Deferred admissions and cancellation retain individual notifications. Add deterministic eventfd tests and native byte-exact/cancellation coverage. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- README.md | 4 + docs/batch-reads.md | 34 +++++++ src/batch_read_tests.rs | 192 ++++++++++++++++++++++++++++++++++++++++ src/driver.rs | 80 +++++++++++++++-- src/lib.rs | 2 +- tests/cancel.rs | 72 ++++++++++++++- 6 files changed, 375 insertions(+), 9 deletions(-) create mode 100644 docs/batch-reads.md create mode 100644 src/batch_read_tests.rs diff --git a/README.md b/README.md index a74b489..8ca1b5a 100644 --- a/README.md +++ b/README.md @@ -161,6 +161,10 @@ cargo test -- --nocapture --test-threads=1 The harness fails on a non-degrading leg 1 or a vacuous-pass leg 2, so a skipped suite can never masquerade as coverage. The cancel-safety contract is pinned by the acceptance tests in `tests/cancel.rs`; the `fault-injection` feature (test-only) drives the panic-abort, bounded-drain-leak, and probe-failure escape hatches in `tests/fault_injection.rs`. +For bounded buffered read groups, see [explicit batch reads](docs/batch-reads.md). +`read_at_batch` shares eager notifications per owning shard while keeping each +read's admission, result and cancellation independent. + ## License Apache-2.0. See [LICENSE](LICENSE). diff --git a/docs/batch-reads.md b/docs/batch-reads.md new file mode 100644 index 0000000..0ee177b --- /dev/null +++ b/docs/batch-reads.md @@ -0,0 +1,34 @@ +# Explicit buffered read batches + +`UringDriver::read_at_batch(Vec)` accepts up to `MAX_BATCH_READS` +(64) buffered positioned reads and returns one `ReadHandle` per input, in input +order. More than 64 requests fails before any submission; an empty batch sends +no wakeup. Invalid individual offsets or lengths become normal asynchronous +handle errors and do not reject other requests. + +Eagerly admitted reads share one eventfd notification per distinct final owning +shard after handle construction. The capacity-aware policy, when selected, can +route several inputs to the same owner, which still receives one notification. +Requests awaiting count or byte admission signal their owner individually when +polled and admitted. No tasks are spawned and existing single-read, direct-read +and stream APIs retain their notification behavior. + +This is notification batching, not an atomic multi-read operation or snapshot. +Kernel submission/completion order may differ from input order; each handle +keeps independent cancellation and result ownership. Construction performs at +most 64 submissions and at most 64-by-64 pointer identity comparisons for wake +deduplication. It does not await capacity. If construction unwinds, already built +handles are dropped, queue their normal cancels, and wake the owning shards. +Driver-owned buffers, FDs and admission permits still survive until final CQE. + +Unit tests count real eventfd notifications using threadless driver queues, and +cover rejection, mixed valid/invalid requests, final-owner routing, deferred byte +admission, dropping handles and interrupted construction. Native integration +tests verify byte-exact results and cancellation/drain conservation. Threadless +tests model CQE resource release without submitting to a kernel. Notification +reduction is verified deterministically; throughput and latency are unmeasured. + +```sh +cargo test --all-features --lib batch_read_tests -- --nocapture +cargo test --all-features --test cancel batch_ -- --nocapture +``` diff --git a/src/batch_read_tests.rs b/src/batch_read_tests.rs new file mode 100644 index 0000000..0796c2e --- /dev/null +++ b/src/batch_read_tests.rs @@ -0,0 +1,192 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +// Threadless driver: message queues own real permits, and eventfd counters let +// tests count notifications without races with a running consumer. +fn driver(shard_count: usize, capacity: usize, bytes: Option) -> (UringDriver, Vec>) { + let byte_admission = bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + let mut receivers = Vec::new(); + let shards = (0..shard_count) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(capacity)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().unwrap()), + } + }) + .collect(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: bytes, + }, + shard_policy: ShardPolicy::default(), + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) +} + +fn requests(count: usize) -> Vec { + let file = Arc::new(File::open("/dev/zero").unwrap()); + (0..count) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: index as u64, + len: 8, + }) + .collect() +} + +fn notifications(event: &EventFd) -> u64 { + let mut count = 0u64; + // SAFETY: read writes one u64 into valid memory; eventfd is nonblocking. + let result = unsafe { libc::read(event.as_raw(), (&mut count as *mut u64).cast(), 8) }; + if result == -1 { + assert_eq!(io::Error::last_os_error().raw_os_error(), Some(libc::EAGAIN)); + return 0; + } + assert_eq!(result, 8); + count +} + +fn poll(handle: &mut ReadHandle) -> Poll>> { + Pin::new(handle).poll(&mut Context::from_waker(std::task::Waker::noop())) +} + +#[test] +fn batch_signals_once_per_owner_while_single_reads_keep_individual_signals() { + let (driver, receivers) = driver(2, 64, None); + let handles = driver.read_at_batch(requests(MAX_BATCH_READS)).unwrap(); + assert_eq!(handles.len(), MAX_BATCH_READS); + for (shard, receiver) in driver.shards.iter().zip(&receivers) { + assert_eq!(notifications(&shard.wake_efd), 1); + assert_eq!(receiver.try_iter().count(), MAX_BATCH_READS / 2); + } + let single: Vec<_> = requests(4) + .into_iter() + .map(|r| driver.read_at(r.file, r.offset, r.len)) + .collect(); + for shard in &driver.shards { + assert_eq!(notifications(&shard.wake_efd), 2); + } + drop(single); + drop(handles); +} + +#[test] +fn oversized_batch_rejects_every_request_before_submission() { + let (driver, receivers) = driver(1, 64, None); + let error = driver.read_at_batch(requests(MAX_BATCH_READS + 1)).err().unwrap(); + assert_eq!(error.kind(), io::ErrorKind::InvalidInput); + assert_eq!(driver.next_id.load(Ordering::Relaxed), 1); + assert_eq!(driver.rr.load(Ordering::Relaxed), 0); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); +} + +#[test] +fn empty_and_invalid_requests_do_not_notify_or_reject_valid_members() { + let (driver, receivers) = driver(1, 4, None); + assert!(driver.read_at_batch(Vec::new()).unwrap().is_empty()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + let mut invalid = requests(1); + invalid[0].offset = u64::MAX; + let mut invalid_handles = driver.read_at_batch(invalid).unwrap(); + assert!(matches!(poll(&mut invalid_handles[0]), Poll::Ready(Err(_)))); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + let mut reads = requests(2); + reads[0].offset = u64::MAX; + let mut handles = driver.read_at_batch(reads).unwrap(); + assert!(matches!(poll(&mut handles[0]), Poll::Ready(Err(e)) if e.kind() == io::ErrorKind::InvalidInput)); + assert!(matches!(&handles[1].state, HandleState::Submitted { .. })); + assert_eq!(receivers[0].try_iter().count(), 1); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); +} + +#[test] +fn capacity_aware_batch_notifies_final_owner_only() { + let (driver, receivers) = driver(2, 4, None); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let _held = Arc::clone(&driver.shards[0].sem).try_acquire_many_owned(4).unwrap(); + let handles = driver.read_at_batch(requests(4)).unwrap(); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + assert_eq!(notifications(&driver.shards[1].wake_efd), 1); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(receivers[1].try_iter().count(), 4); + for handle in &handles { + let HandleState::Submitted { wake } = &handle.state else { panic!("expected eager admission") }; + assert!(Arc::ptr_eq(wake, &driver.shards[1].wake_efd)); + } +} + +#[test] +fn deferred_batch_member_notifies_when_polled_after_byte_permit_returns() { + let (driver, receivers) = driver(1, 2, Some(8)); + let mut handles = driver.read_at_batch(requests(2)).unwrap(); + assert!(matches!(&handles[1].state, HandleState::WaitingPermit { .. })); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + assert!(poll(&mut handles[1]).is_pending()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + let accepted = receivers[0].try_recv().unwrap(); + assert!(matches!(&accepted, Msg::Read { .. })); + drop(accepted); // Simulate completed ownership release; no kernel used. + assert!(poll(&mut handles[1]).is_pending()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { .. }))); +} + +#[test] +fn dropping_batch_routes_cancels_and_keeps_accepted_permits_owned() { + let (driver, receivers) = driver(1, 2, Some(16)); + let handles = driver.read_at_batch(requests(2)).unwrap(); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + let accepted: Vec<_> = receivers[0].try_iter().collect(); + let ids: Vec<_> = handles.iter().map(|handle| handle.id).collect(); + drop(handles); + assert_eq!(notifications(&driver.shards[0].wake_efd), 2); + let cancelled: Vec<_> = receivers[0] + .try_iter() + .map(|msg| match msg { + Msg::Cancel { id } => id, + _ => panic!("expected cancel"), + }) + .collect(); + assert_eq!(cancelled, ids); + assert_eq!(driver.shards[0].sem.available_permits(), 0); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 0); + drop(accepted); + assert_eq!(driver.shards[0].sem.available_permits(), 2); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 16); +} + +#[test] +fn interrupted_construction_drops_handles_and_wakes_previously_queued_reads() { + let (driver, receivers) = driver(1, 2, None); + // Force the existing id-overflow assertion on the second batch member. + driver.next_id.store(CANCEL_BIT - 1, Ordering::Relaxed); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| driver.read_at_batch(requests(2)))); + assert!(result.is_err()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1, "unwind cancellation wakes the owner"); + let accepted = receivers[0].try_recv().unwrap(); + assert!(matches!(&accepted, Msg::Read { id, .. } if *id == CANCEL_BIT - 1)); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Cancel { id }) if id == CANCEL_BIT - 1)); + assert_eq!(driver.shards[0].sem.available_permits(), 1); + drop(accepted); + assert_eq!(driver.shards[0].sem.available_permits(), 2); +} diff --git a/src/driver.rs b/src/driver.rs index bb8cb7c..1ad97be 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -36,6 +36,10 @@ mod admission_tests; #[path = "shard_policy_tests.rs"] mod shard_policy_tests; +#[cfg(test)] +#[path = "batch_read_tests.rs"] +mod batch_read_tests; + #[cfg(feature = "diagnostics")] use crate::diagnostics::{Diagnostics, DiagnosticsSnapshot, Trace}; @@ -854,8 +858,8 @@ struct Shard { /// when the driver thread exits so any waiting `ReadHandle` resolves with a /// driver-gone error instead of hanging (rustfs/backlog#1102). sem: Arc, - /// Signaled after every message send so this shard's loop wakes immediately - /// instead of waiting out the heartbeat (backlog#1102). + /// Signaled after message sends (once per shard for an explicit eager batch) + /// so the loop wakes without waiting out the heartbeat (backlog#1102). wake_efd: Arc, } @@ -901,6 +905,20 @@ pub struct UringDriver { rr: AtomicUsize, } +/// Maximum requests accepted by one [`UringDriver::read_at_batch`] call. +pub const MAX_BATCH_READS: usize = 64; + +/// One buffered positioned read in an explicit notification batch. +#[derive(Debug)] +pub struct ReadRequest { + /// File whose ownership is retained through completion of an accepted read. + pub file: Arc, + /// Positioned byte offset, with the same validation as [`UringDriver::read_at`]. + pub offset: u64, + /// Logical byte count, subject to the driver's configured read limits. + pub len: usize, +} + impl UringDriver { /// Create the ring AND verify a real `IORING_OP_READ` round-trip on a /// temp file before accepting work. `io_uring_setup` succeeding is not @@ -1122,12 +1140,50 @@ impl UringDriver { /// [`Self::read_current`] and is returned as an asynchronous /// `io::ErrorKind::InvalidInput` result rather than panicking. pub fn read_at(&self, file: Arc, offset: u64, len: usize) -> ReadHandle { - self.submit(file, offset, len, 1, false) + self.submit(file, offset, len, 1, false, true) + } + + /// Create handles in input order for up to [`MAX_BATCH_READS`] buffered positioned reads. + /// + /// Eagerly accepted requests share one notification per owning shard after + /// constructing the handles. Saturated requests retain normal asynchronous + /// admission and notify their own shard when polled and admitted. The batch + /// is not atomic and does not guarantee completion order or snapshot reads. + /// Empty batches send no notification. Dropping any handle retains normal + /// cancel safety, including during unwinding of interrupted construction. + /// + /// # Errors + /// + /// More than [`MAX_BATCH_READS`] requests returns `InvalidInput` before any + /// submission. Individual invalid requests return errors from their handles, + /// exactly like [`Self::read_at`]; they do not reject other batch members. + pub fn read_at_batch(&self, requests: Vec) -> io::Result> { + if requests.len() > MAX_BATCH_READS { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "read batch exceeds MAX_BATCH_READS")); + } + let mut handles = Vec::with_capacity(requests.len()); + let mut wakes: Vec> = Vec::with_capacity(requests.len()); + for ReadRequest { file, offset, len } in requests { + let handle = self.submit(file, offset, len, 1, false, false); + if let HandleState::Submitted { wake } = &handle.state + && !wakes.iter().any(|queued| Arc::ptr_eq(queued, wake)) + { + wakes.push(Arc::clone(wake)); + } + // A partially built vector owns every submitted handle. If a later + // construction unwinds, their Drop sends cancels and signals the + // final owner, so delayed batch notification cannot strand reads. + handles.push(handle); + } + for wake in wakes { + wake.signal(); + } + Ok(handles) } /// Read at the file's current position (read(2) semantics) — pipes. pub fn read_current(&self, file: Arc, len: usize) -> ReadHandle { - self.submit(file, CURRENT_POSITION, len, 1, true) + self.submit(file, CURRENT_POSITION, len, 1, true, true) } /// Positioned read from a file opened with `O_DIRECT` (rustfs/backlog#1102). @@ -1146,10 +1202,18 @@ impl UringDriver { /// A non-aligned short read that does not cover the requested range succeeds /// only when metadata confirms EOF; a failed metadata lookup returns an error. pub fn read_at_direct(&self, file: Arc, offset: u64, len: usize, align: usize) -> ReadHandle { - self.submit(file, offset, len, align, false) + self.submit(file, offset, len, align, false, true) } - fn submit(&self, file: Arc, offset: u64, len: usize, align: usize, allow_current_position: bool) -> ReadHandle { + fn submit( + &self, + file: Arc, + offset: u64, + len: usize, + align: usize, + allow_current_position: bool, + notify_now: bool, + ) -> ReadHandle { let id = self.next_id.fetch_add(1, Ordering::Relaxed); assert_eq!(id & CANCEL_BIT, 0, "op id overflowed into the cancel bit"); let (done, rx) = oneshot::channel(); @@ -1346,7 +1410,9 @@ impl UringDriver { }; } // Wake the driver loop so the read starts immediately. - shard.wake_efd.signal(); + if notify_now { + shard.wake_efd.signal(); + } ReadHandle { id, #[cfg(feature = "diagnostics")] diff --git a/src/lib.rs b/src/lib.rs index d714619..405b661 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -55,4 +55,4 @@ mod diagnostics; pub use diagnostics::{DIAGNOSTICS_SAMPLE_INTERVAL, DiagnosticsSnapshot, LatencyHistogram}; #[cfg(target_os = "linux")] -pub use driver::{ProbeFailure, ReadHandle, ReadLimits, ShardPolicy, StatsSnapshot, UringDriver}; +pub use driver::{MAX_BATCH_READS, ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, StatsSnapshot, UringDriver}; diff --git a/tests/cancel.rs b/tests/cancel.rs index bf3716a..4d3b574 100644 --- a/tests/cancel.rs +++ b/tests/cancel.rs @@ -28,7 +28,7 @@ use std::os::fd::{AsRawFd, FromRawFd}; use std::sync::Arc; use std::time::{Duration, Instant}; -use rustfs_uring::UringDriver; +use rustfs_uring::{MAX_BATCH_READS, ReadLimits, ReadRequest, ShardPolicy, UringDriver}; fn driver_or_skip(name: &str) -> Option { match UringDriver::probe_and_start(64) { @@ -80,6 +80,76 @@ fn temp_file_with(content: &[u8], tag: &str) -> (std::path::PathBuf, Arc) (path, file) } +#[tokio::test(flavor = "multi_thread")] +async fn batch_reads_return_byte_exact_results_in_handle_order() { + let Some(driver) = sharded_driver_or_skip("batch_reads_return_byte_exact_results_in_handle_order", 2) else { + return; + }; + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let content = make_content(8192); + let (path, file) = temp_file_with(&content, "batch-exact"); + let requests = (0..MAX_BATCH_READS) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: (index * 17) as u64, + len: 257, + }) + .collect(); + for (index, handle) in driver.read_at_batch(requests).unwrap().into_iter().enumerate() { + let bytes = tokio::time::timeout(Duration::from_secs(5), handle).await.unwrap().unwrap(); + assert_eq!(bytes, &content[index * 17..index * 17 + 257]); + } + let stats = driver.shutdown(); + assert_eq!(stats.submitted, MAX_BATCH_READS as u64); + assert_eq!(stats.delivered, stats.submitted); + assert_eq!(stats.in_flight, 0); + let _ = std::fs::remove_file(path); +} + +#[tokio::test(flavor = "multi_thread")] +async fn batch_cancellation_and_deferred_admission_drain_without_losing_live_results() { + let limits = ReadLimits { + max_read_len: Some(256), + max_in_flight_bytes: Some(512), + }; + let driver = match UringDriver::probe_and_start_with_limits(2, 2, limits) { + Ok(driver) => driver.with_shard_policy(ShardPolicy::CapacityAware), + Err(error) => { + assert!(error.is_expected_restriction(), "{error}"); + eprintln!("SKIP batch_cancellation_and_deferred_admission_drain_without_losing_live_results: {error}"); + return; + } + }; + let content = make_content(8192); + let (path, file) = temp_file_with(&content, "batch-cancel"); + let requests = (0..MAX_BATCH_READS) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: (index * 31) as u64, + len: 256, + }) + .collect(); + let mut retained = Vec::new(); + for (index, handle) in driver.read_at_batch(requests).unwrap().into_iter().enumerate() { + if index % 2 == 0 { + drop(handle); // Accepted reads cancel; deferred reads submit nothing. + } else { + retained.push((index, handle)); + } + } + for (index, handle) in retained { + let bytes = tokio::time::timeout(Duration::from_secs(5), handle).await.unwrap().unwrap(); + assert_eq!(bytes, &content[index * 31..index * 31 + 256]); + } + assert!(wait_until(Duration::from_secs(2), || driver.stats().in_flight == 0).await); + let stats = driver.shutdown(); + assert!(stats.submitted >= (MAX_BATCH_READS / 2) as u64); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + // Completion can win the cancel race; no minimum successful cancel count. + let _ = std::fs::remove_file(path); +} + /// An OS pipe whose read side never completes until we write — the only /// deterministic way to hold an op in flight across a future drop. fn os_pipe() -> (Arc, File) { From 80a7af9055b10167f75429ec2031de23e94c6d51 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 02:30:09 +0800 Subject: [PATCH 11/22] docs(uring): reconcile hardening contracts and validation gates Record native default and all-feature regression results and the rejected A/A calibration without attributing performance benefits. Keep later application and memory-pooling work explicitly pending. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- CHANGELOG.md | 24 +++++++++++++ README.md | 6 ++-- docs/cancellation-efficiency.md | 15 ++++---- docs/driver-fairness.md | 6 ++-- docs/fault-recovery.md | 5 +-- docs/optimization-status.md | 64 +++++++++++++++++++++++++++++++++ src/driver.rs | 2 +- 7 files changed, 109 insertions(+), 13 deletions(-) create mode 100644 docs/optimization-status.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 5035bd3..54f4d98 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,12 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- Opt-in `ReadLimits` for logical read size and driver-wide in-flight read-buffer + bytes, with aligned allocation accounting and terminal-CQE ownership. +- Opt-in capacity-aware shard selection for positioned reads; round-robin remains + the default and stream reads retain their routing semantics. +- `ReadRequest`, `MAX_BATCH_READS` and `read_at_batch` for bounded buffered groups + sharing eager notifications per final owning shard. - Opt-in `diagnostics` feature with per-shard sampled driver-stage histograms, aggregate snapshots, and measurement-interval deltas. Default builds compile out the timing fields and sampling work. @@ -34,6 +40,24 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Changed +- Bound driver intake, allocations and completion work per turn, preserving + progress after notification draining and partial successful submissions. +- Stop explicitly canceled positioned-read continuations after read CQEs and + avoid materializing orphaned results without releasing kernel-owned resources. + +### Fixed + +- Progress empty-SQ CQ overflow/taskrun work and propagate metadata failures + instead of treating an unconfirmed direct-read short prefix as EOF. +- Close count-stage and byte-stage waiters together when shared byte admission + shuts down; retain charged resources for bounded-drain bailout leaks. +- Retry interrupted eventfd operations and suppress repeated unchanged overflow + warnings without changing the cumulative snapshot counter. +- Reap the whole benchmark process group after wrapper failure, including when + the group leader exits before a child process. + +### Benchmarking + - Benchmark CSV schema v2 obtains headers from the executable and reports independent setup, workload, and teardown timings, configurable ring depth, workers, warmup, and instrumentation status. Positional CSV consumers must diff --git a/README.md b/README.md index 8ca1b5a..52260b5 100644 --- a/README.md +++ b/README.md @@ -145,6 +145,8 @@ stage overlap, cancellation, and instrumentation-overhead boundaries. Benchmark configuration, CSV schema, timing boundaries, and performance gates are documented in [the benchmarking guide](docs/benchmarking.md). +See [implementation and acceptance status](docs/optimization-status.md) for +completed correctness work and the still-open performance/integration gates. Linux only; on other hosts `cargo check` builds the empty stub. @@ -153,8 +155,8 @@ Linux only; on other hosts `cargo check` builds the empty stub. cargo test -- --nocapture --test-threads=1 # Two legs in Docker (also on macOS via Docker Desktop / OrbStack): -# leg 1 — io_uring blocked by an explicit seccomp profile → every test MUST -# degrade to a graceful skip; +# leg 1 — io_uring blocked by an explicit seccomp profile → ring-dependent +# tests gracefully skip; kernel-independent unit tests still run; # leg 2 — seccomp=unconfined → real io_uring, and NO test may skip. ./run-docker.sh ``` diff --git a/docs/cancellation-efficiency.md b/docs/cancellation-efficiency.md index 576e7a2..55429f6 100644 --- a/docs/cancellation-efficiency.md +++ b/docs/cancellation-efficiency.md @@ -19,8 +19,10 @@ Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647), skips result memmove/truncation/materialization, including direct-I/O padding. - [x] Keep fd, buffer, and permit ownership until read completion and pending removal. AsyncCancel CQEs still only update cancel statistics. -- [ ] Intake/reap budgets and wake coalescing (separate liveness work). -- [ ] Native Linux execution of the new tests and the full regression suite. +- [x] Bounded intake/reap and explicit finite-batch notifications (see + [driver fairness](driver-fairness.md) and [batch reads](batch-reads.md)). + No state-handshake/global wake coalescing is introduced. +- [x] Native Linux execution of the new tests and the full regression suite. - [ ] Isolated before/after benchmark evidence; no measured speedup is claimed. ## Review and validation @@ -35,12 +37,13 @@ unexpected errors. They do not depend on io_uring availability or accept a skip. Local `cargo fmt --all --check`, `git diff --check`, and Linux-target `cargo check` / `cargo clippy` with all targets and features passed. Cross-checks -compile the Linux-only tests but do not execute them. Native execution remains -an integration gate for this batch. +compile the Linux-only tests but do not execute them. The combined hardening +patch subsequently passed the full all-feature suite on unrestricted Linux, +with no skipped tests and a mandatory positive O_DIRECT marker. Review confirms that no buffer is reclaimed at a cancel CQE or merely because the receiver closed. The optimization only observes cancellation at a read completion, with no live SQE writing into that buffer. A receiver closing after the final `is_closed` check may still incur a copy; this is an accepted race, -not an ownership or correctness failure. Current direct-I/O `fstat` handling -is unchanged and is tracked by the separate integrity work. +not an ownership or correctness failure. Direct-I/O `fstat` error propagation +is covered by the integrated [integrity work](fault-recovery.md). diff --git a/docs/driver-fairness.md b/docs/driver-fairness.md index caa7ea7..054e21b 100644 --- a/docs/driver-fairness.md +++ b/docs/driver-fairness.md @@ -26,8 +26,10 @@ The existing active/idle heartbeats and two bounded submission attempts remain. An exact budget boundary with no actual remaining work costs one empty turn, after which normal idle waiting resumes. -No wakeup coalescing or async-only eventfd registration is introduced. Messages -arriving after the driver's queue check retain their eventfd notification; work +Single-read producer notifications are not coalesced; explicit bounded groups +can share notifications through [read_at_batch](batch-reads.md). No async-only +eventfd registration is introduced. Messages arriving after the driver's queue +check retain their eventfd notification; work left because of a budget retains the explicit continuation flag. Buffers, FDs and permits retain the existing final-CQE ownership rules. diff --git a/docs/fault-recovery.md b/docs/fault-recovery.md index 5b8266c..65fbcb9 100644 --- a/docs/fault-recovery.md +++ b/docs/fault-recovery.md @@ -7,8 +7,9 @@ the driver attempts submission again so newly freed CQ space can receive the kernel's NODROP overflow entries even when no new reads arrive. Each driver turn makes at most two submission attempts. Partial submission, -zero progress, EINTR and EBUSY do not start an unbounded retry loop. The next -event or heartbeat drives further progress. Pending buffers, file descriptors +zero progress, EINTR and EBUSY do not start an unbounded retry loop. Positive +partial progress with queued SQ work starts another bounded turn; errors and +zero progress alone wait for the next event or heartbeat. Pending buffers, file descriptors and admission permits retain their existing final-CQE ownership rules. CQ-overflow warnings are emitted when the observed kernel counter changes to a diff --git a/docs/optimization-status.md b/docs/optimization-status.md new file mode 100644 index 0000000..1768847 --- /dev/null +++ b/docs/optimization-status.md @@ -0,0 +1,64 @@ +# Read-driver optimization status + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647). +Updated 2026-09-23. Implementation, correctness, CI, performance and application +integration are separate gates; a checked code item does not close the roadmap. + +| Step | Implementation and review | Remaining acceptance | +| --- | --- | --- | +| 1.1–1.4: benchmark schema, timing, diagnostics, direct positive gate | Merged in [#15](https://github.com/rustfs/uring/pull/15), CI passed | Diagnostics overhead and real LocalIoBackend baseline | +| 1.5: ABBA evidence tooling | Implemented; 10 gate/cleanup tests pass | Stable calibration, valid comparison and application integration | +| 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | Hardening PR CI/merge; application-wide limits remain separate | +| 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR CI/merge | +| 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR CI/merge | +| 4.2: system-wide budgets and probe offload | Not implemented in this crate | Fresh application-path review and coordinated integration | +| 5: owned buffers and direct FD cache | Not implemented | Profile evidence, lease/pool lifetime design and cache invalidation integration | +| 6: ordered streaming prefetch | Not implemented | Consumer/backpressure contract and end-to-end evidence | +| 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | + +## Correctness and review evidence + +The integrated code revision `7e2f1aa` passed 104 tests on unrestricted Linux: +71 unit, 4 admission, 21 read/cancellation, 6 fault and 2 shard-policy tests. +No test skipped; the mandatory `DIRECT_OK` marker was observed. All-feature +Clippy, formatting, 10 Python evidence-runner tests and the instrumented +eight-strategy benchmark correctness smoke also passed. The default-feature +native suite passed another 90 tests with no skips and the same direct marker. +A Linux-target +default/all-feature Clippy check and warning-denying rustdoc passed locally; +cross-compilation is not additional native execution evidence. + +Independent reviews covered buffer/FD/permit ownership, global closure and +count-stage waiters, cancellation races, final-owner routing, batch unwind and +notifications, fairness/progress, test validity and documentation. Review found +and fixed a global-close waiter gap and benchmark orphan-child cleanup. No +blocking finding remained in the completed component and combined-patch reviews. + +Kernel CQ-overflow and direct-read tests actually execute on Linux. Partial +submit/error classification and metadata failures use deterministic production +decision seams; they do not prove forced real-kernel partial acceptance, hung +syscall recovery, or advanced taskrun-mode behavior. + +## Performance status: not accepted + +The same-binary warm-cache A/A control used three A1/B1/B2/A2 rounds, 10 million +operations per leg and unchanged drift gates (3% throughput, 5% p99). The third +round's p99 baseline drift was **6.35%**, so the whole calibration was rejected +and candidate/diagnostics-overhead comparisons were not started. An earlier +shorter calibration was rejected for insufficient measurement duration; neither +run is performance evidence. The baseline executable predates this hardening +patch: these controls do not measure the new driver's speed. + +Do not infer throughput, latency, CPU or S3 improvements from correctness tests, +eventfd-count assertions, or rejected calibration. Keep diagnostics disabled +and round-robin routing as defaults. Before further performance-dependent work, +establish stable calibration without loosening gates after observing results. + +## Contract references + +- [Public read and admission contracts](../README.md) +- [Completion recovery and direct integrity](fault-recovery.md) +- [Cancellation and eventfd behavior](cancellation-efficiency.md) +- [Driver turn fairness](driver-fairness.md) +- [Explicit buffered batches](batch-reads.md) +- [Measurement boundaries and evidence gates](benchmarking.md) diff --git a/src/driver.rs b/src/driver.rs index 1ad97be..571a51c 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -369,7 +369,7 @@ type AcquireFut = Pin, /// Maximum sum of reserved read-buffer bytes across all shards. /// From 6927bb0adf0ea9de416cc21e1697c39a162ff52c Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 03:26:16 +0800 Subject: [PATCH 12/22] feat(bench): add explicit same-binary calibration gates Require identical artifacts and feature-off rows for opt-in A/A calibration. Gate each middle leg against the endpoint mean without candidate attribution while preserving comparison defaults. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/benchmarking.md | 41 ++++++++++- scripts/bench-abba.py | 45 +++++++++--- scripts/test_bench_abba.py | 140 +++++++++++++++++++++++++++++++++++++ 3 files changed, 214 insertions(+), 12 deletions(-) diff --git a/docs/benchmarking.md b/docs/benchmarking.md index 7119567..dee3852 100644 --- a/docs/benchmarking.md +++ b/docs/benchmarking.md @@ -171,9 +171,40 @@ service can introduce load, supply `--require-inactive-unit UNIT`; the runner refuses to start or continue unless that unit is inactive. Any service stop or restore is an explicit operator action outside this tool. A reservation note and process checks are evidence aids, not a substitute for exclusive resources. -Perform A/A calibration first by passing the baseline binary in both positions -and `--candidate-interval 0`; then use the diagnostics candidate and the default -interval of 64. Keep the workload and thresholds fixed between experiments. +Perform explicit A/A calibration first with `--calibration --candidate-interval 0` +and the same feature-off binary in both positions; then use the diagnostics +candidate without `--calibration` and with the default interval of 64. Keep the +workload and thresholds fixed between experiments. + +For example, use the preceding command with both executable arguments pointing +to the baseline artifact and add: + +```sh +--calibration --candidate-interval 0 +``` + +Calibration requires matching executable SHA-256 values before running either +artifact (identical copies at different paths are allowed), and requires interval +zero for every measurement row. A mismatched hash or nonzero candidate interval +is rejected, including in `--dry-run`. Each leg retains the normal binary-identity, +geometry, resource and duration checks. Dry-run, provenance and summary output +identify `mode` as `calibration` or `comparison`; dry-run does not execute workloads +or establish that runtime gates will pass. + +Calibration preserves at least three A1/B1/B2/A2 rounds and the existing default +thresholds: 3% for IOPS and 5% for p99. Every round checks A2 against A1, then checks +**B1 and B2 separately** against the arithmetic mean of A1/A2, using those same +thresholds in either direction. It never averages B1/B2 before gating: opposite +noise must not cancel out. The first failed round stops the experiment. Only +after all requested rounds pass does the summary say `valid-calibration`. +Calibration records `baseline_drift` and per-leg `middle_drift`, never +`candidate_change_pct`, whether it succeeds or fails. This validates the specified +within-round noise gates, not an optimization benefit or stability under another +workload. Cross-round trends, resources and application SLOs still require review. + +Omitting `--calibration` retains comparison mode and its original endpoint drift +gate. Passing the same binary with `--candidate-interval 0` alone does not enable +the additional middle-leg gates and must not be reported as validated calibration. Every leg must run at least five measured seconds by default. If it is too short, increase operations in a **new** experiment. The tool validates geometry, @@ -190,5 +221,9 @@ and resource reports against the application SLO separately. Duration histograms with too few sampled operations are not reliable tail estimates. Runner gate tests: `python3 -m unittest discover -s scripts -p 'test_bench_abba.py'`. +Calibration regressions use synthetic CSV/resource reports and mocked workload +execution to cover mode validation, matching hashes, endpoint and middle drift, +opposite-noise rejection, three-round success and early failure. They verify the +runner's decisions, not native Linux performance or isolation. Tracking and implementation status: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647). diff --git a/scripts/bench-abba.py b/scripts/bench-abba.py index d99d272..0f8258d 100644 --- a/scripts/bench-abba.py +++ b/scripts/bench-abba.py @@ -66,7 +66,7 @@ def percent_change(after, before): return result -def evaluate_round(rows, throughput_drift, tail_drift): +def evaluate_round(rows, throughput_drift, tail_drift, *, calibration=False): if [row["leg"] for row in rows] != ["A1", "B1", "B2", "A2"]: raise ValueError("incomplete or reordered ABBA round") a1, b1, b2, a2 = [row["measurement"] for row in rows] @@ -76,6 +76,23 @@ def evaluate_round(rows, throughput_drift, tail_drift): } valid = abs(drift["iops_pct"]) <= throughput_drift and abs(drift["p99_pct"]) <= tail_drift result = {"valid": valid, "baseline_drift": drift} + if calibration: + # Each middle leg must be stable on its own: averaging B1/B2 first + # could cancel opposite noise and incorrectly certify calibration. + middle = { + leg: { + "iops_pct": percent_change(row["IOPS"], (a1["IOPS"] + a2["IOPS"]) / 2), + "p99_pct": percent_change(row["p99_us"], (a1["p99_us"] + a2["p99_us"]) / 2), + } + for leg, row in (("B1", b1), ("B2", b2)) + } + result["middle_drift"] = middle + result["valid"] = valid and all( + abs(drift["iops_pct"]) <= throughput_drift and abs(drift["p99_pct"]) <= tail_drift + for drift in middle.values() + ) + # A/A noise is never a candidate benefit, even when all gates pass. + return result # Do not calculate candidate attribution after a failed baseline gate. if valid: result["candidate_change_pct"] = { @@ -166,8 +183,10 @@ def arguments(): parser.add_argument("--ops", type=int, default=1_000_000) parser.add_argument("--warmup-ops", type=int, default=10000) parser.add_argument("--rounds", type=int, default=3) + parser.add_argument("--calibration", action="store_true", + help="same-binary A/A calibration; also gate each middle leg; requires --candidate-interval 0") parser.add_argument("--candidate-interval", type=int, choices=(0, 64), default=64, - help="0 supports A/A calibration; 64 compares diagnostics on against off") + help="0 compares feature-off artifacts; 64 compares diagnostics on against off (default)") parser.add_argument("--throughput-drift-pct", type=float, default=3) parser.add_argument("--p99-drift-pct", type=float, default=5) parser.add_argument("--timeout", type=float, default=300) @@ -175,6 +194,8 @@ def arguments(): parser.add_argument("--min-seconds", type=float, default=5) parser.add_argument("--dry-run", action="store_true") args = parser.parse_args() + if args.calibration and args.candidate_interval != 0: + parser.error("calibration requires --candidate-interval 0 so every leg has the same diagnostics configuration") if not (100000 <= args.ops <= 10_000_000) or args.rounds < 3: parser.error("acceptance requires 100000..10000000 operations and at least three rounds") if any(not math.isfinite(value) or value < 0 for value in ( @@ -196,7 +217,11 @@ def arguments(): def main(): args = arguments() + mode = "calibration" if args.calibration else "comparison" binaries = {"A": args.baseline.resolve(), "B": args.candidate.resolve()} + identities = {key: {"path": str(path), "sha256": sha256(path)} for key, path in binaries.items()} + if args.calibration and identities["A"]["sha256"] != identities["B"]["sha256"]: + raise ValueError("calibration requires the same binary content for baseline and candidate") headers = {key: subprocess.check_output([str(path), "--header"], text=True).strip() for key, path in binaries.items()} if headers["A"] != headers["B"]: raise ValueError("baseline and candidate schema differ") @@ -205,18 +230,18 @@ def main(): ring_entries=args.entries, warmup_ops=args.warmup_ops) plan = [(round_id, leg) for round_id in range(1, args.rounds + 1) for leg in ("A1", "B1", "B2", "A2")] if args.dry_run: - print(json.dumps({"plan": plan, "geometry": expected, "candidate_interval": args.candidate_interval})) + print(json.dumps({"mode": mode, "plan": plan, "geometry": expected, "candidate_interval": args.candidate_interval})) return 0 environment_guard(args.require_inactive_unit) args.run_dir.mkdir(mode=0o700) - provenance = {"source_revision": args.source_revision, "reservation": args.reservation_note, + provenance = {"mode": mode, "source_revision": args.source_revision, "reservation": args.reservation_note, "cache": "warm-preload", "geometry": expected, "cpus": args.cpus, "data_identity": data_identity(args.data_file), - "binaries": {key: {"path": str(path), "sha256": sha256(path)} for key, path in binaries.items()}, + "binaries": identities, "gates": {"throughput_drift_pct": args.throughput_drift_pct, "p99_drift_pct": args.p99_drift_pct, "min_seconds": args.min_seconds}, "candidate_interval": args.candidate_interval} (args.run_dir / "provenance.json").write_text(json.dumps(provenance, indent=2) + "\n") - summary = {"status": "incomplete", "rounds": []} + summary = {"mode": mode, "status": "incomplete", "rounds": []} try: current_round = [] for round_id, leg in plan: @@ -254,12 +279,14 @@ def main(): current_round.append(record) print(f"round={round_id} leg={leg} IOPS={row['IOPS']:.0f} p99_us={row['p99_us']:.0f}", flush=True) if leg == "A2": - result = evaluate_round(current_round, args.throughput_drift_pct, args.p99_drift_pct) + result = evaluate_round(current_round, args.throughput_drift_pct, args.p99_drift_pct, + calibration=args.calibration) summary["rounds"].append(result) current_round = [] if not result["valid"]: - raise RuntimeError("baseline drift failed; stopped before expanding the experiment") - summary["status"] = "valid-comparison" + gate = "calibration drift" if args.calibration else "baseline drift" + raise RuntimeError(f"{gate} failed; stopped before expanding the experiment") + summary["status"] = "valid-calibration" if args.calibration else "valid-comparison" summary["scope"] = "driver-only; resource CSV is whole-process CPU/RSS, not steady-state-only" return 0 except (ValueError, RuntimeError, OSError, subprocess.SubprocessError) as error: diff --git a/scripts/test_bench_abba.py b/scripts/test_bench_abba.py index 9295728..b1283a7 100644 --- a/scripts/test_bench_abba.py +++ b/scripts/test_bench_abba.py @@ -1,6 +1,8 @@ # Copyright 2024 RustFS Team # SPDX-License-Identifier: Apache-2.0 import importlib.util +import io +import json import os from pathlib import Path import select @@ -130,5 +132,143 @@ def test_missing_process_group_preserves_the_benchmark_failure(self): killpg.assert_called_once() +class CalibrationGates(unittest.TestCase): + HEADER = ("schema_version,mode,strategy,shards,file_size,read_size,concurrency,ops,secs,IOPS,MBps," + "p50_us,p99_us,p999_us,startup_secs,shutdown_secs,workers,ring_entries,warmup_ops,diagnostics_interval") + + def rows(self): + return [{"leg": leg, "measurement": dict(IOPS=100, p99_us=20, p999_us=30)} + for leg in ("A1", "B1", "B2", "A2")] + + def argv(self, root, *extra): + (root / "baseline").write_bytes(b"identical executable content") + (root / "candidate").write_bytes(b"identical executable content") + (root / "data").write_bytes(bytes(36864)) + return ["bench-abba.py", "--baseline", str(root / "baseline"), "--candidate", str(root / "candidate"), + "--data-file", str(root / "data"), "--run-dir", str(root / "results"), + "--source-revision", "test-revision", "--reservation-note", "unit test; no benchmark executed", + "--calibration", "--candidate-interval", "0", *extra] + + def test_calibration_success_has_no_candidate_attribution(self): + result = BENCH.evaluate_round(self.rows(), 3, 5, calibration=True) + self.assertTrue(result["valid"]) + self.assertEqual(set(result["middle_drift"]), {"B1", "B2"}) + self.assertNotIn("candidate_change_pct", result) + + def test_each_middle_leg_is_checked_without_opposite_noise_cancellation(self): + for metric, low, high in (("IOPS", 90, 110), ("p99_us", 18, 22)): + with self.subTest(metric=metric): + rows = self.rows() + rows[1]["measurement"][metric] = low + rows[2]["measurement"][metric] = high + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertFalse(result["valid"]) + self.assertNotIn("candidate_change_pct", result) + + def test_middle_gates_use_endpoint_mean(self): + rows = self.rows() + for row, iops in zip(rows, (100, 103, 103, 102)): + row["measurement"]["IOPS"] = iops + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertTrue(result["valid"]) + self.assertAlmostEqual(result["middle_drift"]["B1"]["iops_pct"], 100 * (103 / 101 - 1)) + + def test_calibration_endpoint_drift_still_invalidates_without_attribution(self): + for metric, changed in (("IOPS", 110), ("p99_us", 22)): + with self.subTest(metric=metric): + rows = self.rows() + rows[3]["measurement"][metric] = changed + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertFalse(result["valid"]) + self.assertNotIn("candidate_change_pct", result) + + def test_calibration_requires_zero_interval_and_three_rounds(self): + for extra in (("--candidate-interval", "64"), ("--rounds", "2")): + with self.subTest(extra=extra), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + with patch.object(sys, "argv", self.argv(root, *extra)), patch("sys.stderr", new_callable=io.StringIO): + with self.assertRaises(SystemExit) as error: + BENCH.arguments() + self.assertEqual(error.exception.code, 2) + self.assertFalse((root / "results").exists()) + + def test_default_mode_still_compares_different_feature_states(self): + with tempfile.TemporaryDirectory() as directory: + argv = self.argv(Path(directory)) + argv.remove("--calibration") + del argv[-2:] # Preserve the default candidate interval of 64. + with patch.object(sys, "argv", argv): + args = BENCH.arguments() + self.assertFalse(args.calibration) + self.assertEqual(args.candidate_interval, 64) + + def test_different_binary_hashes_reject_calibration_before_execution(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + argv = self.argv(root) + (root / "candidate").write_bytes(b"different executable content") + with patch.object(sys, "argv", argv), patch.object(BENCH.subprocess, "check_output") as command: + with self.assertRaisesRegex(ValueError, "same binary"): + BENCH.main() + command.assert_not_called() + self.assertFalse((root / "results").exists()) + + def test_calibration_dry_run_validates_identity_and_reports_its_mode(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + with patch.object(sys, "argv", self.argv(root, "--dry-run")), \ + patch.object(BENCH.subprocess, "check_output", return_value=self.HEADER), \ + patch("sys.stdout", new_callable=io.StringIO) as output: + self.assertEqual(BENCH.main(), 0) + plan = json.loads(output.getvalue()) + self.assertEqual(plan["mode"], "calibration") + self.assertEqual(len(plan["plan"]), 12) + self.assertFalse((root / "results").exists()) + + def run_main_with_measurements(self, root, invalid_leg=None): + executed = [] + + def execute(command, env, stdout, stderr, timeout, unit): + executed.append(stdout.stem) + self.assertEqual(env["BENCH_DIAGNOSTICS"], "0") + iops = 90000 if stdout.stem == invalid_leg else 100000 + secs = 1_000_000 / iops + stdout.write_text(f"2,measure,uring_cached_read,2,36864,32768,32,1000000,{secs:.6f},{iops}," + f"{iops / 32:.1f},10,20,30,0.1,0.1,4,64,10000,0\n") + stderr.write_text("") + Path(command[4]).write_text(f"1,1,{secs + 1},1000,3,4\n") + + with patch.object(sys, "argv", self.argv(root)), \ + patch.object(BENCH.subprocess, "check_output", return_value=self.HEADER), \ + patch.object(BENCH, "environment_guard"), patch.object(BENCH.time, "sleep"), \ + patch.object(BENCH, "execute", side_effect=execute), \ + patch("sys.stdout", new_callable=io.StringIO), patch("sys.stderr", new_callable=io.StringIO): + code = BENCH.main() + summary = json.loads((root / "results" / "summary.json").read_text()) + provenance = json.loads((root / "results" / "provenance.json").read_text()) + return code, summary, provenance, executed + + def test_three_valid_rounds_report_valid_calibration(self): + with tempfile.TemporaryDirectory() as directory: + code, summary, provenance, executed = self.run_main_with_measurements(Path(directory)) + self.assertEqual(code, 0) + self.assertEqual(summary["mode"], "calibration") + self.assertEqual(summary["status"], "valid-calibration") + self.assertEqual(len(summary["rounds"]), 3) + self.assertEqual(len(executed), 12) + self.assertEqual(provenance["mode"], "calibration") + self.assertEqual(provenance["binaries"]["A"]["sha256"], provenance["binaries"]["B"]["sha256"]) + self.assertNotIn("candidate_change_pct", json.dumps(summary)) + + def test_middle_failure_stops_calibration_without_attribution_from_earlier_rounds(self): + with tempfile.TemporaryDirectory() as directory: + code, summary, _, executed = self.run_main_with_measurements(Path(directory), "round-2-B1") + self.assertEqual(code, 1) + self.assertEqual(summary["status"], "invalid") + self.assertEqual(len(summary["rounds"]), 2) + self.assertEqual(len(executed), 8) + self.assertNotIn("candidate_change_pct", json.dumps(summary)) + + if __name__ == "__main__": unittest.main() From f135f1dcabe7dd48ad52c19813cfb6ab6f11378a Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 03:32:38 +0800 Subject: [PATCH 13/22] test(prefetch): add bounded ordered-reader correctness experiment Keep ordered prefetch example-local with bounded logical slots and bytes, cancellation-safe next polling, all-slot admission progress and ordered EOF/error termination. Add deterministic semaphore-race coverage and a read-only Linux byte oracle without production API changes or performance claims. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/ordered-prefetch.md | 120 ++++++++++++ examples/ordered_prefetch.rs | 30 +++ examples/ordered_prefetch/linux.rs | 100 ++++++++++ examples/ordered_prefetch/reader.rs | 175 +++++++++++++++++ tests/ordered_prefetch.rs | 286 ++++++++++++++++++++++++++++ 5 files changed, 711 insertions(+) create mode 100644 docs/ordered-prefetch.md create mode 100644 examples/ordered_prefetch.rs create mode 100644 examples/ordered_prefetch/linux.rs create mode 100644 examples/ordered_prefetch/reader.rs create mode 100644 tests/ordered_prefetch.rs diff --git a/docs/ordered-prefetch.md b/docs/ordered-prefetch.md new file mode 100644 index 0000000..9a267f8 --- /dev/null +++ b/docs/ordered-prefetch.md @@ -0,0 +1,120 @@ +# Ordered buffered prefetch: correctness experiment + +This example explores step 6 of [the optimization roadmap](optimization-status.md). +It does not add a public reader API, change production defaults, integrate an +object stream, or establish a performance improvement. Keep it as an example +and test harness until the consumer contract and performance gates are accepted. + +## Run against an existing immutable fixture + +```sh +cargo run --example ordered_prefetch -- FILE OFFSET LENGTH CHUNK WINDOW MAX_BYTES +``` + +All geometry arguments are decimal integers. `CHUNK` is 1 byte through 8 MiB, +`WINDOW` is 1 through 64, and `MAX_BYTES` is between `CHUNK` and 512 MiB. +`OFFSET + LENGTH` must fit `i64::MAX`. Zero length is supported. The file must +already exist and be regular; symlinks are refused. The example never creates, +truncates or modifies the fixture. Keep it immutable during verification: +independent positioned reads do not provide snapshot isolation. + +The example compares every returned chunk, including a short final chunk, +against a synchronous positioned std read outside the async executor. It checks +contiguous output, premature EOF and final driver drain conservation. Successful +execution prints `ORDERED_PREFETCH_OK bytes=N chunks=N`. Invalid input, unavailable +io_uring, read/reference mismatch or incomplete drain returns a nonzero status +without that marker. Non-Linux execution also fails explicitly. + +The driver uses one shard, two count permits and a one-chunk byte budget to +exercise deferred admission; the consumer window uses the supplied geometry. +This configuration and the reference reads are deliberately for correctness, +not a performance comparison. No new dependency or per-chunk task is introduced. + +## State and backpressure contract + +The example-local `OrderedReader::new(Config, source)` stores ordered slots of +`Pending(future)` or `Ready(result)`. The source must implement whole requested +range-or-EOF positioned reads, as `UringDriver::read_at` does. Arbitrary stream +short reads must not be substituted: the reader treats a successful short range +as EOF. Sources must return no more bytes than requested; excess output becomes +`InvalidData`. The generic source exists for deterministic tests, not as a +published production abstraction. + +`next(&mut self)` fills only available slots and logical bytes at the start of +a consumer poll, then polls all active slots, retaining out-of-order results. +It returns only the front slot and never refills after yielding that chunk. +No consumer polls means no replacement submissions. The constructor submits +nothing. A short result or error stops future scheduling and drops later slots; +earlier results remain ordered. A nonempty EOF prefix is yielded once, then +`None`; an empty EOF returns `None`; an ordered error is returned once, then +the reader is fused at `None`. + +Slots remain inside the reader across `Poll::Pending`. Dropping a pending +`next()` future neither advances the delivered offset nor drops the front +handle. Dropping the reader drops all its handles, retaining their normal +driver cancellation and final-CQE ownership rules. Earlier speculative reads +may already have completed, and cancellation is not guaranteed to win that race. + +Both pending and ready slots reserve their original requested logical length +until removal. Thus their combined requested lengths never exceed `MAX_BYTES`, +and their combined count never exceeds `WINDOW`. This is not an RSS or +whole-process memory bound: it excludes allocator overhead/excess Vec capacity, +driver structures, the example's reference buffer and already returned chunks +retained by the caller. Dropped kernel-visible buffers can also remain in the +driver until CQE; dropping a slot does not free them early. A terminal boundary +stops replenishment after dropping its speculative tail. + +## Admission and liveness + +Polling only the front is safe from this particular admission cycle only when +later handles have never been polled: eager later reads progress on the driver +and release permits at CQE, independently of consuming their results. Such a +simple reader can lose prefetch depth after saturation because deferred tails +remain unpolled. + +Once a deferred tail has been polled, switching to front-only polling can +deadlock an active consumer. For example, the head waits for a count permit on +shard A while a tail on shard B waits for shared bytes. Tokio may assign newly +released bytes to the tail before A's count becomes available. The head then +waits for bytes reserved by the tail; the tail must be repolled to submit and +eventually release them. The state machine polls every remaining active slot +on each consumer poll, including tails when the head is pending. Ready futures +are never polled again. Work per poll is bounded by the 64-slot limit; there is +no wait/spin loop or background producer. + +This does not promise global admission fairness while a consumer is paused. +Already-polled deferred handles can retain partial or newly assigned permits +until the consumer polls again or drops the reader. Application integration +must define stall timeouts and drop the reader when abandoning a stalled +consumer; merely dropping a pending `next()` preserves the reader and its +reservations intentionally. Paused consumer tests prove no refill, not continued +background permit release or unlimited progress by unrelated readers. + +## Deterministic coverage and remaining gates + +```sh +cargo test --test ordered_prefetch +``` + +The portable tests use controlled futures and actual Tokio count/byte semaphores +for the deferred-tail reservation race. They cover out-of-order completions, +ready-result byte accounting, slow-consumer refill boundaries, cancellation of +`next`, whole-reader drop, early/empty EOF, ordered errors, final range geometry, +invalid/overflow inputs and an oversized source result. These are state-machine +and admission tests, not native kernel stall or filesystem truncation injection. +Run the example on unrestricted Linux for actual io_uring byte correctness. + +Before any public API or RustFS integration, define whole-range versus exact +object-length EOF policy, stall timeout, quorum abandonment, error mapping and +file-generation consistency. Bitrot verification and checksum framing belong to +the application: raw byte equality in this example does not implement bitrot, +and arbitrary chunk boundaries cannot be assumed to match its verification +units. Consumer-retained results need a separate ownership/budget policy. + +Only after stable calibration should a measurement harness compare a sequential +buffered/readahead baseline and small ordered windows with matched consumer +backpressure, provenance and ABBA drift gates. Measure TTFB, p99/stall behavior, +throughput, CPU, syscall counts and memory residency. The existing unordered +JoinSet benchmark is not evidence for ordered-consumer behavior. Actual S3 +streaming, bitrot/quorum behavior, performance acceptance and production enablement +remain unimplemented/unverified by this experiment. diff --git a/examples/ordered_prefetch.rs b/examples/ordered_prefetch.rs new file mode 100644 index 0000000..450c852 --- /dev/null +++ b/examples/ordered_prefetch.rs @@ -0,0 +1,30 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Correctness-only, example-local ordered buffered prefetch. No timing claims. + +#[cfg(target_os = "linux")] +#[path = "ordered_prefetch/reader.rs"] +mod reader; + +#[cfg(target_os = "linux")] +#[path = "ordered_prefetch/linux.rs"] +mod linux; + +fn main() -> std::process::ExitCode { + #[cfg(target_os = "linux")] + { + match linux::run() { + Ok(()) => std::process::ExitCode::SUCCESS, + Err(error) => { + eprintln!("ordered_prefetch: {error}"); + std::process::ExitCode::FAILURE + } + } + } + #[cfg(not(target_os = "linux"))] + { + eprintln!("ordered_prefetch requires Linux and a usable io_uring driver"); + std::process::ExitCode::FAILURE + } +} diff --git a/examples/ordered_prefetch/linux.rs b/examples/ordered_prefetch/linux.rs new file mode 100644 index 0000000..d7b4dd6 --- /dev/null +++ b/examples/ordered_prefetch/linux.rs @@ -0,0 +1,100 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::reader::{Config, OrderedReader}; +use rustfs_uring::{ReadLimits, UringDriver}; +use std::fs::{File, OpenOptions}; +use std::io; +use std::os::unix::fs::{FileExt, OpenOptionsExt}; +use std::sync::Arc; + +fn invalid(message: impl Into) -> io::Error { + io::Error::new(io::ErrorKind::InvalidInput, message.into()) +} + +fn number(name: &str, value: &str) -> io::Result { + value.parse().map_err(|_| invalid(format!("invalid {name}: {value}"))) +} + +fn reference_read(file: &File, offset: u64, len: usize) -> io::Result> { + let mut bytes = vec![0; len]; + let mut read = 0; + while read < len { + match file.read_at(&mut bytes[read..], offset + read as u64) { + Ok(0) => break, + Ok(count) => read += count, + Err(error) if error.kind() == io::ErrorKind::Interrupted => continue, + Err(error) => return Err(error), + } + } + bytes.truncate(read); + Ok(bytes) +} + +pub fn run() -> io::Result<()> { + let args: Vec = std::env::args().skip(1).collect(); + if args.len() != 6 { + return Err(invalid("usage: ordered_prefetch FILE OFFSET LENGTH CHUNK WINDOW MAX_BYTES")); + } + let cfg = Config { + offset: number("OFFSET", &args[1])?, + length: number("LENGTH", &args[2])?, + chunk: number("CHUNK", &args[3])?, + window: number("WINDOW", &args[4])?, + max_bytes: number("MAX_BYTES", &args[5])?, + }; + let end = cfg.validate()?; + // Never create/write/truncate the fixture or block opening a FIFO. Validate + // the opened descriptor, and refuse symlinks to keep fixture identity clear. + let file = OpenOptions::new() + .read(true) + .custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK) + .open(&args[0])?; + if !file.metadata()?.is_file() { + return Err(invalid("FILE must be an existing regular file")); + } + let file = Arc::new(file); + // Intentionally tight admission exercises deferred reads. This executable + // verifies correctness; it is not a throughput baseline or production tuning. + let driver = UringDriver::probe_and_start_with_limits( + 2, + 1, + ReadLimits { + max_read_len: Some(cfg.chunk), + max_in_flight_bytes: Some(cfg.chunk), + }, + ) + .map_err(io::Error::other)?; + let runtime = tokio::runtime::Builder::new_current_thread().build()?; + let mut reader = OrderedReader::new(cfg, |offset, len| driver.read_at(Arc::clone(&file), offset, len))?; + let result = (|| { + let mut delivered = 0u64; + let mut chunks = 0u64; + // Synchronous reference I/O runs outside the async executor. Pausing + // consumption for verification must not refill the reader's window. + while let Some(chunk) = runtime.block_on(reader.next())? { + let expected_offset = cfg.offset + delivered; + if chunk.offset != expected_offset { + return Err(io::Error::other("non-contiguous output offsets")); + } + let want = (end - expected_offset).min(cfg.chunk as u64) as usize; + if chunk.bytes != reference_read(&file, expected_offset, want)? { + return Err(io::Error::other("byte mismatch against positioned std read")); + } + delivered += chunk.bytes.len() as u64; + chunks += 1; + } + if cfg.offset + delivered < end && !reference_read(&file, cfg.offset + delivered, 1)?.is_empty() { + return Err(io::Error::other("ordered reader terminated before reference EOF")); + } + Ok((delivered, chunks)) + })(); + drop(reader); + let stats = driver.shutdown(); + let (bytes, chunks) = result?; + if stats.in_flight != 0 || stats.submitted != stats.delivered + stats.orphan_reclaimed { + return Err(io::Error::other("driver did not drain with read conservation intact")); + } + println!("ORDERED_PREFETCH_OK bytes={bytes} chunks={chunks}"); + Ok(()) +} diff --git a/examples/ordered_prefetch/reader.rs b/examples/ordered_prefetch/reader.rs new file mode 100644 index 0000000..4f119d2 --- /dev/null +++ b/examples/ordered_prefetch/reader.rs @@ -0,0 +1,175 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Example-only ordered consumer state machine. No production API or runtime. + +use std::collections::VecDeque; +use std::future::{Future, poll_fn}; +use std::io; +use std::pin::Pin; +use std::task::{Context, Poll}; + +const MAX_WINDOW: usize = 64; +const MAX_CHUNK: usize = 8 * 1024 * 1024; + +#[derive(Clone, Copy)] +pub struct Config { + pub offset: u64, + pub length: u64, + pub chunk: usize, + pub window: usize, + pub max_bytes: usize, +} + +impl Config { + pub fn validate(self) -> io::Result { + if self.chunk == 0 || self.chunk > MAX_CHUNK || self.window == 0 || self.window > MAX_WINDOW { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "chunk must be 1..=8 MiB and window 1..=64")); + } + if self.max_bytes < self.chunk || self.max_bytes > MAX_WINDOW * MAX_CHUNK { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "max_bytes must be at least chunk and at most 512 MiB", + )); + } + self.offset + .checked_add(self.length) + .filter(|end| *end <= i64::MAX as u64) + .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "range end must fit signed file offsets")) + } +} + +pub struct Chunk { + pub offset: u64, + pub bytes: Vec, +} + +enum State { + Pending(F), + Ready(io::Result>), +} + +struct Slot { + offset: u64, + len: usize, + state: State, +} + +pub struct OrderedReader { + config: Config, + end: u64, + next_offset: u64, + reserved: usize, + slots: VecDeque>, + source: S, + stop_scheduling: bool, + finished: bool, +} + +impl OrderedReader +where + F: Future>> + Unpin, + S: FnMut(u64, usize) -> F, +{ + pub fn new(config: Config, source: S) -> io::Result { + let end = config.validate()?; + Ok(Self { + config, + end, + next_offset: config.offset, + reserved: 0, + slots: VecDeque::with_capacity(config.window), + source, + stop_scheduling: false, + finished: false, + }) + } + + /// Dropping a pending next future leaves every handle and cursor in self. + /// No slot is removed before Ready, and Ready results are never repolled. + pub async fn next(&mut self) -> io::Result> { + poll_fn(|cx| self.poll_next(cx)).await + } + + fn fill_window(&mut self) { + while !self.stop_scheduling && self.next_offset < self.end && self.slots.len() < self.config.window { + let len = (self.end - self.next_offset).min(self.config.chunk as u64) as usize; + if len > self.config.max_bytes - self.reserved { + break; + } + let handle = (self.source)(self.next_offset, len); + self.slots.push_back(Slot { + offset: self.next_offset, + len, + state: State::Pending(handle), + }); + self.reserved += len; + self.next_offset += len as u64; + } + } + + fn poll_next(&mut self, cx: &mut Context<'_>) -> Poll>> { + if self.finished { + return Poll::Ready(Ok(None)); + } + // Refill only in a consumer poll, never after delivering a chunk. + self.fill_window(); + let mut terminal_index = None; + for (index, slot) in self.slots.iter_mut().enumerate() { + if let State::Pending(handle) = &mut slot.state + && let Poll::Ready(mut result) = Pin::new(handle).poll(cx) + { + if result.as_ref().is_ok_and(|bytes| bytes.len() > slot.len) { + result = Err(io::Error::new(io::ErrorKind::InvalidData, "source returned more bytes than requested")); + } + slot.state = State::Ready(result); + } + if let State::Ready(result) = &slot.state + && !result.as_ref().is_ok_and(|bytes| bytes.len() == slot.len) + { + // Preserve earlier chunks and the first ordered terminal result; + // later reads can no longer contribute and are dropped/cancelled. + terminal_index = Some(index); + break; + } + } + if let Some(index) = terminal_index { + self.stop_scheduling = true; + while self.slots.len() > index + 1 { + if let Some(slot) = self.slots.pop_back() { + self.reserved -= slot.len; + } + } + } + + match self.slots.front() { + Some(Slot { + state: State::Pending(_), + .. + }) => return Poll::Pending, + None => { + self.finished = true; + return Poll::Ready(Ok(None)); + } + Some(_) => {} + } + let slot = self.slots.pop_front().expect("ready front checked above"); + self.reserved -= slot.len; + let State::Ready(result) = slot.state else { unreachable!("ready front checked above") }; + match result { + Ok(bytes) => { + if bytes.len() < slot.len { + self.finished = true; + } + Poll::Ready(Ok((!bytes.is_empty()).then_some(Chunk { + offset: slot.offset, + bytes, + }))) + } + Err(error) => { + self.finished = true; + Poll::Ready(Err(error)) + } + } + } +} diff --git a/tests/ordered_prefetch.rs b/tests/ordered_prefetch.rs new file mode 100644 index 0000000..9b45a76 --- /dev/null +++ b/tests/ordered_prefetch.rs @@ -0,0 +1,286 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +#[path = "../examples/ordered_prefetch/reader.rs"] +mod reader; + +use reader::{Config, OrderedReader}; +use std::cell::RefCell; +use std::collections::BTreeMap; +use std::future::Future; +use std::io; +use std::pin::Pin; +use std::rc::Rc; +use std::sync::Arc; +use std::task::{Context, Poll, Waker}; + +#[derive(Default)] +struct Control { + submitted: Vec<(u64, usize)>, + completed: BTreeMap>>, + polls: BTreeMap, + dropped: Vec, +} + +struct ManualRead { + offset: u64, + control: Rc>, +} + +impl Future for ManualRead { + type Output = io::Result>; + + fn poll(self: Pin<&mut Self>, _: &mut Context<'_>) -> Poll { + let mut control = self.control.borrow_mut(); + *control.polls.entry(self.offset).or_default() += 1; + match control.completed.remove(&self.offset) { + Some(result) => Poll::Ready(result), + None => Poll::Pending, + } + } +} + +impl Drop for ManualRead { + fn drop(&mut self) { + self.control.borrow_mut().dropped.push(self.offset); + } +} + +fn config() -> Config { + Config { + offset: 0, + length: 32, + chunk: 4, + window: 3, + max_bytes: 12, + } +} + +fn source(control: &Rc>) -> impl FnMut(u64, usize) -> ManualRead + use<> { + let control = Rc::clone(control); + move |offset, len| { + control.borrow_mut().submitted.push((offset, len)); + ManualRead { + offset, + control: Rc::clone(&control), + } + } +} + +fn poll(future: &mut F) -> Poll { + Pin::new(future).poll(&mut Context::from_waker(Waker::noop())) +} + +#[test] +fn out_of_order_results_remain_ordered_and_no_refill_occurs_after_yield() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(control.borrow().submitted.is_empty(), "constructor must not start work"); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(8, Ok(vec![8; 4])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 4])); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("head must be ready") + }; + assert_eq!((chunk.offset, chunk.bytes), (0, vec![0; 4])); + assert_eq!(control.borrow().submitted.len(), 3, "slow consumer must not trigger refill"); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("cached second chunk") + }; + assert_eq!((chunk.offset, chunk.bytes), (4, vec![4; 4])); + assert_eq!(control.borrow().polls[&8], 2, "a completed tail must never be repolled"); +} + +#[test] +fn dropping_pending_next_keeps_head_handle_and_offset() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + let mut next = Box::pin(reader.next()); + assert!(poll(&mut next).is_pending()); + drop(next); + assert!(control.borrow().dropped.is_empty()); + control.borrow_mut().completed.insert(0, Ok(vec![1; 4])); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("same head must resume") + }; + assert_eq!(chunk.offset, 0); + assert_eq!(control.borrow().submitted.len(), 3); +} + +#[test] +fn ready_tail_remains_charged_to_the_same_logical_byte_window() { + let control = Rc::new(RefCell::new(Control::default())); + let mut cfg = config(); + cfg.max_bytes = 8; + let mut reader = OrderedReader::new(cfg, source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 4])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(control.borrow().submitted, [(0, 4), (4, 4)]); +} + +#[test] +fn early_tail_eof_cancels_later_slots_but_preserves_earlier_output() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 2])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert!(control.borrow().dropped.contains(&8)); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("short final chunk") + }; + assert_eq!((chunk.offset, chunk.bytes), (4, vec![4; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert_eq!(control.borrow().submitted.len(), 3); +} + +#[test] +fn error_is_delivered_at_its_ordered_position_then_reader_is_fused() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Err(io::Error::other("fault"))); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Err(_)))); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn empty_range_submits_nothing_and_zero_byte_eof_returns_no_chunk() { + let control = Rc::new(RefCell::new(Control::default())); + let mut cfg = config(); + cfg.length = 0; + let mut empty = OrderedReader::new(cfg, source(&control)).unwrap(); + assert!(matches!(poll(&mut Box::pin(empty.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().submitted.is_empty()); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(Vec::new())); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&4)); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn dropping_reader_drops_every_active_handle() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + drop(reader); + assert_eq!(control.borrow().dropped, [0, 4, 8]); +} + +#[test] +fn oversized_source_result_is_invalid_data_and_cancels_tail() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(vec![0; 5])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Err(e)) if e.kind() == io::ErrorKind::InvalidData)); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&4)); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn final_requested_slot_is_shortened_to_range_end_without_extra_reads() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new( + Config { + offset: 7, + length: 6, + ..config() + }, + source(&control), + ) + .unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(control.borrow().submitted, [(7, 4), (11, 2)]); + control.borrow_mut().completed.insert(7, Ok(vec![7; 4])); + control.borrow_mut().completed.insert(11, Ok(vec![11; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("final range slot") + }; + assert_eq!((chunk.offset, chunk.bytes), (11, vec![11; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert_eq!(control.borrow().submitted.len(), 2); +} + +#[test] +fn invalid_limits_and_overflow_fail_before_any_source_call() { + let control = Rc::new(RefCell::new(Control::default())); + let base = config(); + for cfg in [ + Config { chunk: 0, ..base }, + Config { window: 0, ..base }, + Config { window: 65, ..base }, + Config { max_bytes: 3, ..base }, + Config { + offset: u64::MAX, + ..base + }, + Config { + offset: i64::MAX as u64, + length: 1, + ..base + }, + Config { + chunk: usize::MAX, + ..base + }, + ] { + assert!(OrderedReader::new(cfg, source(&control)).is_err()); + } + assert!(control.borrow().submitted.is_empty()); +} + +#[test] +fn deferred_tail_assigned_bytes_is_repolled_even_while_head_waits_on_admission() { + let head_count = Arc::new(tokio::sync::Semaphore::new(1)); + let tail_count = Arc::new(tokio::sync::Semaphore::new(1)); + let bytes = Arc::new(tokio::sync::Semaphore::new(4)); + let held_head = Arc::clone(&head_count).try_acquire_owned().unwrap(); + let held_bytes = Arc::clone(&bytes).try_acquire_many_owned(4).unwrap(); + let submitted = Rc::new(RefCell::new(Vec::new())); + let accepted = Rc::clone(&submitted); + let mut reader = OrderedReader::new( + Config { + window: 2, + max_bytes: 8, + ..config() + }, + move |offset, len| { + let count = Arc::clone(if offset == 0 { &head_count } else { &tail_count }); + let bytes = Arc::clone(&bytes); + let accepted = Rc::clone(&accepted); + Box::pin(async move { + let _count = count.acquire_owned().await.unwrap(); + let _bytes = bytes.acquire_many_owned(len as u32).await.unwrap(); + accepted.borrow_mut().push(offset); + Ok(vec![offset as u8; len]) + }) + }, + ) + .unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + // Tokio assigns bytes to the already queued tail waiter before head can + // acquire its count permit. Polling only head now would strand those bytes. + drop(held_bytes); + drop(held_head); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(*submitted.borrow(), [4], "tail must progress without head completion"); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("head should now acquire released bytes") + }; + assert_eq!(chunk.offset, 0); + assert_eq!(*submitted.borrow(), [4, 0]); +} From bfc2a5de5698d4c427f6b781472347f8fe5c2651 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 03:38:06 +0800 Subject: [PATCH 14/22] test(prefetch): gate native contracts and document application prerequisites Require byte-exact native example execution and precise CLI validation errors. Record current-main application budget, fallback, probe and descriptor-cache integration boundaries separately from the completed driver work. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- .github/workflows/ci.yml | 2 + CHANGELOG.md | 5 ++ README.md | 3 ++ docs/optimization-status.md | 43 ++++++++++++--- docs/rustfs-integration.md | 81 ++++++++++++++++++++++++++++ scripts/test-ordered-prefetch-cli.sh | 53 ++++++++++++++++++ 6 files changed, 181 insertions(+), 6 deletions(-) create mode 100644 docs/rustfs-integration.md create mode 100644 scripts/test-ordered-prefetch-cli.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index bb95bb6..7640e86 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -64,6 +64,8 @@ jobs: exit 1 fi grep -q "DIRECT_OK direct_read_returns_exact_unaligned_ranges" test.out + - name: Ordered-prefetch contract smoke (real io_uring) + run: bash scripts/test-ordered-prefetch-cli.sh - name: Benchmark schema and correctness smoke run: bash scripts/test-benchmark-cli.sh - name: Instrumented benchmark schema and correctness smoke diff --git a/CHANGELOG.md b/CHANGELOG.md index 54f4d98..e4cf99f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,11 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- Explicit same-binary A/A calibration mode for the evidence runner, including + individual middle-leg drift gates and no candidate attribution in control runs. +- Example-only bounded ordered prefetch with deterministic backpressure, + cancellation and deferred-admission tests, plus a native CLI correctness gate. + This does not add a production streaming API or claim end-to-end speedups. - Opt-in `ReadLimits` for logical read size and driver-wide in-flight read-buffer bytes, with aligned allocation accounting and terminal-CQE ownership. - Opt-in capacity-aware shard selection for positioned reads; round-robin remains diff --git a/README.md b/README.md index 52260b5..d5de066 100644 --- a/README.md +++ b/README.md @@ -147,6 +147,9 @@ Benchmark configuration, CSV schema, timing boundaries, and performance gates are documented in [the benchmarking guide](docs/benchmarking.md). See [implementation and acceptance status](docs/optimization-status.md) for completed correctness work and the still-open performance/integration gates. +Application wiring has separate [RustFS integration prerequisites](docs/rustfs-integration.md). +The [ordered-prefetch example](docs/ordered-prefetch.md) is a bounded consumer +contract experiment, not a production streaming API or performance result. Linux only; on other hosts `cargo check` builds the empty stub. diff --git a/docs/optimization-status.md b/docs/optimization-status.md index 1768847..15ee87a 100644 --- a/docs/optimization-status.md +++ b/docs/optimization-status.md @@ -7,13 +7,13 @@ integration are separate gates; a checked code item does not close the roadmap. | Step | Implementation and review | Remaining acceptance | | --- | --- | --- | | 1.1–1.4: benchmark schema, timing, diagnostics, direct positive gate | Merged in [#15](https://github.com/rustfs/uring/pull/15), CI passed | Diagnostics overhead and real LocalIoBackend baseline | -| 1.5: ABBA evidence tooling | Implemented; 10 gate/cleanup tests pass | Stable calibration, valid comparison and application integration | -| 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | Hardening PR CI/merge; application-wide limits remain separate | -| 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR CI/merge | -| 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR CI/merge | -| 4.2: system-wide budgets and probe offload | Not implemented in this crate | Fresh application-path review and coordinated integration | +| 1.5: ABBA evidence tooling | Explicit same-binary calibration implemented/reviewed; 20 gate/cleanup tests pass | Stable native calibration, valid comparison and application integration | +| 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | PR merge; application-wide limits remain separate | +| 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | +| 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | +| 4.2: system-wide budgets and probe offload | Current application main audited; integration not implemented | [Application dependency, fallback and cross-disk budget wiring](rustfs-integration.md) | | 5: owned buffers and direct FD cache | Not implemented | Profile evidence, lease/pool lifetime design and cache invalidation integration | -| 6: ordered streaming prefetch | Not implemented | Consumer/backpressure contract and end-to-end evidence | +| 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests pass | Native example gate, production consumer contract, bitrot/S3 and performance evidence | | 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | ## Correctness and review evidence @@ -28,6 +28,35 @@ A Linux-target default/all-feature Clippy check and warning-denying rustdoc passed locally; cross-compilation is not additional native execution evidence. +The following documentation head `80a7af9` passed all three jobs in +[CI #53](https://github.com/rustfs/uring/actions/runs/35767797346), including +restricted/unrestricted Docker tests and both benchmark smoke modes. Later +follow-up commits need their own CI; the live PR and tracking issue record +final-head CI and merge status separately. + +## Follow-up evidence and examples + +`6927bb0` adds explicit same-binary calibration: identical executable content, +zero diagnostics interval on every leg, unchanged endpoint drift thresholds, +and a separate drift check for each middle leg against the endpoint mean. +Calibration never emits candidate-attribution fields; 20 Python tests pass, +including synthetic three-round execution and early-stop cases. These tests do +not establish a stable native calibration. + +`f135f1d` adds an example-local ordered reader, not a public streaming API. +Eleven portable tests pass, including actual Tokio semaphore reservation order, +cancelled `next`, EOF/error ordering, logical-byte accounting and slow-consumer +boundaries. The CLI uses an immutable fixture and positioned std oracle outside +the executor; the new CI smoke requires successful native io_uring byte checks +and bounds each child process duration. No production driver code or default +behavior changes in these follow-ups. + +Both additions and the current-main application integration plan received +independent review. Dedicated-host native execution is deferred while an existing +CI worker is active; the worker was not stopped and no performance run was made. +Portable checks and Linux-target compilation are not substitutes for the new +native CLI gate. Final-head CI/execution results are tracked in the issue/PR. + Independent reviews covered buffer/FD/permit ownership, global closure and count-stage waiters, cancellation races, final-owner routing, batch unwind and notifications, fairness/progress, test validity and documentation. Review found @@ -62,3 +91,5 @@ establish stable calibration without loosening gates after observing results. - [Driver turn fairness](driver-fairness.md) - [Explicit buffered batches](batch-reads.md) - [Measurement boundaries and evidence gates](benchmarking.md) +- [Current RustFS application integration prerequisites](rustfs-integration.md) +- [Ordered-prefetch contract experiment](ordered-prefetch.md) diff --git a/docs/rustfs-integration.md b/docs/rustfs-integration.md new file mode 100644 index 0000000..fd9bd6a --- /dev/null +++ b/docs/rustfs-integration.md @@ -0,0 +1,81 @@ +# RustFS application integration prerequisites + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647), +steps 4.2 and 5.2. This is a source-backed implementation plan, **not completed +application integration**. The reviewed RustFS main revision is +[`1880b42`](https://github.com/rustfs/rustfs/tree/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad) +(2026-09-23). Recheck the application head before implementation. + +## Current production boundary + +The application's [ecstore dependency](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/Cargo.toml#L230) +still uses registry `rustfs-uring = "0.2.2"`; it does not yet consume the new +`ReadLimits`, batch or shard-policy APIs. The [local backend](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L1087) +keeps io_uring opt-in, a depth of 128 per shard and a 128 MiB read chunk cap. +Configured shard counts multiply across disks; driver-local limits are not a +machine-wide budget. + +| Concern | Verified current behavior | Required integration | +| --- | --- | --- | +| Initialization | `LocalDisk::new` is async but calls synchronous probe/start | Bounded initialization offload plus explicit resource ownership | +| Teardown | Backend Drop already uses a blocking worker | Preserve it; account for retiring instances until real completion | +| Byte limits | Application has no `ReadLimits` wiring | Coordinate chunks, direct padding, per-driver allocation and fallback | +| Buffered descriptors | Existing FD cache with generation fencing | Reuse; do not replace invalidation with TTL-only behavior | +| Direct descriptors | Open/stat on each read; alignment is cached | Separate direct entries with all invalidation paths covered | +| Reclaim | Synchronous fadvise inside async read paths | Measure first; any offload must retain error and ordering semantics | + +Relevant paths: [backend lifecycle](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4186), +[async construction](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L5291), +[direct reads](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4561). + +## Required implementation order + +1. **Deliver the dependency and wire resource-aware chunks.** Use a released or + otherwise explicitly pinned uring version containing the new APIs. Validate + logical read size, actual aligned allocation charge and byte quota together; + the existing 128 MiB chunk cannot remain unconditional under a smaller quota. + Audit the [fallback path](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4676): + it currently retries any uring error through the std backend. Merely setting + a smaller driver limit would cause fallback, not enforce application-wide + memory admission. An application-level budget must also cover fallback and + retained results; changing fallback/error classification needs its own tests. +2. **Budget and offload initialization.** Bound startup/reconnect concurrency + before scheduling blocking probe work. Account driver slots, rings and OS + threads across all disks, including retiring/leaked instances. A timeout + stops waiting but cannot kill an already running blocking syscall. Do not + immediately release its capacity and admit unlimited replacement probes. + Current `ReadLimits` covers one driver only: use conservatively allocated + per-driver quotas or design an explicitly shared application admission layer. + The existing best-effort per-ring io-wq setting is not a process thread cap. +3. **Fix invalidation before adding direct cache hits.** + [Exact invalidation](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L3991) + currently builds a buffered-only key; invalidate both variants before + inserting direct entries. Preserve generation fencing, inode/length and + permission checks, alignment, FD capacity, prefix/volume invalidation and + open-to-insert race protection. Do not reuse a buffered descriptor as direct. + Specify permission freshness explicitly: today's direct path opens/checks on + every read, whereas buffered cache hits may skip some checks within the TTL. + Adopting its cache policy must not silently weaken the direct-read contract. +4. **Measure reclaim and validate the real backend.** If fadvise offload is + justified, use bounded blocking work and await its result; fire-and-forget + changes both errors and page-cache timing. Compare real LocalIoBackend cache + hit/miss, buffered/direct and fallback behavior after correctness gates, + rather than extrapolating from this crate's read microbenchmark. + +## Acceptance checklist (not yet completed) + +- [ ] Current application dependency contains the intended reviewed driver code. +- [ ] Below-quota/at-quota/over-quota and direct-padding cases; zero accidental + budget bypass through fallback; returned-result residency explicitly covered. +- [ ] Many-disk startup, reconnect storms, retiring instances, failed/timed-out + probes and shutdown retain correct capacity accounting and task responsiveness. +- [ ] Direct/buffered exact invalidation, heal same-path replacement, both rename + endpoints, delete, volume removal, open/insert races and permission failures. +- [ ] Real direct I/O executes; unavailable capability is not mistaken for a + passing direct test. Existing bitrot, EOF/range and error semantics preserved. +- [ ] Stable resource-isolated calibration precedes before/after comparisons; + CPU, tail latency, descriptors and memory remain within agreed bounds. + +The ordered-prefetch example in this repository cannot satisfy application +bitrot, quorum cancellation, stall policy or end-to-end stream acceptance. +Owned-buffer pooling and advanced ring modes remain evidence-gated separately. diff --git a/scripts/test-ordered-prefetch-cli.sh b/scripts/test-ordered-prefetch-cli.sh new file mode 100644 index 0000000..3596f71 --- /dev/null +++ b/scripts/test-ordered-prefetch-cli.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Copyright 2024 RustFS Team +# SPDX-License-Identifier: Apache-2.0 +set -euo pipefail +cd "$(dirname "$0")/.." + +if [[ "${BENCH_SKIP_BUILD:-0}" != 1 ]]; then + cargo build --locked --example ordered_prefetch --example streaming_bench +fi +bin="${BENCH_BIN_DIR:-${CARGO_TARGET_DIR:-target}/debug/examples}" +scratch=$(mktemp -d) +trap 'rm -rf -- "$scratch"' EXIT + +# Reuse the existing deterministic fixture generator/untimed byte verifier. +# This smoke checks contracts; no output is performance evidence. +BENCH_VERIFY=1 "$bin/streaming_bench" std_buffered "$scratch/data.bin" \ + 1048583 65536 1 4096 >/dev/null + +check_range() { + local expected=$1 + shift + local output + output=$(timeout --kill-after=5s 30s "$bin/ordered_prefetch" "$scratch/data.bin" "$@") + if [[ "$output" != "ORDERED_PREFETCH_OK bytes=$expected chunks="* ]]; then + echo "missing ordered-prefetch positive marker: $output" >&2 + exit 1 + fi +} + +# A non-aligned range spanning many windows, with a tighter byte than slot cap. +check_range 65543 17 65543 4097 4 8194 +# Early EOF must deliver only real bytes and terminate the ordered tail. +check_range 13 1048570 1000 7 8 56 +check_range 0 0 0 1024 4 4096 + +reject() { + local expected=$1 + shift + local code=0 + timeout --kill-after=5s 30s "$bin/ordered_prefetch" "$scratch/data.bin" "$@" \ + >"$scratch/rejected.stdout" 2>"$scratch/rejected.stderr" || code=$? + if [[ "$code" != 1 || -s "$scratch/rejected.stdout" ]] || + ! grep -Fqx "ordered_prefetch: $expected" "$scratch/rejected.stderr"; then + echo "invalid geometry did not return the normal CLI error (exit=$code)" >&2 + cat "$scratch/rejected.stderr" >&2 + exit 1 + fi +} +reject 'chunk must be 1..=8 MiB and window 1..=64' 0 100 0 4 4096 +reject 'chunk must be 1..=8 MiB and window 1..=64' 0 100 16 0 4096 +reject 'max_bytes must be at least chunk and at most 512 MiB' 0 100 16 4 0 +reject 'range end must fit signed file offsets' 18446744073709551615 2 16 4 4096 +echo "ordered-prefetch range, byte budget, EOF and validation smoke passed" From be66d8e9ca1ee17ca61dcf8d4dc14db04b72e775 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 06:32:09 +0800 Subject: [PATCH 15/22] test(driver): cover nonblocking and async shutdown contracts Exercise registered admission waiters, repeated shutdown requests, new-read rejection and clean conservation on real Linux pipes. Cover eager async shutdown abandonment and isolate the nonclean thread-exit fault in a child process without global environment mutation. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- tests/shutdown.rs | 263 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 263 insertions(+) create mode 100644 tests/shutdown.rs diff --git a/tests/shutdown.rs b/tests/shutdown.rs new file mode 100644 index 0000000..17d638a --- /dev/null +++ b/tests/shutdown.rs @@ -0,0 +1,263 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Native shutdown API tests. Expected io_uring restrictions print SKIP; only +//! an unrestricted Linux run proves the driver paths. Timeouts bound test +//! failures for ordinary pipes, not hung-kernel shutdown guarantees. +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io; +use std::os::fd::FromRawFd; +use std::pin::Pin; +use std::sync::Arc; +use std::time::Duration; + +use rustfs_uring::{ReadHandle, ReadLimits, StatsSnapshot, UringDriver}; + +fn driver_or_skip(name: &str, entries: u32, limits: ReadLimits) -> Option { + match UringDriver::probe_and_start_with_limits(entries, 2, limits) { + Ok(driver) => Some(driver), + Err(error) => { + assert!(error.is_expected_restriction(), "unexpected probe failure: {error}"); + eprintln!("SKIP {name}: restricted environment ({error})"); + None + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid output array; successful pipe2 creates two owned fds. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0, "create pending-read pipe"); + // SAFETY: transfer each fresh descriptor into exactly one File owner. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(5), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("pipe reads should reach the driver"); +} + +async fn wait_complete(driver: &UringDriver) { + tokio::time::timeout(Duration::from_secs(10), async { + while !driver.is_finished() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .expect("ordinary test driver threads should exit"); +} + +async fn assert_read_error(handle: ReadHandle) { + let result = tokio::time::timeout(Duration::from_secs(10), handle) + .await + .expect("shutdown must resolve a pending read handle"); + assert!(result.is_err(), "closed admission or canceled pipe must not return data"); +} + +fn assert_clean_conservation(stats: StatsSnapshot, submitted: u64) { + assert_eq!(stats.submitted, submitted); + assert_eq!(stats.in_flight, 0); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); +} + +async fn request_and_drain(name: &str, limits: ReadLimits) { + let Some(driver) = driver_or_skip(name, 2, limits) else { return }; + assert!(!driver.is_finished(), "idle but live driver threads are not finished"); + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + driver.request_shutdown(); // Idempotent even before a driver processes it. + for _ in 0..2 { + assert_read_error(driver.read_at(Arc::new(File::open("/dev/zero").expect("open zero fixture")), 0, 8)).await; + } + assert_read_error(first).await; + wait_complete(&driver).await; + driver.request_shutdown(); // Repeated request after thread exit is harmless. + assert_clean_conservation(driver.shutdown(), 1); +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_drains_default_driver_and_rejects_new_reads() { + request_and_drain("request_shutdown_drains_default_driver_and_rejects_new_reads", ReadLimits::default()).await; +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_drains_byte_limited_driver_and_rejects_new_reads() { + request_and_drain( + "request_shutdown_drains_byte_limited_driver_and_rejects_new_reads", + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) + .await; +} + +fn start_waiter(mut handle: ReadHandle) -> (tokio::task::JoinHandle>>, tokio::sync::oneshot::Receiver<()>) { + let (entered, ready) = tokio::sync::oneshot::channel(); + let mut entered = Some(entered); + let waiter = tokio::spawn(async move { + std::future::poll_fn(|cx| { + let result = Pin::new(&mut handle).poll(cx); + if result.is_pending() + && let Some(entered) = entered.take() + { + let _ = entered.send(()); + } + result + }) + .await + }); + (waiter, ready) +} + +async fn waiting_admission_is_woken(name: &str, limited: bool) { + let limits = if limited { + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + } + } else { + ReadLimits::default() + }; + let Some(driver) = driver_or_skip(name, 1, limits) else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); // Owner shard 0. + wait_submitted(&driver, 1).await; + let zero = Arc::new(File::open("/dev/zero").expect("open zero fixture")); + let waiting = if limited { + // Owner shard 1 has count capacity, but shared byte capacity is held. + driver.read_at(Arc::clone(&zero), 0, 8) + } else { + // Advance the cursor over shard 1 with an invalid inert read; the real + // waiter binds to saturated shard 0 without introducing another op. + assert_read_error(driver.read_at(Arc::clone(&zero), u64::MAX, 8)).await; + driver.read_at(Arc::clone(&zero), 0, 8) + }; + let (waiter, ready) = start_waiter(waiting); + tokio::time::timeout(Duration::from_secs(5), ready) + .await + .expect("waiter should be polled") + .expect("waiter must actually register Pending before shutdown"); + driver.request_shutdown(); + let result = tokio::time::timeout(Duration::from_secs(5), waiter) + .await + .expect("shutdown must wake the registered admission task") + .expect("admission waiter task must not panic"); + assert!(result.is_err()); + assert_read_error(first).await; + wait_complete(&driver).await; + assert_clean_conservation(driver.shutdown(), 1); +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_wakes_registered_count_waiter() { + waiting_admission_is_woken("request_shutdown_wakes_registered_count_waiter", false).await; +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_wakes_registered_cross_shard_byte_waiter() { + waiting_admission_is_woken("request_shutdown_wakes_registered_cross_shard_byte_waiter", true).await; +} + +#[cfg(feature = "tokio-runtime")] +#[tokio::test(flavor = "current_thread")] +async fn shutdown_async_is_send_and_returns_drained_stats_on_current_thread() { + fn require_send(value: T) -> T { + value + } + let Some(driver) = driver_or_skip( + "shutdown_async_is_send_and_returns_drained_stats_on_current_thread", + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) else { + return; + }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let stats = tokio::time::timeout(Duration::from_secs(10), require_send(driver.shutdown_async())) + .await + .expect("ordinary async shutdown should finish") + .expect("blocking join task should succeed"); + assert_read_error(first).await; + assert_clean_conservation(stats, 1); +} + +#[cfg(feature = "tokio-runtime")] +#[tokio::test(flavor = "current_thread")] +async fn dropping_unpolled_shutdown_async_future_does_not_abandon_cleanup() { + let Some(driver) = driver_or_skip( + "dropping_unpolled_shutdown_async_future_does_not_abandon_cleanup", + 2, + ReadLimits::default(), + ) else { + return; + }; + let (read, _write) = pipe(); + let weak = Arc::downgrade(&read); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let shutdown = driver.shutdown_async(); + drop(shutdown); // Never poll the returned future. + assert_read_error(first).await; + tokio::time::timeout(Duration::from_secs(5), async { + while weak.upgrade().is_some() { + tokio::task::yield_now().await; + } + }) + .await + .expect("detached shutdown still reclaims completed pending resources"); +} + +#[cfg(feature = "fault-injection")] +#[test] +fn is_finished_reports_thread_exit_not_clean_drain() { + const CHILD: &str = "RUSTFS_URING_SHUTDOWN_NONCLEAN_CHILD"; + const MARKER: &str = "SHUTDOWN_THREAD_EXIT_NONCLEAN_OK"; + let name = "is_finished_reports_thread_exit_not_clean_drain"; + if std::env::var_os(CHILD).is_some() { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .expect("build child runtime"); + runtime.block_on(async { + let Some(driver) = driver_or_skip(name, 2, ReadLimits::default()) else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + assert_read_error(first).await; + wait_complete(&driver).await; + assert_eq!(driver.stats().in_flight, 1, "fault seam must leave a leaked pending entry"); + assert_eq!(driver.shutdown().in_flight, 1); + eprintln!("{MARKER}"); + }); + return; + } + let Some(driver) = driver_or_skip(name, 2, ReadLimits::default()) else { return }; + driver.shutdown(); + // Only the child inherits injected drain settings; no process-wide env + // mutation can leak into this test binary's concurrently running tests. + let output = std::process::Command::new(std::env::current_exe().expect("locate test executable")) + .args(["--exact", name, "--nocapture", "--test-threads=1"]) + .env(CHILD, "1") + .env("RUSTFS_URING_FAULT_STUCK_DRAIN", "1") + .env("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400") + .output() + .expect("run isolated nonclean-shutdown test"); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(output.status.success(), "child failed: {stderr}"); + assert!(stderr.contains(MARKER), "child did not execute the injected nonclean path: {stderr}"); +} From c7bb321d7c22b963fe30417e09295a50275b8f14 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 06:34:40 +0800 Subject: [PATCH 16/22] feat(driver): expose shutdown requests and eager Tokio handoff Close count and byte admission before every shutdown request returns, expose advisory thread completion, and reuse the request path for consuming shutdown and Drop. Add an opt-in tokio-runtime adapter that transfers driver ownership before its returned future is polled, with deterministic private lifecycle tests. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- Cargo.toml | 3 + src/driver.rs | 268 +++++++++++++++++++++++++++++++++++++++++++++++--- 2 files changed, 258 insertions(+), 13 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index e91d8ca..39121fc 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,6 +36,9 @@ fault-injection = [] # Opt-in sampled timing. No timestamps, histogram storage, or tracing allocations # are compiled into the default driver. diagnostics = [] +# Optional Tokio blocking-pool shutdown adapter. Default builds only require +# Tokio's synchronization primitives and can drive reads from other executors. +tokio-runtime = ["tokio/rt"] [dependencies] # tracing for driver diagnostics: unifies with the RustFS tracing pipeline for diff --git a/src/driver.rs b/src/driver.rs index 571a51c..76d8a3e 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -1519,20 +1519,80 @@ impl UringDriver { shard.wake_efd.signal(); } - /// Stop accepting work, cancel all in-flight ops, drain every ring to - /// `in_flight == 0`, then join each driver thread. Only after that is a ring - /// dropped/unmapped — the shutdown ordering P2 requires, per shard. + /// Close admission and ask every shard to cancel/drain, without joining + /// driver threads or waiting for pending reads to complete. /// - /// Shards are asked to stop first and joined afterwards, so their bounded - /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. - pub fn shutdown(mut self) -> StatsSnapshot { + /// Count and byte waiters are woken with an admission error. An already + /// kernel-submitted operation still owns its buffer until its read CQE, or retains + /// it through the leak-over-UAF escape path. Concurrent and repeated calls + /// are safe; every call closes all admission before returning. + pub fn request_shutdown(&self) { if let Some(admission) = &self.byte_admission { admission.close(); } + // Also close count-only admission when byte limits are disabled. Do not + // use a once flag: a concurrent caller must not return before this work + // is complete merely because another caller started the request. + for shard in &self.shards { + shard.sem.close(); + } for shard in &self.shards { let _ = shard.tx.send(Msg::Shutdown); shard.wake_efd.signal(); } + } + + /// Advisory query: whether every driver thread's join handle reports + /// finished (or has already been joined). This does not join threads or + /// imply a clean drain; use [`Self::shutdown`] to join and inspect its stats. + /// + /// Inspect [`Self::stats`]: nonzero `in_flight` can remain after the bounded + /// drain leaks kernel-owned resources and the driver threads finish. + /// Like [`std::thread::JoinHandle::is_finished`], this can become true just + /// before final thread teardown completes; it is not proof of a completed + /// synchronous join and does not impose a deadline on a blocked syscall. + pub fn is_finished(&self) -> bool { + self.shards + .iter() + .all(|shard| shard.handle.as_ref().is_none_or(JoinHandle::is_finished)) + } + + /// Request shutdown immediately and move the consuming shutdown/join onto + /// the current Tokio runtime's blocking pool. Await the returned future for + /// the final snapshot, or an error if the blocking task fails to join. + /// + /// Available with `tokio-runtime`. This is deliberately not an `async fn`: + /// the driver is handed off when this method is called, before the returned + /// future is polled. Dropping that future, even unpolled, only detaches the + /// join handle; it does not cancel started blocking work or drop the driver + /// on the caller during normal runtime operation. + /// + /// Runtime shutdown may reject or discard blocking work and synchronously + /// drop its captured driver. Scheduling is not a hard cleanup deadline, and + /// neither this adapter nor a timeout can kill a hung kernel syscall. A + /// successful snapshot may still report leaked reads via nonzero `in_flight`. + /// + /// # Panics + /// + /// Panics if called without an entered Tokio runtime, matching + /// [`tokio::runtime::Handle::current`]. Unwinding then drops the driver using + /// its synchronous cleanup path. + #[cfg(feature = "tokio-runtime")] + pub fn shutdown_async(self) -> impl Future> + Send { + let runtime = tokio::runtime::Handle::current(); + self.request_shutdown(); + let shutdown = runtime.spawn_blocking(move || self.shutdown()); + async move { shutdown.await.map_err(io::Error::other) } + } + + /// Stop accepting work, cancel/drain every ring, and join the driver threads + /// synchronously. A clean drain returns `in_flight == 0`; a bounded-drain + /// escape can return a nonzero count with the ring and buffers still leaked. + /// + /// Shards are asked to stop first and joined afterwards, so their bounded + /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. + pub fn shutdown(mut self) -> StatsSnapshot { + self.request_shutdown(); for shard in &mut self.shards { shard.join(); } @@ -1554,17 +1614,11 @@ impl UringDriver { impl Drop for UringDriver { fn drop(&mut self) { - if let Some(admission) = &self.byte_admission { - admission.close(); - } // Ask every shard to stop before joining any of them, so their bounded // drains overlap. Dropping the `Vec` would instead run each // `Shard::drop` in turn, serializing up to `shards * DRAIN_TIMEOUT` on a // hung device. `Shard::join` is idempotent, so the later drops are no-ops. - for shard in &self.shards { - let _ = shard.tx.send(Msg::Shutdown); - shard.wake_efd.signal(); - } + self.request_shutdown(); for shard in &mut self.shards { shard.join(); } @@ -2365,6 +2419,194 @@ fn drive( } } +#[cfg(test)] +mod shutdown_request_tests { + use super::*; + + fn mock_driver(shards: usize, bytes: Option) -> (UringDriver, Vec>) { + let byte_admission = bytes.map(|limit| Arc::new(ByteAdmission::new(limit))); + let mut receivers = Vec::new(); + let shards = (0..shards) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(1)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().expect("mock wake fd")), + } + }) + .collect(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: bytes, + }, + shard_policy: ShardPolicy::RoundRobin, + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) + } + + fn poll_acquire(future: &mut AcquireFut) -> Poll> { + future.as_mut().poll(&mut Context::from_waker(std::task::Waker::noop())) + } + + #[test] + fn request_shutdown_closes_count_only_admission_before_returning() { + let (driver, messages) = mock_driver(1, None); + let count = Arc::clone(&driver.shards[0].sem); + let held = try_read_permits(&count, None, 1).expect("occupy mock shard"); + let mut waiter = acquire_read_permits(count.clone(), None, 1); + assert!(poll_acquire(&mut waiter).is_pending()); + driver.request_shutdown(); + assert!(count.is_closed()); + assert!(matches!(poll_acquire(&mut waiter), Poll::Ready(Err(_)))); + assert!(matches!(messages[0].try_recv(), Ok(Msg::Shutdown))); + assert_eq!(count.available_permits(), 0, "request must not reclaim accepted work"); + drop(held); + } + + #[test] + fn request_shutdown_wakes_both_admission_stages_without_reclaiming_buffers() { + let (driver, _messages) = mock_driver(2, Some(8)); + let bytes = Arc::clone(&driver.byte_admission.as_ref().expect("byte budget").bytes); + let held = try_read_permits(&driver.shards[0].sem, Some(&bytes), 8).expect("occupy byte budget"); + let mut count_waiter = acquire_read_permits(driver.shards[0].sem.clone(), Some(bytes.clone()), 8); + let mut byte_waiter = acquire_read_permits(driver.shards[1].sem.clone(), Some(bytes.clone()), 8); + assert!(poll_acquire(&mut count_waiter).is_pending()); + assert!(poll_acquire(&mut byte_waiter).is_pending()); + driver.request_shutdown(); + assert!(matches!(poll_acquire(&mut count_waiter), Poll::Ready(Err(_)))); + assert!(matches!(poll_acquire(&mut byte_waiter), Poll::Ready(Err(_)))); + assert!(bytes.is_closed()); + assert_eq!(bytes.available_permits(), 0, "in-flight reservation remains owned"); + drop(held); + } + + #[test] + fn concurrent_shutdown_requests_each_return_with_all_admission_closed() { + for bytes in [None, Some(8)] { + let (driver, _messages) = mock_driver(2, bytes); + let barrier = std::sync::Barrier::new(8); + std::thread::scope(|scope| { + for _ in 0..8 { + scope.spawn(|| { + barrier.wait(); + for _ in 0..2 { + driver.request_shutdown(); + assert!(driver.shards.iter().all(|shard| shard.sem.is_closed())); + if let Some(admission) = &driver.byte_admission { + assert!(admission.bytes.is_closed()); + } + } + }); + } + }); + } + } + + #[tokio::test] + async fn finished_observes_thread_exit_not_request_or_clean_drain() { + let (mut driver, _messages) = mock_driver(1, None); + let (entered_tx, entered_rx) = oneshot::channel(); + let (exit_tx, exit_rx) = mpsc::channel(); + driver.shards[0].handle = Some(std::thread::spawn(move || { + entered_tx.send(()).expect("observe mock thread entry"); + exit_rx.recv_timeout(Duration::from_secs(10)).expect("allow mock thread exit"); + })); + entered_rx.await.expect("mock thread started"); + assert!(!driver.is_finished()); + driver.request_shutdown(); + assert!(!driver.is_finished(), "request must return while a thread is still active"); + // A synthetic leak counter illustrates that thread completion alone is + // not the clean-drain condition. No kernel ever touches this mock state. + driver.shards[0].stats.in_flight.store(1, Ordering::SeqCst); + exit_tx.send(()).expect("finish mock driver"); + tokio::time::timeout(Duration::from_secs(3), async { + while !driver.is_finished() { + tokio::task::yield_now().await; + } + }) + .await + .expect("finished thread must become observable"); + assert_eq!(driver.stats().in_flight, 1); + driver.shards[0].join(); + assert!(driver.is_finished(), "joined handles also report finished"); + } + + #[cfg(feature = "tokio-runtime")] + #[test] + fn async_shutdown_transfers_driver_before_poll_and_unpolled_drop_does_not_join_caller() { + let (mut driver, _messages) = mock_driver(1, None); + let count = driver.shards[0].sem.clone(); + let retired = Arc::downgrade(&driver.shards[0].stats); + let (driver_exit_tx, driver_exit_rx) = mpsc::channel(); + driver.shards[0].handle = Some(std::thread::spawn(move || { + driver_exit_rx + .recv_timeout(Duration::from_secs(10)) + .expect("release mock driver thread"); + })); + let (pool_release_tx, pool_release_rx) = mpsc::channel(); + let (returned_tx, returned_rx) = mpsc::channel(); + let (runtime_exit_tx, runtime_exit_rx) = mpsc::channel(); + let caller = std::thread::spawn(move || { + let runtime = tokio::runtime::Builder::new_current_thread() + .max_blocking_threads(1) + .build() + .expect("shutdown adapter runtime"); + let (pool_entered_tx, pool_entered_rx) = mpsc::channel(); + let _occupied = runtime.spawn_blocking(move || { + pool_entered_tx.send(()).expect("observe occupied blocking pool"); + pool_release_rx + .recv_timeout(Duration::from_secs(10)) + .expect("release blocking pool"); + }); + pool_entered_rx + .recv_timeout(Duration::from_secs(10)) + .expect("pool task started"); + { + let _entered = runtime.enter(); + let future = driver.shutdown_async(); + // The shutdown closure cannot start yet. Admission must already + // be closed at method return, not at the returned future's poll. + let closed_before_poll = count.is_closed(); + drop(future); + returned_tx.send(closed_before_poll).expect("observe unpolled drop return"); + } + runtime_exit_rx + .recv_timeout(Duration::from_secs(10)) + .expect("keep runtime alive until cleanup"); + }); + + let returned = returned_rx.recv_timeout(Duration::from_secs(3)); + // Release every gate before assertions, including on a broken adapter + // that joined on the caller and caused the observation to time out. + pool_release_tx.send(()).expect("allow shutdown closure to run"); + driver_exit_tx.send(()).expect("allow driver join to finish"); + let deadline = Instant::now() + Duration::from_secs(3); + while retired.strong_count() != 0 && Instant::now() < deadline { + std::thread::yield_now(); + } + let cleaned_up = retired.strong_count() == 0; + runtime_exit_tx.send(()).expect("allow runtime teardown"); + caller.join().expect("shutdown caller thread"); + assert!(returned.expect("unpolled drop must return without waiting for the mock driver")); + assert!(cleaned_up, "detached shutdown must still consume and drop the driver"); + } +} + #[cfg(test)] mod cancellation_efficiency_tests { use super::*; From a942ac1bc549c3b5ac3f907c993bac762db9c9a8 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 06:45:10 +0800 Subject: [PATCH 17/22] docs(driver): define shutdown ownership and acceptance boundaries Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- CHANGELOG.md | 3 ++ README.md | 2 + docs/optimization-status.md | 28 +++++++++++-- docs/shutdown.md | 80 +++++++++++++++++++++++++++++++++++++ 4 files changed, 110 insertions(+), 3 deletions(-) create mode 100644 docs/shutdown.md diff --git a/CHANGELOG.md b/CHANGELOG.md index e4cf99f..30a2eeb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,9 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- Non-joining `request_shutdown`, advisory `is_finished`, and a default-off + `tokio-runtime` feature for eager consuming `shutdown_async` ownership handoff. + Synchronous `shutdown`/`Drop` and bounded-drain leak guarantees remain explicit. - Explicit same-binary A/A calibration mode for the evidence runner, including individual middle-leg drift gates and no candidate attribution in control runs. - Example-only bounded ordered prefetch with deterministic backpressure, diff --git a/README.md b/README.md index d5de066..c2e8789 100644 --- a/README.md +++ b/README.md @@ -49,6 +49,8 @@ assert_eq!(snapshot.delivered + snapshot.orphan_reclaimed, snapshot.submitted); - `probe_and_start_sharded(entries, shards)` — several independent rings per disk (each ring caps at one core's memory bandwidth for cache-hit reads); `probe_and_start(entries)` equals `..._sharded(entries, 1)`. - `probe_and_start_with_limits(entries, shards, ReadLimits { max_read_len, max_in_flight_bytes })` — optional logical read-size and driver-wide read-buffer limits. Both fields default to `None`, preserving existing constructor behavior. - `with_shard_policy(ShardPolicy::CapacityAware)` — opt-in capacity-aware routing for positioned reads. Constructors keep `ShardPolicy::RoundRobin` by default. +- `request_shutdown()` — close admission and request cancellation/drain without joining; `is_finished()` reports advisory thread completion, not a clean drain. +- `shutdown_async()` — with the default-off `tokio-runtime` feature, transfer consuming cleanup to Tokio's blocking pool at method call time. See [shutdown ownership and runtime boundaries](docs/shutdown.md). ### Shard selection diff --git a/docs/optimization-status.md b/docs/optimization-status.md index 15ee87a..dc2b679 100644 --- a/docs/optimization-status.md +++ b/docs/optimization-status.md @@ -9,11 +9,12 @@ integration are separate gates; a checked code item does not close the roadmap. | 1.1–1.4: benchmark schema, timing, diagnostics, direct positive gate | Merged in [#15](https://github.com/rustfs/uring/pull/15), CI passed | Diagnostics overhead and real LocalIoBackend baseline | | 1.5: ABBA evidence tooling | Explicit same-binary calibration implemented/reviewed; 20 gate/cleanup tests pass | Stable native calibration, valid comparison and application integration | | 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | PR merge; application-wide limits remain separate | +| 2.4 API follow-up: shutdown control and runtime adapter | Implemented with mock-thread and native regression tests; two independent code reviews passed | New-head native CI and PR merge | | 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | | 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | -| 4.2: system-wide budgets and probe offload | Current application main audited; integration not implemented | [Application dependency, fallback and cross-disk budget wiring](rustfs-integration.md) | -| 5: owned buffers and direct FD cache | Not implemented | Profile evidence, lease/pool lifetime design and cache invalidation integration | -| 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests pass | Native example gate, production consumer contract, bitrot/S3 and performance evidence | +| 4.2: system-wide budgets and probe offload | Application probe offload, driver-thread budget and logical chunk limit implemented/reviewed in draft PRs #8072/#8074/#8076 | Full CI/merge; physical byte/result/io-wq budgets and [dependency wiring](rustfs-integration.md) remain open | +| 5: owned buffers and direct FD cache | Dual-mode exact invalidation implemented/reviewed in application draft PR #8075; direct caching and owned buffers not enabled | Full CI/merge, profile evidence, lease/pool lifetime and complete invalidation integration | +| 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests and native CLI CI #54 passed | Production consumer contract, bitrot/S3 and performance evidence | | 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | ## Correctness and review evidence @@ -36,6 +37,26 @@ final-head CI and merge status separately. ## Follow-up evidence and examples +Application work is tracked in [RustFS #8072](https://github.com/rustfs/rustfs/pull/8072), +[#8074](https://github.com/rustfs/rustfs/pull/8074), +[#8075](https://github.com/rustfs/rustfs/pull/8075), and +[#8076](https://github.com/rustfs/rustfs/pull/8076), not merged into application +main. The first two passed their native io_uring tests but failed workspace and +full E2E gates respectively; the latter two still have checks running. These +partial results do not establish successful application integration. + +`c7bb321` adds non-joining shutdown requests, advisory thread completion and the +default-off Tokio shutdown adapter. Count and byte admission close before a +request returns; the adapter transfers ownership before its result future is +polled. The [shutdown contract](shutdown.md) distinguishes thread completion, +successful join, clean drain, and runtime-shutdown failure boundaries. Tests in +`be66d8e` include real pending reads and an isolated nonclean-drain subprocess. +Two independent code reviews found no blocking issue. Linux-target default +all-target checking, all-feature Clippy, both feature-state rustdoc builds with +warnings denied, formatting and diff checks pass locally. These are compilation +checks, not native execution. New-head native CI is pending; previous CI does not +validate these additions. No throughput or hard cleanup deadline is claimed. + `6927bb0` adds explicit same-binary calibration: identical executable content, zero diagnostics interval on every leg, unchanged endpoint drift thresholds, and a separate drift check for each middle leg against the endpoint mean. @@ -86,6 +107,7 @@ establish stable calibration without loosening gates after observing results. ## Contract references - [Public read and admission contracts](../README.md) +- [Shutdown ownership and Tokio adapter](shutdown.md) - [Completion recovery and direct integrity](fault-recovery.md) - [Cancellation and eventfd behavior](cancellation-efficiency.md) - [Driver turn fairness](driver-fairness.md) diff --git a/docs/shutdown.md b/docs/shutdown.md new file mode 100644 index 0000000..d71cb65 --- /dev/null +++ b/docs/shutdown.md @@ -0,0 +1,80 @@ +# Shutdown ownership and runtime integration + +`request_shutdown`, `is_finished`, `shutdown`, and the optional `shutdown_async` +have different contracts. None can kill a blocked kernel syscall or promise a +hard cleanup deadline. + +## Request, observe, join + +`request_shutdown(&self)` closes every shard's count admission and the shared +byte admission, if configured, before returning. It wakes admission waiters and +sends each driver thread a shutdown request. It does not join or wait for reads. +Repeated and concurrent calls are safe; they can send repeated notifications. +Call this at lifecycle transitions, not in a polling loop. + +Valid reads, including zero-length reads, started after the request cannot gain +admission. Invalid inputs retain their validation errors. A read racing the +request may already hold admission: shutdown still +owns and settles its queued or submitted state. Neither admission closure nor +dropping a read handle releases a kernel-owned buffer early. Submitted buffers, +file descriptors and permits remain owned until the terminal read CQE, or are +retained by the leak-over-use-after-free escape path. + +`is_finished()` is an advisory query over the driver thread join handles. It is +not a join, a synchronization barrier, or proof of a clean drain. It can become +true just before final thread teardown, and can be true with nonzero +`stats().in_flight` after a bounded-drain leak. Do not return process-wide memory +credits merely because it is true. + +`shutdown(self)` requests all shards to stop, then synchronously joins them and +returns the final snapshot. `Drop` uses the same synchronous cleanup ordering. +The bounded drain only advances while the driver loop runs: a syscall that does +not return can also prevent the join from returning. A returned snapshot with +nonzero `in_flight` records retained kernel-owned resources, not clean recovery. + +## Optional Tokio adapter + +Enable `features = ["tokio-runtime"]` for `shutdown_async(self)`. Default builds +still require only Tokio synchronization support; read handles can be driven by +other executors. The adapter does not move each read onto Tokio's blocking pool. + +Call the method from an entered, live Tokio runtime. At **method call time**, it +requests shutdown and moves the driver into one blocking-pool task that performs +the consuming synchronous shutdown. This is deliberately an ordinary function +returning a `Send` future, not an `async fn` whose body starts at first poll. + +```rust +// `driver` is an exclusively owned UringDriver; tokio-runtime is enabled. +let shutdown = driver.shutdown_async(); // ownership transferred here +let stats = shutdown.await?; +if stats.in_flight != 0 { + // Report degraded cleanup and retain any external memory reservation. +} +``` + +Dropping or timing out the returned future, even before its first poll, detaches +the join handle during normal runtime operation; cleanup continues. It does not +abort a started blocking task or recover leaked buffers. Keep the runtime alive +until cleanup completes if that completion is required. Concurrent driver +retirement still needs application-level limits: the runtime's blocking queue +is not an admission budget. + +Calling without an entered runtime panics and drops the driver through its +synchronous cleanup path. A runtime shutting down may reject or discard queued +blocking work and synchronously drop its captured driver. Thus the adapter does +not make runtime teardown, panic unwinding, or arbitrary `Drop` nonblocking. +An `Ok` snapshot means the blocking task joined successfully, not necessarily +that `in_flight` is zero. A join error is returned as an `io::Error`. + +## Validation boundaries + +Private mock-thread tests exercise immediate count/byte closure, concurrent +requests, advisory completion, and eager ownership handoff with an occupied +blocking pool. Native tests exercise pending pipe reads, registered count/byte +waiters, current-thread Tokio cleanup, and dropped unpolled shutdown futures. +An isolated fault-injection subprocess verifies that finished threads can retain +nonzero in-flight resources. These are correctness tests, not performance or +real hung-device recovery evidence. + +See [completion recovery](fault-recovery.md), [integration prerequisites](rustfs-integration.md), +and [acceptance status](optimization-status.md) for the remaining gates. From 1dfd81b15ae0c63ee718d17ba50e344daa47f4f5 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 06:49:01 +0800 Subject: [PATCH 18/22] docs(driver): record native shutdown regression evidence Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/optimization-status.md | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/docs/optimization-status.md b/docs/optimization-status.md index dc2b679..ab82324 100644 --- a/docs/optimization-status.md +++ b/docs/optimization-status.md @@ -9,7 +9,7 @@ integration are separate gates; a checked code item does not close the roadmap. | 1.1–1.4: benchmark schema, timing, diagnostics, direct positive gate | Merged in [#15](https://github.com/rustfs/uring/pull/15), CI passed | Diagnostics overhead and real LocalIoBackend baseline | | 1.5: ABBA evidence tooling | Explicit same-binary calibration implemented/reviewed; 20 gate/cleanup tests pass | Stable native calibration, valid comparison and application integration | | 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | PR merge; application-wide limits remain separate | -| 2.4 API follow-up: shutdown control and runtime adapter | Implemented with mock-thread and native regression tests; two independent code reviews passed | New-head native CI and PR merge | +| 2.4 API follow-up: shutdown control and runtime adapter | Implemented/reviewed; native CI #55 passed at `a942ac1` | PR merge; no hard cleanup deadline or performance claim | | 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | | 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | | 4.2: system-wide budgets and probe offload | Application probe offload, driver-thread budget and logical chunk limit implemented/reviewed in draft PRs #8072/#8074/#8076 | Full CI/merge; physical byte/result/io-wq budgets and [dependency wiring](rustfs-integration.md) remain open | @@ -54,8 +54,15 @@ successful join, clean drain, and runtime-shutdown failure boundaries. Tests in Two independent code reviews found no blocking issue. Linux-target default all-target checking, all-feature Clippy, both feature-state rustdoc builds with warnings denied, formatting and diff checks pass locally. These are compilation -checks, not native execution. New-head native CI is pending; previous CI does not -validate these additions. No throughput or hard cleanup deadline is claimed. +checks, not native execution. Subsequently, [CI #55](https://github.com/rustfs/uring/actions/runs/35794075169) +at `a942ac1` passed all three jobs: 127 all-feature native tests (76 unit, 4 +admission, 21 read/cancel, 6 fault, 11 prefetch, 2 shard-policy, 7 shutdown), the +mandatory O_DIRECT marker, ordered-prefetch and both benchmark CLI smokes, lint +and 20 Python tests. Docker also passed restricted degradation and unrestricted +native legs; its fault-injection-only build does not enable `tokio-runtime`. +Later documentation-only commits do not change this tested implementation; +their own workflow status remains separate. No throughput or hard cleanup +deadline is claimed. `6927bb0` adds explicit same-binary calibration: identical executable content, zero diagnostics interval on every leg, unchanged endpoint drift thresholds, From 7e94f36406ef6673a00e39c90889c1a2739ebc34 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 07:04:03 +0800 Subject: [PATCH 19/22] test(driver): cover shared whole-driver read quota ownership Validate strict quota geometry, refusal classification, independent shutdown, deferred-handle retention and clean recycling. Cover capacity-aware batch ownership and isolate permanent whole-quota retention after a leaked one-byte read in a fault child. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- tests/shared_budget.rs | 384 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 384 insertions(+) create mode 100644 tests/shared_budget.rs diff --git a/tests/shared_budget.rs b/tests/shared_budget.rs new file mode 100644 index 0000000..fd96bb5 --- /dev/null +++ b/tests/shared_budget.rs @@ -0,0 +1,384 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Whole-driver read-quota reservations. Kernel tests print SKIP only for +//! expected restrictions; configuration tests do not require io_uring setup. +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io::{self, Write}; +use std::os::fd::FromRawFd; +use std::pin::Pin; +use std::sync::Arc; +use std::task::{Context, Waker}; +use std::time::Duration; + +use rustfs_uring::{ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, SharedReadBudget, UringDriver}; + +fn limits(bytes: usize) -> ReadLimits { + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(bytes), + } +} + +fn assert_setup_kind(result: Result, kind: io::ErrorKind) { + match result { + Err(ProbeFailure::Setup(error)) => assert_eq!(error.kind(), kind), + Err(other) => panic!("wrong failure boundary: {other}"), + Ok(driver) => { + driver.shutdown(); + panic!("invalid or unavailable quota was admitted"); + } + } +} + +#[test] +fn shared_budget_accepts_large_usize_capacity_without_allocating_read_buffers() { + assert!(matches!(SharedReadBudget::new(0), Err(error) if error.kind() == io::ErrorKind::InvalidInput)); + for total in [1, 16, usize::MAX] { + let budget = SharedReadBudget::new(total).expect("positive accounting capacity"); + let clone = budget.clone(); + assert_eq!(budget.capacity(), total); + assert_eq!(clone.capacity(), total); + assert_eq!(budget.available(), total); + drop(budget); + assert_eq!(clone.available(), total); + } +} + +#[test] +fn shared_constructor_rejects_missing_zero_or_oversized_local_quota_before_probe() { + let budget = SharedReadBudget::new(8).expect("test shared budget"); + for local in [ReadLimits::default(), limits(0), limits(9)] { + assert_setup_kind( + UringDriver::probe_and_start_with_shared_budget(1, 1, local, &budget), + io::ErrorKind::InvalidInput, + ); + assert_eq!(budget.available(), 8, "invalid configuration must not retain quota"); + } + let largest = SharedReadBudget::new(usize::MAX).expect("global capacity has no u32 limit"); + assert_setup_kind( + UringDriver::probe_and_start_with_shared_budget(1, 1, limits(tokio::sync::Semaphore::MAX_PERMITS + 1), &largest), + io::ErrorKind::InvalidInput, + ); + assert_eq!(largest.available(), usize::MAX); +} + +fn driver_or_skip(name: &str, entries: u32, shards: usize, bytes: usize, budget: &SharedReadBudget) -> Option { + let available_before = budget.available(); + match UringDriver::probe_and_start_with_shared_budget(entries, shards, limits(bytes), budget) { + Ok(driver) => Some(driver), + Err(error) => { + // Each test owns this pool; no concurrent constructor changes it. + // A restricted-kernel failure must refund the tentative whole quota. + assert_eq!(budget.available(), available_before, "failed probe retained a whole-driver quota"); + assert!(error.is_expected_restriction(), "unexpected shared-budget probe failure: {error}"); + eprintln!("SKIP {name}: restricted environment ({error})"); + None + } + } +} + +fn assert_would_block(result: Result) { + match result { + Err(error) => { + assert!( + !error.is_expected_restriction(), + "quota exhaustion must not be negatively cached as a kernel restriction" + ); + assert_setup_kind(Err(error), io::ErrorKind::WouldBlock); + } + Ok(driver) => { + driver.shutdown(); + panic!("exhausted shared quota must refuse construction"); + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid two-fd output array; success creates two owned descriptors. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0); + // SAFETY: each fresh descriptor transfers into exactly one File owner. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(5), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("pipe read enters driver pending table"); +} + +async fn wait_finished(driver: &UringDriver) { + tokio::time::timeout(Duration::from_secs(10), async { + while !driver.is_finished() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .expect("ordinary test driver threads should finish"); +} + +async fn read_result(handle: ReadHandle) -> io::Result> { + tokio::time::timeout(Duration::from_secs(10), handle) + .await + .expect("test read should resolve") +} + +#[test] +fn shared_quota_is_reserved_once_per_driver_and_clean_shutdown_recycles_it() { + let budget = SharedReadBudget::new(24).expect("three whole-driver quotas"); + let clone = budget.clone(); + let Some(first) = driver_or_skip( + "shared_quota_is_reserved_once_per_driver_and_clean_shutdown_recycles_it", + 1, + 2, + 8, + &budget, + ) else { + return; + }; + assert_eq!(clone.available(), 16, "two shards share one driver quota, not two reservations"); + let second = driver_or_skip("second shared driver", 1, 1, 8, &clone).expect("kernel precondition already proved"); + assert_eq!(budget.available(), 8); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(9), &budget)); + assert_eq!(budget.available(), 8, "refused reservation must not consume partial capacity"); + let third = driver_or_skip("exact remaining shared quota", 1, 1, 8, &budget).expect("remaining quota fits"); + assert_eq!(budget.available(), 0); + assert_eq!(first.shutdown().in_flight, 0); + assert_eq!(budget.available(), 8); + assert_eq!(second.shutdown().in_flight, 0); + assert_eq!(budget.available(), 16); + assert_eq!(third.shutdown().in_flight, 0); + assert_eq!(budget.available(), 24); +} + +#[cfg(target_pointer_width = "64")] +#[test] +fn driver_quota_larger_than_u32_is_not_truncated_to_per_read_permit_width() { + let quota = u32::MAX as usize + 1; + let budget = SharedReadBudget::new(quota * 2).expect("large accounting capacity"); + let Some(driver) = driver_or_skip( + "driver_quota_larger_than_u32_is_not_truncated_to_per_read_permit_width", + 1, + 1, + quota, + &budget, + ) else { + return; + }; + // Startup only: no application read or multi-gigabyte buffer is allocated. + assert_eq!(driver.stats().submitted, 0); + assert_eq!(budget.available(), quota); + driver.shutdown(); + assert_eq!(budget.available(), quota * 2); +} + +#[tokio::test(flavor = "current_thread")] +async fn shutting_down_one_driver_does_not_close_another_drivers_admission() { + let budget = SharedReadBudget::new(16).expect("two driver quotas"); + let Some(first) = driver_or_skip("shutting_down_one_driver_does_not_close_another_drivers_admission", 1, 1, 8, &budget) + else { + return; + }; + let second = driver_or_skip("independent peer driver", 1, 2, 8, &budget) + .expect("second driver must start") + .with_shard_policy(ShardPolicy::CapacityAware); + let (read, mut write) = pipe(); + let active = second.read_current(read, 8); + wait_submitted(&second, 1).await; + first.request_shutdown(); + let rejected = first.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + assert!(read_result(rejected).await.is_err()); + wait_finished(&first).await; + assert_eq!(first.shutdown().in_flight, 0); + assert_eq!(budget.available(), 8); + write.write_all(b"12345678").expect("complete peer pipe read"); + let retained = read_result(active) + .await + .expect("peer read must survive another driver stopping"); + assert_eq!(retained, b"12345678"); + let zero = Arc::new(File::open("/dev/zero").expect("open fixture")); + let batch = second + .read_at_batch(vec![ + ReadRequest { + file: Arc::clone(&zero), + offset: 0, + len: 4, + }, + ReadRequest { + file: zero, + offset: 4, + len: 4, + }, + ]) + .expect("capacity-aware peer accepts new batch reads"); + for next in batch { + assert_eq!(read_result(next).await.expect("peer remains open to new reads"), [0; 4]); + } + let stats = second.shutdown(); + assert_eq!(stats.submitted, 3); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + assert_eq!(budget.available(), 16, "retained caller Vec is outside driver in-flight quota"); + assert_eq!(retained, b"12345678"); +} + +#[tokio::test(flavor = "current_thread")] +async fn unpolled_deferred_count_handle_retains_whole_quota_after_clean_driver_shutdown() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip( + "unpolled_deferred_count_handle_retains_whole_quota_after_clean_driver_shutdown", + 1, + 1, + 8, + &budget, + ) else { + return; + }; + let (read, _write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let waiting = driver.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + // No poll: this handle has not joined the local count queue or submitted I/O. + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 0); + assert_eq!(budget.available(), 0, "deferred handle still owns the retired driver's receipt"); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(8), &budget)); + drop(waiting); + assert_eq!(budget.available(), 8); + let replacement = driver_or_skip("quota recycled after final deferred owner", 1, 1, 8, &budget).expect("quota must recycle"); + replacement.shutdown(); + assert_eq!(budget.available(), 8); +} + +#[tokio::test(flavor = "current_thread")] +async fn canceled_polled_byte_waiter_releases_local_reservation_but_live_driver_keeps_quota() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip( + "canceled_polled_byte_waiter_releases_local_reservation_but_live_driver_keeps_quota", + 2, + 1, + 8, + &budget, + ) else { + return; + }; + let (read, mut write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let zero = Arc::new(File::open("/dev/zero").expect("open fixture")); + let mut waiting = driver.read_at(Arc::clone(&zero), 0, 8); + assert!( + Pin::new(&mut waiting) + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + drop(waiting); + assert_eq!(budget.available(), 0, "a live idle-or-busy driver retains its entire quota"); + write.write_all(b"12345678").expect("complete original read"); + assert_eq!(read_result(active).await.expect("original read completes"), b"12345678"); + assert_eq!( + read_result(driver.read_at(zero, 0, 8)) + .await + .expect("canceled waiter must not strand local permits"), + [0; 8] + ); + let stats = driver.shutdown(); + assert_eq!(stats.submitted, 2, "canceled deferred request never submitted"); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + assert_eq!(budget.available(), 8); +} + +#[tokio::test(flavor = "current_thread")] +async fn retired_drivers_polled_byte_waiter_returns_final_receipt_when_dropped() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip("retired_drivers_polled_byte_waiter_returns_final_receipt_when_dropped", 2, 1, 8, &budget) + else { + return; + }; + let (read, _write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let mut waiting = driver.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + assert!( + Pin::new(&mut waiting) + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 0); + assert_eq!(budget.available(), 0, "unconsumed closed waiter retains its receipt"); + assert!(read_result(waiting).await.is_err()); // Consumes and drops the handle. + assert_eq!(budget.available(), 8); +} + +#[cfg(feature = "fault-injection")] +#[test] +fn leaked_read_retains_entire_driver_quota_without_closing_shared_budget() { + const CHILD: &str = "RUSTFS_URING_SHARED_QUOTA_LEAK_CHILD"; + const MARKER: &str = "SHARED_QUOTA_LEAK_RETENTION_OK"; + let name = "leaked_read_retains_entire_driver_quota_without_closing_shared_budget"; + if std::env::var_os(CHILD).is_some() { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .expect("child runtime"); + runtime.block_on(async { + let budget = SharedReadBudget::new(12).expect("shared test capacity"); + let Some(driver) = driver_or_skip(name, 2, 1, 8, &budget) else { return }; + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + // The capacity-aware eager path and batch path must carry the same + // receipt as ordinary submit. Leak one byte, retain the whole eight. + let active = driver + .read_at_batch(vec![ReadRequest { + file: Arc::new(File::open("/dev/zero").expect("open leak fixture")), + offset: 0, + len: 1, + }]) + .expect("single-read batch") + .pop() + .expect("batch contains its one read"); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 1, "injected drain must really leak a pending entry"); + assert_eq!(budget.available(), 4, "leak must not refund any of the driver's whole quota"); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(5), &budget)); + // Startup probe is independent of the drive-loop stuck-CQE seam. + // No application read is submitted to this second driver. + let peer = driver_or_skip("healthy peer uses unreserved capacity", 1, 1, 4, &budget) + .expect("remaining capacity must stay usable"); + assert_eq!(budget.available(), 0); + assert_eq!(peer.shutdown().in_flight, 0); + assert_eq!(budget.available(), 4); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(5), &budget)); + eprintln!("{MARKER}"); + }); + return; + } + let budget = SharedReadBudget::new(12).expect("parent precondition capacity"); + let Some(driver) = driver_or_skip(name, 1, 1, 8, &budget) else { return }; + driver.shutdown(); + let output = std::process::Command::new(std::env::current_exe().expect("test executable")) + .args(["--exact", name, "--nocapture", "--test-threads=1"]) + .env(CHILD, "1") + .env("RUSTFS_URING_FAULT_STUCK_DRAIN", "1") + .env("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400") + .output() + .expect("run isolated leaked-quota case"); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(output.status.success(), "child failed: {stderr}"); + assert!(stderr.contains(MARKER), "child did not prove the leaked-quota path: {stderr}"); +} From 39311240ee448c98a81a4e5029f5cba8d3fe5892 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 07:09:09 +0800 Subject: [PATCH 20/22] feat(driver): reserve shared read budgets across driver lifetimes Reserve each explicit local byte limit before probing, reject transient shared shortage without waiting, and retain the whole reservation through deferred admission and accepted reads. Keep local shutdown independent and leaked pending buffers charged, with atomic accounting and deterministic ownership tests. Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- src/admission_tests.rs | 16 +- src/driver.rs | 402 ++++++++++++++++++++++++++++- src/driver_fault_recovery_tests.rs | 1 + src/driver_submit_result_tests.rs | 1 + src/lib.rs | 4 +- 5 files changed, 409 insertions(+), 15 deletions(-) diff --git a/src/admission_tests.rs b/src/admission_tests.rs index 4abd3c4..a984cac 100644 --- a/src/admission_tests.rs +++ b/src/admission_tests.rs @@ -11,7 +11,7 @@ fn byte_saturation_returns_partial_count_reservation_on_drop() { let count = Arc::new(Semaphore::new(2)); let bytes = Arc::new(Semaphore::new(8)); let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); - let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8, None); assert!(poll_once(&mut waiting).is_pending()); assert_eq!(count.available_permits(), 0); drop(waiting); @@ -26,7 +26,7 @@ fn dropping_count_waiter_does_not_reserve_bytes() { let count = Arc::new(Semaphore::new(1)); let bytes = Arc::new(Semaphore::new(8)); let first = try_read_permits(&count, Some(&bytes), 4).unwrap(); - let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 4); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 4, None); assert!(poll_once(&mut waiting).is_pending()); drop(waiting); assert_eq!(bytes.available_permits(), 4); @@ -41,7 +41,7 @@ fn shard_exit_closes_bytes_and_releases_waiting_count_permit() { admission.register(&count); let bytes = admission.bytes.clone(); let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); - let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8, None); assert!(poll_once(&mut waiting).is_pending()); drop(CloseByteAdmission(Some(admission))); assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); @@ -71,7 +71,7 @@ fn global_close_wakes_count_stage_waiter_without_releasing_hung_read() { let admission = Arc::new(ByteAdmission::new(8)); admission.register(&count); let hung = try_read_permits(&count, Some(&admission.bytes), 8).unwrap(); - let mut waiting = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8); + let mut waiting = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8, None); let wake = Arc::new(LockCheckingWake { admission: admission.clone(), wakes: AtomicUsize::new(0), @@ -87,7 +87,7 @@ fn global_close_wakes_count_stage_waiter_without_releasing_hung_read() { try_read_permits(&count, Some(&admission.bytes), 8), Err(TryAcquireError::Closed) )); - let mut new_waiter = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8); + let mut new_waiter = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8, None); assert!(matches!(poll_once(&mut new_waiter), Poll::Ready(Err(_)))); assert_eq!(admission.bytes.available_permits(), 0, "hung read must remain charged"); drop(hung); @@ -127,10 +127,10 @@ fn canceling_partial_fifo_byte_reservation_unblocks_next_waiter() { let count = Arc::new(Semaphore::new(4)); let bytes = Arc::new(Semaphore::new(8)); let held = try_read_permits(&count, Some(&bytes), 4).unwrap(); - let mut large = acquire_read_permits(count.clone(), Some(bytes.clone()), 6); + let mut large = acquire_read_permits(count.clone(), Some(bytes.clone()), 6, None); assert!(poll_once(&mut large).is_pending()); assert_eq!(bytes.available_permits(), 0, "large FIFO waiter should reserve the available four bytes"); - let mut small = acquire_read_permits(count.clone(), Some(bytes.clone()), 2); + let mut small = acquire_read_permits(count.clone(), Some(bytes.clone()), 2, None); assert!(poll_once(&mut small).is_pending()); drop(large); let Poll::Ready(Ok(small)) = poll_once(&mut small) else { @@ -163,7 +163,7 @@ fn shared_byte_budget_serializes_reads_from_distinct_shards() { let first = try_read_permits(&counts[0], Some(&bytes), 8).unwrap(); assert!(matches!(try_read_permits(&counts[1], Some(&bytes), 1), Err(TryAcquireError::NoPermits))); assert_eq!(counts[1].available_permits(), 1); - let mut waiting = acquire_read_permits(counts[1].clone(), Some(bytes.clone()), 8); + let mut waiting = acquire_read_permits(counts[1].clone(), Some(bytes.clone()), 8, None); assert!(poll_once(&mut waiting).is_pending()); drop(first); let Poll::Ready(Ok(second)) = poll_once(&mut waiting) else { panic!("budget was not released") }; diff --git a/src/driver.rs b/src/driver.rs index 76d8a3e..f1f4f5c 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -383,6 +383,122 @@ pub struct ReadLimits { pub max_in_flight_bytes: Option, } +/// Shared whole-driver reservations for in-flight read-buffer budgets. +/// +/// Clone this handle to give multiple drivers the same pool. Each driver reserves +/// its entire configured [`ReadLimits::max_in_flight_bytes`] before probing, even +/// while idle; local per-read admission is unchanged. Reservations are returned +/// only after their driver, deferred admission and accepted reads release them. +/// A leaked pending read retains the driver's whole reservation permanently. +/// +/// This bounds the sum of participating drivers' reserved read-buffer limits, +/// not completed results, allocator overhead, probe/ring allocations or RSS. +/// Shutting down one driver never closes the pool or another driver's admission. +/// Creating a separate pool creates a separate accounting domain. +/// +/// # Example +/// +/// ```no_run +/// use std::{fs::File, sync::Arc}; +/// use rustfs_uring::{ReadLimits, SharedReadBudget, UringDriver}; +/// # fn main() -> Result<(), Box> { +/// let pool = SharedReadBudget::new(24 << 20)?; +/// let first = UringDriver::probe_and_start_with_shared_budget( +/// 64, 1, +/// ReadLimits { max_read_len: None, max_in_flight_bytes: Some(8 << 20) }, +/// &pool, +/// )?; +/// let second = UringDriver::probe_and_start_with_shared_budget( +/// 64, 2, +/// ReadLimits { max_read_len: None, max_in_flight_bytes: Some(16 << 20) }, +/// &pool, +/// )?; +/// assert_eq!(pool.available(), 0); // whole reservations, even while idle +/// let file = Arc::new(File::open("object.bin")?); +/// let reads = [first.read_at(file.clone(), 0, 4096), second.read_at(file, 4096, 4096)]; +/// drop(reads); // also release any deferred admission owners +/// let first_stats = first.shutdown(); +/// let second_stats = second.shutdown(); +/// if first_stats.in_flight == 0 && second_stats.in_flight == 0 { +/// assert_eq!(pool.available(), pool.capacity()); +/// } +/// # Ok(()) +/// # } +/// ``` +#[derive(Clone, Debug)] +pub struct SharedReadBudget { + inner: Arc, +} + +#[derive(Debug)] +struct SharedReadBudgetInner { + capacity: usize, + available: AtomicUsize, +} + +impl SharedReadBudget { + /// Create a pool with `total` bytes of reservation capacity. + /// + /// Zero returns `InvalidInput`. This does not allocate `total` bytes or + /// require a runtime. Individual driver limits must still fit Tokio's + /// `Semaphore::MAX_PERMITS`, but pool capacity is not narrowed to `u32`. + pub fn new(total: usize) -> io::Result { + if total == 0 { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "shared read budget must be nonzero")); + } + Ok(Self { + inner: Arc::new(SharedReadBudgetInner { + capacity: total, + available: AtomicUsize::new(total), + }), + }) + } + + /// The pool's fixed whole-driver reservation capacity, in bytes. + pub fn capacity(&self) -> usize { + self.inner.capacity + } + + /// Advisory snapshot of unreserved capacity, in bytes. + /// + /// This is not a measurement of live buffers: an idle participating driver + /// still holds its whole reservation. Concurrent construction or cleanup + /// can change this value immediately after it is read. + pub fn available(&self) -> usize { + self.inner.available.load(Ordering::Acquire) + } + + fn reserve(&self, bytes: usize) -> io::Result> { + if bytes == 0 || bytes > self.inner.capacity { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "driver read limit must fit the shared read budget", + )); + } + self.inner + .available + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |available| available.checked_sub(bytes)) + .map_err(|_| io::Error::new(io::ErrorKind::WouldBlock, "shared read budget has insufficient unreserved capacity"))?; + Ok(Arc::new(SharedReadReservation { + pool: Arc::clone(&self.inner), + bytes, + })) + } +} + +struct SharedReadReservation { + pool: Arc, + bytes: usize, +} + +impl Drop for SharedReadReservation { + fn drop(&mut self) { + // One receipt refunds its successful checked subtraction exactly once. + // Read-path clones only touch Arc counts, never this global counter. + self.pool.available.fetch_add(self.bytes, Ordering::Release); + } +} + /// Shard selection for positioned reads. Stream reads retain round-robin routing. #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] pub enum ShardPolicy { @@ -400,6 +516,7 @@ pub enum ShardPolicy { struct ReadPermits { _count: OwnedSemaphorePermit, _bytes: Option, + _shared_reservation: Option>, } fn try_read_permits(count: &Arc, bytes: Option<&Arc>, charge: u32) -> Result { @@ -408,10 +525,16 @@ fn try_read_permits(count: &Arc, bytes: Option<&Arc>, char Ok(ReadPermits { _count: count, _bytes: bytes, + _shared_reservation: None, }) } -fn acquire_read_permits(count: Arc, bytes: Option>, charge: u32) -> AcquireFut { +fn acquire_read_permits( + count: Arc, + bytes: Option>, + charge: u32, + shared_reservation: Option>, +) -> AcquireFut { Box::pin(async move { // Every admission takes count before bytes. Pending reads need neither // resource to complete, so there is no inverse acquisition cycle. @@ -423,6 +546,7 @@ fn acquire_read_permits(count: Arc, bytes: Option>, ch Ok(ReadPermits { _count: count, _bytes: bytes, + _shared_reservation: shared_reservation, }) }) } @@ -432,6 +556,7 @@ fn acquire_read_permits(count: Arc, bytes: Option>, ch struct ByteAdmission { bytes: Arc, registry: Mutex, + shared_reservation: Option>, } #[derive(Default)] @@ -445,6 +570,7 @@ impl ByteAdmission { Self { bytes: Arc::new(Semaphore::new(bytes)), registry: Mutex::new(AdmissionRegistry::default()), + shared_reservation: None, } } @@ -958,6 +1084,42 @@ impl UringDriver { /// Saturated admission waits asynchronously and fairly on Tokio semaphores; /// dropping a waiting handle returns any partial reservation. pub fn probe_and_start_with_limits(entries: u32, shards: usize, limits: ReadLimits) -> Result { + Self::validate_read_limits(limits)?; + Self::start_with_limits(entries, shards, limits, None) + } + + /// Reserve a whole driver budget from `budget`, then probe and start it. + /// + /// `limits.max_in_flight_bytes` must be explicitly set and nonzero. It + /// remains the driver's independent local byte limit, shared by its shards. + /// A limit larger than the pool capacity returns `ProbeFailure::Setup` with + /// `InvalidInput`; temporary reservation shortage returns `WouldBlock` + /// immediately, without probing or waiting for another driver to retire. + /// These errors have no restriction errno and do not imply io_uring is + /// unsupported. No fairness or automatic retry is promised. + /// + /// Startup failure refunds its reservation after partial shards are cleaned + /// up. Successful drivers keep it while idle, through shutdown and deferred + /// or accepted reads; a leaked pending read keeps the whole reservation. + /// Completed result buffers are not covered by this accounting. + pub fn probe_and_start_with_shared_budget( + entries: u32, + shards: usize, + limits: ReadLimits, + budget: &SharedReadBudget, + ) -> Result { + Self::validate_read_limits(limits)?; + let bytes = limits.max_in_flight_bytes.ok_or_else(|| { + ProbeFailure::Setup(io::Error::new( + io::ErrorKind::InvalidInput, + "shared read budget requires an explicit local byte limit", + )) + })?; + let reservation = budget.reserve(bytes).map_err(ProbeFailure::Setup)?; + Self::start_with_limits(entries, shards, limits, Some(reservation)) + } + + fn validate_read_limits(limits: ReadLimits) -> Result<(), ProbeFailure> { if limits .max_in_flight_bytes .is_some_and(|bytes| bytes == 0 || bytes > Semaphore::MAX_PERMITS) @@ -967,7 +1129,20 @@ impl UringDriver { "byte budget must be between 1 and Semaphore::MAX_PERMITS", ))); } - let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + Ok(()) + } + + fn start_with_limits( + entries: u32, + shards: usize, + limits: ReadLimits, + shared_reservation: Option>, + ) -> Result { + let byte_admission = limits.max_in_flight_bytes.map(|bytes| { + let mut admission = ByteAdmission::new(bytes); + admission.shared_reservation = shared_reservation; + Arc::new(admission) + }); let mut started = Vec::with_capacity(shards.max(1)); for i in 0..shards.max(1) { // Probe only the first shard (rustfs/backlog#1165): the probe read @@ -1027,6 +1202,7 @@ impl UringDriver { .map(|bytes| ReadPermits { _count: count, _bytes: bytes, + _shared_reservation: None, }); return (shard, permits); } @@ -1374,7 +1550,11 @@ impl UringDriver { // Fast path: a permit was free, so submit eagerly — no allocation, // no await, and the op is in flight the moment `submit` returns, // exactly as with the previous blocking implementation. - Ok(permit) => { + Ok(mut permit) => { + permit._shared_reservation = self + .byte_admission + .as_ref() + .and_then(|admission| admission.shared_reservation.clone()); #[cfg(feature = "diagnostics")] if let Some(timing) = &timing { timing.enqueue(); @@ -1442,6 +1622,9 @@ impl UringDriver { Arc::clone(&shard.sem), self.byte_admission.as_ref().map(|admission| Arc::clone(&admission.bytes)), charge, + self.byte_admission + .as_ref() + .and_then(|admission| admission.shared_reservation.clone()), ), file, offset, @@ -2419,6 +2602,212 @@ fn drive( } } +#[cfg(test)] +mod shared_budget_reservation_tests { + use super::*; + + fn mock_driver(pool: &SharedReadBudget, bytes: usize, policy: ShardPolicy) -> (UringDriver, mpsc::Receiver) { + let mut admission = ByteAdmission::new(bytes); + admission.shared_reservation = Some(pool.reserve(bytes).expect("reserve mock driver")); + let admission = Arc::new(admission); + let sem = Arc::new(Semaphore::new(1)); + admission.register(&sem); + let (tx, rx) = mpsc::channel(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(bytes), + }, + shard_policy: policy, + byte_admission: Some(admission), + shards: vec![Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().expect("mock wake fd")), + }], + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + rx, + ) + } + + #[test] + fn pool_reserves_whole_weights_and_refunds_only_the_last_receipt_owner() { + assert_eq!(SharedReadBudget::new(0).expect_err("zero pool").kind(), io::ErrorKind::InvalidInput); + let pool = SharedReadBudget::new(12).expect("pool"); + let alias = pool.clone(); + let first = pool.reserve(8).expect("first driver"); + let first_read = first.clone(); + let second = alias.reserve(4).expect("second driver"); + assert_eq!(pool.capacity(), 12); + assert_eq!(alias.available(), 0); + assert!(matches!(pool.reserve(1), Err(error) if error.kind() == io::ErrorKind::WouldBlock)); + drop(first); + assert_eq!(pool.available(), 0, "an accepted read still owns the receipt"); + drop(first_read); + assert_eq!(pool.available(), 8); + drop(second); + assert_eq!(pool.available(), 12); + } + + #[cfg(target_pointer_width = "64")] + #[test] + fn whole_driver_reservations_do_not_narrow_to_u32_or_overflow_refunds() { + let pool = SharedReadBudget::new(usize::MAX).expect("large accounting-only pool"); + let bytes = usize::try_from(u32::MAX).expect("64-bit usize") + 1; + let large = pool.reserve(bytes).expect("reservation above u32"); + assert_eq!(pool.available(), usize::MAX - bytes); + let remaining = pool.reserve(usize::MAX - bytes).expect("reserve exact remainder"); + assert_eq!(pool.available(), 0); + assert!(matches!(pool.reserve(1), Err(error) if error.kind() == io::ErrorKind::WouldBlock)); + drop(remaining); + drop(large); + assert_eq!(pool.available(), usize::MAX); + } + + #[test] + fn concurrent_reservations_never_exceed_pool_capacity() { + let pool = SharedReadBudget::new(17).expect("pool"); + let gate = std::sync::Barrier::new(9); + let successful = AtomicUsize::new(0); + let unexpected_error = AtomicBool::new(false); + let observed = std::thread::scope(|scope| { + for _ in 0..8 { + scope.spawn(|| { + gate.wait(); + let reservation = pool.reserve(3); + match &reservation { + Ok(_) => { + successful.fetch_add(1, Ordering::SeqCst); + } + Err(error) if error.kind() == io::ErrorKind::WouldBlock => {} + Err(_) => { + unexpected_error.store(true, Ordering::SeqCst); + } + } + gate.wait(); + gate.wait(); + drop(reservation); + }); + } + gate.wait(); + gate.wait(); + let observed = (pool.available(), successful.load(Ordering::SeqCst)); + gate.wait(); + observed + }); + assert_eq!(observed, (2, 5)); + assert!(!unexpected_error.load(Ordering::SeqCst)); + assert_eq!(pool.available(), pool.capacity()); + } + + #[test] + fn invalid_and_temporarily_unavailable_driver_limits_fail_before_probing() { + let pool = SharedReadBudget::new(8).expect("pool"); + for limit in [None, Some(0), Some(9), Some(Semaphore::MAX_PERMITS + 1)] { + let result = UringDriver::probe_and_start_with_shared_budget( + 0, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: limit, + }, + &pool, + ); + let Err(error) = result else { panic!("invalid constructor must fail") }; + assert!(!error.is_expected_restriction()); + assert!(matches!(error, ProbeFailure::Setup(ref error) if error.kind() == io::ErrorKind::InvalidInput)); + assert_eq!(pool.available(), 8); + } + let occupied = pool.reserve(8).expect("occupy pool"); + let result = UringDriver::probe_and_start_with_shared_budget( + 0, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(1), + }, + &pool, + ); + let Err(error) = result else { panic!("insufficient shared budget must fail") }; + assert!(!error.is_expected_restriction()); + assert!(matches!(error, ProbeFailure::Setup(ref error) if error.kind() == io::ErrorKind::WouldBlock)); + assert_eq!(pool.available(), 0); + drop(occupied); + assert_eq!(pool.available(), 8); + } + + #[test] + fn eager_messages_keep_reservations_after_driver_and_caller_drop() { + for policy in [ShardPolicy::RoundRobin, ShardPolicy::CapacityAware] { + let pool = SharedReadBudget::new(8).expect("pool"); + let (driver, messages) = mock_driver(&pool, 8, policy); + let handle = driver + .read_at(Arc::new(File::open("/dev/zero").expect("fixture file")), 0, 4) + .without_cancel_on_drop(); + let message = messages.try_recv().expect("eager read enqueued"); + assert!(matches!(message, Msg::Read { .. })); + drop(handle); + drop(driver); + assert_eq!(pool.available(), 0, "accepted message must retain the whole driver reservation"); + drop(message); + assert_eq!(pool.available(), 8); + } + } + + #[test] + fn deferred_admission_keeps_reservation_until_closed_poll_or_drop() { + for poll_closed in [false, true] { + let pool = SharedReadBudget::new(8).expect("pool"); + let (driver, messages) = mock_driver(&pool, 8, ShardPolicy::RoundRobin); + let file = Arc::new(File::open("/dev/zero").expect("fixture file")); + let first = driver.read_at(file.clone(), 0, 8).without_cancel_on_drop(); + let message = messages.try_recv().expect("first read owns local permits"); + let mut waiting = driver.read_at(file, 0, 8); + assert!(matches!(waiting.state, HandleState::WaitingPermit { .. })); + drop(first); + drop(message); + drop(driver); + assert_eq!(pool.available(), 0, "deferred acquire owns the reservation before its first poll"); + if poll_closed { + let result = Pin::new(&mut waiting).poll(&mut Context::from_waker(std::task::Waker::noop())); + assert!(matches!(result, Poll::Ready(Err(_)))); + assert_eq!(pool.available(), 8, "closed acquire releases its receipt"); + } + drop(waiting); + assert_eq!(pool.available(), 8); + } + } + + #[test] + fn local_shutdown_does_not_close_another_participating_driver() { + let pool = SharedReadBudget::new(8).expect("pool"); + let (first, _first_messages) = mock_driver(&pool, 4, ShardPolicy::RoundRobin); + let (second, second_messages) = mock_driver(&pool, 4, ShardPolicy::RoundRobin); + first.request_shutdown(); + assert!(first.shards[0].sem.is_closed()); + assert!(!second.shards[0].sem.is_closed()); + assert!(!second.byte_admission.as_ref().expect("local budget").bytes.is_closed()); + assert_eq!(pool.available(), 0, "shutdown request alone does not return an idle driver's reservation"); + let read = second + .read_at(Arc::new(File::open("/dev/zero").expect("fixture file")), 0, 4) + .without_cancel_on_drop(); + let message = second_messages.try_recv().expect("other driver still accepts reads"); + assert!(matches!(message, Msg::Read { .. })); + drop(first); + assert_eq!(pool.available(), 4); + drop(second); + drop(read); + assert_eq!(pool.available(), 4, "second driver's queued operation still owns its reservation"); + drop(message); + assert_eq!(pool.available(), 8); + } +} + #[cfg(test)] mod shutdown_request_tests { use super::*; @@ -2468,7 +2857,7 @@ mod shutdown_request_tests { let (driver, messages) = mock_driver(1, None); let count = Arc::clone(&driver.shards[0].sem); let held = try_read_permits(&count, None, 1).expect("occupy mock shard"); - let mut waiter = acquire_read_permits(count.clone(), None, 1); + let mut waiter = acquire_read_permits(count.clone(), None, 1, None); assert!(poll_acquire(&mut waiter).is_pending()); driver.request_shutdown(); assert!(count.is_closed()); @@ -2483,8 +2872,8 @@ mod shutdown_request_tests { let (driver, _messages) = mock_driver(2, Some(8)); let bytes = Arc::clone(&driver.byte_admission.as_ref().expect("byte budget").bytes); let held = try_read_permits(&driver.shards[0].sem, Some(&bytes), 8).expect("occupy byte budget"); - let mut count_waiter = acquire_read_permits(driver.shards[0].sem.clone(), Some(bytes.clone()), 8); - let mut byte_waiter = acquire_read_permits(driver.shards[1].sem.clone(), Some(bytes.clone()), 8); + let mut count_waiter = acquire_read_permits(driver.shards[0].sem.clone(), Some(bytes.clone()), 8, None); + let mut byte_waiter = acquire_read_permits(driver.shards[1].sem.clone(), Some(bytes.clone()), 8, None); assert!(poll_acquire(&mut count_waiter).is_pending()); assert!(poll_acquire(&mut byte_waiter).is_pending()); driver.request_shutdown(); @@ -2627,6 +3016,7 @@ mod cancellation_efficiency_tests { _permit: ReadPermits { _count: Arc::clone(&sem).try_acquire_owned().expect("fixture permit"), _bytes: None, + _shared_reservation: None, }, pad: 0, head: 0, diff --git a/src/driver_fault_recovery_tests.rs b/src/driver_fault_recovery_tests.rs index ff3123d..d5d3ee5 100644 --- a/src/driver_fault_recovery_tests.rs +++ b/src/driver_fault_recovery_tests.rs @@ -49,6 +49,7 @@ fn direct_prefix() -> (Pending, oneshot::Receiver>>) { _permit: ReadPermits { _count: Arc::new(Semaphore::new(1)).try_acquire_owned().unwrap(), _bytes: None, + _shared_reservation: None, }, pad: 1, head: 2, diff --git a/src/driver_submit_result_tests.rs b/src/driver_submit_result_tests.rs index 71de099..6f580ab 100644 --- a/src/driver_submit_result_tests.rs +++ b/src/driver_submit_result_tests.rs @@ -49,6 +49,7 @@ fn pending_read(file: Arc, count: &Arc, bytes: &Arc) _permit: ReadPermits { _count: Arc::clone(count).try_acquire_owned().unwrap(), _bytes: Some(Arc::clone(bytes).try_acquire_many_owned(16).unwrap()), + _shared_reservation: None, }, pad: 0, head: 0, diff --git a/src/lib.rs b/src/lib.rs index 405b661..6b965e1 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -55,4 +55,6 @@ mod diagnostics; pub use diagnostics::{DIAGNOSTICS_SAMPLE_INTERVAL, DiagnosticsSnapshot, LatencyHistogram}; #[cfg(target_os = "linux")] -pub use driver::{MAX_BATCH_READS, ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, StatsSnapshot, UringDriver}; +pub use driver::{ + MAX_BATCH_READS, ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, SharedReadBudget, StatsSnapshot, UringDriver, +}; From b09b966de92e388b1c1cec9e82d08f6b84edb676 Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 07:10:55 +0800 Subject: [PATCH 21/22] docs(driver): specify shared read quota ownership and acceptance Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- CHANGELOG.md | 3 ++ README.md | 1 + docs/optimization-status.md | 15 +++++++ docs/rustfs-integration.md | 6 +++ docs/shared-read-budget.md | 89 +++++++++++++++++++++++++++++++++++++ 5 files changed, 114 insertions(+) create mode 100644 docs/shared-read-budget.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 30a2eeb..abdf337 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,9 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- Opt-in `SharedReadBudget` whole-driver quota reservations, retained through + deferred and kernel-owned reads, including bounded-drain leaks. Independent + drivers keep independent admission and shutdown; default constructors are unchanged. - Non-joining `request_shutdown`, advisory `is_finished`, and a default-off `tokio-runtime` feature for eager consuming `shutdown_async` ownership handoff. Synchronous `shutdown`/`Drop` and bounded-drain leak guarantees remain explicit. diff --git a/README.md b/README.md index c2e8789..03dadae 100644 --- a/README.md +++ b/README.md @@ -48,6 +48,7 @@ assert_eq!(snapshot.delivered + snapshot.orphan_reclaimed, snapshot.submitted); - `read_current(file, len)` — `read(2)` semantics from the current position, for pipes and other non-seekable fds (a short read is a valid final result). - `probe_and_start_sharded(entries, shards)` — several independent rings per disk (each ring caps at one core's memory bandwidth for cache-hit reads); `probe_and_start(entries)` equals `..._sharded(entries, 1)`. - `probe_and_start_with_limits(entries, shards, ReadLimits { max_read_len, max_in_flight_bytes })` — optional logical read-size and driver-wide read-buffer limits. Both fields default to `None`, preserving existing constructor behavior. +- `probe_and_start_with_shared_budget(entries, shards, limits, &pool)` — reserve the driver's whole configured byte quota from a cloneable `SharedReadBudget` before startup. See [shared reservation ownership](docs/shared-read-budget.md); this is not dynamic per-read sharing or an RSS limit. - `with_shard_policy(ShardPolicy::CapacityAware)` — opt-in capacity-aware routing for positioned reads. Constructors keep `ShardPolicy::RoundRobin` by default. - `request_shutdown()` — close admission and request cancellation/drain without joining; `is_finished()` reports advisory thread completion, not a clean drain. - `shutdown_async()` — with the default-off `tokio-runtime` feature, transfer consuming cleanup to Tokio's blocking pool at method call time. See [shutdown ownership and runtime boundaries](docs/shutdown.md). diff --git a/docs/optimization-status.md b/docs/optimization-status.md index ab82324..1cc9237 100644 --- a/docs/optimization-status.md +++ b/docs/optimization-status.md @@ -13,6 +13,7 @@ integration are separate gates; a checked code item does not close the roadmap. | 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | | 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | | 4.2: system-wide budgets and probe offload | Application probe offload, driver-thread budget and logical chunk limit implemented/reviewed in draft PRs #8072/#8074/#8076 | Full CI/merge; physical byte/result/io-wq budgets and [dependency wiring](rustfs-integration.md) remain open | +| 4.2 pool prerequisite | Library whole-driver shared quota implemented; two independent reviews and local compilation checks pass | Native CI, application dependency/fallback/result ownership wiring and merge | | 5: owned buffers and direct FD cache | Dual-mode exact invalidation implemented/reviewed in application draft PR #8075; direct caching and owned buffers not enabled | Full CI/merge, profile evidence, lease/pool lifetime and complete invalidation integration | | 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests and native CLI CI #54 passed | Production consumer contract, bitrot/S3 and performance evidence | | 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | @@ -37,6 +38,19 @@ final-head CI and merge status separately. ## Follow-up evidence and examples +`3931124` implements `SharedReadBudget` whole-driver reservations, with public +tests in `7e94f36`. Seven deterministic private tests and nine public/native +tests cover concurrent accounting, large `usize` quotas, message/deferred +ownership, independent shutdown, retained results and leaked whole quotas. +Two independent code/test/document reviews found no blocking issue. Local +Linux-target default/all-feature all-target checks, all-feature Clippy, +default/all-feature warning-denying rustdoc, formatting and diff checks pass; +the default library/dependency graph remains Tokio sync-only. The initial RED +was missing-API compilation, not a behavioral run. Native execution for this +addition is pending. Its [contract](shared-read-budget.md) documents the +later-shard startup-failure coverage boundary and conservative idle/retirement +reservation costs. Application wiring and performance remain separate gates. + Application work is tracked in [RustFS #8072](https://github.com/rustfs/rustfs/pull/8072), [#8074](https://github.com/rustfs/rustfs/pull/8074), [#8075](https://github.com/rustfs/rustfs/pull/8075), and @@ -115,6 +129,7 @@ establish stable calibration without loosening gates after observing results. - [Public read and admission contracts](../README.md) - [Shutdown ownership and Tokio adapter](shutdown.md) +- [Shared whole-driver read-budget reservations](shared-read-budget.md) - [Completion recovery and direct integrity](fault-recovery.md) - [Cancellation and eventfd behavior](cancellation-efficiency.md) - [Driver turn fairness](driver-fairness.md) diff --git a/docs/rustfs-integration.md b/docs/rustfs-integration.md index fd9bd6a..f75d912 100644 --- a/docs/rustfs-integration.md +++ b/docs/rustfs-integration.md @@ -47,6 +47,12 @@ Relevant paths: [backend lifecycle](https://github.com/rustfs/rustfs/blob/1880b4 Current `ReadLimits` covers one driver only: use conservatively allocated per-driver quotas or design an explicitly shared application admission layer. The existing best-effort per-ring io-wq setting is not a process thread cap. + The library's [shared reservation pool](shared-read-budget.md) offers the + conservative whole-driver quota option: pass the same pool across reconnects + and generations; leaked pending reads keep the entire reservation. It does + not wire the application's dependency, std fallback, result assembly, or + retained results into any budget. A temporary `WouldBlock` reservation error + must not enter the permanent unsupported-disk cache. 3. **Fix invalidation before adding direct cache hits.** [Exact invalidation](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L3991) currently builds a buffered-only key; invalidate both variants before diff --git a/docs/shared-read-budget.md b/docs/shared-read-budget.md new file mode 100644 index 0000000..0763e1f --- /dev/null +++ b/docs/shared-read-budget.md @@ -0,0 +1,89 @@ +# Shared driver read-budget reservations + +`SharedReadBudget` is an opt-in pool for **whole-driver reservations**. It bounds +the sum of configured in-flight read-buffer quotas across drivers using the same +pool. It is not a dynamic shared per-read byte queue, returned-result limit, or +process RSS limit. + +## Admission + +Create a pool with `SharedReadBudget::new(total)` and pass a reference to +`UringDriver::probe_and_start_with_shared_budget(entries, shards, limits, &pool)`. +Cloning the pool shares its capacity; constructing a separate pool does not. +The new constructor requires an explicit `limits.max_in_flight_bytes = Some(B)`. +It validates the local limit and reserves all `B` bytes before ring startup. + +Zero total capacity, missing/invalid local byte limits, or `B > total` are +`InvalidInput`. Insufficient currently available capacity returns `WouldBlock` +without starting a driver; it is not an environment restriction and must not +poison a disk's unsupported-io_uring cache. Admission does not wait, queue a +blocking task, or partially reserve a driver. The caller decides when to retry +or use a separately budgeted fallback. Budgets use `usize`; whole-driver +reservations must not truncate through a per-read `u32` permit count. + +`capacity()` returns the fixed pool size. `available()` is an advisory snapshot +of capacity not reserved by drivers or their surviving resource owners. Do not +check it and then assume creation must succeed; another constructor can reserve +the same capacity before this caller's atomic reservation. + +Once admitted, a driver keeps the existing local count and byte semaphores. +Buffered reads charge logical length; direct reads charge the aligned superset +plus allocation padding. Read admission does not acquire from the global pool. +Each driver therefore retains its full reservation even while idle, and cannot +borrow unused capacity already reserved by another driver. This deliberately +trades utilization for a smaller shutdown/cancellation contract. + +## Ownership and retirement + +The same reservation receipt follows driver-owned and deferred read resources +through queued messages and pending kernel operations. Canceling the caller does +not refund a reservation while the kernel can still access a buffer. Normal +terminal CQEs release their owners; the whole block returns only after its last +owner disappears. A deferred handle retained after driver shutdown can therefore +keep the entire block reserved until its ownership is released. Drop unused +handles during retirement instead of treating `is_finished()` or a clean +snapshot as a pool-refund signal. + +A bounded-drain escape retains the receipt with the leaked pending resources, +so the entire `B` stays reserved, even if the leaked read is much smaller. A +replacement driver cannot reuse that capacity. There is no forced refund API. +Creating a new independent pool to bypass retained reservations also bypasses +the aggregate bound; the application must keep a stable shared pool across +reconnects and generations. + +Closing one driver closes only its local admission and wakes its local waiters. +It does not close the pool or reject reads on another admitted driver. Startup +failures roll back the reservation after resource cleanup; existing probe/setup +failure and synchronous cleanup contracts still apply. + +## Scope and cost + +This accounts configured driver read-buffer capacity, including direct padding +and retained pending reads. It excludes probe buffers, rings, kernel/io-wq +resources, allocator overhead, handle metadata, completion/result copies, +caller-held results and application std fallback allocations. Startup concurrency +and these other resources require separate controls. It does not constrain +drivers created without this pool. + +Pool atomics run at reservation/final release and explicit observation. Opt-in +reads retain a shared receipt; this is not a performance improvement claim. +Existing constructors and default limits keep their behavior. No application +dependency source or production pool wiring is changed by this library API. + +## Verification boundaries + +Deterministic tests cover concurrent weighted reservations, `usize` arithmetic, +receipt ownership in eager messages and deferred handles, and independent local +closure. Native tests cover multi-driver admission and retirement, retained +results, deferred count/byte cancellation, and a subprocess-injected leak that +keeps the entire quota. Restricted-environment tests also check that failed setup +refunds its reservation before reporting a capability skip. + +Failure after one or more earlier shards have started is covered by the existing +RAII cleanup path and code review, not a new later-shard fault-injection test. +The simulated stuck completion does not prove recovery from a real hung syscall. +See [acceptance status](optimization-status.md) for execution evidence; compiled +tests alone are not native I/O coverage. + +See [read admission](../README.md#read-allocation-admission), +[shutdown ownership](shutdown.md), and [application integration](rustfs-integration.md). From e1992ce1087db7ec2523a1bf68b7276e39d72dbc Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 23 Sep 2026 07:14:54 +0800 Subject: [PATCH 22/22] docs(driver): record native shared quota verification Co-Authored-By: heihutu Co-Authored-By: zhi22915 --- docs/optimization-status.md | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/docs/optimization-status.md b/docs/optimization-status.md index 1cc9237..ff25684 100644 --- a/docs/optimization-status.md +++ b/docs/optimization-status.md @@ -13,7 +13,7 @@ integration are separate gates; a checked code item does not close the roadmap. | 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | | 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | | 4.2: system-wide budgets and probe offload | Application probe offload, driver-thread budget and logical chunk limit implemented/reviewed in draft PRs #8072/#8074/#8076 | Full CI/merge; physical byte/result/io-wq budgets and [dependency wiring](rustfs-integration.md) remain open | -| 4.2 pool prerequisite | Library whole-driver shared quota implemented; two independent reviews and local compilation checks pass | Native CI, application dependency/fallback/result ownership wiring and merge | +| 4.2 pool prerequisite | Library whole-driver shared quota implemented/reviewed; native CI #57 passed at `b09b966` | Application dependency/fallback/result ownership wiring and merge | | 5: owned buffers and direct FD cache | Dual-mode exact invalidation implemented/reviewed in application draft PR #8075; direct caching and owned buffers not enabled | Full CI/merge, profile evidence, lease/pool lifetime and complete invalidation integration | | 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests and native CLI CI #54 passed | Production consumer contract, bitrot/S3 and performance evidence | | 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | @@ -46,8 +46,14 @@ Two independent code/test/document reviews found no blocking issue. Local Linux-target default/all-feature all-target checks, all-feature Clippy, default/all-feature warning-denying rustdoc, formatting and diff checks pass; the default library/dependency graph remains Tokio sync-only. The initial RED -was missing-API compilation, not a behavioral run. Native execution for this -addition is pending. Its [contract](shared-read-budget.md) documents the +was missing-API compilation, not a behavioral run. Subsequently, [CI #57](https://github.com/rustfs/uring/actions/runs/35796196402) +at `b09b966` passed all three jobs: 143 native all-feature tests plus one compiled +`no_run` documentation example, mandatory O_DIRECT, ordered-prefetch and both +benchmark CLI smokes, lint, and restricted/unrestricted Docker legs. All seven +private and nine public shared-budget tests ran successfully, including the +isolated whole-quota leak and greater-than-`u32` reservation cases. Later +evidence-only documentation commits have their own workflow status. Its +[contract](shared-read-budget.md) documents the later-shard startup-failure coverage boundary and conservative idle/retirement reservation costs. Application wiring and performance remain separate gates.