diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index bb95bb6..7640e86 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -64,6 +64,8 @@ jobs: exit 1 fi grep -q "DIRECT_OK direct_read_returns_exact_unaligned_ranges" test.out + - name: Ordered-prefetch contract smoke (real io_uring) + run: bash scripts/test-ordered-prefetch-cli.sh - name: Benchmark schema and correctness smoke run: bash scripts/test-benchmark-cli.sh - name: Instrumented benchmark schema and correctness smoke diff --git a/CHANGELOG.md b/CHANGELOG.md index 5035bd3..abdf337 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,23 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Added +- Opt-in `SharedReadBudget` whole-driver quota reservations, retained through + deferred and kernel-owned reads, including bounded-drain leaks. Independent + drivers keep independent admission and shutdown; default constructors are unchanged. +- Non-joining `request_shutdown`, advisory `is_finished`, and a default-off + `tokio-runtime` feature for eager consuming `shutdown_async` ownership handoff. + Synchronous `shutdown`/`Drop` and bounded-drain leak guarantees remain explicit. +- Explicit same-binary A/A calibration mode for the evidence runner, including + individual middle-leg drift gates and no candidate attribution in control runs. +- Example-only bounded ordered prefetch with deterministic backpressure, + cancellation and deferred-admission tests, plus a native CLI correctness gate. + This does not add a production streaming API or claim end-to-end speedups. +- Opt-in `ReadLimits` for logical read size and driver-wide in-flight read-buffer + bytes, with aligned allocation accounting and terminal-CQE ownership. +- Opt-in capacity-aware shard selection for positioned reads; round-robin remains + the default and stream reads retain their routing semantics. +- `ReadRequest`, `MAX_BATCH_READS` and `read_at_batch` for bounded buffered groups + sharing eager notifications per final owning shard. - Opt-in `diagnostics` feature with per-shard sampled driver-stage histograms, aggregate snapshots, and measurement-interval deltas. Default builds compile out the timing fields and sampling work. @@ -34,6 +51,24 @@ aims to follow [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Changed +- Bound driver intake, allocations and completion work per turn, preserving + progress after notification draining and partial successful submissions. +- Stop explicitly canceled positioned-read continuations after read CQEs and + avoid materializing orphaned results without releasing kernel-owned resources. + +### Fixed + +- Progress empty-SQ CQ overflow/taskrun work and propagate metadata failures + instead of treating an unconfirmed direct-read short prefix as EOF. +- Close count-stage and byte-stage waiters together when shared byte admission + shuts down; retain charged resources for bounded-drain bailout leaks. +- Retry interrupted eventfd operations and suppress repeated unchanged overflow + warnings without changing the cumulative snapshot counter. +- Reap the whole benchmark process group after wrapper failure, including when + the group leader exits before a child process. + +### Benchmarking + - Benchmark CSV schema v2 obtains headers from the executable and reports independent setup, workload, and teardown timings, configurable ring depth, workers, warmup, and instrumentation status. Positional CSV consumers must diff --git a/Cargo.toml b/Cargo.toml index e91d8ca..39121fc 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,6 +36,9 @@ fault-injection = [] # Opt-in sampled timing. No timestamps, histogram storage, or tracing allocations # are compiled into the default driver. diagnostics = [] +# Optional Tokio blocking-pool shutdown adapter. Default builds only require +# Tokio's synchronization primitives and can drive reads from other executors. +tokio-runtime = ["tokio/rt"] [dependencies] # tracing for driver diagnostics: unifies with the RustFS tracing pipeline for diff --git a/README.md b/README.md index d30fc77..03dadae 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,67 @@ assert_eq!(snapshot.delivered + snapshot.orphan_reclaimed, snapshot.submitted); - `read_at_direct(file, offset, len, align)` — the same for an `O_DIRECT` fd; `offset`/`len` need not be aligned (the driver reads a block-aligned superset and returns exactly the requested range). - `read_current(file, len)` — `read(2)` semantics from the current position, for pipes and other non-seekable fds (a short read is a valid final result). - `probe_and_start_sharded(entries, shards)` — several independent rings per disk (each ring caps at one core's memory bandwidth for cache-hit reads); `probe_and_start(entries)` equals `..._sharded(entries, 1)`. +- `probe_and_start_with_limits(entries, shards, ReadLimits { max_read_len, max_in_flight_bytes })` — optional logical read-size and driver-wide read-buffer limits. Both fields default to `None`, preserving existing constructor behavior. +- `probe_and_start_with_shared_budget(entries, shards, limits, &pool)` — reserve the driver's whole configured byte quota from a cloneable `SharedReadBudget` before startup. See [shared reservation ownership](docs/shared-read-budget.md); this is not dynamic per-read sharing or an RSS limit. +- `with_shard_policy(ShardPolicy::CapacityAware)` — opt-in capacity-aware routing for positioned reads. Constructors keep `ShardPolicy::RoundRobin` by default. +- `request_shutdown()` — close admission and request cancellation/drain without joining; `is_finished()` reports advisory thread completion, not a clean drain. +- `shutdown_async()` — with the default-off `tokio-runtime` feature, transfer consuming cleanup to Tokio's blocking pool at method call time. See [shutdown ownership and runtime boundaries](docs/shutdown.md). + +### Shard selection + +Round-robin binds each read to the next shard, even if that shard is busy or +closed. Capacity-aware selection starts at the same cursor and tries each shard's +count permit at most once, skipping closed count semaphores. It uses actual +permit acquisition, not a free-capacity snapshot. If all healthy shards are busy, +the handle waits on the first healthy candidate using Tokio's fair semaphore. +The wait is local to that shard; it does not rebalance when another shard frees. + +A shared byte-budget shortage waits on the first shard whose count permit was +available, returning that temporary count reservation before constructing the +waiter. A closed shared byte budget rejects admission globally. Once a read is +accepted or deferred, its read, wakeup, retry, and cancel retain the same owning +shard. `read_current` always keeps the original round-robin behavior; concurrent +stream reads still require caller serialization when ordering matters. + +Enable the policy explicitly on the constructed driver before sharing it: + +```rust,ignore +let driver = UringDriver::probe_and_start_sharded(128, 4)? + .with_shard_policy(rustfs_uring::ShardPolicy::CapacityAware); +``` + +This policy has additional admission work under contention. Throughput, CPU cost, +and tail-latency acceptance remain pending target-hardware measurements; it is +not enabled by default. + +### Read allocation admission + +With `max_in_flight_bytes: Some(budget)`, all shards share one byte budget. +Buffered reads reserve `len` bytes; direct reads reserve the block-aligned +superset length plus `align - 1` bytes of allocation padding, including for +zero-length direct reads. A request whose allocation exceeds the entire budget, +or whose logical length exceeds `max_read_len`, returns `InvalidInput` before +allocation. A byte budget of zero or above `tokio::sync::Semaphore::MAX_PERMITS` +is rejected at construction. `max_read_len: Some(0)` allows only zero-length reads. + +Admission acquires the shard's count permit before its byte permits. Saturated +handles wait asynchronously, holding no read buffer; a byte waiter may hold a +count permit, and Tokio's fair byte semaphore can put small reads behind a large +waiter. Dropping a waiting handle returns all partial reservations. After enqueue, +both permits travel with the read until its terminal CQE, even if its caller is +canceled. Short-read retries retain the same reservation. A leaked read retains +its charge. Shutdown or any shard-thread exit closes both the shared byte +semaphore and every shard's count semaphore when byte limits are enabled. +This rejects further admission and wakes waiters at either acquisition stage, +even when another shard has a hung read or takes a bounded-drain escape. +Registration and terminal closure are synchronized at startup/shutdown; ordinary +read admission does not take a registry lock. + +This limits reserved driver read-buffer allocation bytes, **not process RSS**. +It excludes queued handle/FD metadata, allocator overhead, result copies and +completed `Vec` results retained in channels or by callers. The caller must bound +its task fan-out and result queue separately. Completion releases admission even +when the returned result remains alive. ## API contract @@ -87,6 +148,11 @@ stage overlap, cancellation, and instrumentation-overhead boundaries. Benchmark configuration, CSV schema, timing boundaries, and performance gates are documented in [the benchmarking guide](docs/benchmarking.md). +See [implementation and acceptance status](docs/optimization-status.md) for +completed correctness work and the still-open performance/integration gates. +Application wiring has separate [RustFS integration prerequisites](docs/rustfs-integration.md). +The [ordered-prefetch example](docs/ordered-prefetch.md) is a bounded consumer +contract experiment, not a production streaming API or performance result. Linux only; on other hosts `cargo check` builds the empty stub. @@ -95,14 +161,18 @@ Linux only; on other hosts `cargo check` builds the empty stub. cargo test -- --nocapture --test-threads=1 # Two legs in Docker (also on macOS via Docker Desktop / OrbStack): -# leg 1 — io_uring blocked by an explicit seccomp profile → every test MUST -# degrade to a graceful skip; +# leg 1 — io_uring blocked by an explicit seccomp profile → ring-dependent +# tests gracefully skip; kernel-independent unit tests still run; # leg 2 — seccomp=unconfined → real io_uring, and NO test may skip. ./run-docker.sh ``` The harness fails on a non-degrading leg 1 or a vacuous-pass leg 2, so a skipped suite can never masquerade as coverage. The cancel-safety contract is pinned by the acceptance tests in `tests/cancel.rs`; the `fault-injection` feature (test-only) drives the panic-abort, bounded-drain-leak, and probe-failure escape hatches in `tests/fault_injection.rs`. +For bounded buffered read groups, see [explicit batch reads](docs/batch-reads.md). +`read_at_batch` shares eager notifications per owning shard while keeping each +read's admission, result and cancellation independent. + ## License Apache-2.0. See [LICENSE](LICENSE). diff --git a/docs/batch-reads.md b/docs/batch-reads.md new file mode 100644 index 0000000..0ee177b --- /dev/null +++ b/docs/batch-reads.md @@ -0,0 +1,34 @@ +# Explicit buffered read batches + +`UringDriver::read_at_batch(Vec)` accepts up to `MAX_BATCH_READS` +(64) buffered positioned reads and returns one `ReadHandle` per input, in input +order. More than 64 requests fails before any submission; an empty batch sends +no wakeup. Invalid individual offsets or lengths become normal asynchronous +handle errors and do not reject other requests. + +Eagerly admitted reads share one eventfd notification per distinct final owning +shard after handle construction. The capacity-aware policy, when selected, can +route several inputs to the same owner, which still receives one notification. +Requests awaiting count or byte admission signal their owner individually when +polled and admitted. No tasks are spawned and existing single-read, direct-read +and stream APIs retain their notification behavior. + +This is notification batching, not an atomic multi-read operation or snapshot. +Kernel submission/completion order may differ from input order; each handle +keeps independent cancellation and result ownership. Construction performs at +most 64 submissions and at most 64-by-64 pointer identity comparisons for wake +deduplication. It does not await capacity. If construction unwinds, already built +handles are dropped, queue their normal cancels, and wake the owning shards. +Driver-owned buffers, FDs and admission permits still survive until final CQE. + +Unit tests count real eventfd notifications using threadless driver queues, and +cover rejection, mixed valid/invalid requests, final-owner routing, deferred byte +admission, dropping handles and interrupted construction. Native integration +tests verify byte-exact results and cancellation/drain conservation. Threadless +tests model CQE resource release without submitting to a kernel. Notification +reduction is verified deterministically; throughput and latency are unmeasured. + +```sh +cargo test --all-features --lib batch_read_tests -- --nocapture +cargo test --all-features --test cancel batch_ -- --nocapture +``` diff --git a/docs/benchmarking.md b/docs/benchmarking.md index 6373202..dee3852 100644 --- a/docs/benchmarking.md +++ b/docs/benchmarking.md @@ -85,6 +85,15 @@ sample toward shard zero. Invalid requests can consume a sampling position without recording stages. Deterministic sampling is diagnostic, not an unbiased estimate for every possible periodic workload. +With opt-in `ShardPolicy::CapacityAware`, positioned reads choose their owner +before consuming that shard's sample sequence. Rejected geometry and unsuccessful +candidate attempts do not consume sample positions; closed admission records no +sample. For valid capacity-aware positioned reads, feature-on builds take one +timestamp before selection even for unsampled reads, so sampled admission covers +selection and permit acquisition (but excludes preceding geometry validation). +Round-robin and stream sampling retain their original behavior. Measure this +additional diagnostics cost with the selected routing policy. + `UringDriver::diagnostics()` aggregates shards; `UringDriver::shard_diagnostics()` preserves shard identity. Each stage has a count, total nanoseconds, and 64 log2 nanosecond buckets. Bucket zero covers @@ -117,7 +126,7 @@ even on std strategies (which do not use the instrumented driver). The default build compiles out timing fields, clock reads, histogram storage and sample allocations. Enabled builds add a per-shard atomic sampling counter per -handle and an Arc/timestamps/histogram updates for sampled operations. Measure +eligible handle and an Arc/timestamps/histogram updates for sampled operations. Measure that overhead on target hardware with separate feature-off/feature-on artifacts; do not assume it is free. Runtime schedule-latency and blocking-pool metrics must still be correlated in the application, which owns the Tokio runtime. @@ -162,9 +171,40 @@ service can introduce load, supply `--require-inactive-unit UNIT`; the runner refuses to start or continue unless that unit is inactive. Any service stop or restore is an explicit operator action outside this tool. A reservation note and process checks are evidence aids, not a substitute for exclusive resources. -Perform A/A calibration first by passing the baseline binary in both positions -and `--candidate-interval 0`; then use the diagnostics candidate and the default -interval of 64. Keep the workload and thresholds fixed between experiments. +Perform explicit A/A calibration first with `--calibration --candidate-interval 0` +and the same feature-off binary in both positions; then use the diagnostics +candidate without `--calibration` and with the default interval of 64. Keep the +workload and thresholds fixed between experiments. + +For example, use the preceding command with both executable arguments pointing +to the baseline artifact and add: + +```sh +--calibration --candidate-interval 0 +``` + +Calibration requires matching executable SHA-256 values before running either +artifact (identical copies at different paths are allowed), and requires interval +zero for every measurement row. A mismatched hash or nonzero candidate interval +is rejected, including in `--dry-run`. Each leg retains the normal binary-identity, +geometry, resource and duration checks. Dry-run, provenance and summary output +identify `mode` as `calibration` or `comparison`; dry-run does not execute workloads +or establish that runtime gates will pass. + +Calibration preserves at least three A1/B1/B2/A2 rounds and the existing default +thresholds: 3% for IOPS and 5% for p99. Every round checks A2 against A1, then checks +**B1 and B2 separately** against the arithmetic mean of A1/A2, using those same +thresholds in either direction. It never averages B1/B2 before gating: opposite +noise must not cancel out. The first failed round stops the experiment. Only +after all requested rounds pass does the summary say `valid-calibration`. +Calibration records `baseline_drift` and per-leg `middle_drift`, never +`candidate_change_pct`, whether it succeeds or fails. This validates the specified +within-round noise gates, not an optimization benefit or stability under another +workload. Cross-round trends, resources and application SLOs still require review. + +Omitting `--calibration` retains comparison mode and its original endpoint drift +gate. Passing the same binary with `--candidate-interval 0` alone does not enable +the additional middle-leg gates and must not be reported as validated calibration. Every leg must run at least five measured seconds by default. If it is too short, increase operations in a **new** experiment. The tool validates geometry, @@ -181,5 +221,9 @@ and resource reports against the application SLO separately. Duration histograms with too few sampled operations are not reliable tail estimates. Runner gate tests: `python3 -m unittest discover -s scripts -p 'test_bench_abba.py'`. +Calibration regressions use synthetic CSV/resource reports and mocked workload +execution to cover mode validation, matching hashes, endpoint and middle drift, +opposite-noise rejection, three-round success and early failure. They verify the +runner's decisions, not native Linux performance or isolation. Tracking and implementation status: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647). diff --git a/docs/cancellation-efficiency.md b/docs/cancellation-efficiency.md new file mode 100644 index 0000000..55429f6 --- /dev/null +++ b/docs/cancellation-efficiency.md @@ -0,0 +1,49 @@ +# Cancellation and eventfd efficiency + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647), step 3 (partial). + +## Implemented scope + +- [x] Retry interrupted eventfd signal/drain syscalls. Treat `EAGAIN` as an + already-pending wake on write, or an already-empty counter on read. Log an + unexpected error at most once per eventfd; the existing heartbeat remains. +- [x] Drain one successful counter read per turn. A concurrent later signal + stays readable; intake and completion processing still run after draining. +- [x] Record explicit drop-cancel intent separately from receiver closure. + After the read CQE arrives, stop positioned short-read and transient-error + continuations on explicit cancellation or shutdown with `ECANCELED`. +- [x] Preserve successful complete reads, EOF results, and `read_current` + short-read semantics when they race with cancellation. +- [x] Preserve `without_cancel_on_drop`: closing the receiver alone does not + stop positioned continuation. At the terminal completion, a closed receiver + skips result memmove/truncation/materialization, including direct-I/O padding. +- [x] Keep fd, buffer, and permit ownership until read completion and pending + removal. AsyncCancel CQEs still only update cancel statistics. +- [x] Bounded intake/reap and explicit finite-batch notifications (see + [driver fairness](driver-fairness.md) and [batch reads](batch-reads.md)). + No state-handshake/global wake coalescing is introduced. +- [x] Native Linux execution of the new tests and the full regression suite. +- [ ] Isolated before/after benchmark evidence; no measured speedup is claimed. + +## Review and validation + +The completion decision now lives in `reap_read` so deterministic tests can +provide short reads and transient errnos without regular-file timing races. +Tests cover explicit cancellation, shutdown, abandoned receivers without +cancellation, stream semantics, successful cancel races, direct result ranges, +unchanged orphan allocations, and resource ownership at terminal decisions. +Eventfd tests cover interrupted retries, saturated counters, empty drains, and +unexpected errors. They do not depend on io_uring availability or accept a skip. + +Local `cargo fmt --all --check`, `git diff --check`, and Linux-target +`cargo check` / `cargo clippy` with all targets and features passed. Cross-checks +compile the Linux-only tests but do not execute them. The combined hardening +patch subsequently passed the full all-feature suite on unrestricted Linux, +with no skipped tests and a mandatory positive O_DIRECT marker. + +Review confirms that no buffer is reclaimed at a cancel CQE or merely because +the receiver closed. The optimization only observes cancellation at a read +completion, with no live SQE writing into that buffer. A receiver closing after +the final `is_closed` check may still incur a copy; this is an accepted race, +not an ownership or correctness failure. Direct-I/O `fstat` error propagation +is covered by the integrated [integrity work](fault-recovery.md). diff --git a/docs/driver-fairness.md b/docs/driver-fairness.md new file mode 100644 index 0000000..054e21b --- /dev/null +++ b/docs/driver-fairness.md @@ -0,0 +1,48 @@ +# Driver turn fairness + +Each driver turn processes at most 64 input messages and 64 completion entries. +Intake also ends once the turn has allocated 8 MiB of read buffers, including +alignment padding. These private constants bound batches; they are not claimed +to be optimal throughput settings. Public configuration and ring flags remain +unchanged. + +An individual accepted read may exceed 8 MiB: it is allocated once and then +intake yields to submission and reap. Consequently the allocation threshold may +be exceeded by one allocation. These limits do not preempt a single allocation, +direct-read copy, metadata lookup, syscall, or shutdown's cancellation sweep. +Request-count and optional byte admission limits continue to govern ownership. + +The driver alternates bounded intake, submission, bounded reap, and a second +submission. This prevents a continuously replenished input or completion queue +from starving the other phase or shutdown-deadline checks. Messages remain FIFO; +this does not move cancellation or shutdown ahead of previously queued reads. + +Hitting either phase budget remembers that work may remain and starts the next +turn without waiting for another eventfd edge. This matters because eventfd has +already been drained even when input/CQ entries remain. A ready CQ also prevents +waiting. A partial successful submission continues immediately while queued SQ +work remains; failed or zero-progress submissions alone do not prevent waiting. +The existing active/idle heartbeats and two bounded submission attempts remain. +An exact budget boundary with no actual remaining work costs one empty turn, +after which normal idle waiting resumes. + +Single-read producer notifications are not coalesced; explicit bounded groups +can share notifications through [read_at_batch](batch-reads.md). No async-only +eventfd registration is introduced. Messages arriving after the driver's queue +check retain their eventfd notification; work +left because of a budget retains the explicit continuation flag. Buffers, FDs +and permits retain the existing final-CQE ownership rules. + +## Validation + +```sh +cargo test --all-features --lib loop_budget_tests -- --nocapture +``` + +Tests exercise a real eventfd plus an input queue whose wake signal is drained +before the intake limit, allocation fairness including an oversized read, +completion budget continuation, idle boundaries and submission retry decisions. +Submission-error/partial-submit decisions and CQ batches are deterministic +models, not injected kernel failures. Native read/cancellation/fault suites +remain necessary integration coverage. Throughput and tail-latency changes +require isolated before/after measurements and are currently unmeasured. diff --git a/docs/fault-recovery.md b/docs/fault-recovery.md new file mode 100644 index 0000000..65fbcb9 --- /dev/null +++ b/docs/fault-recovery.md @@ -0,0 +1,55 @@ +# Completion progress and direct-read integrity + +The driver skips `io_uring_enter` only when the submission queue is empty and +neither `IORING_SQ_CQ_OVERFLOW` nor `IORING_SQ_TASKRUN` requests kernel work. +The io-uring 0.7.15 submitter supplies `GETEVENTS` for these flags. After reaping, +the driver attempts submission again so newly freed CQ space can receive the +kernel's NODROP overflow entries even when no new reads arrive. + +Each driver turn makes at most two submission attempts. Partial submission, +zero progress, EINTR and EBUSY do not start an unbounded retry loop. Positive +partial progress with queued SQ work starts another bounded turn; errors and +zero progress alone wait for the next event or heartbeat. Pending buffers, file descriptors +and admission permits retain their existing final-CQE ownership rules. + +CQ-overflow warnings are emitted when the observed kernel counter changes to a +nonzero value, not on every turn while the cumulative counter remains nonzero. +The snapshot continues to report the kernel counter, including its u32 wrap: +a wrap to zero updates the snapshot silently, and a later nonzero value warns +again. This prevents repeated log output without changing completion recovery. + +A non-block-aligned O_DIRECT short read that does not cover the logical range +cannot be resumed from its unaligned endpoint. The driver checks file metadata: +confirmed EOF returns the initialized logical prefix; a mid-file endpoint +returns an error; metadata failure propagates its original I/O error. A metadata +failure is not evidence of EOF. Concurrent file mutation retains ordinary read +semantics, without providing snapshot isolation. + +## Validation + +On Linux, run: + +```sh +cargo test --all-features --lib fault_recovery_tests -- --nocapture +cargo test --all-features --lib submit_result_tests -- --nocapture +``` + +The metadata-result tests inject a completed prefix and metadata outcomes into +the production decision helper. They do not simulate a kernel short read or a +real `fstat` failure. The NOP overflow test creates a two-entry CQ, overflows it, +drains it, and verifies the empty-SQ submission path recovers the third CQE. It +uses actual kernel CQEs but no user buffers. The backlog test covers SQ-capacity +backpressure, not forced partial acceptance by `io_uring_enter`. + +Submit-result tests inject `io::Result` into the production classification +helper: partial positive acceptance, zero progress, EINTR/EBUSY and persistent +nontransient errors. They verify counter reset, the exact shutdown threshold, +one shutdown transition, deduplicated cancels and retained pending buffer/FD/ +count-permit/byte-permit ownership. They also check the loop's retry decision. +This models syscall return handling without inducing a partial return or failure +from a real kernel; it is not native kernel partial-submission evidence. + +Kernel tests print `SKIP` when setup returns an expected restriction error; +such a run is not evidence of overflow recovery. Run on an unrestricted Linux +host for kernel coverage. Advanced taskrun ring modes remain disabled by default +and are not enabled or kernel-validated by these tests. diff --git a/docs/optimization-status.md b/docs/optimization-status.md new file mode 100644 index 0000000..ff25684 --- /dev/null +++ b/docs/optimization-status.md @@ -0,0 +1,145 @@ +# Read-driver optimization status + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647). +Updated 2026-09-23. Implementation, correctness, CI, performance and application +integration are separate gates; a checked code item does not close the roadmap. + +| Step | Implementation and review | Remaining acceptance | +| --- | --- | --- | +| 1.1–1.4: benchmark schema, timing, diagnostics, direct positive gate | Merged in [#15](https://github.com/rustfs/uring/pull/15), CI passed | Diagnostics overhead and real LocalIoBackend baseline | +| 1.5: ABBA evidence tooling | Explicit same-binary calibration implemented/reviewed; 20 gate/cleanup tests pass | Stable native calibration, valid comparison and application integration | +| 2.1–2.4: completion recovery, direct integrity, byte admission, guarantee boundaries | Implemented and independently reviewed; native regression suite passed | PR merge; application-wide limits remain separate | +| 2.4 API follow-up: shutdown control and runtime adapter | Implemented/reviewed; native CI #55 passed at `a942ac1` | PR merge; no hard cleanup deadline or performance claim | +| 3.1–3.3: bounded turns, explicit batch notifications, cancellation efficiency | Implemented and independently reviewed; native regression suite passed | CPU/syscall and tail-latency comparison; PR merge | +| 4.1: capacity-aware routing | Implemented and independently reviewed, opt-in; native tests passed | Controlled slow-shard/mixed-load performance; PR merge | +| 4.2: system-wide budgets and probe offload | Application probe offload, driver-thread budget and logical chunk limit implemented/reviewed in draft PRs #8072/#8074/#8076 | Full CI/merge; physical byte/result/io-wq budgets and [dependency wiring](rustfs-integration.md) remain open | +| 4.2 pool prerequisite | Library whole-driver shared quota implemented/reviewed; native CI #57 passed at `b09b966` | Application dependency/fallback/result ownership wiring and merge | +| 5: owned buffers and direct FD cache | Dual-mode exact invalidation implemented/reviewed in application draft PR #8075; direct caching and owned buffers not enabled | Full CI/merge, profile evidence, lease/pool lifetime and complete invalidation integration | +| 6: ordered streaming prefetch | [Example-only contract experiment](ordered-prefetch.md) implemented/reviewed; 11 portable tests and native CLI CI #54 passed | Production consumer contract, bitrot/S3 and performance evidence | +| 7: advanced ring/runtime modes | Not enabled or implemented | Earlier gates, capability/fallback and isolated benefit evidence | + +## Correctness and review evidence + +The integrated code revision `7e2f1aa` passed 104 tests on unrestricted Linux: +71 unit, 4 admission, 21 read/cancellation, 6 fault and 2 shard-policy tests. +No test skipped; the mandatory `DIRECT_OK` marker was observed. All-feature +Clippy, formatting, 10 Python evidence-runner tests and the instrumented +eight-strategy benchmark correctness smoke also passed. The default-feature +native suite passed another 90 tests with no skips and the same direct marker. +A Linux-target +default/all-feature Clippy check and warning-denying rustdoc passed locally; +cross-compilation is not additional native execution evidence. + +The following documentation head `80a7af9` passed all three jobs in +[CI #53](https://github.com/rustfs/uring/actions/runs/35767797346), including +restricted/unrestricted Docker tests and both benchmark smoke modes. Later +follow-up commits need their own CI; the live PR and tracking issue record +final-head CI and merge status separately. + +## Follow-up evidence and examples + +`3931124` implements `SharedReadBudget` whole-driver reservations, with public +tests in `7e94f36`. Seven deterministic private tests and nine public/native +tests cover concurrent accounting, large `usize` quotas, message/deferred +ownership, independent shutdown, retained results and leaked whole quotas. +Two independent code/test/document reviews found no blocking issue. Local +Linux-target default/all-feature all-target checks, all-feature Clippy, +default/all-feature warning-denying rustdoc, formatting and diff checks pass; +the default library/dependency graph remains Tokio sync-only. The initial RED +was missing-API compilation, not a behavioral run. Subsequently, [CI #57](https://github.com/rustfs/uring/actions/runs/35796196402) +at `b09b966` passed all three jobs: 143 native all-feature tests plus one compiled +`no_run` documentation example, mandatory O_DIRECT, ordered-prefetch and both +benchmark CLI smokes, lint, and restricted/unrestricted Docker legs. All seven +private and nine public shared-budget tests ran successfully, including the +isolated whole-quota leak and greater-than-`u32` reservation cases. Later +evidence-only documentation commits have their own workflow status. Its +[contract](shared-read-budget.md) documents the +later-shard startup-failure coverage boundary and conservative idle/retirement +reservation costs. Application wiring and performance remain separate gates. + +Application work is tracked in [RustFS #8072](https://github.com/rustfs/rustfs/pull/8072), +[#8074](https://github.com/rustfs/rustfs/pull/8074), +[#8075](https://github.com/rustfs/rustfs/pull/8075), and +[#8076](https://github.com/rustfs/rustfs/pull/8076), not merged into application +main. The first two passed their native io_uring tests but failed workspace and +full E2E gates respectively; the latter two still have checks running. These +partial results do not establish successful application integration. + +`c7bb321` adds non-joining shutdown requests, advisory thread completion and the +default-off Tokio shutdown adapter. Count and byte admission close before a +request returns; the adapter transfers ownership before its result future is +polled. The [shutdown contract](shutdown.md) distinguishes thread completion, +successful join, clean drain, and runtime-shutdown failure boundaries. Tests in +`be66d8e` include real pending reads and an isolated nonclean-drain subprocess. +Two independent code reviews found no blocking issue. Linux-target default +all-target checking, all-feature Clippy, both feature-state rustdoc builds with +warnings denied, formatting and diff checks pass locally. These are compilation +checks, not native execution. Subsequently, [CI #55](https://github.com/rustfs/uring/actions/runs/35794075169) +at `a942ac1` passed all three jobs: 127 all-feature native tests (76 unit, 4 +admission, 21 read/cancel, 6 fault, 11 prefetch, 2 shard-policy, 7 shutdown), the +mandatory O_DIRECT marker, ordered-prefetch and both benchmark CLI smokes, lint +and 20 Python tests. Docker also passed restricted degradation and unrestricted +native legs; its fault-injection-only build does not enable `tokio-runtime`. +Later documentation-only commits do not change this tested implementation; +their own workflow status remains separate. No throughput or hard cleanup +deadline is claimed. + +`6927bb0` adds explicit same-binary calibration: identical executable content, +zero diagnostics interval on every leg, unchanged endpoint drift thresholds, +and a separate drift check for each middle leg against the endpoint mean. +Calibration never emits candidate-attribution fields; 20 Python tests pass, +including synthetic three-round execution and early-stop cases. These tests do +not establish a stable native calibration. + +`f135f1d` adds an example-local ordered reader, not a public streaming API. +Eleven portable tests pass, including actual Tokio semaphore reservation order, +cancelled `next`, EOF/error ordering, logical-byte accounting and slow-consumer +boundaries. The CLI uses an immutable fixture and positioned std oracle outside +the executor; the new CI smoke requires successful native io_uring byte checks +and bounds each child process duration. No production driver code or default +behavior changes in these follow-ups. + +Both additions and the current-main application integration plan received +independent review. Dedicated-host native execution is deferred while an existing +CI worker is active; the worker was not stopped and no performance run was made. +Portable checks and Linux-target compilation are not substitutes for the new +native CLI gate. Final-head CI/execution results are tracked in the issue/PR. + +Independent reviews covered buffer/FD/permit ownership, global closure and +count-stage waiters, cancellation races, final-owner routing, batch unwind and +notifications, fairness/progress, test validity and documentation. Review found +and fixed a global-close waiter gap and benchmark orphan-child cleanup. No +blocking finding remained in the completed component and combined-patch reviews. + +Kernel CQ-overflow and direct-read tests actually execute on Linux. Partial +submit/error classification and metadata failures use deterministic production +decision seams; they do not prove forced real-kernel partial acceptance, hung +syscall recovery, or advanced taskrun-mode behavior. + +## Performance status: not accepted + +The same-binary warm-cache A/A control used three A1/B1/B2/A2 rounds, 10 million +operations per leg and unchanged drift gates (3% throughput, 5% p99). The third +round's p99 baseline drift was **6.35%**, so the whole calibration was rejected +and candidate/diagnostics-overhead comparisons were not started. An earlier +shorter calibration was rejected for insufficient measurement duration; neither +run is performance evidence. The baseline executable predates this hardening +patch: these controls do not measure the new driver's speed. + +Do not infer throughput, latency, CPU or S3 improvements from correctness tests, +eventfd-count assertions, or rejected calibration. Keep diagnostics disabled +and round-robin routing as defaults. Before further performance-dependent work, +establish stable calibration without loosening gates after observing results. + +## Contract references + +- [Public read and admission contracts](../README.md) +- [Shutdown ownership and Tokio adapter](shutdown.md) +- [Shared whole-driver read-budget reservations](shared-read-budget.md) +- [Completion recovery and direct integrity](fault-recovery.md) +- [Cancellation and eventfd behavior](cancellation-efficiency.md) +- [Driver turn fairness](driver-fairness.md) +- [Explicit buffered batches](batch-reads.md) +- [Measurement boundaries and evidence gates](benchmarking.md) +- [Current RustFS application integration prerequisites](rustfs-integration.md) +- [Ordered-prefetch contract experiment](ordered-prefetch.md) diff --git a/docs/ordered-prefetch.md b/docs/ordered-prefetch.md new file mode 100644 index 0000000..9a267f8 --- /dev/null +++ b/docs/ordered-prefetch.md @@ -0,0 +1,120 @@ +# Ordered buffered prefetch: correctness experiment + +This example explores step 6 of [the optimization roadmap](optimization-status.md). +It does not add a public reader API, change production defaults, integrate an +object stream, or establish a performance improvement. Keep it as an example +and test harness until the consumer contract and performance gates are accepted. + +## Run against an existing immutable fixture + +```sh +cargo run --example ordered_prefetch -- FILE OFFSET LENGTH CHUNK WINDOW MAX_BYTES +``` + +All geometry arguments are decimal integers. `CHUNK` is 1 byte through 8 MiB, +`WINDOW` is 1 through 64, and `MAX_BYTES` is between `CHUNK` and 512 MiB. +`OFFSET + LENGTH` must fit `i64::MAX`. Zero length is supported. The file must +already exist and be regular; symlinks are refused. The example never creates, +truncates or modifies the fixture. Keep it immutable during verification: +independent positioned reads do not provide snapshot isolation. + +The example compares every returned chunk, including a short final chunk, +against a synchronous positioned std read outside the async executor. It checks +contiguous output, premature EOF and final driver drain conservation. Successful +execution prints `ORDERED_PREFETCH_OK bytes=N chunks=N`. Invalid input, unavailable +io_uring, read/reference mismatch or incomplete drain returns a nonzero status +without that marker. Non-Linux execution also fails explicitly. + +The driver uses one shard, two count permits and a one-chunk byte budget to +exercise deferred admission; the consumer window uses the supplied geometry. +This configuration and the reference reads are deliberately for correctness, +not a performance comparison. No new dependency or per-chunk task is introduced. + +## State and backpressure contract + +The example-local `OrderedReader::new(Config, source)` stores ordered slots of +`Pending(future)` or `Ready(result)`. The source must implement whole requested +range-or-EOF positioned reads, as `UringDriver::read_at` does. Arbitrary stream +short reads must not be substituted: the reader treats a successful short range +as EOF. Sources must return no more bytes than requested; excess output becomes +`InvalidData`. The generic source exists for deterministic tests, not as a +published production abstraction. + +`next(&mut self)` fills only available slots and logical bytes at the start of +a consumer poll, then polls all active slots, retaining out-of-order results. +It returns only the front slot and never refills after yielding that chunk. +No consumer polls means no replacement submissions. The constructor submits +nothing. A short result or error stops future scheduling and drops later slots; +earlier results remain ordered. A nonempty EOF prefix is yielded once, then +`None`; an empty EOF returns `None`; an ordered error is returned once, then +the reader is fused at `None`. + +Slots remain inside the reader across `Poll::Pending`. Dropping a pending +`next()` future neither advances the delivered offset nor drops the front +handle. Dropping the reader drops all its handles, retaining their normal +driver cancellation and final-CQE ownership rules. Earlier speculative reads +may already have completed, and cancellation is not guaranteed to win that race. + +Both pending and ready slots reserve their original requested logical length +until removal. Thus their combined requested lengths never exceed `MAX_BYTES`, +and their combined count never exceeds `WINDOW`. This is not an RSS or +whole-process memory bound: it excludes allocator overhead/excess Vec capacity, +driver structures, the example's reference buffer and already returned chunks +retained by the caller. Dropped kernel-visible buffers can also remain in the +driver until CQE; dropping a slot does not free them early. A terminal boundary +stops replenishment after dropping its speculative tail. + +## Admission and liveness + +Polling only the front is safe from this particular admission cycle only when +later handles have never been polled: eager later reads progress on the driver +and release permits at CQE, independently of consuming their results. Such a +simple reader can lose prefetch depth after saturation because deferred tails +remain unpolled. + +Once a deferred tail has been polled, switching to front-only polling can +deadlock an active consumer. For example, the head waits for a count permit on +shard A while a tail on shard B waits for shared bytes. Tokio may assign newly +released bytes to the tail before A's count becomes available. The head then +waits for bytes reserved by the tail; the tail must be repolled to submit and +eventually release them. The state machine polls every remaining active slot +on each consumer poll, including tails when the head is pending. Ready futures +are never polled again. Work per poll is bounded by the 64-slot limit; there is +no wait/spin loop or background producer. + +This does not promise global admission fairness while a consumer is paused. +Already-polled deferred handles can retain partial or newly assigned permits +until the consumer polls again or drops the reader. Application integration +must define stall timeouts and drop the reader when abandoning a stalled +consumer; merely dropping a pending `next()` preserves the reader and its +reservations intentionally. Paused consumer tests prove no refill, not continued +background permit release or unlimited progress by unrelated readers. + +## Deterministic coverage and remaining gates + +```sh +cargo test --test ordered_prefetch +``` + +The portable tests use controlled futures and actual Tokio count/byte semaphores +for the deferred-tail reservation race. They cover out-of-order completions, +ready-result byte accounting, slow-consumer refill boundaries, cancellation of +`next`, whole-reader drop, early/empty EOF, ordered errors, final range geometry, +invalid/overflow inputs and an oversized source result. These are state-machine +and admission tests, not native kernel stall or filesystem truncation injection. +Run the example on unrestricted Linux for actual io_uring byte correctness. + +Before any public API or RustFS integration, define whole-range versus exact +object-length EOF policy, stall timeout, quorum abandonment, error mapping and +file-generation consistency. Bitrot verification and checksum framing belong to +the application: raw byte equality in this example does not implement bitrot, +and arbitrary chunk boundaries cannot be assumed to match its verification +units. Consumer-retained results need a separate ownership/budget policy. + +Only after stable calibration should a measurement harness compare a sequential +buffered/readahead baseline and small ordered windows with matched consumer +backpressure, provenance and ABBA drift gates. Measure TTFB, p99/stall behavior, +throughput, CPU, syscall counts and memory residency. The existing unordered +JoinSet benchmark is not evidence for ordered-consumer behavior. Actual S3 +streaming, bitrot/quorum behavior, performance acceptance and production enablement +remain unimplemented/unverified by this experiment. diff --git a/docs/rustfs-integration.md b/docs/rustfs-integration.md new file mode 100644 index 0000000..f75d912 --- /dev/null +++ b/docs/rustfs-integration.md @@ -0,0 +1,87 @@ +# RustFS application integration prerequisites + +Tracking: [rustfs/backlog#2647](https://github.com/rustfs/backlog/issues/2647), +steps 4.2 and 5.2. This is a source-backed implementation plan, **not completed +application integration**. The reviewed RustFS main revision is +[`1880b42`](https://github.com/rustfs/rustfs/tree/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad) +(2026-09-23). Recheck the application head before implementation. + +## Current production boundary + +The application's [ecstore dependency](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/Cargo.toml#L230) +still uses registry `rustfs-uring = "0.2.2"`; it does not yet consume the new +`ReadLimits`, batch or shard-policy APIs. The [local backend](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L1087) +keeps io_uring opt-in, a depth of 128 per shard and a 128 MiB read chunk cap. +Configured shard counts multiply across disks; driver-local limits are not a +machine-wide budget. + +| Concern | Verified current behavior | Required integration | +| --- | --- | --- | +| Initialization | `LocalDisk::new` is async but calls synchronous probe/start | Bounded initialization offload plus explicit resource ownership | +| Teardown | Backend Drop already uses a blocking worker | Preserve it; account for retiring instances until real completion | +| Byte limits | Application has no `ReadLimits` wiring | Coordinate chunks, direct padding, per-driver allocation and fallback | +| Buffered descriptors | Existing FD cache with generation fencing | Reuse; do not replace invalidation with TTL-only behavior | +| Direct descriptors | Open/stat on each read; alignment is cached | Separate direct entries with all invalidation paths covered | +| Reclaim | Synchronous fadvise inside async read paths | Measure first; any offload must retain error and ordering semantics | + +Relevant paths: [backend lifecycle](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4186), +[async construction](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L5291), +[direct reads](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4561). + +## Required implementation order + +1. **Deliver the dependency and wire resource-aware chunks.** Use a released or + otherwise explicitly pinned uring version containing the new APIs. Validate + logical read size, actual aligned allocation charge and byte quota together; + the existing 128 MiB chunk cannot remain unconditional under a smaller quota. + Audit the [fallback path](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L4676): + it currently retries any uring error through the std backend. Merely setting + a smaller driver limit would cause fallback, not enforce application-wide + memory admission. An application-level budget must also cover fallback and + retained results; changing fallback/error classification needs its own tests. +2. **Budget and offload initialization.** Bound startup/reconnect concurrency + before scheduling blocking probe work. Account driver slots, rings and OS + threads across all disks, including retiring/leaked instances. A timeout + stops waiting but cannot kill an already running blocking syscall. Do not + immediately release its capacity and admit unlimited replacement probes. + Current `ReadLimits` covers one driver only: use conservatively allocated + per-driver quotas or design an explicitly shared application admission layer. + The existing best-effort per-ring io-wq setting is not a process thread cap. + The library's [shared reservation pool](shared-read-budget.md) offers the + conservative whole-driver quota option: pass the same pool across reconnects + and generations; leaked pending reads keep the entire reservation. It does + not wire the application's dependency, std fallback, result assembly, or + retained results into any budget. A temporary `WouldBlock` reservation error + must not enter the permanent unsupported-disk cache. +3. **Fix invalidation before adding direct cache hits.** + [Exact invalidation](https://github.com/rustfs/rustfs/blob/1880b42169bf26b15d8ca6bea3d8f1e4203b1bad/crates/ecstore/src/disk/local.rs#L3991) + currently builds a buffered-only key; invalidate both variants before + inserting direct entries. Preserve generation fencing, inode/length and + permission checks, alignment, FD capacity, prefix/volume invalidation and + open-to-insert race protection. Do not reuse a buffered descriptor as direct. + Specify permission freshness explicitly: today's direct path opens/checks on + every read, whereas buffered cache hits may skip some checks within the TTL. + Adopting its cache policy must not silently weaken the direct-read contract. +4. **Measure reclaim and validate the real backend.** If fadvise offload is + justified, use bounded blocking work and await its result; fire-and-forget + changes both errors and page-cache timing. Compare real LocalIoBackend cache + hit/miss, buffered/direct and fallback behavior after correctness gates, + rather than extrapolating from this crate's read microbenchmark. + +## Acceptance checklist (not yet completed) + +- [ ] Current application dependency contains the intended reviewed driver code. +- [ ] Below-quota/at-quota/over-quota and direct-padding cases; zero accidental + budget bypass through fallback; returned-result residency explicitly covered. +- [ ] Many-disk startup, reconnect storms, retiring instances, failed/timed-out + probes and shutdown retain correct capacity accounting and task responsiveness. +- [ ] Direct/buffered exact invalidation, heal same-path replacement, both rename + endpoints, delete, volume removal, open/insert races and permission failures. +- [ ] Real direct I/O executes; unavailable capability is not mistaken for a + passing direct test. Existing bitrot, EOF/range and error semantics preserved. +- [ ] Stable resource-isolated calibration precedes before/after comparisons; + CPU, tail latency, descriptors and memory remain within agreed bounds. + +The ordered-prefetch example in this repository cannot satisfy application +bitrot, quorum cancellation, stall policy or end-to-end stream acceptance. +Owned-buffer pooling and advanced ring modes remain evidence-gated separately. diff --git a/docs/shared-read-budget.md b/docs/shared-read-budget.md new file mode 100644 index 0000000..0763e1f --- /dev/null +++ b/docs/shared-read-budget.md @@ -0,0 +1,89 @@ +# Shared driver read-budget reservations + +`SharedReadBudget` is an opt-in pool for **whole-driver reservations**. It bounds +the sum of configured in-flight read-buffer quotas across drivers using the same +pool. It is not a dynamic shared per-read byte queue, returned-result limit, or +process RSS limit. + +## Admission + +Create a pool with `SharedReadBudget::new(total)` and pass a reference to +`UringDriver::probe_and_start_with_shared_budget(entries, shards, limits, &pool)`. +Cloning the pool shares its capacity; constructing a separate pool does not. +The new constructor requires an explicit `limits.max_in_flight_bytes = Some(B)`. +It validates the local limit and reserves all `B` bytes before ring startup. + +Zero total capacity, missing/invalid local byte limits, or `B > total` are +`InvalidInput`. Insufficient currently available capacity returns `WouldBlock` +without starting a driver; it is not an environment restriction and must not +poison a disk's unsupported-io_uring cache. Admission does not wait, queue a +blocking task, or partially reserve a driver. The caller decides when to retry +or use a separately budgeted fallback. Budgets use `usize`; whole-driver +reservations must not truncate through a per-read `u32` permit count. + +`capacity()` returns the fixed pool size. `available()` is an advisory snapshot +of capacity not reserved by drivers or their surviving resource owners. Do not +check it and then assume creation must succeed; another constructor can reserve +the same capacity before this caller's atomic reservation. + +Once admitted, a driver keeps the existing local count and byte semaphores. +Buffered reads charge logical length; direct reads charge the aligned superset +plus allocation padding. Read admission does not acquire from the global pool. +Each driver therefore retains its full reservation even while idle, and cannot +borrow unused capacity already reserved by another driver. This deliberately +trades utilization for a smaller shutdown/cancellation contract. + +## Ownership and retirement + +The same reservation receipt follows driver-owned and deferred read resources +through queued messages and pending kernel operations. Canceling the caller does +not refund a reservation while the kernel can still access a buffer. Normal +terminal CQEs release their owners; the whole block returns only after its last +owner disappears. A deferred handle retained after driver shutdown can therefore +keep the entire block reserved until its ownership is released. Drop unused +handles during retirement instead of treating `is_finished()` or a clean +snapshot as a pool-refund signal. + +A bounded-drain escape retains the receipt with the leaked pending resources, +so the entire `B` stays reserved, even if the leaked read is much smaller. A +replacement driver cannot reuse that capacity. There is no forced refund API. +Creating a new independent pool to bypass retained reservations also bypasses +the aggregate bound; the application must keep a stable shared pool across +reconnects and generations. + +Closing one driver closes only its local admission and wakes its local waiters. +It does not close the pool or reject reads on another admitted driver. Startup +failures roll back the reservation after resource cleanup; existing probe/setup +failure and synchronous cleanup contracts still apply. + +## Scope and cost + +This accounts configured driver read-buffer capacity, including direct padding +and retained pending reads. It excludes probe buffers, rings, kernel/io-wq +resources, allocator overhead, handle metadata, completion/result copies, +caller-held results and application std fallback allocations. Startup concurrency +and these other resources require separate controls. It does not constrain +drivers created without this pool. + +Pool atomics run at reservation/final release and explicit observation. Opt-in +reads retain a shared receipt; this is not a performance improvement claim. +Existing constructors and default limits keep their behavior. No application +dependency source or production pool wiring is changed by this library API. + +## Verification boundaries + +Deterministic tests cover concurrent weighted reservations, `usize` arithmetic, +receipt ownership in eager messages and deferred handles, and independent local +closure. Native tests cover multi-driver admission and retirement, retained +results, deferred count/byte cancellation, and a subprocess-injected leak that +keeps the entire quota. Restricted-environment tests also check that failed setup +refunds its reservation before reporting a capability skip. + +Failure after one or more earlier shards have started is covered by the existing +RAII cleanup path and code review, not a new later-shard fault-injection test. +The simulated stuck completion does not prove recovery from a real hung syscall. +See [acceptance status](optimization-status.md) for execution evidence; compiled +tests alone are not native I/O coverage. + +See [read admission](../README.md#read-allocation-admission), +[shutdown ownership](shutdown.md), and [application integration](rustfs-integration.md). diff --git a/docs/shutdown.md b/docs/shutdown.md new file mode 100644 index 0000000..d71cb65 --- /dev/null +++ b/docs/shutdown.md @@ -0,0 +1,80 @@ +# Shutdown ownership and runtime integration + +`request_shutdown`, `is_finished`, `shutdown`, and the optional `shutdown_async` +have different contracts. None can kill a blocked kernel syscall or promise a +hard cleanup deadline. + +## Request, observe, join + +`request_shutdown(&self)` closes every shard's count admission and the shared +byte admission, if configured, before returning. It wakes admission waiters and +sends each driver thread a shutdown request. It does not join or wait for reads. +Repeated and concurrent calls are safe; they can send repeated notifications. +Call this at lifecycle transitions, not in a polling loop. + +Valid reads, including zero-length reads, started after the request cannot gain +admission. Invalid inputs retain their validation errors. A read racing the +request may already hold admission: shutdown still +owns and settles its queued or submitted state. Neither admission closure nor +dropping a read handle releases a kernel-owned buffer early. Submitted buffers, +file descriptors and permits remain owned until the terminal read CQE, or are +retained by the leak-over-use-after-free escape path. + +`is_finished()` is an advisory query over the driver thread join handles. It is +not a join, a synchronization barrier, or proof of a clean drain. It can become +true just before final thread teardown, and can be true with nonzero +`stats().in_flight` after a bounded-drain leak. Do not return process-wide memory +credits merely because it is true. + +`shutdown(self)` requests all shards to stop, then synchronously joins them and +returns the final snapshot. `Drop` uses the same synchronous cleanup ordering. +The bounded drain only advances while the driver loop runs: a syscall that does +not return can also prevent the join from returning. A returned snapshot with +nonzero `in_flight` records retained kernel-owned resources, not clean recovery. + +## Optional Tokio adapter + +Enable `features = ["tokio-runtime"]` for `shutdown_async(self)`. Default builds +still require only Tokio synchronization support; read handles can be driven by +other executors. The adapter does not move each read onto Tokio's blocking pool. + +Call the method from an entered, live Tokio runtime. At **method call time**, it +requests shutdown and moves the driver into one blocking-pool task that performs +the consuming synchronous shutdown. This is deliberately an ordinary function +returning a `Send` future, not an `async fn` whose body starts at first poll. + +```rust +// `driver` is an exclusively owned UringDriver; tokio-runtime is enabled. +let shutdown = driver.shutdown_async(); // ownership transferred here +let stats = shutdown.await?; +if stats.in_flight != 0 { + // Report degraded cleanup and retain any external memory reservation. +} +``` + +Dropping or timing out the returned future, even before its first poll, detaches +the join handle during normal runtime operation; cleanup continues. It does not +abort a started blocking task or recover leaked buffers. Keep the runtime alive +until cleanup completes if that completion is required. Concurrent driver +retirement still needs application-level limits: the runtime's blocking queue +is not an admission budget. + +Calling without an entered runtime panics and drops the driver through its +synchronous cleanup path. A runtime shutting down may reject or discard queued +blocking work and synchronously drop its captured driver. Thus the adapter does +not make runtime teardown, panic unwinding, or arbitrary `Drop` nonblocking. +An `Ok` snapshot means the blocking task joined successfully, not necessarily +that `in_flight` is zero. A join error is returned as an `io::Error`. + +## Validation boundaries + +Private mock-thread tests exercise immediate count/byte closure, concurrent +requests, advisory completion, and eager ownership handoff with an occupied +blocking pool. Native tests exercise pending pipe reads, registered count/byte +waiters, current-thread Tokio cleanup, and dropped unpolled shutdown futures. +An isolated fault-injection subprocess verifies that finished threads can retain +nonzero in-flight resources. These are correctness tests, not performance or +real hung-device recovery evidence. + +See [completion recovery](fault-recovery.md), [integration prerequisites](rustfs-integration.md), +and [acceptance status](optimization-status.md) for the remaining gates. diff --git a/examples/ordered_prefetch.rs b/examples/ordered_prefetch.rs new file mode 100644 index 0000000..450c852 --- /dev/null +++ b/examples/ordered_prefetch.rs @@ -0,0 +1,30 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Correctness-only, example-local ordered buffered prefetch. No timing claims. + +#[cfg(target_os = "linux")] +#[path = "ordered_prefetch/reader.rs"] +mod reader; + +#[cfg(target_os = "linux")] +#[path = "ordered_prefetch/linux.rs"] +mod linux; + +fn main() -> std::process::ExitCode { + #[cfg(target_os = "linux")] + { + match linux::run() { + Ok(()) => std::process::ExitCode::SUCCESS, + Err(error) => { + eprintln!("ordered_prefetch: {error}"); + std::process::ExitCode::FAILURE + } + } + } + #[cfg(not(target_os = "linux"))] + { + eprintln!("ordered_prefetch requires Linux and a usable io_uring driver"); + std::process::ExitCode::FAILURE + } +} diff --git a/examples/ordered_prefetch/linux.rs b/examples/ordered_prefetch/linux.rs new file mode 100644 index 0000000..d7b4dd6 --- /dev/null +++ b/examples/ordered_prefetch/linux.rs @@ -0,0 +1,100 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::reader::{Config, OrderedReader}; +use rustfs_uring::{ReadLimits, UringDriver}; +use std::fs::{File, OpenOptions}; +use std::io; +use std::os::unix::fs::{FileExt, OpenOptionsExt}; +use std::sync::Arc; + +fn invalid(message: impl Into) -> io::Error { + io::Error::new(io::ErrorKind::InvalidInput, message.into()) +} + +fn number(name: &str, value: &str) -> io::Result { + value.parse().map_err(|_| invalid(format!("invalid {name}: {value}"))) +} + +fn reference_read(file: &File, offset: u64, len: usize) -> io::Result> { + let mut bytes = vec![0; len]; + let mut read = 0; + while read < len { + match file.read_at(&mut bytes[read..], offset + read as u64) { + Ok(0) => break, + Ok(count) => read += count, + Err(error) if error.kind() == io::ErrorKind::Interrupted => continue, + Err(error) => return Err(error), + } + } + bytes.truncate(read); + Ok(bytes) +} + +pub fn run() -> io::Result<()> { + let args: Vec = std::env::args().skip(1).collect(); + if args.len() != 6 { + return Err(invalid("usage: ordered_prefetch FILE OFFSET LENGTH CHUNK WINDOW MAX_BYTES")); + } + let cfg = Config { + offset: number("OFFSET", &args[1])?, + length: number("LENGTH", &args[2])?, + chunk: number("CHUNK", &args[3])?, + window: number("WINDOW", &args[4])?, + max_bytes: number("MAX_BYTES", &args[5])?, + }; + let end = cfg.validate()?; + // Never create/write/truncate the fixture or block opening a FIFO. Validate + // the opened descriptor, and refuse symlinks to keep fixture identity clear. + let file = OpenOptions::new() + .read(true) + .custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK) + .open(&args[0])?; + if !file.metadata()?.is_file() { + return Err(invalid("FILE must be an existing regular file")); + } + let file = Arc::new(file); + // Intentionally tight admission exercises deferred reads. This executable + // verifies correctness; it is not a throughput baseline or production tuning. + let driver = UringDriver::probe_and_start_with_limits( + 2, + 1, + ReadLimits { + max_read_len: Some(cfg.chunk), + max_in_flight_bytes: Some(cfg.chunk), + }, + ) + .map_err(io::Error::other)?; + let runtime = tokio::runtime::Builder::new_current_thread().build()?; + let mut reader = OrderedReader::new(cfg, |offset, len| driver.read_at(Arc::clone(&file), offset, len))?; + let result = (|| { + let mut delivered = 0u64; + let mut chunks = 0u64; + // Synchronous reference I/O runs outside the async executor. Pausing + // consumption for verification must not refill the reader's window. + while let Some(chunk) = runtime.block_on(reader.next())? { + let expected_offset = cfg.offset + delivered; + if chunk.offset != expected_offset { + return Err(io::Error::other("non-contiguous output offsets")); + } + let want = (end - expected_offset).min(cfg.chunk as u64) as usize; + if chunk.bytes != reference_read(&file, expected_offset, want)? { + return Err(io::Error::other("byte mismatch against positioned std read")); + } + delivered += chunk.bytes.len() as u64; + chunks += 1; + } + if cfg.offset + delivered < end && !reference_read(&file, cfg.offset + delivered, 1)?.is_empty() { + return Err(io::Error::other("ordered reader terminated before reference EOF")); + } + Ok((delivered, chunks)) + })(); + drop(reader); + let stats = driver.shutdown(); + let (bytes, chunks) = result?; + if stats.in_flight != 0 || stats.submitted != stats.delivered + stats.orphan_reclaimed { + return Err(io::Error::other("driver did not drain with read conservation intact")); + } + println!("ORDERED_PREFETCH_OK bytes={bytes} chunks={chunks}"); + Ok(()) +} diff --git a/examples/ordered_prefetch/reader.rs b/examples/ordered_prefetch/reader.rs new file mode 100644 index 0000000..4f119d2 --- /dev/null +++ b/examples/ordered_prefetch/reader.rs @@ -0,0 +1,175 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Example-only ordered consumer state machine. No production API or runtime. + +use std::collections::VecDeque; +use std::future::{Future, poll_fn}; +use std::io; +use std::pin::Pin; +use std::task::{Context, Poll}; + +const MAX_WINDOW: usize = 64; +const MAX_CHUNK: usize = 8 * 1024 * 1024; + +#[derive(Clone, Copy)] +pub struct Config { + pub offset: u64, + pub length: u64, + pub chunk: usize, + pub window: usize, + pub max_bytes: usize, +} + +impl Config { + pub fn validate(self) -> io::Result { + if self.chunk == 0 || self.chunk > MAX_CHUNK || self.window == 0 || self.window > MAX_WINDOW { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "chunk must be 1..=8 MiB and window 1..=64")); + } + if self.max_bytes < self.chunk || self.max_bytes > MAX_WINDOW * MAX_CHUNK { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "max_bytes must be at least chunk and at most 512 MiB", + )); + } + self.offset + .checked_add(self.length) + .filter(|end| *end <= i64::MAX as u64) + .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "range end must fit signed file offsets")) + } +} + +pub struct Chunk { + pub offset: u64, + pub bytes: Vec, +} + +enum State { + Pending(F), + Ready(io::Result>), +} + +struct Slot { + offset: u64, + len: usize, + state: State, +} + +pub struct OrderedReader { + config: Config, + end: u64, + next_offset: u64, + reserved: usize, + slots: VecDeque>, + source: S, + stop_scheduling: bool, + finished: bool, +} + +impl OrderedReader +where + F: Future>> + Unpin, + S: FnMut(u64, usize) -> F, +{ + pub fn new(config: Config, source: S) -> io::Result { + let end = config.validate()?; + Ok(Self { + config, + end, + next_offset: config.offset, + reserved: 0, + slots: VecDeque::with_capacity(config.window), + source, + stop_scheduling: false, + finished: false, + }) + } + + /// Dropping a pending next future leaves every handle and cursor in self. + /// No slot is removed before Ready, and Ready results are never repolled. + pub async fn next(&mut self) -> io::Result> { + poll_fn(|cx| self.poll_next(cx)).await + } + + fn fill_window(&mut self) { + while !self.stop_scheduling && self.next_offset < self.end && self.slots.len() < self.config.window { + let len = (self.end - self.next_offset).min(self.config.chunk as u64) as usize; + if len > self.config.max_bytes - self.reserved { + break; + } + let handle = (self.source)(self.next_offset, len); + self.slots.push_back(Slot { + offset: self.next_offset, + len, + state: State::Pending(handle), + }); + self.reserved += len; + self.next_offset += len as u64; + } + } + + fn poll_next(&mut self, cx: &mut Context<'_>) -> Poll>> { + if self.finished { + return Poll::Ready(Ok(None)); + } + // Refill only in a consumer poll, never after delivering a chunk. + self.fill_window(); + let mut terminal_index = None; + for (index, slot) in self.slots.iter_mut().enumerate() { + if let State::Pending(handle) = &mut slot.state + && let Poll::Ready(mut result) = Pin::new(handle).poll(cx) + { + if result.as_ref().is_ok_and(|bytes| bytes.len() > slot.len) { + result = Err(io::Error::new(io::ErrorKind::InvalidData, "source returned more bytes than requested")); + } + slot.state = State::Ready(result); + } + if let State::Ready(result) = &slot.state + && !result.as_ref().is_ok_and(|bytes| bytes.len() == slot.len) + { + // Preserve earlier chunks and the first ordered terminal result; + // later reads can no longer contribute and are dropped/cancelled. + terminal_index = Some(index); + break; + } + } + if let Some(index) = terminal_index { + self.stop_scheduling = true; + while self.slots.len() > index + 1 { + if let Some(slot) = self.slots.pop_back() { + self.reserved -= slot.len; + } + } + } + + match self.slots.front() { + Some(Slot { + state: State::Pending(_), + .. + }) => return Poll::Pending, + None => { + self.finished = true; + return Poll::Ready(Ok(None)); + } + Some(_) => {} + } + let slot = self.slots.pop_front().expect("ready front checked above"); + self.reserved -= slot.len; + let State::Ready(result) = slot.state else { unreachable!("ready front checked above") }; + match result { + Ok(bytes) => { + if bytes.len() < slot.len { + self.finished = true; + } + Poll::Ready(Ok((!bytes.is_empty()).then_some(Chunk { + offset: slot.offset, + bytes, + }))) + } + Err(error) => { + self.finished = true; + Poll::Ready(Err(error)) + } + } + } +} diff --git a/scripts/bench-abba.py b/scripts/bench-abba.py index 11e787b..0f8258d 100644 --- a/scripts/bench-abba.py +++ b/scripts/bench-abba.py @@ -66,7 +66,7 @@ def percent_change(after, before): return result -def evaluate_round(rows, throughput_drift, tail_drift): +def evaluate_round(rows, throughput_drift, tail_drift, *, calibration=False): if [row["leg"] for row in rows] != ["A1", "B1", "B2", "A2"]: raise ValueError("incomplete or reordered ABBA round") a1, b1, b2, a2 = [row["measurement"] for row in rows] @@ -76,6 +76,23 @@ def evaluate_round(rows, throughput_drift, tail_drift): } valid = abs(drift["iops_pct"]) <= throughput_drift and abs(drift["p99_pct"]) <= tail_drift result = {"valid": valid, "baseline_drift": drift} + if calibration: + # Each middle leg must be stable on its own: averaging B1/B2 first + # could cancel opposite noise and incorrectly certify calibration. + middle = { + leg: { + "iops_pct": percent_change(row["IOPS"], (a1["IOPS"] + a2["IOPS"]) / 2), + "p99_pct": percent_change(row["p99_us"], (a1["p99_us"] + a2["p99_us"]) / 2), + } + for leg, row in (("B1", b1), ("B2", b2)) + } + result["middle_drift"] = middle + result["valid"] = valid and all( + abs(drift["iops_pct"]) <= throughput_drift and abs(drift["p99_pct"]) <= tail_drift + for drift in middle.values() + ) + # A/A noise is never a candidate benefit, even when all gates pass. + return result # Do not calculate candidate attribution after a failed baseline gate. if valid: result["candidate_change_pct"] = { @@ -113,13 +130,15 @@ def execute(command, env, stdout, stderr, timeout, unit): if code != 0: raise RuntimeError(f"benchmark failed with exit code {code}; inspect the leg stderr") except BaseException: - if process.poll() is None: - try: - os.killpg(process.pid, signal.SIGKILL) - except ProcessLookupError: - pass - finally: - process.wait() + # The time wrapper can exit before its benchmark child. The owned + # process group can therefore still need cleanup after wait/poll + # has reaped its leader. + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + finally: + process.wait() raise @@ -164,8 +183,10 @@ def arguments(): parser.add_argument("--ops", type=int, default=1_000_000) parser.add_argument("--warmup-ops", type=int, default=10000) parser.add_argument("--rounds", type=int, default=3) + parser.add_argument("--calibration", action="store_true", + help="same-binary A/A calibration; also gate each middle leg; requires --candidate-interval 0") parser.add_argument("--candidate-interval", type=int, choices=(0, 64), default=64, - help="0 supports A/A calibration; 64 compares diagnostics on against off") + help="0 compares feature-off artifacts; 64 compares diagnostics on against off (default)") parser.add_argument("--throughput-drift-pct", type=float, default=3) parser.add_argument("--p99-drift-pct", type=float, default=5) parser.add_argument("--timeout", type=float, default=300) @@ -173,6 +194,8 @@ def arguments(): parser.add_argument("--min-seconds", type=float, default=5) parser.add_argument("--dry-run", action="store_true") args = parser.parse_args() + if args.calibration and args.candidate_interval != 0: + parser.error("calibration requires --candidate-interval 0 so every leg has the same diagnostics configuration") if not (100000 <= args.ops <= 10_000_000) or args.rounds < 3: parser.error("acceptance requires 100000..10000000 operations and at least three rounds") if any(not math.isfinite(value) or value < 0 for value in ( @@ -194,7 +217,11 @@ def arguments(): def main(): args = arguments() + mode = "calibration" if args.calibration else "comparison" binaries = {"A": args.baseline.resolve(), "B": args.candidate.resolve()} + identities = {key: {"path": str(path), "sha256": sha256(path)} for key, path in binaries.items()} + if args.calibration and identities["A"]["sha256"] != identities["B"]["sha256"]: + raise ValueError("calibration requires the same binary content for baseline and candidate") headers = {key: subprocess.check_output([str(path), "--header"], text=True).strip() for key, path in binaries.items()} if headers["A"] != headers["B"]: raise ValueError("baseline and candidate schema differ") @@ -203,18 +230,18 @@ def main(): ring_entries=args.entries, warmup_ops=args.warmup_ops) plan = [(round_id, leg) for round_id in range(1, args.rounds + 1) for leg in ("A1", "B1", "B2", "A2")] if args.dry_run: - print(json.dumps({"plan": plan, "geometry": expected, "candidate_interval": args.candidate_interval})) + print(json.dumps({"mode": mode, "plan": plan, "geometry": expected, "candidate_interval": args.candidate_interval})) return 0 environment_guard(args.require_inactive_unit) args.run_dir.mkdir(mode=0o700) - provenance = {"source_revision": args.source_revision, "reservation": args.reservation_note, + provenance = {"mode": mode, "source_revision": args.source_revision, "reservation": args.reservation_note, "cache": "warm-preload", "geometry": expected, "cpus": args.cpus, "data_identity": data_identity(args.data_file), - "binaries": {key: {"path": str(path), "sha256": sha256(path)} for key, path in binaries.items()}, + "binaries": identities, "gates": {"throughput_drift_pct": args.throughput_drift_pct, "p99_drift_pct": args.p99_drift_pct, "min_seconds": args.min_seconds}, "candidate_interval": args.candidate_interval} (args.run_dir / "provenance.json").write_text(json.dumps(provenance, indent=2) + "\n") - summary = {"status": "incomplete", "rounds": []} + summary = {"mode": mode, "status": "incomplete", "rounds": []} try: current_round = [] for round_id, leg in plan: @@ -252,12 +279,14 @@ def main(): current_round.append(record) print(f"round={round_id} leg={leg} IOPS={row['IOPS']:.0f} p99_us={row['p99_us']:.0f}", flush=True) if leg == "A2": - result = evaluate_round(current_round, args.throughput_drift_pct, args.p99_drift_pct) + result = evaluate_round(current_round, args.throughput_drift_pct, args.p99_drift_pct, + calibration=args.calibration) summary["rounds"].append(result) current_round = [] if not result["valid"]: - raise RuntimeError("baseline drift failed; stopped before expanding the experiment") - summary["status"] = "valid-comparison" + gate = "calibration drift" if args.calibration else "baseline drift" + raise RuntimeError(f"{gate} failed; stopped before expanding the experiment") + summary["status"] = "valid-calibration" if args.calibration else "valid-comparison" summary["scope"] = "driver-only; resource CSV is whole-process CPU/RSS, not steady-state-only" return 0 except (ValueError, RuntimeError, OSError, subprocess.SubprocessError) as error: diff --git a/scripts/test-ordered-prefetch-cli.sh b/scripts/test-ordered-prefetch-cli.sh new file mode 100644 index 0000000..3596f71 --- /dev/null +++ b/scripts/test-ordered-prefetch-cli.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Copyright 2024 RustFS Team +# SPDX-License-Identifier: Apache-2.0 +set -euo pipefail +cd "$(dirname "$0")/.." + +if [[ "${BENCH_SKIP_BUILD:-0}" != 1 ]]; then + cargo build --locked --example ordered_prefetch --example streaming_bench +fi +bin="${BENCH_BIN_DIR:-${CARGO_TARGET_DIR:-target}/debug/examples}" +scratch=$(mktemp -d) +trap 'rm -rf -- "$scratch"' EXIT + +# Reuse the existing deterministic fixture generator/untimed byte verifier. +# This smoke checks contracts; no output is performance evidence. +BENCH_VERIFY=1 "$bin/streaming_bench" std_buffered "$scratch/data.bin" \ + 1048583 65536 1 4096 >/dev/null + +check_range() { + local expected=$1 + shift + local output + output=$(timeout --kill-after=5s 30s "$bin/ordered_prefetch" "$scratch/data.bin" "$@") + if [[ "$output" != "ORDERED_PREFETCH_OK bytes=$expected chunks="* ]]; then + echo "missing ordered-prefetch positive marker: $output" >&2 + exit 1 + fi +} + +# A non-aligned range spanning many windows, with a tighter byte than slot cap. +check_range 65543 17 65543 4097 4 8194 +# Early EOF must deliver only real bytes and terminate the ordered tail. +check_range 13 1048570 1000 7 8 56 +check_range 0 0 0 1024 4 4096 + +reject() { + local expected=$1 + shift + local code=0 + timeout --kill-after=5s 30s "$bin/ordered_prefetch" "$scratch/data.bin" "$@" \ + >"$scratch/rejected.stdout" 2>"$scratch/rejected.stderr" || code=$? + if [[ "$code" != 1 || -s "$scratch/rejected.stdout" ]] || + ! grep -Fqx "ordered_prefetch: $expected" "$scratch/rejected.stderr"; then + echo "invalid geometry did not return the normal CLI error (exit=$code)" >&2 + cat "$scratch/rejected.stderr" >&2 + exit 1 + fi +} +reject 'chunk must be 1..=8 MiB and window 1..=64' 0 100 0 4 4096 +reject 'chunk must be 1..=8 MiB and window 1..=64' 0 100 16 0 4096 +reject 'max_bytes must be at least chunk and at most 512 MiB' 0 100 16 4 0 +reject 'range end must fit signed file offsets' 18446744073709551615 2 16 4 4096 +echo "ordered-prefetch range, byte budget, EOF and validation smoke passed" diff --git a/scripts/test_bench_abba.py b/scripts/test_bench_abba.py index 5339ae5..b1283a7 100644 --- a/scripts/test_bench_abba.py +++ b/scripts/test_bench_abba.py @@ -1,8 +1,12 @@ # Copyright 2024 RustFS Team # SPDX-License-Identifier: Apache-2.0 import importlib.util +import io +import json import os from pathlib import Path +import select +import signal import sys import tempfile import unittest @@ -78,6 +82,193 @@ def test_timeout_terminates_the_owned_process_group(self): with self.assertRaises(ProcessLookupError): os.kill(pid, 0) + def test_failed_leader_does_not_leave_its_child_running(self): + # The child holds a FIFO writer open before allowing the leader to exit. + # EOF proves it exited without depending on orphan/zombie reaping timing. + child = """ +import os, sys, time +print(os.getpid(), flush=True) +ready_read, ready_write = os.pipe() +if os.fork() == 0: + os.close(ready_read) + writer = os.open(sys.argv[1], os.O_WRONLY) + os.write(writer, b"ready") + os.write(ready_write, b"ready") + time.sleep(30) + os._exit(0) +os.close(ready_write) +os.read(ready_read, 5) +os._exit(7) +""" + with tempfile.TemporaryDirectory() as directory, patch.object(BENCH, "environment_guard"): + out = Path(directory) / "out" + err = Path(directory) / "err" + fifo = Path(directory) / "child-lifetime" + os.mkfifo(fifo) + reader = os.open(fifo, os.O_RDONLY | os.O_NONBLOCK) + try: + with self.assertRaisesRegex(RuntimeError, "exit code 7"): + BENCH.execute([sys.executable, "-c", child, str(fifo)], os.environ.copy(), out, err, 5, None) + self.assertEqual(os.read(reader, 5), b"ready") + readable, _, _ = select.select([reader], [], [], 5) + self.assertEqual(readable, [reader], "benchmark descendant still holds the FIFO open") + self.assertEqual(os.read(reader, 1), b"") + finally: + os.close(reader) + # Also clean up when this regression is run against broken code. + if out.exists() and out.read_text().strip(): + try: + os.killpg(int(out.read_text().strip()), signal.SIGKILL) + except ProcessLookupError: + pass + + def test_missing_process_group_preserves_the_benchmark_failure(self): + with tempfile.TemporaryDirectory() as directory, patch.object(BENCH, "environment_guard"), \ + patch.object(BENCH.os, "killpg", side_effect=ProcessLookupError) as killpg: + out = Path(directory) / "out" + err = Path(directory) / "err" + with self.assertRaisesRegex(RuntimeError, "exit code 7"): + BENCH.execute([sys.executable, "-c", "raise SystemExit(7)"], os.environ.copy(), out, err, 5, None) + killpg.assert_called_once() + + +class CalibrationGates(unittest.TestCase): + HEADER = ("schema_version,mode,strategy,shards,file_size,read_size,concurrency,ops,secs,IOPS,MBps," + "p50_us,p99_us,p999_us,startup_secs,shutdown_secs,workers,ring_entries,warmup_ops,diagnostics_interval") + + def rows(self): + return [{"leg": leg, "measurement": dict(IOPS=100, p99_us=20, p999_us=30)} + for leg in ("A1", "B1", "B2", "A2")] + + def argv(self, root, *extra): + (root / "baseline").write_bytes(b"identical executable content") + (root / "candidate").write_bytes(b"identical executable content") + (root / "data").write_bytes(bytes(36864)) + return ["bench-abba.py", "--baseline", str(root / "baseline"), "--candidate", str(root / "candidate"), + "--data-file", str(root / "data"), "--run-dir", str(root / "results"), + "--source-revision", "test-revision", "--reservation-note", "unit test; no benchmark executed", + "--calibration", "--candidate-interval", "0", *extra] + + def test_calibration_success_has_no_candidate_attribution(self): + result = BENCH.evaluate_round(self.rows(), 3, 5, calibration=True) + self.assertTrue(result["valid"]) + self.assertEqual(set(result["middle_drift"]), {"B1", "B2"}) + self.assertNotIn("candidate_change_pct", result) + + def test_each_middle_leg_is_checked_without_opposite_noise_cancellation(self): + for metric, low, high in (("IOPS", 90, 110), ("p99_us", 18, 22)): + with self.subTest(metric=metric): + rows = self.rows() + rows[1]["measurement"][metric] = low + rows[2]["measurement"][metric] = high + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertFalse(result["valid"]) + self.assertNotIn("candidate_change_pct", result) + + def test_middle_gates_use_endpoint_mean(self): + rows = self.rows() + for row, iops in zip(rows, (100, 103, 103, 102)): + row["measurement"]["IOPS"] = iops + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertTrue(result["valid"]) + self.assertAlmostEqual(result["middle_drift"]["B1"]["iops_pct"], 100 * (103 / 101 - 1)) + + def test_calibration_endpoint_drift_still_invalidates_without_attribution(self): + for metric, changed in (("IOPS", 110), ("p99_us", 22)): + with self.subTest(metric=metric): + rows = self.rows() + rows[3]["measurement"][metric] = changed + result = BENCH.evaluate_round(rows, 3, 5, calibration=True) + self.assertFalse(result["valid"]) + self.assertNotIn("candidate_change_pct", result) + + def test_calibration_requires_zero_interval_and_three_rounds(self): + for extra in (("--candidate-interval", "64"), ("--rounds", "2")): + with self.subTest(extra=extra), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + with patch.object(sys, "argv", self.argv(root, *extra)), patch("sys.stderr", new_callable=io.StringIO): + with self.assertRaises(SystemExit) as error: + BENCH.arguments() + self.assertEqual(error.exception.code, 2) + self.assertFalse((root / "results").exists()) + + def test_default_mode_still_compares_different_feature_states(self): + with tempfile.TemporaryDirectory() as directory: + argv = self.argv(Path(directory)) + argv.remove("--calibration") + del argv[-2:] # Preserve the default candidate interval of 64. + with patch.object(sys, "argv", argv): + args = BENCH.arguments() + self.assertFalse(args.calibration) + self.assertEqual(args.candidate_interval, 64) + + def test_different_binary_hashes_reject_calibration_before_execution(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + argv = self.argv(root) + (root / "candidate").write_bytes(b"different executable content") + with patch.object(sys, "argv", argv), patch.object(BENCH.subprocess, "check_output") as command: + with self.assertRaisesRegex(ValueError, "same binary"): + BENCH.main() + command.assert_not_called() + self.assertFalse((root / "results").exists()) + + def test_calibration_dry_run_validates_identity_and_reports_its_mode(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + with patch.object(sys, "argv", self.argv(root, "--dry-run")), \ + patch.object(BENCH.subprocess, "check_output", return_value=self.HEADER), \ + patch("sys.stdout", new_callable=io.StringIO) as output: + self.assertEqual(BENCH.main(), 0) + plan = json.loads(output.getvalue()) + self.assertEqual(plan["mode"], "calibration") + self.assertEqual(len(plan["plan"]), 12) + self.assertFalse((root / "results").exists()) + + def run_main_with_measurements(self, root, invalid_leg=None): + executed = [] + + def execute(command, env, stdout, stderr, timeout, unit): + executed.append(stdout.stem) + self.assertEqual(env["BENCH_DIAGNOSTICS"], "0") + iops = 90000 if stdout.stem == invalid_leg else 100000 + secs = 1_000_000 / iops + stdout.write_text(f"2,measure,uring_cached_read,2,36864,32768,32,1000000,{secs:.6f},{iops}," + f"{iops / 32:.1f},10,20,30,0.1,0.1,4,64,10000,0\n") + stderr.write_text("") + Path(command[4]).write_text(f"1,1,{secs + 1},1000,3,4\n") + + with patch.object(sys, "argv", self.argv(root)), \ + patch.object(BENCH.subprocess, "check_output", return_value=self.HEADER), \ + patch.object(BENCH, "environment_guard"), patch.object(BENCH.time, "sleep"), \ + patch.object(BENCH, "execute", side_effect=execute), \ + patch("sys.stdout", new_callable=io.StringIO), patch("sys.stderr", new_callable=io.StringIO): + code = BENCH.main() + summary = json.loads((root / "results" / "summary.json").read_text()) + provenance = json.loads((root / "results" / "provenance.json").read_text()) + return code, summary, provenance, executed + + def test_three_valid_rounds_report_valid_calibration(self): + with tempfile.TemporaryDirectory() as directory: + code, summary, provenance, executed = self.run_main_with_measurements(Path(directory)) + self.assertEqual(code, 0) + self.assertEqual(summary["mode"], "calibration") + self.assertEqual(summary["status"], "valid-calibration") + self.assertEqual(len(summary["rounds"]), 3) + self.assertEqual(len(executed), 12) + self.assertEqual(provenance["mode"], "calibration") + self.assertEqual(provenance["binaries"]["A"]["sha256"], provenance["binaries"]["B"]["sha256"]) + self.assertNotIn("candidate_change_pct", json.dumps(summary)) + + def test_middle_failure_stops_calibration_without_attribution_from_earlier_rounds(self): + with tempfile.TemporaryDirectory() as directory: + code, summary, _, executed = self.run_main_with_measurements(Path(directory), "round-2-B1") + self.assertEqual(code, 1) + self.assertEqual(summary["status"], "invalid") + self.assertEqual(len(summary["rounds"]), 2) + self.assertEqual(len(executed), 8) + self.assertNotIn("candidate_change_pct", json.dumps(summary)) + if __name__ == "__main__": unittest.main() diff --git a/src/admission_tests.rs b/src/admission_tests.rs new file mode 100644 index 0000000..a984cac --- /dev/null +++ b/src/admission_tests.rs @@ -0,0 +1,257 @@ +//! Deterministic admission tests; no kernel ring is needed. +use super::*; +use std::task::Waker; + +fn poll_once(future: &mut AcquireFut) -> Poll> { + future.as_mut().poll(&mut Context::from_waker(Waker::noop())) +} + +#[test] +fn byte_saturation_returns_partial_count_reservation_on_drop() { + let count = Arc::new(Semaphore::new(2)); + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8, None); + assert!(poll_once(&mut waiting).is_pending()); + assert_eq!(count.available_permits(), 0); + drop(waiting); + assert_eq!(count.available_permits(), 1); + assert_eq!(bytes.available_permits(), 0); + drop(first); + assert_eq!(bytes.available_permits(), 8); +} + +#[test] +fn dropping_count_waiter_does_not_reserve_bytes() { + let count = Arc::new(Semaphore::new(1)); + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&count, Some(&bytes), 4).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 4, None); + assert!(poll_once(&mut waiting).is_pending()); + drop(waiting); + assert_eq!(bytes.available_permits(), 4); + drop(first); + assert_eq!(count.available_permits(), 1); +} + +#[test] +fn shard_exit_closes_bytes_and_releases_waiting_count_permit() { + let count = Arc::new(Semaphore::new(2)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let bytes = admission.bytes.clone(); + let first = try_read_permits(&count, Some(&bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(bytes.clone()), 8, None); + assert!(poll_once(&mut waiting).is_pending()); + drop(CloseByteAdmission(Some(admission))); + assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); + assert_eq!(count.available_permits(), 1); + assert_eq!(bytes.available_permits(), 0); + drop(first); +} + +#[test] +fn global_close_wakes_count_stage_waiter_without_releasing_hung_read() { + struct LockCheckingWake { + admission: Arc, + wakes: AtomicUsize, + } + impl std::task::Wake for LockCheckingWake { + fn wake(self: Arc) { + self.wake_by_ref(); + } + + fn wake_by_ref(self: &Arc) { + assert!(self.admission.registry.try_lock().is_ok(), "registry lock held across task wake"); + self.wakes.fetch_add(1, Ordering::SeqCst); + } + } + + let count = Arc::new(Semaphore::new(1)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let hung = try_read_permits(&count, Some(&admission.bytes), 8).unwrap(); + let mut waiting = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8, None); + let wake = Arc::new(LockCheckingWake { + admission: admission.clone(), + wakes: AtomicUsize::new(0), + }); + let waker = Waker::from(wake.clone()); + assert!(waiting.as_mut().poll(&mut Context::from_waker(&waker)).is_pending()); + + // Model a different shard exiting while this shard's read stays hung. + drop(CloseByteAdmission(Some(admission.clone()))); + assert!(wake.wakes.load(Ordering::SeqCst) > 0, "count waiter was not woken"); + assert!(matches!(poll_once(&mut waiting), Poll::Ready(Err(_)))); + assert!(matches!( + try_read_permits(&count, Some(&admission.bytes), 8), + Err(TryAcquireError::Closed) + )); + let mut new_waiter = acquire_read_permits(count.clone(), Some(admission.bytes.clone()), 8, None); + assert!(matches!(poll_once(&mut new_waiter), Poll::Ready(Err(_)))); + assert_eq!(admission.bytes.available_permits(), 0, "hung read must remain charged"); + drop(hung); +} + +#[test] +fn late_registration_is_closed_after_terminal_admission_close() { + let admission = ByteAdmission::new(8); + admission.close(); + let count = Arc::new(Semaphore::new(1)); + admission.register(&count); + assert!(count.is_closed()); + assert!(admission.registry.lock().unwrap().counts.is_empty()); +} + +#[test] +fn concurrent_registration_cannot_escape_terminal_close() { + for _ in 0..16 { + let admission = ByteAdmission::new(8); + let count = Arc::new(Semaphore::new(1)); + let barrier = std::sync::Barrier::new(2); + std::thread::scope(|scope| { + scope.spawn(|| { + barrier.wait(); + admission.register(&count); + }); + barrier.wait(); + admission.close(); + }); + assert!(count.is_closed()); + assert!(admission.bytes.is_closed()); + } +} + +#[test] +fn canceling_partial_fifo_byte_reservation_unblocks_next_waiter() { + let count = Arc::new(Semaphore::new(4)); + let bytes = Arc::new(Semaphore::new(8)); + let held = try_read_permits(&count, Some(&bytes), 4).unwrap(); + let mut large = acquire_read_permits(count.clone(), Some(bytes.clone()), 6, None); + assert!(poll_once(&mut large).is_pending()); + assert_eq!(bytes.available_permits(), 0, "large FIFO waiter should reserve the available four bytes"); + let mut small = acquire_read_permits(count.clone(), Some(bytes.clone()), 2, None); + assert!(poll_once(&mut small).is_pending()); + drop(large); + let Poll::Ready(Ok(small)) = poll_once(&mut small) else { + panic!("canceling partial reservation did not advance FIFO") + }; + assert_eq!(bytes.available_permits(), 2); + drop(held); + drop(small); + assert_eq!(bytes.available_permits(), 8); + assert_eq!(count.available_permits(), 4); +} + +#[test] +fn permits_retained_by_leaked_operation_remain_charged_after_close() { + let count = Arc::new(Semaphore::new(1)); + let admission = Arc::new(ByteAdmission::new(8)); + admission.register(&count); + let bytes = admission.bytes.clone(); + let leaked = try_read_permits(&count, Some(&bytes), 8).unwrap(); + std::mem::forget(leaked); + drop(CloseByteAdmission(Some(admission))); + assert_eq!(bytes.available_permits(), 0); + assert_eq!(count.available_permits(), 0); +} + +#[test] +fn shared_byte_budget_serializes_reads_from_distinct_shards() { + let counts = [Arc::new(Semaphore::new(1)), Arc::new(Semaphore::new(1))]; + let bytes = Arc::new(Semaphore::new(8)); + let first = try_read_permits(&counts[0], Some(&bytes), 8).unwrap(); + assert!(matches!(try_read_permits(&counts[1], Some(&bytes), 1), Err(TryAcquireError::NoPermits))); + assert_eq!(counts[1].available_permits(), 1); + let mut waiting = acquire_read_permits(counts[1].clone(), Some(bytes.clone()), 8, None); + assert!(poll_once(&mut waiting).is_pending()); + drop(first); + let Poll::Ready(Ok(second)) = poll_once(&mut waiting) else { panic!("budget was not released") }; + drop(second); + assert_eq!(bytes.available_permits(), 8); +} + +fn fake_driver(limits: ReadLimits) -> (UringDriver, mpsc::Receiver) { + let (tx, rx) = mpsc::channel(); + let count = Arc::new(Semaphore::new(8)); + let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + if let Some(admission) = &byte_admission { + admission.register(&count); + } + ( + UringDriver { + limits, + shard_policy: ShardPolicy::default(), + byte_admission, + shards: vec![Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem: count, + wake_efd: Arc::new(EventFd::new().unwrap()), + }], + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + rx, + ) +} + +#[tokio::test] +async fn logical_limit_rejects_before_message_or_allocation() { + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: Some(4), + max_in_flight_bytes: Some(8192), + }); + let err = driver + .read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 5) + .await + .unwrap_err(); + assert_eq!(err.kind(), io::ErrorKind::InvalidInput); + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); +} + +#[tokio::test] +async fn direct_budget_includes_superset_padding_and_zero_length_geometry() { + for (offset, len, charge) in [(1, 4096, 12287), (0, 0, 4095), (1, 0, 8191)] { + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(charge - 1), + }); + let file = Arc::new(File::open("/dev/zero").unwrap()); + let err = driver.read_at_direct(file.clone(), offset, len, 4096).await.unwrap_err(); + assert_eq!(err.kind(), io::ErrorKind::InvalidInput); + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); + + let (driver, rx) = fake_driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(charge), + }); + let handle = driver.read_at_direct(file, offset, len, 4096).without_cancel_on_drop(); + let msg = rx.try_recv().unwrap(); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 0); + drop(handle); + assert_eq!( + driver.byte_admission.as_ref().unwrap().bytes.available_permits(), + 0, + "caller drop released bytes" + ); + drop(msg); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), charge); + } +} + +#[test] +fn invalid_budget_is_rejected_before_kernel_probe() { + for budget in [0, Semaphore::MAX_PERMITS + 1] { + let result = UringDriver::probe_and_start_with_limits( + 8, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(budget), + }, + ); + assert!(matches!(result, Err(ProbeFailure::Setup(err)) if err.kind() == io::ErrorKind::InvalidInput)); + } +} diff --git a/src/batch_read_tests.rs b/src/batch_read_tests.rs new file mode 100644 index 0000000..0796c2e --- /dev/null +++ b/src/batch_read_tests.rs @@ -0,0 +1,192 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +// Threadless driver: message queues own real permits, and eventfd counters let +// tests count notifications without races with a running consumer. +fn driver(shard_count: usize, capacity: usize, bytes: Option) -> (UringDriver, Vec>) { + let byte_admission = bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + let mut receivers = Vec::new(); + let shards = (0..shard_count) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(capacity)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().unwrap()), + } + }) + .collect(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: bytes, + }, + shard_policy: ShardPolicy::default(), + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) +} + +fn requests(count: usize) -> Vec { + let file = Arc::new(File::open("/dev/zero").unwrap()); + (0..count) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: index as u64, + len: 8, + }) + .collect() +} + +fn notifications(event: &EventFd) -> u64 { + let mut count = 0u64; + // SAFETY: read writes one u64 into valid memory; eventfd is nonblocking. + let result = unsafe { libc::read(event.as_raw(), (&mut count as *mut u64).cast(), 8) }; + if result == -1 { + assert_eq!(io::Error::last_os_error().raw_os_error(), Some(libc::EAGAIN)); + return 0; + } + assert_eq!(result, 8); + count +} + +fn poll(handle: &mut ReadHandle) -> Poll>> { + Pin::new(handle).poll(&mut Context::from_waker(std::task::Waker::noop())) +} + +#[test] +fn batch_signals_once_per_owner_while_single_reads_keep_individual_signals() { + let (driver, receivers) = driver(2, 64, None); + let handles = driver.read_at_batch(requests(MAX_BATCH_READS)).unwrap(); + assert_eq!(handles.len(), MAX_BATCH_READS); + for (shard, receiver) in driver.shards.iter().zip(&receivers) { + assert_eq!(notifications(&shard.wake_efd), 1); + assert_eq!(receiver.try_iter().count(), MAX_BATCH_READS / 2); + } + let single: Vec<_> = requests(4) + .into_iter() + .map(|r| driver.read_at(r.file, r.offset, r.len)) + .collect(); + for shard in &driver.shards { + assert_eq!(notifications(&shard.wake_efd), 2); + } + drop(single); + drop(handles); +} + +#[test] +fn oversized_batch_rejects_every_request_before_submission() { + let (driver, receivers) = driver(1, 64, None); + let error = driver.read_at_batch(requests(MAX_BATCH_READS + 1)).err().unwrap(); + assert_eq!(error.kind(), io::ErrorKind::InvalidInput); + assert_eq!(driver.next_id.load(Ordering::Relaxed), 1); + assert_eq!(driver.rr.load(Ordering::Relaxed), 0); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); +} + +#[test] +fn empty_and_invalid_requests_do_not_notify_or_reject_valid_members() { + let (driver, receivers) = driver(1, 4, None); + assert!(driver.read_at_batch(Vec::new()).unwrap().is_empty()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + let mut invalid = requests(1); + invalid[0].offset = u64::MAX; + let mut invalid_handles = driver.read_at_batch(invalid).unwrap(); + assert!(matches!(poll(&mut invalid_handles[0]), Poll::Ready(Err(_)))); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + let mut reads = requests(2); + reads[0].offset = u64::MAX; + let mut handles = driver.read_at_batch(reads).unwrap(); + assert!(matches!(poll(&mut handles[0]), Poll::Ready(Err(e)) if e.kind() == io::ErrorKind::InvalidInput)); + assert!(matches!(&handles[1].state, HandleState::Submitted { .. })); + assert_eq!(receivers[0].try_iter().count(), 1); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); +} + +#[test] +fn capacity_aware_batch_notifies_final_owner_only() { + let (driver, receivers) = driver(2, 4, None); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let _held = Arc::clone(&driver.shards[0].sem).try_acquire_many_owned(4).unwrap(); + let handles = driver.read_at_batch(requests(4)).unwrap(); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + assert_eq!(notifications(&driver.shards[1].wake_efd), 1); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(receivers[1].try_iter().count(), 4); + for handle in &handles { + let HandleState::Submitted { wake } = &handle.state else { panic!("expected eager admission") }; + assert!(Arc::ptr_eq(wake, &driver.shards[1].wake_efd)); + } +} + +#[test] +fn deferred_batch_member_notifies_when_polled_after_byte_permit_returns() { + let (driver, receivers) = driver(1, 2, Some(8)); + let mut handles = driver.read_at_batch(requests(2)).unwrap(); + assert!(matches!(&handles[1].state, HandleState::WaitingPermit { .. })); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + assert!(poll(&mut handles[1]).is_pending()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 0); + let accepted = receivers[0].try_recv().unwrap(); + assert!(matches!(&accepted, Msg::Read { .. })); + drop(accepted); // Simulate completed ownership release; no kernel used. + assert!(poll(&mut handles[1]).is_pending()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { .. }))); +} + +#[test] +fn dropping_batch_routes_cancels_and_keeps_accepted_permits_owned() { + let (driver, receivers) = driver(1, 2, Some(16)); + let handles = driver.read_at_batch(requests(2)).unwrap(); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1); + let accepted: Vec<_> = receivers[0].try_iter().collect(); + let ids: Vec<_> = handles.iter().map(|handle| handle.id).collect(); + drop(handles); + assert_eq!(notifications(&driver.shards[0].wake_efd), 2); + let cancelled: Vec<_> = receivers[0] + .try_iter() + .map(|msg| match msg { + Msg::Cancel { id } => id, + _ => panic!("expected cancel"), + }) + .collect(); + assert_eq!(cancelled, ids); + assert_eq!(driver.shards[0].sem.available_permits(), 0); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 0); + drop(accepted); + assert_eq!(driver.shards[0].sem.available_permits(), 2); + assert_eq!(driver.byte_admission.as_ref().unwrap().bytes.available_permits(), 16); +} + +#[test] +fn interrupted_construction_drops_handles_and_wakes_previously_queued_reads() { + let (driver, receivers) = driver(1, 2, None); + // Force the existing id-overflow assertion on the second batch member. + driver.next_id.store(CANCEL_BIT - 1, Ordering::Relaxed); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| driver.read_at_batch(requests(2)))); + assert!(result.is_err()); + assert_eq!(notifications(&driver.shards[0].wake_efd), 1, "unwind cancellation wakes the owner"); + let accepted = receivers[0].try_recv().unwrap(); + assert!(matches!(&accepted, Msg::Read { id, .. } if *id == CANCEL_BIT - 1)); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Cancel { id }) if id == CANCEL_BIT - 1)); + assert_eq!(driver.shards[0].sem.available_permits(), 1); + drop(accepted); + assert_eq!(driver.shards[0].sem.available_permits(), 2); +} diff --git a/src/diagnostics.rs b/src/diagnostics.rs index 8c82834..b478e05 100644 --- a/src/diagnostics.rs +++ b/src/diagnostics.rs @@ -179,12 +179,20 @@ pub(crate) struct Trace { impl Trace { pub(crate) fn sample(diagnostics: &Arc) -> Option> { + Self::sample_with_start(diagnostics, Instant::now) + } + + pub(crate) fn sample_since(diagnostics: &Arc, started: Instant) -> Option> { + Self::sample_with_start(diagnostics, || started) + } + + fn sample_with_start(diagnostics: &Arc, start: impl FnOnce() -> Instant) -> Option> { // A global id modulo 64 would sample only shard zero for power-of-two // round-robin sharding. Each shard therefore owns its sampling sequence. let sequence = diagnostics.requests.fetch_add(1, Ordering::Relaxed); sequence.is_multiple_of(DIAGNOSTICS_SAMPLE_INTERVAL).then(|| { Arc::new(Self { - start: Instant::now(), + start: start(), diagnostics: Arc::clone(diagnostics), queued: AtomicU64::new(0), entered: AtomicU64::new(0), diff --git a/src/driver.rs b/src/driver.rs index 4072c3f..f1f4f5c 100644 --- a/src/driver.rs +++ b/src/driver.rs @@ -19,15 +19,27 @@ use std::io::Write as _; use std::os::fd::{AsRawFd, FromRawFd}; use std::os::unix::ffi::OsStrExt; use std::pin::Pin; -use std::sync::Arc; -use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; use std::sync::mpsc::{self, TryRecvError}; +use std::sync::{Arc, Mutex}; use std::task::{Context, Poll}; use std::thread::JoinHandle; use std::time::{Duration, Instant}; use io_uring::{IoUring, opcode, types}; +#[cfg(test)] +#[path = "admission_tests.rs"] +mod admission_tests; + +#[cfg(test)] +#[path = "shard_policy_tests.rs"] +mod shard_policy_tests; + +#[cfg(test)] +#[path = "batch_read_tests.rs"] +mod batch_read_tests; + #[cfg(feature = "diagnostics")] use crate::diagnostics::{Diagnostics, DiagnosticsSnapshot, Trace}; @@ -112,12 +124,48 @@ const MAX_CONSECUTIVE_SUBMIT_ERRORS: u32 = 128; /// pathological storm cannot spin the driver thread (rustfs/backlog#1166). const MAX_TRANSIENT_RETRIES: u32 = 16; +// Bound each phase so intake/allocation cannot indefinitely defer reap, and a +// CQ burst cannot indefinitely defer cancel/shutdown intake. These are fairness +// limits, not tuned throughput settings (rustfs/backlog#2647). +const TURN_MESSAGES: usize = 64; +const TURN_COMPLETIONS: usize = 64; +const TURN_ALLOCATION_BYTES: usize = 8 * 1024 * 1024; + +#[derive(Default)] +struct TurnBudget { + messages: usize, + completions: usize, + allocation_bytes: usize, +} + +impl TurnBudget { + fn can_take_message(&self) -> bool { + self.messages < TURN_MESSAGES && self.allocation_bytes < TURN_ALLOCATION_BYTES + } + + fn can_reap(&self) -> bool { + self.completions < TURN_COMPLETIONS + } + + fn continue_without_wait(&self, ready_cqe: bool, queued_submission: bool, submitted: usize) -> bool { + // A hit budget can leave work behind after its eventfd edge was drained. + // One extra empty turn at an exact boundary is harmless. Merely having + // unaccepted SQEs must not spin after EBUSY, EINTR, errors or Ok(0). + !self.can_take_message() || !self.can_reap() || ready_cqe || (queued_submission && submitted > 0) + } +} + +#[cfg(test)] +#[path = "driver_loop_budget_tests.rs"] +mod loop_budget_tests; + /// Owned `eventfd(2)` used to wake the driver loop (backlog#1102): one is /// registered with the ring so the kernel signals it on every CQE, the other is /// signaled by `submit`/shutdown so a new message wakes the loop immediately — /// together they replace the spike's 200 µs busy-poll. struct EventFd { fd: std::os::fd::RawFd, + error_logged: AtomicBool, } impl EventFd { @@ -127,7 +175,10 @@ impl EventFd { if fd < 0 { return Err(io::Error::last_os_error()); } - Ok(Self { fd }) + Ok(Self { + fd, + error_logged: AtomicBool::new(false), + }) } fn as_raw(&self) -> std::os::fd::RawFd { @@ -138,18 +189,53 @@ impl EventFd { /// already readable, which is all a wakeup needs. fn signal(&self) { let v: u64 = 1; - // SAFETY: writing 8 bytes from a valid u64 to an eventfd we own. - unsafe { - libc::write(self.fd, (&v as *const u64).cast(), 8); - } + self.transfer("signal", || { + // SAFETY: writing 8 bytes from a valid u64 to an eventfd we own. + unsafe { libc::write(self.fd, (&v as *const u64).cast(), 8) } + }); } /// Reset the counter. EFD_NONBLOCK guarantees this never blocks; a single - /// successful read drains the whole counter, the next returns EAGAIN. + /// successful read drains the whole counter. A later concurrent signal is + /// left readable for the next turn; intake and CQ are checked after drain. fn drain(&self) { let mut v: u64 = 0; - // SAFETY: reading 8 bytes into a valid u64 from an eventfd we own. - while unsafe { libc::read(self.fd, (&mut v as *mut u64).cast(), 8) } == 8 {} + self.transfer("drain", || { + // SAFETY: reading 8 bytes into a valid u64 from an eventfd we own. + unsafe { libc::read(self.fd, (&mut v as *mut u64).cast(), 8) } + }); + } + + fn transfer(&self, operation: &'static str, mut syscall: impl FnMut() -> isize) { + let result = eventfd_transfer(|| { + let count = syscall(); + if count < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(count as usize) + } + }); + if let Err(error) = result + && !self.error_logged.swap(true, Ordering::Relaxed) + { + // At most one warning per fd, including shared producer wake fds. + // The heartbeat still checks queued work if a wake syscall fails. + tracing::warn!(operation, %error, "uring driver: eventfd operation failed; heartbeat remains active"); + } + } +} + +/// An eventfd transfer is all-or-nothing. Retry interrupted syscalls; EAGAIN +/// means a signal is already pending (write) or no signal remains (read). +fn eventfd_transfer(mut syscall: impl FnMut() -> io::Result) -> io::Result<()> { + loop { + match syscall() { + Ok(8) => return Ok(()), + Ok(_) => return Err(io::Error::other("eventfd transferred an unexpected byte count")), + Err(error) if error.raw_os_error() == Some(libc::EINTR) => continue, + Err(error) if error.raw_os_error() == Some(libc::EAGAIN) => return Ok(()), + Err(error) => return Err(error), + } } } @@ -262,7 +348,7 @@ impl ProbeFailure { // resident memory and reopening the memory-DoS surface. // // That rule is now enforced by the type system rather than by a manual -// `release()` call: the `OwnedSemaphorePermit` travels with `Msg::Read` into the +// `release()` call: the count and optional byte permits travel with `Msg::Read` into the // `Pending` entry and is dropped exactly when the entry is removed at the final // CQE. A short-read resubmit keeps the entry — and thus the permit. // @@ -272,7 +358,261 @@ impl ProbeFailure { // returned `ReadHandle`, which awaits it on its first poll and submits then. /// Boxed `Semaphore::acquire_owned` future held by a saturated `ReadHandle`. -type AcquireFut = Pin> + Send>>; +type AcquireFut = Pin> + Send>>; + +/// Optional resource limits shared by all shards of one driver. +/// +/// Defaults preserve the existing count-only admission policy. Limits cover +/// driver-owned read allocations, not queued handle metadata, allocator overhead, +/// completion copies, or returned results retained by the caller. They are not +/// a whole-process RSS bound. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct ReadLimits { + /// Maximum logical length of one read. `None` retains the kernel read cap. + /// A request exceeding this limit returns `InvalidInput` without read-buffer allocation. + pub max_read_len: Option, + /// Maximum sum of reserved read-buffer bytes across all shards. + /// + /// Direct reads charge their aligned superset plus `align - 1` allocation + /// padding. Reservations start before allocation and end at terminal read + /// completion, including canceled operations; leaked operations stay charged. + /// Requests larger than this budget return `InvalidInput`, never wait. + /// Zero and values above Tokio's `Semaphore::MAX_PERMITS` are invalid. + /// Shutdown or any shard exit closes count and byte admission for the entire + /// driver when enabled, waking all waiters even if buffers must be leaked. + pub max_in_flight_bytes: Option, +} + +/// Shared whole-driver reservations for in-flight read-buffer budgets. +/// +/// Clone this handle to give multiple drivers the same pool. Each driver reserves +/// its entire configured [`ReadLimits::max_in_flight_bytes`] before probing, even +/// while idle; local per-read admission is unchanged. Reservations are returned +/// only after their driver, deferred admission and accepted reads release them. +/// A leaked pending read retains the driver's whole reservation permanently. +/// +/// This bounds the sum of participating drivers' reserved read-buffer limits, +/// not completed results, allocator overhead, probe/ring allocations or RSS. +/// Shutting down one driver never closes the pool or another driver's admission. +/// Creating a separate pool creates a separate accounting domain. +/// +/// # Example +/// +/// ```no_run +/// use std::{fs::File, sync::Arc}; +/// use rustfs_uring::{ReadLimits, SharedReadBudget, UringDriver}; +/// # fn main() -> Result<(), Box> { +/// let pool = SharedReadBudget::new(24 << 20)?; +/// let first = UringDriver::probe_and_start_with_shared_budget( +/// 64, 1, +/// ReadLimits { max_read_len: None, max_in_flight_bytes: Some(8 << 20) }, +/// &pool, +/// )?; +/// let second = UringDriver::probe_and_start_with_shared_budget( +/// 64, 2, +/// ReadLimits { max_read_len: None, max_in_flight_bytes: Some(16 << 20) }, +/// &pool, +/// )?; +/// assert_eq!(pool.available(), 0); // whole reservations, even while idle +/// let file = Arc::new(File::open("object.bin")?); +/// let reads = [first.read_at(file.clone(), 0, 4096), second.read_at(file, 4096, 4096)]; +/// drop(reads); // also release any deferred admission owners +/// let first_stats = first.shutdown(); +/// let second_stats = second.shutdown(); +/// if first_stats.in_flight == 0 && second_stats.in_flight == 0 { +/// assert_eq!(pool.available(), pool.capacity()); +/// } +/// # Ok(()) +/// # } +/// ``` +#[derive(Clone, Debug)] +pub struct SharedReadBudget { + inner: Arc, +} + +#[derive(Debug)] +struct SharedReadBudgetInner { + capacity: usize, + available: AtomicUsize, +} + +impl SharedReadBudget { + /// Create a pool with `total` bytes of reservation capacity. + /// + /// Zero returns `InvalidInput`. This does not allocate `total` bytes or + /// require a runtime. Individual driver limits must still fit Tokio's + /// `Semaphore::MAX_PERMITS`, but pool capacity is not narrowed to `u32`. + pub fn new(total: usize) -> io::Result { + if total == 0 { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "shared read budget must be nonzero")); + } + Ok(Self { + inner: Arc::new(SharedReadBudgetInner { + capacity: total, + available: AtomicUsize::new(total), + }), + }) + } + + /// The pool's fixed whole-driver reservation capacity, in bytes. + pub fn capacity(&self) -> usize { + self.inner.capacity + } + + /// Advisory snapshot of unreserved capacity, in bytes. + /// + /// This is not a measurement of live buffers: an idle participating driver + /// still holds its whole reservation. Concurrent construction or cleanup + /// can change this value immediately after it is read. + pub fn available(&self) -> usize { + self.inner.available.load(Ordering::Acquire) + } + + fn reserve(&self, bytes: usize) -> io::Result> { + if bytes == 0 || bytes > self.inner.capacity { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "driver read limit must fit the shared read budget", + )); + } + self.inner + .available + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |available| available.checked_sub(bytes)) + .map_err(|_| io::Error::new(io::ErrorKind::WouldBlock, "shared read budget has insufficient unreserved capacity"))?; + Ok(Arc::new(SharedReadReservation { + pool: Arc::clone(&self.inner), + bytes, + })) + } +} + +struct SharedReadReservation { + pool: Arc, + bytes: usize, +} + +impl Drop for SharedReadReservation { + fn drop(&mut self) { + // One receipt refunds its successful checked subtraction exactly once. + // Read-path clones only touch Arc counts, never this global counter. + self.pool.available.fetch_add(self.bytes, Ordering::Release); + } +} + +/// Shard selection for positioned reads. Stream reads retain round-robin routing. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum ShardPolicy { + /// Bind to the next shard even when its admission is full or closed. + #[default] + RoundRobin, + /// Starting at the round-robin cursor, try each shard's count permit once, + /// skipping closed shards. If every healthy shard is full, wait fairly on + /// the first healthy shard. A shared byte-budget shortage waits on the first + /// shard with count capacity; a closed byte budget terminates admission. + /// Accepted and waiting reads never migrate to another shard. + CapacityAware, +} + +struct ReadPermits { + _count: OwnedSemaphorePermit, + _bytes: Option, + _shared_reservation: Option>, +} + +fn try_read_permits(count: &Arc, bytes: Option<&Arc>, charge: u32) -> Result { + let count = Arc::clone(count).try_acquire_owned()?; + let bytes = bytes.map(|sem| Arc::clone(sem).try_acquire_many_owned(charge)).transpose()?; + Ok(ReadPermits { + _count: count, + _bytes: bytes, + _shared_reservation: None, + }) +} + +fn acquire_read_permits( + count: Arc, + bytes: Option>, + charge: u32, + shared_reservation: Option>, +) -> AcquireFut { + Box::pin(async move { + // Every admission takes count before bytes. Pending reads need neither + // resource to complete, so there is no inverse acquisition cycle. + let count = count.acquire_owned().await?; + let bytes = match bytes { + Some(sem) => Some(sem.acquire_many_owned(charge).await?), + None => None, + }; + Ok(ReadPermits { + _count: count, + _bytes: bytes, + _shared_reservation: shared_reservation, + }) + }) +} + +/// Registration and terminal closure happen only at startup/shutdown, never on +/// read admission. The shared lock orders registration against shard failure. +struct ByteAdmission { + bytes: Arc, + registry: Mutex, + shared_reservation: Option>, +} + +#[derive(Default)] +struct AdmissionRegistry { + closed: bool, + counts: Vec>, +} + +impl ByteAdmission { + fn new(bytes: usize) -> Self { + Self { + bytes: Arc::new(Semaphore::new(bytes)), + registry: Mutex::new(AdmissionRegistry::default()), + shared_reservation: None, + } + } + + fn register(&self, count: &Arc) { + let close = { + let mut registry = self.registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if registry.closed { + true + } else { + registry.counts.push(Arc::clone(count)); + false + } + }; + // Semaphore::close may run arbitrary task wakers. Never hold the + // registry lock across it, including registration after terminal close. + if close { + count.close(); + } + } + + fn close(&self) { + let counts = { + let mut registry = self.registry.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + registry.closed = true; + std::mem::take(&mut registry.counts) + }; + self.bytes.close(); + for count in counts { + count.close(); + } + } +} + +struct CloseByteAdmission(Option>); + +impl Drop for CloseByteAdmission { + fn drop(&mut self) { + if let Some(admission) = &self.0 { + admission.close(); + } + } +} #[derive(Default)] struct DriverStats { @@ -342,7 +682,7 @@ enum Msg { /// released only when the pending entry is dropped at the final CQE /// (rustfs/backlog#1060/#1102). If the driver rejects the op (shutting /// down) the permit is dropped with the message — released immediately. - permit: OwnedSemaphorePermit, + permit: ReadPermits, /// Block size the read must be aligned to. `1` means a normal buffered /// read; `> 1` means the file was opened `O_DIRECT` and the driver must /// read the block-aligned superset range into a block-aligned buffer @@ -401,7 +741,7 @@ struct Pending { offset: u64, /// Bytes already read into the read region (`buf[pad..]`). nread: usize, - _permit: OwnedSemaphorePermit, + _permit: ReadPermits, /// Offset inside `buf` where the block-aligned read region starts. pad: usize, /// Bytes of the read region that precede the caller's logical range. @@ -416,6 +756,9 @@ struct Pending { /// progress, bounded by `MAX_TRANSIENT_RETRIES` so a storm cannot spin the /// driver thread (rustfs/backlog#1166). Reset whenever a read makes progress. transient_retries: u32, + /// Explicit cancel intent, independent of receiver closure: opting out of + /// drop-cancel must still finish positioned reads after abandoning results. + cancel_requested: bool, } impl Pending { @@ -601,8 +944,8 @@ impl Drop for ReadHandle { fn drop(&mut self) { // The buffer is deliberately NOT touched here: the driver owns it // until the CQE. All we may do is ask the kernel to hurry up. A handle - // dropped before it was submitted (Inert / WaitingPermit) has no buffer, - // no permit and no SQE, so there is nothing to cancel. + // dropped before it was submitted (Inert / WaitingPermit) has no buffer + // and no SQE. A waiting handle releases any partial reservation by drop. if let HandleState::Submitted { wake } = &self.state && !self.finished && self.cancel_on_drop @@ -641,8 +984,8 @@ struct Shard { /// when the driver thread exits so any waiting `ReadHandle` resolves with a /// driver-gone error instead of hanging (rustfs/backlog#1102). sem: Arc, - /// Signaled after every message send so this shard's loop wakes immediately - /// instead of waiting out the heartbeat (backlog#1102). + /// Signaled after message sends (once per shard for an explicit eager batch) + /// so the loop wakes without waiting out the heartbeat (backlog#1102). wake_efd: Arc, } @@ -672,6 +1015,9 @@ impl Drop for Shard { /// kernel. Construct it through [`UringDriver::probe_and_start`] so restricted /// environments can fall back to a blocking backend before serving traffic. pub struct UringDriver { + limits: ReadLimits, + shard_policy: ShardPolicy, + byte_admission: Option>, /// One or more independent rings. A cache-hit buffered read completes inline /// inside `io_uring_enter`, so the thread driving a ring performs that /// read's memcpy — which caps a single-ring driver at one core's memory @@ -685,6 +1031,20 @@ pub struct UringDriver { rr: AtomicUsize, } +/// Maximum requests accepted by one [`UringDriver::read_at_batch`] call. +pub const MAX_BATCH_READS: usize = 64; + +/// One buffered positioned read in an explicit notification batch. +#[derive(Debug)] +pub struct ReadRequest { + /// File whose ownership is retained through completion of an accepted read. + pub file: Arc, + /// Positioned byte offset, with the same validation as [`UringDriver::read_at`]. + pub offset: u64, + /// Logical byte count, subject to the driver's configured read limits. + pub len: usize, +} + impl UringDriver { /// Create the ring AND verify a real `IORING_OP_READ` round-trip on a /// temp file before accepting work. `io_uring_setup` succeeding is not @@ -714,6 +1074,75 @@ impl UringDriver { /// later shard fails to start, the ones already running are shut down and /// joined before the error is returned. pub fn probe_and_start_sharded(entries: u32, shards: usize) -> Result { + Self::probe_and_start_with_limits(entries, shards, ReadLimits::default()) + } + + /// Start one or more shards with opt-in logical-size and read-allocation limits. + /// + /// Ring setup and probing follow [`Self::probe_and_start_sharded`]. Invalid + /// limits return [`ProbeFailure::Setup`] with `InvalidInput` before probing. + /// Saturated admission waits asynchronously and fairly on Tokio semaphores; + /// dropping a waiting handle returns any partial reservation. + pub fn probe_and_start_with_limits(entries: u32, shards: usize, limits: ReadLimits) -> Result { + Self::validate_read_limits(limits)?; + Self::start_with_limits(entries, shards, limits, None) + } + + /// Reserve a whole driver budget from `budget`, then probe and start it. + /// + /// `limits.max_in_flight_bytes` must be explicitly set and nonzero. It + /// remains the driver's independent local byte limit, shared by its shards. + /// A limit larger than the pool capacity returns `ProbeFailure::Setup` with + /// `InvalidInput`; temporary reservation shortage returns `WouldBlock` + /// immediately, without probing or waiting for another driver to retire. + /// These errors have no restriction errno and do not imply io_uring is + /// unsupported. No fairness or automatic retry is promised. + /// + /// Startup failure refunds its reservation after partial shards are cleaned + /// up. Successful drivers keep it while idle, through shutdown and deferred + /// or accepted reads; a leaked pending read keeps the whole reservation. + /// Completed result buffers are not covered by this accounting. + pub fn probe_and_start_with_shared_budget( + entries: u32, + shards: usize, + limits: ReadLimits, + budget: &SharedReadBudget, + ) -> Result { + Self::validate_read_limits(limits)?; + let bytes = limits.max_in_flight_bytes.ok_or_else(|| { + ProbeFailure::Setup(io::Error::new( + io::ErrorKind::InvalidInput, + "shared read budget requires an explicit local byte limit", + )) + })?; + let reservation = budget.reserve(bytes).map_err(ProbeFailure::Setup)?; + Self::start_with_limits(entries, shards, limits, Some(reservation)) + } + + fn validate_read_limits(limits: ReadLimits) -> Result<(), ProbeFailure> { + if limits + .max_in_flight_bytes + .is_some_and(|bytes| bytes == 0 || bytes > Semaphore::MAX_PERMITS) + { + return Err(ProbeFailure::Setup(io::Error::new( + io::ErrorKind::InvalidInput, + "byte budget must be between 1 and Semaphore::MAX_PERMITS", + ))); + } + Ok(()) + } + + fn start_with_limits( + entries: u32, + shards: usize, + limits: ReadLimits, + shared_reservation: Option>, + ) -> Result { + let byte_admission = limits.max_in_flight_bytes.map(|bytes| { + let mut admission = ByteAdmission::new(bytes); + admission.shared_reservation = shared_reservation; + Arc::new(admission) + }); let mut started = Vec::with_capacity(shards.max(1)); for i in 0..shards.max(1) { // Probe only the first shard (rustfs/backlog#1165): the probe read @@ -722,24 +1151,88 @@ impl UringDriver { // verify NODROP — this avoids `shards - 1` extra O_TMPFILE // create+write+read round-trips per disk on every start and renew. // `?` drops `started`, whose `Shard::drop` joins each running thread. - started.push(Self::start_shard(entries, i == 0)?); + started.push(Self::start_shard(entries, i == 0, byte_admission.clone())?); } Ok(Self { + limits, + shard_policy: ShardPolicy::default(), + byte_admission, shards: started, next_id: AtomicU64::new(1), rr: AtomicUsize::new(0), }) } + /// Set the policy for future positioned reads, independently of resource limits. + /// + /// Existing handles retain their owning shard. [`Self::read_current`] keeps + /// round-robin behavior under either policy; callers must still serialize + /// stream reads themselves when ordering matters. Capacity-aware selection + /// is opt-in and has not established a throughput or latency improvement. + #[must_use] + pub fn with_shard_policy(mut self, policy: ShardPolicy) -> Self { + self.shard_policy = policy; + self + } + + fn select_read_shard( + &self, + start: usize, + charge: u32, + capacity_aware: bool, + ) -> (&Shard, Result) { + let first = &self.shards[start]; + let bytes = self.byte_admission.as_ref().map(|admission| &admission.bytes); + if !capacity_aware { + return (first, try_read_permits(&first.sem, bytes, charge)); + } + if bytes.is_some_and(|sem| sem.is_closed()) { + return (first, Err(TryAcquireError::Closed)); + } + let mut waiting = None; + for index in (start..self.shards.len()).chain(0..start) { + let shard = &self.shards[index]; + match Arc::clone(&shard.sem).try_acquire_owned() { + Ok(count) => { + // A shared byte shortage cannot be solved on another shard. + // map drops count on either byte error, before async waiting. + let permits = bytes + .map(|sem| Arc::clone(sem).try_acquire_many_owned(charge)) + .transpose() + .map(|bytes| ReadPermits { + _count: count, + _bytes: bytes, + _shared_reservation: None, + }); + return (shard, permits); + } + Err(TryAcquireError::NoPermits) => { + waiting.get_or_insert(shard); + } + Err(TryAcquireError::Closed) => {} + } + } + // Closure racing with the scan is terminal, even if an earlier count + // attempt saw NoPermits. Later closure wakes the deferred acquire future. + if bytes.is_some_and(|sem| sem.is_closed()) { + return (first, Err(TryAcquireError::Closed)); + } + match waiting { + Some(shard) => (shard, Err(TryAcquireError::NoPermits)), + None => (first, Err(TryAcquireError::Closed)), + } + } + /// Pick the shard for the next op. Round-robin spreads the inline-completion /// memcpy across driver threads; correctness does not depend on the choice, /// because the handle remembers which shard took the op. + #[cfg(feature = "fault-injection")] fn shard(&self) -> &Shard { let n = self.shards.len(); &self.shards[self.rr.fetch_add(1, Ordering::Relaxed) % n] } - fn start_shard(entries: u32, probe: bool) -> Result { + fn start_shard(entries: u32, probe: bool, byte_admission: Option>) -> Result { let mut ring = IoUring::new(entries).map_err(ProbeFailure::Setup)?; // Require the NODROP feature (kernel >= 5.5). Without it, CQ overflow // silently drops CQEs, stranding pending entries forever and hanging @@ -782,6 +1275,9 @@ impl UringDriver { // Cap in-flight at the SQ depth (entries), which is < CQ capacity // (2*entries), so CQ overflow is structurally unreachable (C5/C10). let sem = Arc::new(Semaphore::new(entries as usize)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } let thread_sem = Arc::clone(&sem); // Deterministic spawn-failure seam (rustfs/backlog#1164): exercise the // degrade-not-panic path without a real cgroup pids-limit. Never present @@ -799,7 +1295,10 @@ impl UringDriver { // (moved into the closure) drop cleanly with no SQE in flight. let handle = std::thread::Builder::new() .name("uring-spike-driver".into()) - .spawn(move || drive(ring, rx, thread_stats, thread_sem, cq_efd, thread_wake)) + .spawn(move || { + let _close_bytes = CloseByteAdmission(byte_admission); + drive(ring, rx, thread_stats, thread_sem, cq_efd, thread_wake); + }) .map_err(ProbeFailure::Setup)?; Ok(Shard { @@ -817,12 +1316,50 @@ impl UringDriver { /// [`Self::read_current`] and is returned as an asynchronous /// `io::ErrorKind::InvalidInput` result rather than panicking. pub fn read_at(&self, file: Arc, offset: u64, len: usize) -> ReadHandle { - self.submit(file, offset, len, 1, false) + self.submit(file, offset, len, 1, false, true) + } + + /// Create handles in input order for up to [`MAX_BATCH_READS`] buffered positioned reads. + /// + /// Eagerly accepted requests share one notification per owning shard after + /// constructing the handles. Saturated requests retain normal asynchronous + /// admission and notify their own shard when polled and admitted. The batch + /// is not atomic and does not guarantee completion order or snapshot reads. + /// Empty batches send no notification. Dropping any handle retains normal + /// cancel safety, including during unwinding of interrupted construction. + /// + /// # Errors + /// + /// More than [`MAX_BATCH_READS`] requests returns `InvalidInput` before any + /// submission. Individual invalid requests return errors from their handles, + /// exactly like [`Self::read_at`]; they do not reject other batch members. + pub fn read_at_batch(&self, requests: Vec) -> io::Result> { + if requests.len() > MAX_BATCH_READS { + return Err(io::Error::new(io::ErrorKind::InvalidInput, "read batch exceeds MAX_BATCH_READS")); + } + let mut handles = Vec::with_capacity(requests.len()); + let mut wakes: Vec> = Vec::with_capacity(requests.len()); + for ReadRequest { file, offset, len } in requests { + let handle = self.submit(file, offset, len, 1, false, false); + if let HandleState::Submitted { wake } = &handle.state + && !wakes.iter().any(|queued| Arc::ptr_eq(queued, wake)) + { + wakes.push(Arc::clone(wake)); + } + // A partially built vector owns every submitted handle. If a later + // construction unwinds, their Drop sends cancels and signals the + // final owner, so delayed batch notification cannot strand reads. + handles.push(handle); + } + for wake in wakes { + wake.signal(); + } + Ok(handles) } /// Read at the file's current position (read(2) semantics) — pipes. pub fn read_current(&self, file: Arc, len: usize) -> ReadHandle { - self.submit(file, CURRENT_POSITION, len, 1, true) + self.submit(file, CURRENT_POSITION, len, 1, true, true) } /// Positioned read from a file opened with `O_DIRECT` (rustfs/backlog#1102). @@ -838,23 +1375,32 @@ impl UringDriver { /// just a (correct but pointless) buffered read of the superset range. /// Invalid alignment, range, or reserved-offset inputs are returned through /// the awaited result as `io::ErrorKind::InvalidInput`. + /// A non-aligned short read that does not cover the requested range succeeds + /// only when metadata confirms EOF; a failed metadata lookup returns an error. pub fn read_at_direct(&self, file: Arc, offset: u64, len: usize, align: usize) -> ReadHandle { - self.submit(file, offset, len, align, false) + self.submit(file, offset, len, align, false, true) } - fn submit(&self, file: Arc, offset: u64, len: usize, align: usize, allow_current_position: bool) -> ReadHandle { + fn submit( + &self, + file: Arc, + offset: u64, + len: usize, + align: usize, + allow_current_position: bool, + notify_now: bool, + ) -> ReadHandle { let id = self.next_id.fetch_add(1, Ordering::Relaxed); assert_eq!(id & CANCEL_BIT, 0, "op id overflowed into the cancel bit"); let (done, rx) = oneshot::channel(); - // Bind the op to one shard for its whole life: the permit, the message, - // the wake, and any later cancel all go to this ring. The handle holds - // clones of that shard's `tx`/`wake_efd`, so nothing can route a cancel - // to a ring whose pending table does not hold the op. The rejection paths - // below return an `Inert` handle that never sends, but still need a `tx`. - let shard = self.shard(); + // Rejected requests retain the legacy round-robin cursor behavior. + // Valid capacity-aware reads choose their final owner after validation. + let start = self.rr.fetch_add(1, Ordering::Relaxed) % self.shards.len(); + let shard = &self.shards[start]; + let capacity_aware = self.shard_policy == ShardPolicy::CapacityAware && !allow_current_position; #[cfg(feature = "diagnostics")] - let timing = Trace::sample(&shard.stats.diagnostics); + let timing = (!capacity_aware).then(|| Trace::sample(&shard.stats.diagnostics)).flatten(); // `CURRENT_POSITION` is an internal sentinel used only by // `read_current`; accepting it through a positioned API would silently @@ -907,10 +1453,10 @@ impl UringDriver { // rustfs/backlog#1057). Failing fast here also removes the caller- // controlled `vec![0u8; len]` capacity-overflow panic that made the // unwind-UAF (rustfs/backlog#1054) reachable. P2 must chunk instead. - if len > MAX_READ_LEN { + if len > MAX_READ_LEN || self.limits.max_read_len.is_some_and(|max| len > max) { let _ = done.send(Err(io::Error::new( io::ErrorKind::InvalidInput, - "read length exceeds MAX_RW_COUNT (2 GiB - 4 KiB); caller must chunk", + "read length exceeds MAX_RW_COUNT or configured logical read limit; caller must chunk", ))); return ReadHandle { id, @@ -932,7 +1478,7 @@ impl UringDriver { // submit (rustfs/backlog#1102, #1166). Pre-empting it here also makes // every resubmit's `next_off < kernel_offset + region_len` provably // <= i64::MAX. `align == 1` (buffered) always passes the alignment part. - match aligned_geometry(offset, len, align) { + let allocation_bytes = match aligned_geometry(offset, len, align) { // CURRENT_POSITION (stream) reads use no positional offset — the // kernel reads from the current file position — so the i64::MAX end // check does not apply to them (their sentinel offset would overflow @@ -942,7 +1488,10 @@ impl UringDriver { && (allow_current_position && offset == CURRENT_POSITION || kernel_offset .checked_add(region_len as u64) - .is_some_and(|end| end <= i64::MAX as u64)) => {} + .is_some_and(|end| end <= i64::MAX as u64)) => + { + region_len + align - 1 + } _ => { let _ = done.send(Err(io::Error::new( io::ErrorKind::InvalidInput, @@ -959,17 +1508,53 @@ impl UringDriver { state: HandleState::Inert, }; } + }; + + if self.limits.max_in_flight_bytes.is_some_and(|max| allocation_bytes > max) { + let _ = done.send(Err(io::Error::new(io::ErrorKind::InvalidInput, "aligned allocation exceeds byte budget"))); + return ReadHandle { + id, + #[cfg(feature = "diagnostics")] + timing, + rx, + tx: shard.tx.clone(), + finished: false, + cancel_on_drop: false, + state: HandleState::Inert, + }; } + // Both region and alignment are capped at MAX_READ_LEN, so their sum + // fits u32, even on 32-bit Linux. Charge the exact Vec allocation length. + let charge = allocation_bytes as u32; + + #[cfg(feature = "diagnostics")] + let selection_started = capacity_aware.then(Instant::now); + let (shard, permits) = self.select_read_shard(start, charge, capacity_aware); + // Sample only the final owner, never the unsuccessful candidates. Default + // routing keeps its original sample sequence, including invalid requests. + #[cfg(feature = "diagnostics")] + let timing = match selection_started { + Some(started) if !matches!(&permits, Err(TryAcquireError::Closed)) => { + Trace::sample_since(&shard.stats.diagnostics, started) + } + Some(_) => None, + None => timing, + }; // Take a backpressure permit BEFORE the op reaches the driver; it is // released only when the pending entry is dropped at the CQE (C10, // rustfs/backlog#1060). Acquisition never blocks the caller's thread // (rustfs/backlog#1102). - match Arc::clone(&shard.sem).try_acquire_owned() { + // This owner is fixed for admission, message send, wake, and cancellation. + match permits { // Fast path: a permit was free, so submit eagerly — no allocation, // no await, and the op is in flight the moment `submit` returns, // exactly as with the previous blocking implementation. - Ok(permit) => { + Ok(mut permit) => { + permit._shared_reservation = self + .byte_admission + .as_ref() + .and_then(|admission| admission.shared_reservation.clone()); #[cfg(feature = "diagnostics")] if let Some(timing) = &timing { timing.enqueue(); @@ -1005,7 +1590,9 @@ impl UringDriver { }; } // Wake the driver loop so the read starts immediately. - shard.wake_efd.signal(); + if notify_now { + shard.wake_efd.signal(); + } ReadHandle { id, #[cfg(feature = "diagnostics")] @@ -1031,7 +1618,14 @@ impl UringDriver { finished: false, cancel_on_drop: true, state: HandleState::WaitingPermit { - acquire: Box::pin(Arc::clone(&shard.sem).acquire_owned()), + acquire: acquire_read_permits( + Arc::clone(&shard.sem), + self.byte_admission.as_ref().map(|admission| Arc::clone(&admission.bytes)), + charge, + self.byte_admission + .as_ref() + .and_then(|admission| admission.shared_reservation.clone()), + ), file, offset, len, @@ -1108,17 +1702,80 @@ impl UringDriver { shard.wake_efd.signal(); } - /// Stop accepting work, cancel all in-flight ops, drain every ring to - /// `in_flight == 0`, then join each driver thread. Only after that is a ring - /// dropped/unmapped — the shutdown ordering P2 requires, per shard. + /// Close admission and ask every shard to cancel/drain, without joining + /// driver threads or waiting for pending reads to complete. /// - /// Shards are asked to stop first and joined afterwards, so their bounded - /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. - pub fn shutdown(mut self) -> StatsSnapshot { + /// Count and byte waiters are woken with an admission error. An already + /// kernel-submitted operation still owns its buffer until its read CQE, or retains + /// it through the leak-over-UAF escape path. Concurrent and repeated calls + /// are safe; every call closes all admission before returning. + pub fn request_shutdown(&self) { + if let Some(admission) = &self.byte_admission { + admission.close(); + } + // Also close count-only admission when byte limits are disabled. Do not + // use a once flag: a concurrent caller must not return before this work + // is complete merely because another caller started the request. + for shard in &self.shards { + shard.sem.close(); + } for shard in &self.shards { let _ = shard.tx.send(Msg::Shutdown); shard.wake_efd.signal(); } + } + + /// Advisory query: whether every driver thread's join handle reports + /// finished (or has already been joined). This does not join threads or + /// imply a clean drain; use [`Self::shutdown`] to join and inspect its stats. + /// + /// Inspect [`Self::stats`]: nonzero `in_flight` can remain after the bounded + /// drain leaks kernel-owned resources and the driver threads finish. + /// Like [`std::thread::JoinHandle::is_finished`], this can become true just + /// before final thread teardown completes; it is not proof of a completed + /// synchronous join and does not impose a deadline on a blocked syscall. + pub fn is_finished(&self) -> bool { + self.shards + .iter() + .all(|shard| shard.handle.as_ref().is_none_or(JoinHandle::is_finished)) + } + + /// Request shutdown immediately and move the consuming shutdown/join onto + /// the current Tokio runtime's blocking pool. Await the returned future for + /// the final snapshot, or an error if the blocking task fails to join. + /// + /// Available with `tokio-runtime`. This is deliberately not an `async fn`: + /// the driver is handed off when this method is called, before the returned + /// future is polled. Dropping that future, even unpolled, only detaches the + /// join handle; it does not cancel started blocking work or drop the driver + /// on the caller during normal runtime operation. + /// + /// Runtime shutdown may reject or discard blocking work and synchronously + /// drop its captured driver. Scheduling is not a hard cleanup deadline, and + /// neither this adapter nor a timeout can kill a hung kernel syscall. A + /// successful snapshot may still report leaked reads via nonzero `in_flight`. + /// + /// # Panics + /// + /// Panics if called without an entered Tokio runtime, matching + /// [`tokio::runtime::Handle::current`]. Unwinding then drops the driver using + /// its synchronous cleanup path. + #[cfg(feature = "tokio-runtime")] + pub fn shutdown_async(self) -> impl Future> + Send { + let runtime = tokio::runtime::Handle::current(); + self.request_shutdown(); + let shutdown = runtime.spawn_blocking(move || self.shutdown()); + async move { shutdown.await.map_err(io::Error::other) } + } + + /// Stop accepting work, cancel/drain every ring, and join the driver threads + /// synchronously. A clean drain returns `in_flight == 0`; a bounded-drain + /// escape can return a nonzero count with the ring and buffers still leaked. + /// + /// Shards are asked to stop first and joined afterwards, so their bounded + /// drains overlap instead of serializing `shards * DRAIN_TIMEOUT`. + pub fn shutdown(mut self) -> StatsSnapshot { + self.request_shutdown(); for shard in &mut self.shards { shard.join(); } @@ -1144,10 +1801,7 @@ impl Drop for UringDriver { // drains overlap. Dropping the `Vec` would instead run each // `Shard::drop` in turn, serializing up to `shards * DRAIN_TIMEOUT` on a // hung device. `Shard::join` is idempotent, so the later drops are no-ops. - for shard in &self.shards { - let _ = shard.tx.send(Msg::Shutdown); - shard.wake_efd.signal(); - } + self.request_shutdown(); for shard in &mut self.shards { shard.join(); } @@ -1347,13 +2001,15 @@ impl Drop for DriverState { } } -/// Best-effort file length via `fstat` on the driver thread, used to tell a -/// genuine O_DIRECT tail short read from a non-block-multiple short read that -/// happened mid-file on a stacked filesystem (rustfs/backlog#1168). `None` when -/// the stat fails, in which case the caller keeps the conservative EOF -/// assumption rather than risk a wrong error or an unbounded resubmit loop. -fn file_len(file: &File) -> Option { - file.metadata().ok().map(|m| m.len()) +/// Finish a non-aligned O_DIRECT short read only after confirming the file tail. +/// Keeping the metadata result explicit lets tests inject errors without closing +/// a live fd or changing process-wide environment state (rustfs/backlog#2647). +fn finish_direct_short_read(p: &mut Pending, file_len: io::Result) -> io::Result> { + let file_len = file_len?; + if p.offset + (p.nread as u64) < file_len { + return Err(io::Error::other("io_uring O_DIRECT: non-block-aligned short read before EOF")); + } + Ok(deliver(p)) } /// Hand the caller exactly the logical range `[head, head + want)` of the read @@ -1366,6 +2022,11 @@ fn file_len(file: &File) -> Option { /// clamped to `nread`, so the zero-filled remainder of the buffer stays hidden /// (content hygiene, C12 / rustfs/backlog#1062). fn deliver(p: &mut Pending) -> Vec { + if p.done.as_ref().is_none_or(oneshot::Sender::is_closed) { + // Only called after the terminal read CQE. Keep the allocation in the + // entry for normal reclamation, avoiding an orphan's O_DIRECT memmove. + return Vec::new(); + } let avail = p.nread.saturating_sub(p.head).min(p.want); let start = p.pad + p.head; // The buffered path (`align == 1`) has `pad == 0` and `head == 0`, so the @@ -1386,6 +2047,52 @@ enum ReapStep { Resubmit(io_uring::squeue::Entry), } +/// Decide what follows a READ completion, never an AsyncCancel completion. +/// Cancellation only suppresses continuation: an already-complete successful +/// read still wins the race, and streams keep their read(2) short-read result. +fn reap_read(p: &mut Pending, id: u64, res: i32, shutting_down: bool) -> ReapStep { + let stop_continuation = p.cancel_requested || shutting_down; + if res < 0 { + let err = -res; + let transient = err == libc::EINTR || err == libc::EAGAIN; + if transient && p.offset != CURRENT_POSITION && p.nread < p.region_len { + if stop_continuation { + return ReapStep::Finish(Err(io::Error::from_raw_os_error(libc::ECANCELED))); + } + if p.transient_retries < MAX_TRANSIENT_RETRIES { + p.transient_retries += 1; + return ReapStep::Resubmit(p.read_sqe(id)); + } + } + return ReapStep::Finish(Err(io::Error::from_raw_os_error(err))); + } + if res == 0 { + return ReapStep::Finish(Ok(deliver(p))); + } + p.nread += res as usize; + // Progress resets the transient retry budget (rustfs/backlog#1166). + p.transient_retries = 0; + let is_stream = p.offset == CURRENT_POSITION; + let covered = p.nread >= p.head + p.want; + if is_stream || covered || p.nread >= p.region_len { + return ReapStep::Finish(Ok(deliver(p))); + } + if stop_continuation { + // This read SQE has completed, so reclamation is safe now. Do not start + // another positioned read for an explicitly cancelled logical request. + // Never report its incomplete prefix as successful whole-range output. + return ReapStep::Finish(Err(io::Error::from_raw_os_error(libc::ECANCELED))); + } + if p.align > 1 && !p.nread.is_multiple_of(p.align) { + // Disambiguate a genuine direct-I/O tail from a non-aligned short read + // before EOF; that offset cannot be resubmitted (rustfs/backlog#1168). + let file_len = p.file.metadata().map(|metadata| metadata.len()); + ReapStep::Finish(finish_direct_short_read(p, file_len)) + } else { + ReapStep::Resubmit(p.read_sqe(id)) + } +} + /// Queue at most one `AsyncCancel` per op (rustfs/backlog#1167): a drop-cancel /// followed by a shutdown, or the submit-error shutdown, must not enqueue a /// second cancel for the same id. The set is bounded by the pending table @@ -1410,12 +2117,28 @@ fn flush_backlog(ring: &mut IoUring, backlog: &mut VecDeque Option> { + let needs_enter = { + let sq = ring.submission(); + !sq.is_empty() || sq.cq_overflow() || sq.taskrun() + }; + needs_enter.then(|| ring.submit()) +} + +#[cfg(test)] +#[path = "driver_fault_recovery_tests.rs"] +mod fault_recovery_tests; + /// Flush the backlog into the SQ and submit it, with submit-error classification /// (rustfs/backlog#1162). The single submit path for the whole loop: called once -/// after intake and once more after reap when resubmits were queued. Skips the -/// `io_uring_enter` syscall on an empty SQ (rustfs/backlog#1169). EINTR/EBUSY are +/// after intake and once more after reap. Skips the `io_uring_enter` syscall only +/// when both submissions and kernel completion work are absent. EINTR/EBUSY are /// transient; any other errno is counted and, after a bounded run, transitions /// the shard to shutdown so callers fall back to the std backend. +/// Returns the accepted SQE count, or zero when idle or submission failed. fn submit_ring( state: &mut DriverState, stats: &DriverStats, @@ -1423,21 +2146,44 @@ fn submit_ring( submit_error_logged: &mut bool, shutting_down: &mut bool, queued_cancels: &mut HashSet, -) { +) -> usize { flush_backlog(&mut state.ring, &mut state.backlog); - if state.ring.submission().is_empty() { - return; - } - match state.ring.submit() { - Ok(_) => *consecutive_submit_errors = 0, + let Some(result) = submit_if_needed(&mut state.ring) else { + return 0; + }; + handle_submit_result(result, stats, consecutive_submit_errors, submit_error_logged, shutting_down, || { + let ids: Vec = state.pending.keys().copied().collect(); + for id in ids { + queue_cancel(&mut state.backlog, queued_cancels, id); + } + }) +} + +/// Classify a completed submit syscall without inferring which buffers the +/// kernel owns. The callback only queues cancellation on the first shutdown +/// transition; it never reclaims pending reads. Tests inject syscall results at +/// this boundary, not into a real kernel ring. +fn handle_submit_result( + result: io::Result, + stats: &DriverStats, + consecutive_submit_errors: &mut u32, + submit_error_logged: &mut bool, + shutting_down: &mut bool, + on_shutdown: impl FnOnce(), +) -> usize { + match result { + Ok(submitted) => { + *consecutive_submit_errors = 0; + return submitted; + } // CQ-overflow backpressure (EBUSY) and signal interruption (EINTR) are // transient — retry next turn without counting them (C5, backlog#1056). Err(e) if matches!(e.raw_os_error(), Some(libc::EBUSY) | Some(libc::EINTR)) => *consecutive_submit_errors = 0, Err(e) => { - // The queued SQEs were not accepted, so their CQEs never arrive. A - // brief run may be transient (EAGAIN); a persistent one (e.g. EPERM - // from a seccomp/LSM policy applied after startup) must not be retried - // forever in silence. + // Submission or kernel completion progress failed. Do not infer + // per-op acceptance or reclaim buffers from this syscall error. + // A brief run may be transient (EAGAIN); a persistent one (e.g. + // EPERM from a later seccomp policy) must not retry forever silently. stats.submit_errors.fetch_add(1, Ordering::SeqCst); *consecutive_submit_errors += 1; if !*submit_error_logged { @@ -1450,13 +2196,27 @@ fn submit_ring( "uring driver: consecutive submit failures; shutting down so callers fall back to the std backend" ); *shutting_down = true; - let ids: Vec = state.pending.keys().copied().collect(); - for id in ids { - queue_cancel(&mut state.backlog, queued_cancels, id); - } + on_shutdown(); } } } + 0 +} + +#[cfg(test)] +#[path = "driver_submit_result_tests.rs"] +mod submit_result_tests; + +/// Publish changes to the kernel's cumulative u32 counter. Warn only for a new +/// nonzero observation; a wrap to zero updates the snapshot and rearms logging +/// for the next increment without emitting a misleading zero-overflow warning. +fn update_cq_overflow(stats: &DriverStats, previous: &mut u32, observed: u32) -> bool { + if observed == *previous { + return false; + } + *previous = observed; + stats.cq_overflow.store(u64::from(observed), Ordering::SeqCst); + observed != 0 } fn drive( @@ -1478,6 +2238,7 @@ fn drive( // the persistent-submit-failure escape hatch (rustfs/backlog#1162). let mut consecutive_submit_errors: u32 = 0; let mut submit_error_logged = false; + let mut previous_cq_overflow = 0; // Ids with an AsyncCancel already queued, so a drop-cancel followed by a // shutdown (or vice versa) does not enqueue a second cancel for the same op — // keeping total completions <= 2*entries and CQ overflow unreachable @@ -1503,6 +2264,7 @@ fn drive( #[cfg(feature = "fault-injection")] let fault_stuck_drain = std::env::var_os("RUSTFS_URING_FAULT_STUCK_DRAIN").is_some(); + let mut continue_without_wait = false; loop { // Block until a CQE is ready (the ring's registered eventfd), a new // message arrives (the wakeup eventfd), or the heartbeat elapses — @@ -1520,13 +2282,16 @@ fn drive( } else { IDLE_HEARTBEAT }; - wait_for_events(&cq_efd, &wake_efd, heartbeat); + if !continue_without_wait { + wait_for_events(&cq_efd, &wake_efd, heartbeat); + } cq_efd.drain(); wake_efd.drain(); - // 1. Intake: drain all queued messages (the wait above did the blocking, - // so this is purely non-blocking). - loop { + let mut budget = TurnBudget::default(); + // 1. Bounded intake. An individual large accepted read still makes + // progress; its allocation ends this phase rather than deferring forever. + while budget.can_take_message() { let msg = match rx.try_recv() { Ok(m) => m, Err(TryRecvError::Empty) => break, @@ -1535,6 +2300,7 @@ fn drive( break; } }; + budget.messages += 1; match msg { Msg::Read { id, @@ -1578,6 +2344,7 @@ fn drive( } }; let buf = vec![0u8; cap]; + budget.allocation_bytes = budget.allocation_bytes.saturating_add(cap); let pad = buf.as_ptr().align_offset(align); // Runtime guard (not a debug-only assert): if the allocator // ever returned a block `align_offset` cannot satisfy, refuse @@ -1613,6 +2380,7 @@ fn drive( region_len, align, transient_retries: 0, + cancel_requested: false, }, ); let sqe = state.pending.get(&id).expect("just inserted").read_sqe(id); @@ -1625,7 +2393,8 @@ fn drive( } } Msg::Cancel { id } => { - if state.pending.contains_key(&id) { + if let Some(pending) = state.pending.get_mut(&id) { + pending.cancel_requested = true; queue_cancel(&mut state.backlog, &mut queued_cancels, id); } } @@ -1651,7 +2420,7 @@ fn drive( // 2. Flush the backlog into the SQ and submit it (the single submit path; // see `submit_ring`). - submit_ring( + let submitted_before_reap = submit_ring( &mut state, &stats, &mut consecutive_submit_errors, @@ -1663,7 +2432,12 @@ fn drive( // 3. Reap. A Pending entry (and thus its buffer) is dropped ONLY when // the logical read finishes; a short read is resubmitted for the // remainder and the entry stays put (C9, rustfs/backlog#1058). - while let Some(cqe) = state.ring.completion().next() { + while budget.can_reap() { + let Some(cqe) = state.ring.completion().next() else { + break; + }; + // Count every consumed CQE, including cancels and test-only drops. + budget.completions += 1; let ud = cqe.user_data(); if ud & CANCEL_BIT != 0 { // Result of the AsyncCancel op itself; the read's own CQE @@ -1704,75 +2478,7 @@ fn drive( // the borrow ends (finish removes it; resubmit re-queues an SQE). let step = { let p = state.pending.get_mut(&ud).expect("checked above"); - if res < 0 { - let err = -res; - // C7 three-class contract (rustfs/backlog#1166): a transient - // errno (EINTR/EAGAIN) must be retried, not surfaced as the - // read's final result — surfacing it would also discard the - // already-read prefix of a resubmit. Bounded per logical read - // so a storm cannot spin the driver thread. Streams - // (CURRENT_POSITION) cannot resubmit positionally; ECANCELED - // and every other errno terminate the logical read. - let transient = err == libc::EINTR || err == libc::EAGAIN; - if transient - && p.offset != CURRENT_POSITION - && p.nread < p.region_len - && p.transient_retries < MAX_TRANSIENT_RETRIES - { - p.transient_retries += 1; - ReapStep::Resubmit(p.read_sqe(ud)) - } else { - // Error (incl. ECANCELED, or a transient errno past its - // retry budget) terminates the logical read. - ReapStep::Finish(Err(io::Error::from_raw_os_error(err))) - } - } else if res == 0 { - // Real EOF: deliver whatever of the logical range was read. - ReapStep::Finish(Ok(deliver(p))) - } else { - p.nread += res as usize; - // Progress resets the transient-retry budget (rustfs/backlog#1166). - p.transient_retries = 0; - // Only POSITIONED reads (read_at / read_at_direct, whole-range - // pread contract) resubmit a short read. CURRENT_POSITION - // reads (read_current on pipes/streams) follow read(2) - // semantics: a short read is a valid final result and must be - // delivered as-is — resubmitting would block forever waiting - // for stream data that may never come. - let is_stream = p.offset == CURRENT_POSITION; - let covered = p.nread >= p.head + p.want; - if is_stream || covered || p.nread >= p.region_len { - ReapStep::Finish(Ok(deliver(p))) - } else if p.align > 1 && !p.nread.is_multiple_of(p.align) { - // O_DIRECT non-block-multiple short read below the covered - // range. The kernel returns block multiples EXCEPT at the - // file tail — but a stacked filesystem (NFS/FUSE, or a - // signal-split direct I/O) can legally return a non-multiple - // mid-file, and assuming EOF there would silently truncate - // the delivered range. Disambiguate with the actual file - // length instead of inferring it (rustfs/backlog#1168). - match file_len(&p.file) { - // Genuine tail: at or past EOF — deliver what we read. - Some(len) if p.offset + p.nread as u64 >= len => ReapStep::Finish(Ok(deliver(p))), - // Mid-file non-multiple: an O_DIRECT read cannot resubmit - // from a non-block-aligned offset, so surface an error - // rather than truncate. The integration falls back to - // the std backend for this read, preserving correctness. - Some(_) => ReapStep::Finish(Err(io::Error::other( - "io_uring O_DIRECT: non-block-aligned short read before EOF", - ))), - // fstat failed: keep the conservative EOF assumption - // rather than risk a wrong error or an infinite loop. - None => ReapStep::Finish(Ok(deliver(p))), - } - } else { - // Positioned short read, not EOF, block-aligned: resubmit - // the remainder into the read region. The buffer stays - // owned by the driver and in_flight is unchanged — one - // logical op. - ReapStep::Resubmit(p.read_sqe(ud)) - } - } + reap_read(p, ud, res, shutting_down) }; #[cfg(feature = "diagnostics")] @@ -1810,20 +2516,19 @@ fn drive( } } - // A short-read resubmit queued during reap must reach the kernel in THIS - // turn, not wait out the next heartbeat (rustfs/backlog#1163). Reap runs - // after the submit above, so re-run the single submit path when reap left - // work in the backlog; an idle turn leaves it empty and skips the call. - if !state.backlog.is_empty() { - submit_ring( - &mut state, - &stats, - &mut consecutive_submit_errors, - &mut submit_error_logged, - &mut shutting_down, - &mut queued_cancels, - ); - } + // Reap freed CQ space: flush kernel overflow/task work even with an + // empty SQ/backlog. Also retry SQEs left by partial submission and submit + // short-read continuations this turn. Two bounded attempts per loop + // preserve heartbeat pacing on EBUSY/EINTR/zero-progress submission; + // fully idle calls skip the syscall (rustfs/backlog#2647). + let submitted_after_reap = submit_ring( + &mut state, + &stats, + &mut consecutive_submit_errors, + &mut submit_error_logged, + &mut shutting_down, + &mut queued_cancels, + ); // Monitor CQ overflow. With NODROP (asserted at probe) overflowed CQEs // are BUFFERED in the kernel overflow list and flushed on the next enter, @@ -1832,8 +2537,7 @@ fn drive( // `entries` and cancels are deduped (at most one per op), keeping total // completions <= 2*entries, so this should stay 0 in practice. let overflow = state.ring.completion().overflow(); - if overflow != 0 { - stats.cq_overflow.store(overflow as u64, Ordering::SeqCst); + if update_cq_overflow(&stats, &mut previous_cq_overflow, overflow) { tracing::warn!( overflow, "uring driver: CQ overflow; CQEs buffered (NODROP), not lost — backpressure warning" @@ -1889,7 +2593,620 @@ fn drive( return; } } - // No pacing sleep: `wait_for_events` at the top of the loop blocks until - // the next CQE, message, or heartbeat (backlog#1102). + let ready_cqe = !state.ring.completion().is_empty(); + continue_without_wait = budget.continue_without_wait( + ready_cqe, + !state.backlog.is_empty() || !state.ring.submission().is_empty(), + submitted_before_reap.saturating_add(submitted_after_reap), + ); + } +} + +#[cfg(test)] +mod shared_budget_reservation_tests { + use super::*; + + fn mock_driver(pool: &SharedReadBudget, bytes: usize, policy: ShardPolicy) -> (UringDriver, mpsc::Receiver) { + let mut admission = ByteAdmission::new(bytes); + admission.shared_reservation = Some(pool.reserve(bytes).expect("reserve mock driver")); + let admission = Arc::new(admission); + let sem = Arc::new(Semaphore::new(1)); + admission.register(&sem); + let (tx, rx) = mpsc::channel(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(bytes), + }, + shard_policy: policy, + byte_admission: Some(admission), + shards: vec![Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().expect("mock wake fd")), + }], + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + rx, + ) + } + + #[test] + fn pool_reserves_whole_weights_and_refunds_only_the_last_receipt_owner() { + assert_eq!(SharedReadBudget::new(0).expect_err("zero pool").kind(), io::ErrorKind::InvalidInput); + let pool = SharedReadBudget::new(12).expect("pool"); + let alias = pool.clone(); + let first = pool.reserve(8).expect("first driver"); + let first_read = first.clone(); + let second = alias.reserve(4).expect("second driver"); + assert_eq!(pool.capacity(), 12); + assert_eq!(alias.available(), 0); + assert!(matches!(pool.reserve(1), Err(error) if error.kind() == io::ErrorKind::WouldBlock)); + drop(first); + assert_eq!(pool.available(), 0, "an accepted read still owns the receipt"); + drop(first_read); + assert_eq!(pool.available(), 8); + drop(second); + assert_eq!(pool.available(), 12); + } + + #[cfg(target_pointer_width = "64")] + #[test] + fn whole_driver_reservations_do_not_narrow_to_u32_or_overflow_refunds() { + let pool = SharedReadBudget::new(usize::MAX).expect("large accounting-only pool"); + let bytes = usize::try_from(u32::MAX).expect("64-bit usize") + 1; + let large = pool.reserve(bytes).expect("reservation above u32"); + assert_eq!(pool.available(), usize::MAX - bytes); + let remaining = pool.reserve(usize::MAX - bytes).expect("reserve exact remainder"); + assert_eq!(pool.available(), 0); + assert!(matches!(pool.reserve(1), Err(error) if error.kind() == io::ErrorKind::WouldBlock)); + drop(remaining); + drop(large); + assert_eq!(pool.available(), usize::MAX); + } + + #[test] + fn concurrent_reservations_never_exceed_pool_capacity() { + let pool = SharedReadBudget::new(17).expect("pool"); + let gate = std::sync::Barrier::new(9); + let successful = AtomicUsize::new(0); + let unexpected_error = AtomicBool::new(false); + let observed = std::thread::scope(|scope| { + for _ in 0..8 { + scope.spawn(|| { + gate.wait(); + let reservation = pool.reserve(3); + match &reservation { + Ok(_) => { + successful.fetch_add(1, Ordering::SeqCst); + } + Err(error) if error.kind() == io::ErrorKind::WouldBlock => {} + Err(_) => { + unexpected_error.store(true, Ordering::SeqCst); + } + } + gate.wait(); + gate.wait(); + drop(reservation); + }); + } + gate.wait(); + gate.wait(); + let observed = (pool.available(), successful.load(Ordering::SeqCst)); + gate.wait(); + observed + }); + assert_eq!(observed, (2, 5)); + assert!(!unexpected_error.load(Ordering::SeqCst)); + assert_eq!(pool.available(), pool.capacity()); + } + + #[test] + fn invalid_and_temporarily_unavailable_driver_limits_fail_before_probing() { + let pool = SharedReadBudget::new(8).expect("pool"); + for limit in [None, Some(0), Some(9), Some(Semaphore::MAX_PERMITS + 1)] { + let result = UringDriver::probe_and_start_with_shared_budget( + 0, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: limit, + }, + &pool, + ); + let Err(error) = result else { panic!("invalid constructor must fail") }; + assert!(!error.is_expected_restriction()); + assert!(matches!(error, ProbeFailure::Setup(ref error) if error.kind() == io::ErrorKind::InvalidInput)); + assert_eq!(pool.available(), 8); + } + let occupied = pool.reserve(8).expect("occupy pool"); + let result = UringDriver::probe_and_start_with_shared_budget( + 0, + 1, + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(1), + }, + &pool, + ); + let Err(error) = result else { panic!("insufficient shared budget must fail") }; + assert!(!error.is_expected_restriction()); + assert!(matches!(error, ProbeFailure::Setup(ref error) if error.kind() == io::ErrorKind::WouldBlock)); + assert_eq!(pool.available(), 0); + drop(occupied); + assert_eq!(pool.available(), 8); + } + + #[test] + fn eager_messages_keep_reservations_after_driver_and_caller_drop() { + for policy in [ShardPolicy::RoundRobin, ShardPolicy::CapacityAware] { + let pool = SharedReadBudget::new(8).expect("pool"); + let (driver, messages) = mock_driver(&pool, 8, policy); + let handle = driver + .read_at(Arc::new(File::open("/dev/zero").expect("fixture file")), 0, 4) + .without_cancel_on_drop(); + let message = messages.try_recv().expect("eager read enqueued"); + assert!(matches!(message, Msg::Read { .. })); + drop(handle); + drop(driver); + assert_eq!(pool.available(), 0, "accepted message must retain the whole driver reservation"); + drop(message); + assert_eq!(pool.available(), 8); + } + } + + #[test] + fn deferred_admission_keeps_reservation_until_closed_poll_or_drop() { + for poll_closed in [false, true] { + let pool = SharedReadBudget::new(8).expect("pool"); + let (driver, messages) = mock_driver(&pool, 8, ShardPolicy::RoundRobin); + let file = Arc::new(File::open("/dev/zero").expect("fixture file")); + let first = driver.read_at(file.clone(), 0, 8).without_cancel_on_drop(); + let message = messages.try_recv().expect("first read owns local permits"); + let mut waiting = driver.read_at(file, 0, 8); + assert!(matches!(waiting.state, HandleState::WaitingPermit { .. })); + drop(first); + drop(message); + drop(driver); + assert_eq!(pool.available(), 0, "deferred acquire owns the reservation before its first poll"); + if poll_closed { + let result = Pin::new(&mut waiting).poll(&mut Context::from_waker(std::task::Waker::noop())); + assert!(matches!(result, Poll::Ready(Err(_)))); + assert_eq!(pool.available(), 8, "closed acquire releases its receipt"); + } + drop(waiting); + assert_eq!(pool.available(), 8); + } + } + + #[test] + fn local_shutdown_does_not_close_another_participating_driver() { + let pool = SharedReadBudget::new(8).expect("pool"); + let (first, _first_messages) = mock_driver(&pool, 4, ShardPolicy::RoundRobin); + let (second, second_messages) = mock_driver(&pool, 4, ShardPolicy::RoundRobin); + first.request_shutdown(); + assert!(first.shards[0].sem.is_closed()); + assert!(!second.shards[0].sem.is_closed()); + assert!(!second.byte_admission.as_ref().expect("local budget").bytes.is_closed()); + assert_eq!(pool.available(), 0, "shutdown request alone does not return an idle driver's reservation"); + let read = second + .read_at(Arc::new(File::open("/dev/zero").expect("fixture file")), 0, 4) + .without_cancel_on_drop(); + let message = second_messages.try_recv().expect("other driver still accepts reads"); + assert!(matches!(message, Msg::Read { .. })); + drop(first); + assert_eq!(pool.available(), 4); + drop(second); + drop(read); + assert_eq!(pool.available(), 4, "second driver's queued operation still owns its reservation"); + drop(message); + assert_eq!(pool.available(), 8); + } +} + +#[cfg(test)] +mod shutdown_request_tests { + use super::*; + + fn mock_driver(shards: usize, bytes: Option) -> (UringDriver, Vec>) { + let byte_admission = bytes.map(|limit| Arc::new(ByteAdmission::new(limit))); + let mut receivers = Vec::new(); + let shards = (0..shards) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(1)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().expect("mock wake fd")), + } + }) + .collect(); + ( + UringDriver { + limits: ReadLimits { + max_read_len: None, + max_in_flight_bytes: bytes, + }, + shard_policy: ShardPolicy::RoundRobin, + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) + } + + fn poll_acquire(future: &mut AcquireFut) -> Poll> { + future.as_mut().poll(&mut Context::from_waker(std::task::Waker::noop())) + } + + #[test] + fn request_shutdown_closes_count_only_admission_before_returning() { + let (driver, messages) = mock_driver(1, None); + let count = Arc::clone(&driver.shards[0].sem); + let held = try_read_permits(&count, None, 1).expect("occupy mock shard"); + let mut waiter = acquire_read_permits(count.clone(), None, 1, None); + assert!(poll_acquire(&mut waiter).is_pending()); + driver.request_shutdown(); + assert!(count.is_closed()); + assert!(matches!(poll_acquire(&mut waiter), Poll::Ready(Err(_)))); + assert!(matches!(messages[0].try_recv(), Ok(Msg::Shutdown))); + assert_eq!(count.available_permits(), 0, "request must not reclaim accepted work"); + drop(held); + } + + #[test] + fn request_shutdown_wakes_both_admission_stages_without_reclaiming_buffers() { + let (driver, _messages) = mock_driver(2, Some(8)); + let bytes = Arc::clone(&driver.byte_admission.as_ref().expect("byte budget").bytes); + let held = try_read_permits(&driver.shards[0].sem, Some(&bytes), 8).expect("occupy byte budget"); + let mut count_waiter = acquire_read_permits(driver.shards[0].sem.clone(), Some(bytes.clone()), 8, None); + let mut byte_waiter = acquire_read_permits(driver.shards[1].sem.clone(), Some(bytes.clone()), 8, None); + assert!(poll_acquire(&mut count_waiter).is_pending()); + assert!(poll_acquire(&mut byte_waiter).is_pending()); + driver.request_shutdown(); + assert!(matches!(poll_acquire(&mut count_waiter), Poll::Ready(Err(_)))); + assert!(matches!(poll_acquire(&mut byte_waiter), Poll::Ready(Err(_)))); + assert!(bytes.is_closed()); + assert_eq!(bytes.available_permits(), 0, "in-flight reservation remains owned"); + drop(held); + } + + #[test] + fn concurrent_shutdown_requests_each_return_with_all_admission_closed() { + for bytes in [None, Some(8)] { + let (driver, _messages) = mock_driver(2, bytes); + let barrier = std::sync::Barrier::new(8); + std::thread::scope(|scope| { + for _ in 0..8 { + scope.spawn(|| { + barrier.wait(); + for _ in 0..2 { + driver.request_shutdown(); + assert!(driver.shards.iter().all(|shard| shard.sem.is_closed())); + if let Some(admission) = &driver.byte_admission { + assert!(admission.bytes.is_closed()); + } + } + }); + } + }); + } + } + + #[tokio::test] + async fn finished_observes_thread_exit_not_request_or_clean_drain() { + let (mut driver, _messages) = mock_driver(1, None); + let (entered_tx, entered_rx) = oneshot::channel(); + let (exit_tx, exit_rx) = mpsc::channel(); + driver.shards[0].handle = Some(std::thread::spawn(move || { + entered_tx.send(()).expect("observe mock thread entry"); + exit_rx.recv_timeout(Duration::from_secs(10)).expect("allow mock thread exit"); + })); + entered_rx.await.expect("mock thread started"); + assert!(!driver.is_finished()); + driver.request_shutdown(); + assert!(!driver.is_finished(), "request must return while a thread is still active"); + // A synthetic leak counter illustrates that thread completion alone is + // not the clean-drain condition. No kernel ever touches this mock state. + driver.shards[0].stats.in_flight.store(1, Ordering::SeqCst); + exit_tx.send(()).expect("finish mock driver"); + tokio::time::timeout(Duration::from_secs(3), async { + while !driver.is_finished() { + tokio::task::yield_now().await; + } + }) + .await + .expect("finished thread must become observable"); + assert_eq!(driver.stats().in_flight, 1); + driver.shards[0].join(); + assert!(driver.is_finished(), "joined handles also report finished"); + } + + #[cfg(feature = "tokio-runtime")] + #[test] + fn async_shutdown_transfers_driver_before_poll_and_unpolled_drop_does_not_join_caller() { + let (mut driver, _messages) = mock_driver(1, None); + let count = driver.shards[0].sem.clone(); + let retired = Arc::downgrade(&driver.shards[0].stats); + let (driver_exit_tx, driver_exit_rx) = mpsc::channel(); + driver.shards[0].handle = Some(std::thread::spawn(move || { + driver_exit_rx + .recv_timeout(Duration::from_secs(10)) + .expect("release mock driver thread"); + })); + let (pool_release_tx, pool_release_rx) = mpsc::channel(); + let (returned_tx, returned_rx) = mpsc::channel(); + let (runtime_exit_tx, runtime_exit_rx) = mpsc::channel(); + let caller = std::thread::spawn(move || { + let runtime = tokio::runtime::Builder::new_current_thread() + .max_blocking_threads(1) + .build() + .expect("shutdown adapter runtime"); + let (pool_entered_tx, pool_entered_rx) = mpsc::channel(); + let _occupied = runtime.spawn_blocking(move || { + pool_entered_tx.send(()).expect("observe occupied blocking pool"); + pool_release_rx + .recv_timeout(Duration::from_secs(10)) + .expect("release blocking pool"); + }); + pool_entered_rx + .recv_timeout(Duration::from_secs(10)) + .expect("pool task started"); + { + let _entered = runtime.enter(); + let future = driver.shutdown_async(); + // The shutdown closure cannot start yet. Admission must already + // be closed at method return, not at the returned future's poll. + let closed_before_poll = count.is_closed(); + drop(future); + returned_tx.send(closed_before_poll).expect("observe unpolled drop return"); + } + runtime_exit_rx + .recv_timeout(Duration::from_secs(10)) + .expect("keep runtime alive until cleanup"); + }); + + let returned = returned_rx.recv_timeout(Duration::from_secs(3)); + // Release every gate before assertions, including on a broken adapter + // that joined on the caller and caused the observation to time out. + pool_release_tx.send(()).expect("allow shutdown closure to run"); + driver_exit_tx.send(()).expect("allow driver join to finish"); + let deadline = Instant::now() + Duration::from_secs(3); + while retired.strong_count() != 0 && Instant::now() < deadline { + std::thread::yield_now(); + } + let cleaned_up = retired.strong_count() == 0; + runtime_exit_tx.send(()).expect("allow runtime teardown"); + caller.join().expect("shutdown caller thread"); + assert!(returned.expect("unpolled drop must return without waiting for the mock driver")); + assert!(cleaned_up, "detached shutdown must still consume and drop the driver"); + } +} + +#[cfg(test)] +mod cancellation_efficiency_tests { + use super::*; + + // No real I/O is submitted: completion decisions get deterministic short + // reads / errno results, so a naturally completed file cannot mask a retry. + fn pending() -> (Pending, oneshot::Receiver>>, Arc) { + let sem = Arc::new(Semaphore::new(1)); + let (done, rx) = oneshot::channel(); + let p = Pending { + #[cfg(feature = "diagnostics")] + timing: None, + buf: (0..16).collect(), + file: Arc::new(File::open("/dev/null").expect("open fixture fd")), + done: Some(done), + offset: 0, + nread: 0, + _permit: ReadPermits { + _count: Arc::clone(&sem).try_acquire_owned().expect("fixture permit"), + _bytes: None, + _shared_reservation: None, + }, + pad: 0, + head: 0, + want: 16, + region_len: 16, + align: 1, + transient_retries: 0, + cancel_requested: false, + }; + (p, rx, sem) + } + + fn assert_cancelled(step: ReapStep) { + match step { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::ECANCELED)), + _ => panic!("cancelled positioned continuation must finish with ECANCELED"), + } + } + + #[test] + fn explicit_cancel_stops_short_read_without_releasing_resources_early() { + let (mut p, _rx, sem) = pending(); + let file = Arc::downgrade(&p.file); + let buffer = p.buf.as_ptr(); + p.cancel_requested = true; + assert_eq!(sem.available_permits(), 0); + assert_cancelled(reap_read(&mut p, 1, 4, false)); + assert_eq!(p.nread, 4); + assert_eq!(p.buf.as_ptr(), buffer); + assert_eq!(sem.available_permits(), 0); + assert!(file.upgrade().is_some()); + drop(p); + assert_eq!(sem.available_permits(), 1); + assert!(file.upgrade().is_none()); + } + + #[test] + fn explicit_cancel_stops_both_transient_errno_retries() { + for errno in [libc::EINTR, libc::EAGAIN] { + let (mut p, _rx, _sem) = pending(); + p.cancel_requested = true; + assert_cancelled(reap_read(&mut p, 1, -errno, false)); + assert_eq!(p.transient_retries, 0); + } + } + + #[test] + fn shutdown_stops_short_read_and_transient_continuations() { + for result in [4, -libc::EINTR, -libc::EAGAIN] { + let (mut p, _rx, _sem) = pending(); + assert_cancelled(reap_read(&mut p, 1, result, true)); + } + } + + #[test] + fn closed_receiver_without_cancel_still_continues_positioned_reads() { + for result in [4, -libc::EINTR, -libc::EAGAIN] { + let (mut p, rx, sem) = pending(); + drop(rx); + assert!(matches!(reap_read(&mut p, 1, result, false), ReapStep::Resubmit(_))); + assert_eq!(sem.available_permits(), 0); + } + } + + #[test] + fn current_position_short_read_survives_cancel_and_shutdown_race() { + let (mut p, _rx, _sem) = pending(); + p.offset = CURRENT_POSITION; + p.cancel_requested = true; + match reap_read(&mut p, 1, 4, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [0, 1, 2, 3]), + _ => panic!("stream short read must keep its read(2) result"), + } + } + + #[test] + fn successful_complete_read_wins_cancel_race() { + let (mut p, _rx, _sem) = pending(); + p.cancel_requested = true; + match reap_read(&mut p, 1, 16, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, (0..16).collect::>()), + _ => panic!("complete read must retain success"), + } + } + + #[test] + fn eof_after_prefix_preserves_success_during_shutdown() { + let (mut p, _rx, _sem) = pending(); + p.nread = 4; + p.cancel_requested = true; + match reap_read(&mut p, 1, 0, true) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [0, 1, 2, 3]), + _ => panic!("observed EOF must preserve the completed prefix"), + } + } + + #[test] + fn transient_retry_keeps_prefix_and_exhausts_budget() { + let (mut p, _rx, sem) = pending(); + p.nread = 4; + for _ in 0..MAX_TRANSIENT_RETRIES { + assert!(matches!(reap_read(&mut p, 1, -libc::EAGAIN, false), ReapStep::Resubmit(_))); + assert_eq!(p.nread, 4); + assert_eq!(sem.available_permits(), 0); + } + match reap_read(&mut p, 1, -libc::EAGAIN, false) { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::EAGAIN)), + _ => panic!("transient retry budget must remain bounded"), + } + } + + #[test] + fn current_position_transient_error_never_retries() { + let (mut p, _rx, _sem) = pending(); + p.offset = CURRENT_POSITION; + p.cancel_requested = true; + match reap_read(&mut p, 1, -libc::EINTR, true) { + ReapStep::Finish(Err(error)) => assert_eq!(error.raw_os_error(), Some(libc::EINTR)), + _ => panic!("stream errors preserve read(2) semantics"), + } + } + + #[test] + fn closed_receiver_skips_direct_result_copy_and_materialization() { + let (mut p, rx, _sem) = pending(); + p.pad = 2; + p.head = 3; + p.want = 4; + p.region_len = 8; + p.align = 4; + let before = p.buf.clone(); + let ptr = p.buf.as_ptr(); + drop(rx); + match reap_read(&mut p, 1, 8, false) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes.capacity(), 0), + _ => panic!("terminal direct read must finish"), + } + assert_eq!(p.buf, before, "orphan result must not be memmoved or truncated"); + assert_eq!(p.buf.as_ptr(), ptr); + } + + #[test] + fn live_receiver_gets_exact_direct_result_range() { + let (mut p, _rx, _sem) = pending(); + p.pad = 2; + p.head = 3; + p.want = 4; + p.region_len = 8; + p.align = 4; + match reap_read(&mut p, 1, 8, false) { + ReapStep::Finish(Ok(bytes)) => assert_eq!(bytes, [5, 6, 7, 8]), + _ => panic!("direct result must contain exactly the logical range"), + } + } + + #[test] + fn eventfd_retries_interruption_until_transfer_or_would_block() { + for terminal in [Ok(8), Err(io::Error::from_raw_os_error(libc::EAGAIN))] { + let mut calls = [ + Err(io::Error::from_raw_os_error(libc::EINTR)), + Err(io::Error::from_raw_os_error(libc::EINTR)), + terminal, + ] + .into_iter(); + eventfd_transfer(|| calls.next().expect("unexpected retry")).expect("transfer or readiness satisfied"); + assert!(calls.next().is_none()); + } + } + + #[test] + fn eventfd_propagates_unexpected_error_and_short_transfer() { + let error = eventfd_transfer(|| Err(io::Error::from_raw_os_error(libc::EBADF))).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EBADF)); + assert!(eventfd_transfer(|| Ok(4)).is_err()); + } + + #[test] + fn saturated_eventfd_signal_keeps_readiness_and_drain_clears_it() { + let event = EventFd::new().expect("eventfd"); + let value = u64::MAX - 1; + // SAFETY: event owns the fd; value is an initialized eight-byte counter. + assert_eq!(unsafe { libc::write(event.as_raw(), (&value as *const u64).cast(), 8) }, 8); + event.signal(); // EAGAIN: the already readable saturated fd is sufficient. + assert!(!event.error_logged.load(Ordering::Relaxed)); + event.drain(); + let mut read = 0_u64; + // SAFETY: event owns the fd; read is a valid eight-byte output buffer. + assert_eq!(unsafe { libc::read(event.as_raw(), (&mut read as *mut u64).cast(), 8) }, -1); + assert_eq!(io::Error::last_os_error().raw_os_error(), Some(libc::EAGAIN)); + event.drain(); // Empty EAGAIN is normal too. + assert!(!event.error_logged.load(Ordering::Relaxed)); } } diff --git a/src/driver_fault_recovery_tests.rs b/src/driver_fault_recovery_tests.rs new file mode 100644 index 0000000..d5d3ee5 --- /dev/null +++ b/src/driver_fault_recovery_tests.rs @@ -0,0 +1,166 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +#[test] +fn unchanged_cumulative_overflow_warns_only_once() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(!update_cq_overflow(&stats, &mut previous, 0)); + assert!(update_cq_overflow(&stats, &mut previous, 7)); + assert!(!update_cq_overflow(&stats, &mut previous, 7)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 7); + assert!(update_cq_overflow(&stats, &mut previous, 8)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 8); +} + +#[test] +fn overflow_counter_wrap_to_zero_updates_snapshot_and_rearms_warning() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(update_cq_overflow(&stats, &mut previous, u32::MAX)); + assert!(!update_cq_overflow(&stats, &mut previous, 0)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 0); + assert!(update_cq_overflow(&stats, &mut previous, 1)); + assert!(!update_cq_overflow(&stats, &mut previous, 1)); +} + +#[test] +fn overflow_counter_wrap_to_nonzero_warns_for_new_observation() { + let stats = DriverStats::default(); + let mut previous = 0; + assert!(update_cq_overflow(&stats, &mut previous, u32::MAX)); + assert!(update_cq_overflow(&stats, &mut previous, 2)); + assert_eq!(stats.cq_overflow.load(Ordering::SeqCst), 2); +} + +fn direct_prefix() -> (Pending, oneshot::Receiver>>) { + let (done, receiver) = oneshot::channel(); + let pending = Pending { + #[cfg(feature = "diagnostics")] + timing: None, + // Simulated completed CQE: no kernel ever references this allocation. + buf: vec![99, 10, 11, 12, 13, 14, 88, 88, 88], + file: Arc::new(File::open("/dev/null").unwrap()), + done: Some(done), + offset: 4096, + nread: 5, + _permit: ReadPermits { + _count: Arc::new(Semaphore::new(1)).try_acquire_owned().unwrap(), + _bytes: None, + _shared_reservation: None, + }, + pad: 1, + head: 2, + want: 6, + region_len: 8, + align: 4, + transient_retries: 0, + cancel_requested: false, + }; + (pending, receiver) +} + +#[test] +fn direct_metadata_failure_preserves_errno_without_delivering_prefix() { + let (mut pending, _receiver) = direct_prefix(); + let before = pending.buf.clone(); + let error = finish_direct_short_read(&mut pending, Err(io::Error::from_raw_os_error(libc::EIO))).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EIO)); + assert_eq!(pending.buf, before, "metadata failure must not turn into prefix delivery"); +} + +#[test] +fn direct_mid_file_short_read_errors_without_delivering_prefix() { + let (mut pending, _receiver) = direct_prefix(); + let error = finish_direct_short_read(&mut pending, Ok(8192)).unwrap_err(); + assert_eq!(error.kind(), io::ErrorKind::Other); + assert!(!pending.buf.is_empty()); +} + +#[test] +fn direct_confirmed_tail_returns_only_initialized_logical_bytes() { + let (mut pending, _receiver) = direct_prefix(); + assert_eq!(finish_direct_short_read(&mut pending, Ok(4101)).unwrap(), [12, 13, 14]); +} + +#[test] +fn direct_concurrent_truncation_preserves_completed_prefix() { + let (mut pending, _receiver) = direct_prefix(); + // A truncate after the CQE does not erase bytes the read already completed. + assert_eq!(finish_direct_short_read(&mut pending, Ok(4096)).unwrap(), [12, 13, 14]); +} + +#[test] +fn direct_eof_before_logical_start_returns_empty() { + let (mut pending, _receiver) = direct_prefix(); + pending.nread = 1; + assert!(finish_direct_short_read(&mut pending, Ok(4097)).unwrap().is_empty()); +} + +fn small_ring() -> Option { + match IoUring::builder().setup_cqsize(2).build(2) { + Ok(ring) => { + assert!(ring.params().is_feature_nodrop(), "overflow recovery requires NODROP"); + Some(ring) + } + Err(error) => { + let failure = ProbeFailure::Setup(error); + assert!(failure.is_expected_restriction(), "unexpected io_uring setup failure: {failure}"); + eprintln!("SKIP fault recovery kernel test: io_uring restricted ({failure})"); + None + } + } +} + +fn push_nop(ring: &mut IoUring, id: u64) { + // SAFETY: NOP has no pointers or external resources, including on unwind. + unsafe { ring.submission().push(&opcode::Nop::new().build().user_data(id)) }.unwrap(); +} + +#[test] +fn idle_ring_skips_enter() { + let Some(mut ring) = small_ring() else { return }; + assert!(submit_if_needed(&mut ring).is_none()); +} + +#[test] +fn empty_sq_flushes_real_kernel_overflow_after_reap() { + let Some(mut ring) = small_ring() else { return }; + assert_eq!(ring.completion().capacity(), 2); + push_nop(&mut ring, 1); + push_nop(&mut ring, 2); + assert_eq!(ring.submit_and_wait(2).unwrap(), 2); + // Leave the CQ full, then complete a third NOP into the NODROP overflow list. + push_nop(&mut ring, 3); + assert_eq!(ring.submit().unwrap(), 1); + assert!(ring.submission().is_empty()); + assert!(ring.submission().cq_overflow(), "test must exercise real kernel overflow"); + let first: Vec<_> = ring.completion().map(|cqe| (cqe.user_data(), cqe.result())).collect(); + assert_eq!(first, [(1, 0), (2, 0)]); + + // Regression: the old empty-SQ early return skipped exactly this enter. + assert_eq!(submit_if_needed(&mut ring).expect("overflow must enter").unwrap(), 0); + let recovered: Vec<_> = ring.completion().map(|cqe| (cqe.user_data(), cqe.result())).collect(); + assert_eq!(recovered, [(3, 0)]); + assert!(submit_if_needed(&mut ring).is_none()); +} + +#[test] +fn sq_capacity_leaves_backlog_owned_until_later_submission() { + let Some(mut ring) = small_ring() else { return }; + let mut backlog: VecDeque<_> = (1..=5).map(|id| opcode::Nop::new().build().user_data(id)).collect(); + let mut completed = Vec::new(); + for remaining in [3, 1, 0] { + flush_backlog(&mut ring, &mut backlog); + assert_eq!(backlog.len(), remaining); + submit_if_needed(&mut ring).expect("queued NOPs must submit").unwrap(); + completed.extend(ring.completion().map(|cqe| { + assert_eq!(cqe.result(), 0); + cqe.user_data() + })); + } + assert_eq!(completed, [1, 2, 3, 4, 5]); + assert!(submit_if_needed(&mut ring).is_none()); +} diff --git a/src/driver_loop_budget_tests.rs b/src/driver_loop_budget_tests.rs new file mode 100644 index 0000000..0c0f94a --- /dev/null +++ b/src/driver_loop_budget_tests.rs @@ -0,0 +1,113 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +fn readable(event: &EventFd) -> bool { + let mut fd = libc::pollfd { + fd: event.as_raw(), + events: libc::POLLIN, + revents: 0, + }; + // SAFETY: initialized pollfd, zero timeout, no pointer retained by kernel. + unsafe { libc::poll(&mut fd, 1, 0) > 0 } +} + +#[test] +fn drained_eventfd_does_not_strand_messages_beyond_intake_budget() { + let wake = EventFd::new().unwrap(); + let (tx, rx) = mpsc::channel(); + for id in 0..TURN_MESSAGES + 1 { + tx.send(id).unwrap(); + wake.signal(); + } + wake.drain(); + assert!(!readable(&wake)); + + let mut first = TurnBudget::default(); + while first.can_take_message() { + rx.try_recv().unwrap(); + first.messages += 1; + } + assert!(first.continue_without_wait(false, false, 0)); + // No additional producer or wake signal: the next turn consumes the tail. + assert_eq!(rx.try_recv().unwrap(), TURN_MESSAGES); + let second = TurnBudget { + messages: 1, + ..TurnBudget::default() + }; + assert!(matches!(rx.try_recv(), Err(TryRecvError::Empty))); + assert!(!second.continue_without_wait(false, false, 0)); +} + +#[test] +fn exact_message_boundary_costs_one_empty_turn_then_sleeps() { + let full = TurnBudget { + messages: TURN_MESSAGES, + ..TurnBudget::default() + }; + assert!(full.continue_without_wait(false, false, 0)); + assert!(!TurnBudget::default().continue_without_wait(false, false, 0)); +} + +#[test] +fn oversized_allocation_makes_progress_then_yields_to_reap() { + let mut budget = TurnBudget::default(); + assert!(budget.can_take_message()); + budget.messages += 1; + // Record one already-accepted read, larger than the fairness threshold. + // It is not rejected or repeatedly deferred by the turn budget. + budget.allocation_bytes = TURN_ALLOCATION_BYTES * 2; + assert!(!budget.can_take_message()); + assert!(budget.can_reap()); + assert!(budget.continue_without_wait(false, false, 0)); + assert!(TurnBudget::default().can_take_message()); +} + +#[test] +fn allocation_volume_yields_before_message_limit() { + let mut budget = TurnBudget::default(); + while budget.can_take_message() { + budget.messages += 1; + budget.allocation_bytes += TURN_ALLOCATION_BYTES / 4; + } + assert_eq!(budget.messages, 4); + assert!(budget.continue_without_wait(false, false, 0)); +} + +#[test] +fn completion_burst_yields_to_intake_and_continues_without_a_new_edge() { + let mut cq: VecDeque<_> = (0..TURN_COMPLETIONS + 1).collect(); + let mut budget = TurnBudget::default(); + while budget.can_reap() { + cq.pop_front().unwrap(); + budget.completions += 1; + } + assert_eq!(cq.len(), 1); + assert!(budget.continue_without_wait(true, false, 0)); + // A new turn starts with intake, so shutdown/cancel can be observed before + // the next chunk of completions rather than waiting for CQ to empty. + assert!(TurnBudget::default().can_take_message()); +} + +#[test] +fn failed_or_zero_progress_submissions_cannot_spin_from_queued_sqes_alone() { + let budget = TurnBudget::default(); + // submit_ring maps EBUSY/EINTR/other errors and Ok(0) to zero; each must + // return to the eventfd/heartbeat wait when there is no other ready work. + assert!(!budget.continue_without_wait(false, true, 0)); + // A partial positive submission allows a bounded immediate follow-up; if + // that next turn makes no progress, it waits again. + assert!(budget.continue_without_wait(false, true, 1)); + assert!(!budget.continue_without_wait(false, true, 0)); +} + +#[test] +fn completion_readiness_wins_even_when_last_submit_failed() { + assert!(TurnBudget::default().continue_without_wait(true, true, 0)); +} + +#[test] +fn submission_progress_without_remaining_work_does_not_busy_poll() { + assert!(!TurnBudget::default().continue_without_wait(false, false, 2)); +} diff --git a/src/driver_submit_result_tests.rs b/src/driver_submit_result_tests.rs new file mode 100644 index 0000000..6f580ab --- /dev/null +++ b/src/driver_submit_result_tests.rs @@ -0,0 +1,159 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +#[derive(Default)] +struct ResultHarness { + stats: DriverStats, + consecutive_errors: u32, + logged: bool, + shutting_down: bool, + shutdown_calls: usize, + pending: HashMap, + backlog: VecDeque, + queued_cancels: HashSet, +} + +impl ResultHarness { + fn apply(&mut self, result: io::Result) -> usize { + handle_submit_result( + result, + &self.stats, + &mut self.consecutive_errors, + &mut self.logged, + &mut self.shutting_down, + || { + self.shutdown_calls += 1; + for id in self.pending.keys() { + queue_cancel(&mut self.backlog, &mut self.queued_cancels, *id); + } + }, + ) + } + + fn fail(&mut self, errno: i32) -> usize { + self.apply(Err(io::Error::from_raw_os_error(errno))) + } +} + +fn pending_read(file: Arc, count: &Arc, bytes: &Arc) -> Pending { + Pending { + #[cfg(feature = "diagnostics")] + timing: None, + buf: vec![42; 16], + file, + done: None, + offset: 0, + nread: 0, + _permit: ReadPermits { + _count: Arc::clone(count).try_acquire_owned().unwrap(), + _bytes: Some(Arc::clone(bytes).try_acquire_many_owned(16).unwrap()), + _shared_reservation: None, + }, + pad: 0, + head: 0, + want: 16, + region_len: 16, + align: 1, + transient_retries: 0, + cancel_requested: false, + } +} + +#[test] +fn partial_positive_acceptance_resets_errors_and_allows_bounded_retry() { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.apply(Ok(2)); // Model 2 accepted out of a larger SQ. + assert_eq!(submitted, 2); + assert_eq!(harness.consecutive_errors, 0); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); + assert!(TurnBudget::default().continue_without_wait(false, true, submitted)); + assert_eq!(harness.shutdown_calls, 0); +} + +#[test] +fn zero_acceptance_resets_error_streak_but_waits_with_remaining_sqes() { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.apply(Ok(0)); + assert_eq!(harness.consecutive_errors, 0); + assert!(!TurnBudget::default().continue_without_wait(false, true, submitted)); + assert_eq!(harness.shutdown_calls, 0); +} + +#[test] +fn interrupted_or_busy_submission_resets_streak_without_counting_or_spinning() { + for errno in [libc::EINTR, libc::EBUSY] { + let mut harness = ResultHarness::default(); + harness.fail(libc::EIO); + let submitted = harness.fail(errno); + assert_eq!(harness.consecutive_errors, 0); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); + assert!(!TurnBudget::default().continue_without_wait(false, true, submitted)); + assert!(!harness.shutting_down); + } +} + +#[test] +fn persistent_failures_shutdown_once_and_keep_every_pending_resource() { + let mut harness = ResultHarness::default(); + let file = Arc::new(File::open("/dev/null").unwrap()); + let count = Arc::new(Semaphore::new(2)); + let bytes = Arc::new(Semaphore::new(32)); + harness.pending.insert(1, pending_read(Arc::clone(&file), &count, &bytes)); + harness.pending.insert(2, pending_read(Arc::clone(&file), &count, &bytes)); + let ptrs: Vec<_> = (1..=2).map(|id| harness.pending[&id].buf.as_ptr()).collect(); + // Model an already queued drop-cancel; shutdown must not duplicate it. + queue_cancel(&mut harness.backlog, &mut harness.queued_cancels, 1); + + for attempt in 1..=MAX_CONSECUTIVE_SUBMIT_ERRORS + 2 { + assert_eq!(harness.fail(libc::EPERM), 0); + assert_eq!(harness.consecutive_errors, attempt); + assert_eq!(harness.shutting_down, attempt >= MAX_CONSECUTIVE_SUBMIT_ERRORS); + assert_eq!(harness.shutdown_calls, usize::from(attempt >= MAX_CONSECUTIVE_SUBMIT_ERRORS)); + assert_eq!(harness.pending.len(), 2); + for id in 1..=2 { + assert_eq!(harness.pending[&id].buf.as_ptr(), ptrs[(id - 1) as usize]); + assert_eq!(harness.pending[&id].buf, [42; 16]); + } + assert_eq!(count.available_permits(), 0); + assert_eq!(bytes.available_permits(), 0); + assert_eq!(Arc::strong_count(&file), 3); + } + assert!(harness.logged); + assert_eq!(harness.backlog.len(), 2); + assert_eq!(harness.queued_cancels, HashSet::from([1, 2])); + assert_eq!( + harness.stats.submit_errors.load(Ordering::SeqCst), + u64::from(MAX_CONSECUTIVE_SUBMIT_ERRORS + 2) + ); + // No SQEs were submitted to a kernel in this model, so dropping is safe. + drop(harness); + assert_eq!(count.available_permits(), 2); + assert_eq!(bytes.available_permits(), 32); + assert_eq!(Arc::strong_count(&file), 1); +} + +#[test] +fn success_and_transient_errors_break_the_consecutive_shutdown_threshold() { + for reset in [Ok(1), Ok(0), Err(libc::EINTR), Err(libc::EBUSY)] { + let mut harness = ResultHarness::default(); + for _ in 1..MAX_CONSECUTIVE_SUBMIT_ERRORS { + harness.fail(libc::EIO); + } + harness.apply(reset.map_err(io::Error::from_raw_os_error)); + harness.fail(libc::EIO); + assert_eq!(harness.consecutive_errors, 1); + assert_eq!(harness.shutdown_calls, 0); + } +} + +#[test] +fn eagain_is_counted_as_a_bounded_submit_failure() { + let mut harness = ResultHarness::default(); + assert_eq!(harness.fail(libc::EAGAIN), 0); + assert_eq!(harness.consecutive_errors, 1); + assert_eq!(harness.stats.submit_errors.load(Ordering::SeqCst), 1); +} diff --git a/src/lib.rs b/src/lib.rs index d75e314..6b965e1 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -55,4 +55,6 @@ mod diagnostics; pub use diagnostics::{DIAGNOSTICS_SAMPLE_INTERVAL, DiagnosticsSnapshot, LatencyHistogram}; #[cfg(target_os = "linux")] -pub use driver::{ProbeFailure, ReadHandle, StatsSnapshot, UringDriver}; +pub use driver::{ + MAX_BATCH_READS, ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, SharedReadBudget, StatsSnapshot, UringDriver, +}; diff --git a/src/shard_policy_tests.rs b/src/shard_policy_tests.rs new file mode 100644 index 0000000..689981c --- /dev/null +++ b/src/shard_policy_tests.rs @@ -0,0 +1,215 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +use super::*; + +// No ring or driver thread: queued messages retain real admission permits, and +// tests observe the exact destination of reads and cancels without I/O races. +fn driver(limits: ReadLimits) -> (UringDriver, Vec>) { + let byte_admission = limits.max_in_flight_bytes.map(|bytes| Arc::new(ByteAdmission::new(bytes))); + let mut receivers = Vec::new(); + let shards = (0..2) + .map(|_| { + let (tx, rx) = mpsc::channel(); + receivers.push(rx); + let sem = Arc::new(Semaphore::new(1)); + if let Some(admission) = &byte_admission { + admission.register(&sem); + } + Shard { + tx, + handle: None, + stats: Arc::new(DriverStats::default()), + sem, + wake_efd: Arc::new(EventFd::new().unwrap()), + } + }) + .collect(); + ( + UringDriver { + limits, + shard_policy: ShardPolicy::default(), + byte_admission, + shards, + next_id: AtomicU64::new(1), + rr: AtomicUsize::new(0), + }, + receivers, + ) +} + +fn file() -> Arc { + Arc::new(File::open("/dev/zero").unwrap()) +} + +fn poll(handle: &mut ReadHandle) -> Poll>> { + Pin::new(handle).poll(&mut Context::from_waker(std::task::Waker::noop())) +} + +#[test] +fn capacity_aware_read_and_cancel_use_the_acquired_shard() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = driver.read_at(file(), 0, 8); + let id = handle.id; + let HandleState::Submitted { wake } = &handle.state else { panic!("free shard was not selected") }; + assert!(Arc::ptr_eq(wake, &driver.shards[1].wake_efd)); + let message = receivers[1].try_recv().unwrap(); + assert!(matches!(&message, Msg::Read { id: actual, .. } if *actual == id)); + drop(handle); + assert!(matches!(receivers[1].try_recv(), Ok(Msg::Cancel { id: actual }) if actual == id)); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert_eq!(driver.shards[1].sem.available_permits(), 0, "cancel must not release accepted permits"); + drop(message); + drop(held); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); +} + +#[test] +fn default_and_stream_routing_still_wait_on_the_round_robin_shard() { + for stream in [false, true] { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = if stream { + driver.with_shard_policy(ShardPolicy::CapacityAware) + } else { + driver + }; + let _held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = if stream { + driver.read_current(file(), 8) + } else { + driver.read_at(file(), 0, 8) + }; + let HandleState::WaitingPermit { wake, .. } = &handle.state else { panic!("legacy route changed") }; + assert!(Arc::ptr_eq(wake, &driver.shards[0].wake_efd)); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); + assert_eq!(driver.shards[1].sem.available_permits(), 1); + drop(handle); + } +} + +#[test] +fn capacity_aware_skips_closed_shards_but_default_does_not() { + for policy in [ShardPolicy::RoundRobin, ShardPolicy::CapacityAware] { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(policy); + driver.shards[0].sem.close(); + let mut handle = driver.read_at(file(), 0, 8); + if policy == ShardPolicy::CapacityAware { + assert!(matches!(receivers[1].try_recv(), Ok(Msg::Read { .. }))); + } else { + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); + } + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + } +} + +#[test] +fn capacity_aware_rejects_when_every_shard_is_closed() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + for shard in &driver.shards { + shard.sem.close(); + } + let mut handle = driver.read_at(file(), 0, 8); + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[test] +fn capacity_aware_deferred_owner_remains_fixed_when_another_shard_frees() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let first = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let second = Arc::clone(&driver.shards[1].sem).try_acquire_owned().unwrap(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(poll(&mut handle).is_pending()); + drop(second); + assert!(poll(&mut handle).is_pending()); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); + drop(first); + assert!(poll(&mut handle).is_pending()); // submitted, awaiting its CQE + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { .. }))); + drop(handle); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Cancel { .. }))); + assert!(matches!(receivers[1].try_recv(), Err(TryRecvError::Empty))); +} + +#[test] +fn capacity_aware_deferred_waiters_keep_semaphore_queue_order() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let _other = Arc::clone(&driver.shards[1].sem).try_acquire_owned().unwrap(); + let mut first = driver.read_at(file(), 0, 8); + driver.rr.store(0, Ordering::Relaxed); + let mut second = driver.read_at(file(), 8, 8); + assert!(poll(&mut first).is_pending()); + assert!(poll(&mut second).is_pending()); + drop(held); + assert!(poll(&mut second).is_pending(), "later waiter must not overtake the first"); + assert!(matches!(receivers[0].try_recv(), Err(TryRecvError::Empty))); + assert!(poll(&mut first).is_pending()); + let message = receivers[0].try_recv().unwrap(); + assert!(matches!(&message, Msg::Read { id, .. } if *id == first.id)); + assert!(poll(&mut second).is_pending()); + drop(message); + assert!(poll(&mut second).is_pending()); + assert!(matches!(receivers[0].try_recv(), Ok(Msg::Read { id, .. }) if id == second.id)); +} + +#[test] +fn byte_shortage_releases_failed_count_reservation_and_waiter_drop_releases_partial_admission() { + let (driver, receivers) = driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(8), + }); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let admission = driver.byte_admission.as_ref().unwrap(); + let bytes = Arc::clone(&admission.bytes).try_acquire_many_owned(8).unwrap(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert!(poll(&mut handle).is_pending()); + assert_eq!(driver.shards[0].sem.available_permits(), 0); + assert_eq!(driver.shards[1].sem.available_permits(), 1); + drop(handle); + drop(bytes); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert_eq!(admission.bytes.available_permits(), 8); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[test] +fn closed_byte_admission_is_terminal_even_with_available_count_permits() { + let (driver, receivers) = driver(ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(8), + }); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + // Close just bytes to exercise the race before registry count closure. + driver.byte_admission.as_ref().unwrap().bytes.close(); + let mut handle = driver.read_at(file(), 0, 8); + assert!(matches!(poll(&mut handle), Poll::Ready(Err(_)))); + assert!(driver.shards.iter().all(|shard| shard.sem.available_permits() == 1)); + assert!(receivers.iter().all(|rx| matches!(rx.try_recv(), Err(TryRecvError::Empty)))); +} + +#[cfg(feature = "diagnostics")] +#[test] +fn capacity_aware_diagnostics_sample_only_the_final_owner() { + let (driver, receivers) = driver(ReadLimits::default()); + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let held = Arc::clone(&driver.shards[0].sem).try_acquire_owned().unwrap(); + let handle = driver.read_at(file(), 0, 8); // rr 0, accepted on 1 + assert_eq!(driver.shard_diagnostics()[0].admission.count, 0); + assert_eq!(driver.shard_diagnostics()[1].admission.count, 1); + drop(receivers[1].try_recv().unwrap()); + drop(handle); + drop(held); + // The failed candidate did not consume shard 0's first sample position. + driver.rr.store(0, Ordering::Relaxed); + let _handle = driver.read_at(file(), 0, 8); + assert_eq!(driver.shard_diagnostics()[0].admission.count, 1); +} diff --git a/tests/admission.rs b/tests/admission.rs new file mode 100644 index 0000000..bd9d42d --- /dev/null +++ b/tests/admission.rs @@ -0,0 +1,120 @@ +//! Native Linux resource-admission tests (rustfs/backlog#2647). +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io::Write; +use std::os::fd::FromRawFd; +use std::sync::Arc; +use std::task::Poll; +use std::time::Duration; + +use rustfs_uring::{ReadLimits, UringDriver}; + +fn driver_or_skip() -> Option { + match UringDriver::probe_and_start_with_limits( + 8, + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) { + Ok(driver) => Some(driver), + Err(err) => { + assert!(err.is_expected_restriction(), "unexpected probe failure: {err}"); + eprintln!("SKIP admission: restricted environment ({err})"); + None + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid output array; successful pipe creates two owned descriptors. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0); + // SAFETY: each fresh descriptor transfers ownership exactly once. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("read did not enter the driver"); +} + +async fn assert_pending(future: &mut rustfs_uring::ReadHandle) { + std::future::poll_fn(|cx| { + assert!(std::pin::Pin::new(&mut *future).poll(cx).is_pending()); + Poll::Ready(()) + }) + .await; +} + +#[tokio::test] +async fn orphan_holds_shared_bytes_until_its_real_read_completes() { + let Some(driver) = driver_or_skip() else { return }; + let (read, mut write) = pipe(); + let first = driver.read_current(read, 8).without_cancel_on_drop(); + wait_submitted(&driver, 1).await; + let mut next = driver.read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 8); + assert_pending(&mut next).await; + drop(first); + assert_pending(&mut next).await; + assert_eq!(driver.stats().submitted, 1); + write.write_all(b"12345678").unwrap(); + assert_eq!(tokio::time::timeout(Duration::from_secs(2), next).await.unwrap().unwrap(), vec![0; 8]); + let stats = driver.shutdown(); + assert_eq!(stats.orphan_reclaimed, 1); + assert_eq!(stats.submitted, 2); + assert_eq!(stats.in_flight, 0); +} + +#[tokio::test] +async fn dropping_a_byte_waiter_does_not_submit_or_strand_capacity() { + let Some(driver) = driver_or_skip() else { return }; + let (read, mut write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let file = Arc::new(File::open("/dev/zero").unwrap()); + let mut canceled = driver.read_at(file.clone(), 0, 8); + assert_pending(&mut canceled).await; + drop(canceled); + write.write_all(b"12345678").unwrap(); + assert_eq!(first.await.unwrap(), b"12345678"); + let result = tokio::time::timeout(Duration::from_secs(2), driver.read_at(file, 0, 8)) + .await + .unwrap() + .unwrap(); + assert_eq!(result, vec![0; 8]); + assert_eq!(driver.shutdown().submitted, 2); +} + +#[tokio::test] +async fn shutdown_wakes_byte_waiters_while_canceling_pending_read() { + let Some(driver) = driver_or_skip() else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let mut waiting = driver.read_at(Arc::new(File::open("/dev/zero").unwrap()), 0, 8); + assert_pending(&mut waiting).await; + driver.shutdown(); + assert!(tokio::time::timeout(Duration::from_secs(2), waiting).await.unwrap().is_err()); + assert!(tokio::time::timeout(Duration::from_secs(2), first).await.unwrap().is_err()); +} + +#[tokio::test] +async fn completed_results_do_not_hold_driver_byte_admission() { + let Some(driver) = driver_or_skip() else { return }; + let file = Arc::new(File::open("/dev/zero").unwrap()); + let retained = driver.read_at(file.clone(), 0, 8).await.unwrap(); + let next = tokio::time::timeout(Duration::from_secs(2), driver.read_at(file, 0, 8)) + .await + .unwrap() + .unwrap(); + assert_eq!(retained, next); + assert_eq!(driver.shutdown().submitted, 2); +} diff --git a/tests/cancel.rs b/tests/cancel.rs index bf3716a..4d3b574 100644 --- a/tests/cancel.rs +++ b/tests/cancel.rs @@ -28,7 +28,7 @@ use std::os::fd::{AsRawFd, FromRawFd}; use std::sync::Arc; use std::time::{Duration, Instant}; -use rustfs_uring::UringDriver; +use rustfs_uring::{MAX_BATCH_READS, ReadLimits, ReadRequest, ShardPolicy, UringDriver}; fn driver_or_skip(name: &str) -> Option { match UringDriver::probe_and_start(64) { @@ -80,6 +80,76 @@ fn temp_file_with(content: &[u8], tag: &str) -> (std::path::PathBuf, Arc) (path, file) } +#[tokio::test(flavor = "multi_thread")] +async fn batch_reads_return_byte_exact_results_in_handle_order() { + let Some(driver) = sharded_driver_or_skip("batch_reads_return_byte_exact_results_in_handle_order", 2) else { + return; + }; + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + let content = make_content(8192); + let (path, file) = temp_file_with(&content, "batch-exact"); + let requests = (0..MAX_BATCH_READS) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: (index * 17) as u64, + len: 257, + }) + .collect(); + for (index, handle) in driver.read_at_batch(requests).unwrap().into_iter().enumerate() { + let bytes = tokio::time::timeout(Duration::from_secs(5), handle).await.unwrap().unwrap(); + assert_eq!(bytes, &content[index * 17..index * 17 + 257]); + } + let stats = driver.shutdown(); + assert_eq!(stats.submitted, MAX_BATCH_READS as u64); + assert_eq!(stats.delivered, stats.submitted); + assert_eq!(stats.in_flight, 0); + let _ = std::fs::remove_file(path); +} + +#[tokio::test(flavor = "multi_thread")] +async fn batch_cancellation_and_deferred_admission_drain_without_losing_live_results() { + let limits = ReadLimits { + max_read_len: Some(256), + max_in_flight_bytes: Some(512), + }; + let driver = match UringDriver::probe_and_start_with_limits(2, 2, limits) { + Ok(driver) => driver.with_shard_policy(ShardPolicy::CapacityAware), + Err(error) => { + assert!(error.is_expected_restriction(), "{error}"); + eprintln!("SKIP batch_cancellation_and_deferred_admission_drain_without_losing_live_results: {error}"); + return; + } + }; + let content = make_content(8192); + let (path, file) = temp_file_with(&content, "batch-cancel"); + let requests = (0..MAX_BATCH_READS) + .map(|index| ReadRequest { + file: Arc::clone(&file), + offset: (index * 31) as u64, + len: 256, + }) + .collect(); + let mut retained = Vec::new(); + for (index, handle) in driver.read_at_batch(requests).unwrap().into_iter().enumerate() { + if index % 2 == 0 { + drop(handle); // Accepted reads cancel; deferred reads submit nothing. + } else { + retained.push((index, handle)); + } + } + for (index, handle) in retained { + let bytes = tokio::time::timeout(Duration::from_secs(5), handle).await.unwrap().unwrap(); + assert_eq!(bytes, &content[index * 31..index * 31 + 256]); + } + assert!(wait_until(Duration::from_secs(2), || driver.stats().in_flight == 0).await); + let stats = driver.shutdown(); + assert!(stats.submitted >= (MAX_BATCH_READS / 2) as u64); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + // Completion can win the cancel race; no minimum successful cancel count. + let _ = std::fs::remove_file(path); +} + /// An OS pipe whose read side never completes until we write — the only /// deterministic way to hold an op in flight across a future drop. fn os_pipe() -> (Arc, File) { diff --git a/tests/fault_injection.rs b/tests/fault_injection.rs index 3500d2c..aaea7fa 100644 --- a/tests/fault_injection.rs +++ b/tests/fault_injection.rs @@ -32,7 +32,7 @@ use std::os::unix::process::ExitStatusExt; use std::sync::Arc; use std::time::{Duration, Instant}; -use rustfs_uring::UringDriver; +use rustfs_uring::{ReadLimits, UringDriver}; /// A pipe whose read side blocks until the write side is written or closed — the /// only portable way to hold an op provably in flight. @@ -247,6 +247,79 @@ fn stranded_handle_errors_after_bounded_drain_bailout() { drop(pipe_write); } +/// #2647: exercise the real Pending ownership path with both admission stages +/// waiting. This seam discards a real EOF/cancel CQE; it models a lost completion +/// for bounded-drain testing, not an actual hung kernel read or storage device. +#[test] +fn byte_budget_bailout_leaks_pending_and_errors_both_admission_waiters() { + // SAFETY: this test binary is run with --test-threads=1. Configure the seam + // before driver creation; remove it only after every driver thread joins. + unsafe { + std::env::set_var("RUSTFS_URING_FAULT_STUCK_DRAIN", "1"); + std::env::set_var("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400"); + } + let driver = match UringDriver::probe_and_start_with_limits( + 1, + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) { + Ok(driver) => driver, + Err(error) => { + clear_stuck_env(); + assert!(error.is_expected_restriction(), "unexpected probe failure: {error}"); + eprintln!("SKIP byte_budget_bailout: restricted environment ({error})"); + return; + } + }; + + let (pipe_read, pipe_write) = os_pipe(); + let leaked_file = Arc::downgrade(&pipe_read); + // Round-robin request 0 holds shard 0's sole count permit and all bytes. + let stranded = driver.read_current(Arc::clone(&pipe_read), 8).without_cancel_on_drop(); + assert!(wait_until(Duration::from_secs(2), || driver.stats().in_flight == 1)); + let waiting_file = Arc::new(File::open("/dev/zero").unwrap()); + // Request 1 acquires shard 1's count but waits on shared bytes; request 2 + // waits on shard 0's count before it can reach the byte semaphore. + let mut byte_waiter = driver.read_at(Arc::clone(&waiting_file), 0, 8); + let mut count_waiter = driver.read_at(waiting_file, 0, 8); + let mut cx = std::task::Context::from_waker(std::task::Waker::noop()); + assert!(std::pin::Pin::new(&mut byte_waiter).poll(&mut cx).is_pending()); + assert!(std::pin::Pin::new(&mut count_waiter).poll(&mut cx).is_pending()); + assert_eq!(driver.stats().submitted, 1, "waiting reads must not allocate or submit"); + + // EOF makes the real pipe read finite. The seam deliberately suppresses its + // completion bookkeeping so the Pending and its reservations must be leaked. + drop(pipe_write); + drop(pipe_read); + let start = Instant::now(); + let stats = driver.shutdown(); + let elapsed = start.elapsed(); + clear_stuck_env(); + assert!(elapsed >= Duration::from_millis(300), "bailout path was not taken: {elapsed:?}"); + assert!(elapsed < Duration::from_secs(3), "bounded drain exceeded its deadline: {elapsed:?}"); + assert_eq!(stats.in_flight, 1, "Pending reservation must remain retained after bailout"); + assert_eq!(stats.submitted, 1, "neither admission waiter may reach the driver"); + assert_eq!(stats.delivered, 0, "discarded EOF CQE must not report successful delivery"); + + let rt = tokio::runtime::Builder::new_current_thread().enable_time().build().unwrap(); + rt.block_on(async { + for (name, handle) in [ + ("stranded", stranded), + ("byte-stage", byte_waiter), + ("count-stage", count_waiter), + ] { + let result = tokio::time::timeout(Duration::from_secs(2), handle).await; + assert!(matches!(result, Ok(Err(_))), "{name} handle must fail after bailout, got {result:?}"); + } + }); + // All caller references and handles are gone. This sole remaining strong + // reference belongs to the leaked Pending (whose ReadPermits stay with it). + assert_eq!(leaked_file.strong_count(), 1, "Pending's owned file must survive the bailout"); +} + fn clear_stuck_env() { // SAFETY: `--test-threads=1` teardown; no concurrent environment readers. unsafe { diff --git a/tests/ordered_prefetch.rs b/tests/ordered_prefetch.rs new file mode 100644 index 0000000..9b45a76 --- /dev/null +++ b/tests/ordered_prefetch.rs @@ -0,0 +1,286 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +#[path = "../examples/ordered_prefetch/reader.rs"] +mod reader; + +use reader::{Config, OrderedReader}; +use std::cell::RefCell; +use std::collections::BTreeMap; +use std::future::Future; +use std::io; +use std::pin::Pin; +use std::rc::Rc; +use std::sync::Arc; +use std::task::{Context, Poll, Waker}; + +#[derive(Default)] +struct Control { + submitted: Vec<(u64, usize)>, + completed: BTreeMap>>, + polls: BTreeMap, + dropped: Vec, +} + +struct ManualRead { + offset: u64, + control: Rc>, +} + +impl Future for ManualRead { + type Output = io::Result>; + + fn poll(self: Pin<&mut Self>, _: &mut Context<'_>) -> Poll { + let mut control = self.control.borrow_mut(); + *control.polls.entry(self.offset).or_default() += 1; + match control.completed.remove(&self.offset) { + Some(result) => Poll::Ready(result), + None => Poll::Pending, + } + } +} + +impl Drop for ManualRead { + fn drop(&mut self) { + self.control.borrow_mut().dropped.push(self.offset); + } +} + +fn config() -> Config { + Config { + offset: 0, + length: 32, + chunk: 4, + window: 3, + max_bytes: 12, + } +} + +fn source(control: &Rc>) -> impl FnMut(u64, usize) -> ManualRead + use<> { + let control = Rc::clone(control); + move |offset, len| { + control.borrow_mut().submitted.push((offset, len)); + ManualRead { + offset, + control: Rc::clone(&control), + } + } +} + +fn poll(future: &mut F) -> Poll { + Pin::new(future).poll(&mut Context::from_waker(Waker::noop())) +} + +#[test] +fn out_of_order_results_remain_ordered_and_no_refill_occurs_after_yield() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(control.borrow().submitted.is_empty(), "constructor must not start work"); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(8, Ok(vec![8; 4])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 4])); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("head must be ready") + }; + assert_eq!((chunk.offset, chunk.bytes), (0, vec![0; 4])); + assert_eq!(control.borrow().submitted.len(), 3, "slow consumer must not trigger refill"); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("cached second chunk") + }; + assert_eq!((chunk.offset, chunk.bytes), (4, vec![4; 4])); + assert_eq!(control.borrow().polls[&8], 2, "a completed tail must never be repolled"); +} + +#[test] +fn dropping_pending_next_keeps_head_handle_and_offset() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + let mut next = Box::pin(reader.next()); + assert!(poll(&mut next).is_pending()); + drop(next); + assert!(control.borrow().dropped.is_empty()); + control.borrow_mut().completed.insert(0, Ok(vec![1; 4])); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("same head must resume") + }; + assert_eq!(chunk.offset, 0); + assert_eq!(control.borrow().submitted.len(), 3); +} + +#[test] +fn ready_tail_remains_charged_to_the_same_logical_byte_window() { + let control = Rc::new(RefCell::new(Control::default())); + let mut cfg = config(); + cfg.max_bytes = 8; + let mut reader = OrderedReader::new(cfg, source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 4])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(control.borrow().submitted, [(0, 4), (4, 4)]); +} + +#[test] +fn early_tail_eof_cancels_later_slots_but_preserves_earlier_output() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Ok(vec![4; 2])); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert!(control.borrow().dropped.contains(&8)); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("short final chunk") + }; + assert_eq!((chunk.offset, chunk.bytes), (4, vec![4; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert_eq!(control.borrow().submitted.len(), 3); +} + +#[test] +fn error_is_delivered_at_its_ordered_position_then_reader_is_fused() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(4, Err(io::Error::other("fault"))); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(vec![0; 4])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Err(_)))); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn empty_range_submits_nothing_and_zero_byte_eof_returns_no_chunk() { + let control = Rc::new(RefCell::new(Control::default())); + let mut cfg = config(); + cfg.length = 0; + let mut empty = OrderedReader::new(cfg, source(&control)).unwrap(); + assert!(matches!(poll(&mut Box::pin(empty.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().submitted.is_empty()); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(Vec::new())); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&4)); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn dropping_reader_drops_every_active_handle() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + drop(reader); + assert_eq!(control.borrow().dropped, [0, 4, 8]); +} + +#[test] +fn oversized_source_result_is_invalid_data_and_cancels_tail() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new(config(), source(&control)).unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + control.borrow_mut().completed.insert(0, Ok(vec![0; 5])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Err(e)) if e.kind() == io::ErrorKind::InvalidData)); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert!(control.borrow().dropped.contains(&4)); + assert!(control.borrow().dropped.contains(&8)); +} + +#[test] +fn final_requested_slot_is_shortened_to_range_end_without_extra_reads() { + let control = Rc::new(RefCell::new(Control::default())); + let mut reader = OrderedReader::new( + Config { + offset: 7, + length: 6, + ..config() + }, + source(&control), + ) + .unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(control.borrow().submitted, [(7, 4), (11, 2)]); + control.borrow_mut().completed.insert(7, Ok(vec![7; 4])); + control.borrow_mut().completed.insert(11, Ok(vec![11; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(Some(_))))); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("final range slot") + }; + assert_eq!((chunk.offset, chunk.bytes), (11, vec![11; 2])); + assert!(matches!(poll(&mut Box::pin(reader.next())), Poll::Ready(Ok(None)))); + assert_eq!(control.borrow().submitted.len(), 2); +} + +#[test] +fn invalid_limits_and_overflow_fail_before_any_source_call() { + let control = Rc::new(RefCell::new(Control::default())); + let base = config(); + for cfg in [ + Config { chunk: 0, ..base }, + Config { window: 0, ..base }, + Config { window: 65, ..base }, + Config { max_bytes: 3, ..base }, + Config { + offset: u64::MAX, + ..base + }, + Config { + offset: i64::MAX as u64, + length: 1, + ..base + }, + Config { + chunk: usize::MAX, + ..base + }, + ] { + assert!(OrderedReader::new(cfg, source(&control)).is_err()); + } + assert!(control.borrow().submitted.is_empty()); +} + +#[test] +fn deferred_tail_assigned_bytes_is_repolled_even_while_head_waits_on_admission() { + let head_count = Arc::new(tokio::sync::Semaphore::new(1)); + let tail_count = Arc::new(tokio::sync::Semaphore::new(1)); + let bytes = Arc::new(tokio::sync::Semaphore::new(4)); + let held_head = Arc::clone(&head_count).try_acquire_owned().unwrap(); + let held_bytes = Arc::clone(&bytes).try_acquire_many_owned(4).unwrap(); + let submitted = Rc::new(RefCell::new(Vec::new())); + let accepted = Rc::clone(&submitted); + let mut reader = OrderedReader::new( + Config { + window: 2, + max_bytes: 8, + ..config() + }, + move |offset, len| { + let count = Arc::clone(if offset == 0 { &head_count } else { &tail_count }); + let bytes = Arc::clone(&bytes); + let accepted = Rc::clone(&accepted); + Box::pin(async move { + let _count = count.acquire_owned().await.unwrap(); + let _bytes = bytes.acquire_many_owned(len as u32).await.unwrap(); + accepted.borrow_mut().push(offset); + Ok(vec![offset as u8; len]) + }) + }, + ) + .unwrap(); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + // Tokio assigns bytes to the already queued tail waiter before head can + // acquire its count permit. Polling only head now would strand those bytes. + drop(held_bytes); + drop(held_head); + assert!(poll(&mut Box::pin(reader.next())).is_pending()); + assert_eq!(*submitted.borrow(), [4], "tail must progress without head completion"); + let Poll::Ready(Ok(Some(chunk))) = poll(&mut Box::pin(reader.next())) else { + panic!("head should now acquire released bytes") + }; + assert_eq!(chunk.offset, 0); + assert_eq!(*submitted.borrow(), [4, 0]); +} diff --git a/tests/shard_policy.rs b/tests/shard_policy.rs new file mode 100644 index 0000000..732cce8 --- /dev/null +++ b/tests/shard_policy.rs @@ -0,0 +1,92 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::os::fd::FromRawFd; +use std::sync::Arc; +use std::time::Duration; + +use rustfs_uring::{ShardPolicy, UringDriver}; + +async fn exercise_busy_shard(policy: ShardPolicy) { + let driver = match UringDriver::probe_and_start_sharded(1, 2) { + Ok(driver) => driver.with_shard_policy(policy), + Err(error) => { + assert!(error.is_expected_restriction(), "unexpected setup error: {error}"); + eprintln!("SKIP shard_policy {policy:?}: {error}"); + return; + } + }; + let mut fds = [0; 2]; + // SAFETY: the array holds two fds; each successful fd is owned by one File. + assert_eq!(unsafe { libc::pipe(fds.as_mut_ptr()) }, 0); + let pipe_read = Arc::new(unsafe { File::from_raw_fd(fds[0]) }); + let pipe_write = unsafe { File::from_raw_fd(fds[1]) }; + let held = driver.read_current(pipe_read, 8); // shard 0 stays blocked + let path = std::env::temp_dir().join(format!("uring-shard-policy-{policy:?}-{}.bin", std::process::id())); + std::fs::write(&path, [42; 8]).unwrap(); + let file = Arc::new(File::open(&path).unwrap()); + // Advance the cursor over shard 1 without occupying its permit. Waiting for + // stats.in_flight after a real read would race that read's permit destructor. + assert_eq!( + driver.read_at(Arc::clone(&file), u64::MAX, 8).await.unwrap_err().kind(), + std::io::ErrorKind::InvalidInput + ); + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().in_flight != 1 { + tokio::task::yield_now().await; + } + }) + .await + .expect("only the blocked pipe should remain"); + + let mut read = driver.read_at(file, 0, 8); // round-robin starts at busy shard 0 + match policy { + ShardPolicy::CapacityAware => { + assert_eq!( + tokio::time::timeout(Duration::from_secs(2), &mut read) + .await + .expect("must use the idle shard") + .unwrap(), + [42; 8] + ); + // The pipe writer remains open and empty throughout this read. + drop(held); + } + ShardPolicy::RoundRobin => { + assert!(tokio::time::timeout(Duration::from_millis(100), &mut read).await.is_err()); + drop(held); + assert_eq!( + tokio::time::timeout(Duration::from_secs(2), read) + .await + .expect("cancellation should release the original shard") + .unwrap(), + [42; 8] + ); + } + } + tokio::time::timeout(Duration::from_secs(2), async { + while driver.stats().in_flight != 0 { + tokio::task::yield_now().await; + } + }) + .await + .expect("reads should drain after cancellation"); + let snapshot = driver.shutdown(); + assert_eq!(snapshot.submitted, 2); + assert_eq!(snapshot.delivered, 1); + assert_eq!(snapshot.orphan_reclaimed, 1); + drop(pipe_write); + std::fs::remove_file(path).unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn capacity_aware_bypasses_a_shard_saturated_by_a_blocked_pipe() { + exercise_busy_shard(ShardPolicy::CapacityAware).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn round_robin_still_waits_for_its_original_busy_shard() { + exercise_busy_shard(ShardPolicy::RoundRobin).await; +} diff --git a/tests/shared_budget.rs b/tests/shared_budget.rs new file mode 100644 index 0000000..fd96bb5 --- /dev/null +++ b/tests/shared_budget.rs @@ -0,0 +1,384 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Whole-driver read-quota reservations. Kernel tests print SKIP only for +//! expected restrictions; configuration tests do not require io_uring setup. +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io::{self, Write}; +use std::os::fd::FromRawFd; +use std::pin::Pin; +use std::sync::Arc; +use std::task::{Context, Waker}; +use std::time::Duration; + +use rustfs_uring::{ProbeFailure, ReadHandle, ReadLimits, ReadRequest, ShardPolicy, SharedReadBudget, UringDriver}; + +fn limits(bytes: usize) -> ReadLimits { + ReadLimits { + max_read_len: None, + max_in_flight_bytes: Some(bytes), + } +} + +fn assert_setup_kind(result: Result, kind: io::ErrorKind) { + match result { + Err(ProbeFailure::Setup(error)) => assert_eq!(error.kind(), kind), + Err(other) => panic!("wrong failure boundary: {other}"), + Ok(driver) => { + driver.shutdown(); + panic!("invalid or unavailable quota was admitted"); + } + } +} + +#[test] +fn shared_budget_accepts_large_usize_capacity_without_allocating_read_buffers() { + assert!(matches!(SharedReadBudget::new(0), Err(error) if error.kind() == io::ErrorKind::InvalidInput)); + for total in [1, 16, usize::MAX] { + let budget = SharedReadBudget::new(total).expect("positive accounting capacity"); + let clone = budget.clone(); + assert_eq!(budget.capacity(), total); + assert_eq!(clone.capacity(), total); + assert_eq!(budget.available(), total); + drop(budget); + assert_eq!(clone.available(), total); + } +} + +#[test] +fn shared_constructor_rejects_missing_zero_or_oversized_local_quota_before_probe() { + let budget = SharedReadBudget::new(8).expect("test shared budget"); + for local in [ReadLimits::default(), limits(0), limits(9)] { + assert_setup_kind( + UringDriver::probe_and_start_with_shared_budget(1, 1, local, &budget), + io::ErrorKind::InvalidInput, + ); + assert_eq!(budget.available(), 8, "invalid configuration must not retain quota"); + } + let largest = SharedReadBudget::new(usize::MAX).expect("global capacity has no u32 limit"); + assert_setup_kind( + UringDriver::probe_and_start_with_shared_budget(1, 1, limits(tokio::sync::Semaphore::MAX_PERMITS + 1), &largest), + io::ErrorKind::InvalidInput, + ); + assert_eq!(largest.available(), usize::MAX); +} + +fn driver_or_skip(name: &str, entries: u32, shards: usize, bytes: usize, budget: &SharedReadBudget) -> Option { + let available_before = budget.available(); + match UringDriver::probe_and_start_with_shared_budget(entries, shards, limits(bytes), budget) { + Ok(driver) => Some(driver), + Err(error) => { + // Each test owns this pool; no concurrent constructor changes it. + // A restricted-kernel failure must refund the tentative whole quota. + assert_eq!(budget.available(), available_before, "failed probe retained a whole-driver quota"); + assert!(error.is_expected_restriction(), "unexpected shared-budget probe failure: {error}"); + eprintln!("SKIP {name}: restricted environment ({error})"); + None + } + } +} + +fn assert_would_block(result: Result) { + match result { + Err(error) => { + assert!( + !error.is_expected_restriction(), + "quota exhaustion must not be negatively cached as a kernel restriction" + ); + assert_setup_kind(Err(error), io::ErrorKind::WouldBlock); + } + Ok(driver) => { + driver.shutdown(); + panic!("exhausted shared quota must refuse construction"); + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid two-fd output array; success creates two owned descriptors. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0); + // SAFETY: each fresh descriptor transfers into exactly one File owner. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(5), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("pipe read enters driver pending table"); +} + +async fn wait_finished(driver: &UringDriver) { + tokio::time::timeout(Duration::from_secs(10), async { + while !driver.is_finished() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .expect("ordinary test driver threads should finish"); +} + +async fn read_result(handle: ReadHandle) -> io::Result> { + tokio::time::timeout(Duration::from_secs(10), handle) + .await + .expect("test read should resolve") +} + +#[test] +fn shared_quota_is_reserved_once_per_driver_and_clean_shutdown_recycles_it() { + let budget = SharedReadBudget::new(24).expect("three whole-driver quotas"); + let clone = budget.clone(); + let Some(first) = driver_or_skip( + "shared_quota_is_reserved_once_per_driver_and_clean_shutdown_recycles_it", + 1, + 2, + 8, + &budget, + ) else { + return; + }; + assert_eq!(clone.available(), 16, "two shards share one driver quota, not two reservations"); + let second = driver_or_skip("second shared driver", 1, 1, 8, &clone).expect("kernel precondition already proved"); + assert_eq!(budget.available(), 8); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(9), &budget)); + assert_eq!(budget.available(), 8, "refused reservation must not consume partial capacity"); + let third = driver_or_skip("exact remaining shared quota", 1, 1, 8, &budget).expect("remaining quota fits"); + assert_eq!(budget.available(), 0); + assert_eq!(first.shutdown().in_flight, 0); + assert_eq!(budget.available(), 8); + assert_eq!(second.shutdown().in_flight, 0); + assert_eq!(budget.available(), 16); + assert_eq!(third.shutdown().in_flight, 0); + assert_eq!(budget.available(), 24); +} + +#[cfg(target_pointer_width = "64")] +#[test] +fn driver_quota_larger_than_u32_is_not_truncated_to_per_read_permit_width() { + let quota = u32::MAX as usize + 1; + let budget = SharedReadBudget::new(quota * 2).expect("large accounting capacity"); + let Some(driver) = driver_or_skip( + "driver_quota_larger_than_u32_is_not_truncated_to_per_read_permit_width", + 1, + 1, + quota, + &budget, + ) else { + return; + }; + // Startup only: no application read or multi-gigabyte buffer is allocated. + assert_eq!(driver.stats().submitted, 0); + assert_eq!(budget.available(), quota); + driver.shutdown(); + assert_eq!(budget.available(), quota * 2); +} + +#[tokio::test(flavor = "current_thread")] +async fn shutting_down_one_driver_does_not_close_another_drivers_admission() { + let budget = SharedReadBudget::new(16).expect("two driver quotas"); + let Some(first) = driver_or_skip("shutting_down_one_driver_does_not_close_another_drivers_admission", 1, 1, 8, &budget) + else { + return; + }; + let second = driver_or_skip("independent peer driver", 1, 2, 8, &budget) + .expect("second driver must start") + .with_shard_policy(ShardPolicy::CapacityAware); + let (read, mut write) = pipe(); + let active = second.read_current(read, 8); + wait_submitted(&second, 1).await; + first.request_shutdown(); + let rejected = first.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + assert!(read_result(rejected).await.is_err()); + wait_finished(&first).await; + assert_eq!(first.shutdown().in_flight, 0); + assert_eq!(budget.available(), 8); + write.write_all(b"12345678").expect("complete peer pipe read"); + let retained = read_result(active) + .await + .expect("peer read must survive another driver stopping"); + assert_eq!(retained, b"12345678"); + let zero = Arc::new(File::open("/dev/zero").expect("open fixture")); + let batch = second + .read_at_batch(vec![ + ReadRequest { + file: Arc::clone(&zero), + offset: 0, + len: 4, + }, + ReadRequest { + file: zero, + offset: 4, + len: 4, + }, + ]) + .expect("capacity-aware peer accepts new batch reads"); + for next in batch { + assert_eq!(read_result(next).await.expect("peer remains open to new reads"), [0; 4]); + } + let stats = second.shutdown(); + assert_eq!(stats.submitted, 3); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + assert_eq!(budget.available(), 16, "retained caller Vec is outside driver in-flight quota"); + assert_eq!(retained, b"12345678"); +} + +#[tokio::test(flavor = "current_thread")] +async fn unpolled_deferred_count_handle_retains_whole_quota_after_clean_driver_shutdown() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip( + "unpolled_deferred_count_handle_retains_whole_quota_after_clean_driver_shutdown", + 1, + 1, + 8, + &budget, + ) else { + return; + }; + let (read, _write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let waiting = driver.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + // No poll: this handle has not joined the local count queue or submitted I/O. + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 0); + assert_eq!(budget.available(), 0, "deferred handle still owns the retired driver's receipt"); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(8), &budget)); + drop(waiting); + assert_eq!(budget.available(), 8); + let replacement = driver_or_skip("quota recycled after final deferred owner", 1, 1, 8, &budget).expect("quota must recycle"); + replacement.shutdown(); + assert_eq!(budget.available(), 8); +} + +#[tokio::test(flavor = "current_thread")] +async fn canceled_polled_byte_waiter_releases_local_reservation_but_live_driver_keeps_quota() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip( + "canceled_polled_byte_waiter_releases_local_reservation_but_live_driver_keeps_quota", + 2, + 1, + 8, + &budget, + ) else { + return; + }; + let (read, mut write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let zero = Arc::new(File::open("/dev/zero").expect("open fixture")); + let mut waiting = driver.read_at(Arc::clone(&zero), 0, 8); + assert!( + Pin::new(&mut waiting) + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + drop(waiting); + assert_eq!(budget.available(), 0, "a live idle-or-busy driver retains its entire quota"); + write.write_all(b"12345678").expect("complete original read"); + assert_eq!(read_result(active).await.expect("original read completes"), b"12345678"); + assert_eq!( + read_result(driver.read_at(zero, 0, 8)) + .await + .expect("canceled waiter must not strand local permits"), + [0; 8] + ); + let stats = driver.shutdown(); + assert_eq!(stats.submitted, 2, "canceled deferred request never submitted"); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); + assert_eq!(stats.in_flight, 0); + assert_eq!(budget.available(), 8); +} + +#[tokio::test(flavor = "current_thread")] +async fn retired_drivers_polled_byte_waiter_returns_final_receipt_when_dropped() { + let budget = SharedReadBudget::new(8).expect("one driver quota"); + let Some(driver) = driver_or_skip("retired_drivers_polled_byte_waiter_returns_final_receipt_when_dropped", 2, 1, 8, &budget) + else { + return; + }; + let (read, _write) = pipe(); + let active = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let mut waiting = driver.read_at(Arc::new(File::open("/dev/zero").expect("open fixture")), 0, 8); + assert!( + Pin::new(&mut waiting) + .poll(&mut Context::from_waker(Waker::noop())) + .is_pending() + ); + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 0); + assert_eq!(budget.available(), 0, "unconsumed closed waiter retains its receipt"); + assert!(read_result(waiting).await.is_err()); // Consumes and drops the handle. + assert_eq!(budget.available(), 8); +} + +#[cfg(feature = "fault-injection")] +#[test] +fn leaked_read_retains_entire_driver_quota_without_closing_shared_budget() { + const CHILD: &str = "RUSTFS_URING_SHARED_QUOTA_LEAK_CHILD"; + const MARKER: &str = "SHARED_QUOTA_LEAK_RETENTION_OK"; + let name = "leaked_read_retains_entire_driver_quota_without_closing_shared_budget"; + if std::env::var_os(CHILD).is_some() { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .expect("child runtime"); + runtime.block_on(async { + let budget = SharedReadBudget::new(12).expect("shared test capacity"); + let Some(driver) = driver_or_skip(name, 2, 1, 8, &budget) else { return }; + let driver = driver.with_shard_policy(ShardPolicy::CapacityAware); + // The capacity-aware eager path and batch path must carry the same + // receipt as ordinary submit. Leak one byte, retain the whole eight. + let active = driver + .read_at_batch(vec![ReadRequest { + file: Arc::new(File::open("/dev/zero").expect("open leak fixture")), + offset: 0, + len: 1, + }]) + .expect("single-read batch") + .pop() + .expect("batch contains its one read"); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + assert!(read_result(active).await.is_err()); + wait_finished(&driver).await; + assert_eq!(driver.shutdown().in_flight, 1, "injected drain must really leak a pending entry"); + assert_eq!(budget.available(), 4, "leak must not refund any of the driver's whole quota"); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(5), &budget)); + // Startup probe is independent of the drive-loop stuck-CQE seam. + // No application read is submitted to this second driver. + let peer = driver_or_skip("healthy peer uses unreserved capacity", 1, 1, 4, &budget) + .expect("remaining capacity must stay usable"); + assert_eq!(budget.available(), 0); + assert_eq!(peer.shutdown().in_flight, 0); + assert_eq!(budget.available(), 4); + assert_would_block(UringDriver::probe_and_start_with_shared_budget(1, 1, limits(5), &budget)); + eprintln!("{MARKER}"); + }); + return; + } + let budget = SharedReadBudget::new(12).expect("parent precondition capacity"); + let Some(driver) = driver_or_skip(name, 1, 1, 8, &budget) else { return }; + driver.shutdown(); + let output = std::process::Command::new(std::env::current_exe().expect("test executable")) + .args(["--exact", name, "--nocapture", "--test-threads=1"]) + .env(CHILD, "1") + .env("RUSTFS_URING_FAULT_STUCK_DRAIN", "1") + .env("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400") + .output() + .expect("run isolated leaked-quota case"); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(output.status.success(), "child failed: {stderr}"); + assert!(stderr.contains(MARKER), "child did not prove the leaked-quota path: {stderr}"); +} diff --git a/tests/shutdown.rs b/tests/shutdown.rs new file mode 100644 index 0000000..17d638a --- /dev/null +++ b/tests/shutdown.rs @@ -0,0 +1,263 @@ +// Copyright 2024 RustFS Team +// SPDX-License-Identifier: Apache-2.0 + +//! Native shutdown API tests. Expected io_uring restrictions print SKIP; only +//! an unrestricted Linux run proves the driver paths. Timeouts bound test +//! failures for ordinary pipes, not hung-kernel shutdown guarantees. +#![cfg(target_os = "linux")] + +use std::fs::File; +use std::io; +use std::os::fd::FromRawFd; +use std::pin::Pin; +use std::sync::Arc; +use std::time::Duration; + +use rustfs_uring::{ReadHandle, ReadLimits, StatsSnapshot, UringDriver}; + +fn driver_or_skip(name: &str, entries: u32, limits: ReadLimits) -> Option { + match UringDriver::probe_and_start_with_limits(entries, 2, limits) { + Ok(driver) => Some(driver), + Err(error) => { + assert!(error.is_expected_restriction(), "unexpected probe failure: {error}"); + eprintln!("SKIP {name}: restricted environment ({error})"); + None + } + } +} + +fn pipe() -> (Arc, File) { + let mut fds = [0; 2]; + // SAFETY: valid output array; successful pipe2 creates two owned fds. + assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_CLOEXEC) }, 0, "create pending-read pipe"); + // SAFETY: transfer each fresh descriptor into exactly one File owner. + unsafe { (Arc::new(File::from_raw_fd(fds[0])), File::from_raw_fd(fds[1])) } +} + +async fn wait_submitted(driver: &UringDriver, count: u64) { + tokio::time::timeout(Duration::from_secs(5), async { + while driver.stats().submitted != count { + tokio::task::yield_now().await; + } + }) + .await + .expect("pipe reads should reach the driver"); +} + +async fn wait_complete(driver: &UringDriver) { + tokio::time::timeout(Duration::from_secs(10), async { + while !driver.is_finished() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .expect("ordinary test driver threads should exit"); +} + +async fn assert_read_error(handle: ReadHandle) { + let result = tokio::time::timeout(Duration::from_secs(10), handle) + .await + .expect("shutdown must resolve a pending read handle"); + assert!(result.is_err(), "closed admission or canceled pipe must not return data"); +} + +fn assert_clean_conservation(stats: StatsSnapshot, submitted: u64) { + assert_eq!(stats.submitted, submitted); + assert_eq!(stats.in_flight, 0); + assert_eq!(stats.submitted, stats.delivered + stats.orphan_reclaimed); +} + +async fn request_and_drain(name: &str, limits: ReadLimits) { + let Some(driver) = driver_or_skip(name, 2, limits) else { return }; + assert!(!driver.is_finished(), "idle but live driver threads are not finished"); + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + driver.request_shutdown(); // Idempotent even before a driver processes it. + for _ in 0..2 { + assert_read_error(driver.read_at(Arc::new(File::open("/dev/zero").expect("open zero fixture")), 0, 8)).await; + } + assert_read_error(first).await; + wait_complete(&driver).await; + driver.request_shutdown(); // Repeated request after thread exit is harmless. + assert_clean_conservation(driver.shutdown(), 1); +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_drains_default_driver_and_rejects_new_reads() { + request_and_drain("request_shutdown_drains_default_driver_and_rejects_new_reads", ReadLimits::default()).await; +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_drains_byte_limited_driver_and_rejects_new_reads() { + request_and_drain( + "request_shutdown_drains_byte_limited_driver_and_rejects_new_reads", + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) + .await; +} + +fn start_waiter(mut handle: ReadHandle) -> (tokio::task::JoinHandle>>, tokio::sync::oneshot::Receiver<()>) { + let (entered, ready) = tokio::sync::oneshot::channel(); + let mut entered = Some(entered); + let waiter = tokio::spawn(async move { + std::future::poll_fn(|cx| { + let result = Pin::new(&mut handle).poll(cx); + if result.is_pending() + && let Some(entered) = entered.take() + { + let _ = entered.send(()); + } + result + }) + .await + }); + (waiter, ready) +} + +async fn waiting_admission_is_woken(name: &str, limited: bool) { + let limits = if limited { + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + } + } else { + ReadLimits::default() + }; + let Some(driver) = driver_or_skip(name, 1, limits) else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); // Owner shard 0. + wait_submitted(&driver, 1).await; + let zero = Arc::new(File::open("/dev/zero").expect("open zero fixture")); + let waiting = if limited { + // Owner shard 1 has count capacity, but shared byte capacity is held. + driver.read_at(Arc::clone(&zero), 0, 8) + } else { + // Advance the cursor over shard 1 with an invalid inert read; the real + // waiter binds to saturated shard 0 without introducing another op. + assert_read_error(driver.read_at(Arc::clone(&zero), u64::MAX, 8)).await; + driver.read_at(Arc::clone(&zero), 0, 8) + }; + let (waiter, ready) = start_waiter(waiting); + tokio::time::timeout(Duration::from_secs(5), ready) + .await + .expect("waiter should be polled") + .expect("waiter must actually register Pending before shutdown"); + driver.request_shutdown(); + let result = tokio::time::timeout(Duration::from_secs(5), waiter) + .await + .expect("shutdown must wake the registered admission task") + .expect("admission waiter task must not panic"); + assert!(result.is_err()); + assert_read_error(first).await; + wait_complete(&driver).await; + assert_clean_conservation(driver.shutdown(), 1); +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_wakes_registered_count_waiter() { + waiting_admission_is_woken("request_shutdown_wakes_registered_count_waiter", false).await; +} + +#[tokio::test(flavor = "current_thread")] +async fn request_shutdown_wakes_registered_cross_shard_byte_waiter() { + waiting_admission_is_woken("request_shutdown_wakes_registered_cross_shard_byte_waiter", true).await; +} + +#[cfg(feature = "tokio-runtime")] +#[tokio::test(flavor = "current_thread")] +async fn shutdown_async_is_send_and_returns_drained_stats_on_current_thread() { + fn require_send(value: T) -> T { + value + } + let Some(driver) = driver_or_skip( + "shutdown_async_is_send_and_returns_drained_stats_on_current_thread", + 2, + ReadLimits { + max_read_len: Some(8), + max_in_flight_bytes: Some(8), + }, + ) else { + return; + }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let stats = tokio::time::timeout(Duration::from_secs(10), require_send(driver.shutdown_async())) + .await + .expect("ordinary async shutdown should finish") + .expect("blocking join task should succeed"); + assert_read_error(first).await; + assert_clean_conservation(stats, 1); +} + +#[cfg(feature = "tokio-runtime")] +#[tokio::test(flavor = "current_thread")] +async fn dropping_unpolled_shutdown_async_future_does_not_abandon_cleanup() { + let Some(driver) = driver_or_skip( + "dropping_unpolled_shutdown_async_future_does_not_abandon_cleanup", + 2, + ReadLimits::default(), + ) else { + return; + }; + let (read, _write) = pipe(); + let weak = Arc::downgrade(&read); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + let shutdown = driver.shutdown_async(); + drop(shutdown); // Never poll the returned future. + assert_read_error(first).await; + tokio::time::timeout(Duration::from_secs(5), async { + while weak.upgrade().is_some() { + tokio::task::yield_now().await; + } + }) + .await + .expect("detached shutdown still reclaims completed pending resources"); +} + +#[cfg(feature = "fault-injection")] +#[test] +fn is_finished_reports_thread_exit_not_clean_drain() { + const CHILD: &str = "RUSTFS_URING_SHUTDOWN_NONCLEAN_CHILD"; + const MARKER: &str = "SHUTDOWN_THREAD_EXIT_NONCLEAN_OK"; + let name = "is_finished_reports_thread_exit_not_clean_drain"; + if std::env::var_os(CHILD).is_some() { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .expect("build child runtime"); + runtime.block_on(async { + let Some(driver) = driver_or_skip(name, 2, ReadLimits::default()) else { return }; + let (read, _write) = pipe(); + let first = driver.read_current(read, 8); + wait_submitted(&driver, 1).await; + driver.request_shutdown(); + assert_read_error(first).await; + wait_complete(&driver).await; + assert_eq!(driver.stats().in_flight, 1, "fault seam must leave a leaked pending entry"); + assert_eq!(driver.shutdown().in_flight, 1); + eprintln!("{MARKER}"); + }); + return; + } + let Some(driver) = driver_or_skip(name, 2, ReadLimits::default()) else { return }; + driver.shutdown(); + // Only the child inherits injected drain settings; no process-wide env + // mutation can leak into this test binary's concurrently running tests. + let output = std::process::Command::new(std::env::current_exe().expect("locate test executable")) + .args(["--exact", name, "--nocapture", "--test-threads=1"]) + .env(CHILD, "1") + .env("RUSTFS_URING_FAULT_STUCK_DRAIN", "1") + .env("RUSTFS_URING_FAULT_DRAIN_TIMEOUT_MS", "400") + .output() + .expect("run isolated nonclean-shutdown test"); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(output.status.success(), "child failed: {stderr}"); + assert!(stderr.contains(MARKER), "child did not execute the injected nonclean path: {stderr}"); +}