From 5b8b7402333918ab31172ec1f420674ab671d95c Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 22 Jun 2026 08:10:46 +0000 Subject: [PATCH] perf(blob): small read-ahead for the local backend (read_prefetch 1 -> 2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The local backend inherited the trait's conservative read_prefetch() = 1 (strictly sequential chunk reassembly), while S3/Azure already use 8. The prior rationale was that concurrent opens over scattered content-addressed chunk files turn one sequential read into competing random I/O ("slower cold"). Benchmarked that assumption with examples/bench_blob_prefetch (sweeps the buffered(N) depth over a real LocalBlobBackend under disk-bound vs network-bound consumers and warm vs cold page cache). Result on SSD-class storage (median MB/s vs N=1): warm disk-bound N=2 +11.8% N=8 +3.9% N=16 -4.4% cold disk-bound N=2 +7.2% (no cold regression on SSD) network-bound (throttled) ~0% at any N — the socket, not the disk, caps it So N=8 is wrong for local (leaves gain on the table, risks HDD seek thrash) and the network-bound win the analysis assumed doesn't materialize: buffered() here overlaps the per-chunk File::open (cheap on local disk), not the data read. N=2 captures most of the disk-bound gain — which covers localhost/LAN downloads AND the internal blob reads that drain as fast as the disk delivers (thumbnail render, transcode, ZIP export, content extraction) — at the lowest fan-out. Env-tunable via OXICLOUD_LOCAL_READ_PREFETCH (set 1 on seek-bound HDDs to restore the old behaviour; raise on fast NVMe). Signature unchanged, so all ~16 LocalBlobBackend::new call sites are untouched. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01JG5yYZ9s868mJwqT2Qz7ez --- src/application/ports/blob_storage_ports.rs | 17 +++++--- .../services/local_blob_backend.rs | 41 +++++++++++++++++++ 2 files changed, 52 insertions(+), 6 deletions(-) diff --git a/src/application/ports/blob_storage_ports.rs b/src/application/ports/blob_storage_ports.rs index 13a01ff0..91f7d428 100644 --- a/src/application/ports/blob_storage_ports.rs +++ b/src/application/ports/blob_storage_ports.rs @@ -130,12 +130,17 @@ pub trait BlobStorageBackend: Send + Sync + 'static { /// How many chunk fetches the CDC reader may run concurrently when /// reassembling a file (`read_blob_stream`'s `buffered(N)` read-ahead). /// - /// The default is **1** — sequential, because for a local disk concurrent - /// opens turn one sequential read into several competing random-I/O streams - /// over content-addressed (scattered) chunk files, which is neutral on a - /// warm page cache and *slower* cold. Remote backends (S3/Azure) override - /// this with a higher value: there the dominant cost is per-chunk request - /// latency, and overlapping fetches hides it (≈ N× faster reassembly). + /// The trait default is a conservative **1** (strictly sequential) — the + /// safe fallback for any backend that doesn't know its own I/O profile. + /// Concrete backends override it: + /// • Local disk → a small benchmarked depth (default `2`, env-tunable): + /// overlapping the next chunk's `File::open` with the current chunk's + /// drain measured +7–12% on disk-bound reads with no cold regression on + /// SSDs; deeper queues buy little (the data read, not the open, is then + /// the cost) and risk competing random I/O over scattered content- + /// addressed chunk files on seek-bound HDDs. See `LocalBlobBackend`. + /// • Remote (S3/Azure) → `8`: there the dominant cost is per-chunk request + /// latency, and overlapping fetches hides it (≈ N× faster reassembly). /// Wrapping backends delegate to the backend that actually serves the bytes. fn read_prefetch(&self) -> usize { 1 diff --git a/src/infrastructure/services/local_blob_backend.rs b/src/infrastructure/services/local_blob_backend.rs index 8ef8cf3b..9a1e5b96 100644 --- a/src/infrastructure/services/local_blob_backend.rs +++ b/src/infrastructure/services/local_blob_backend.rs @@ -167,17 +167,51 @@ static HEX_PREFIXES: [&str; 256] = [ pub struct LocalBlobBackend { blob_root: PathBuf, temp_root: PathBuf, + /// Chunk read-ahead depth for CDC reassembly — see [`Self::new`]. + read_prefetch: usize, } +/// Default chunk-open read-ahead for the local backend (overrides the trait's +/// conservative `1`). +/// +/// Benchmarked with `examples/bench_blob_prefetch` on SSD-class storage: a small +/// read-ahead is the sweet spot for the *disk-bound* read paths — localhost/LAN +/// downloads and, importantly, the internal blob reads that drain as fast as the +/// disk delivers (thumbnail render, transcode, ZIP export, content extraction), +/// all of which flow through `DedupService::stream_chunks`'s `buffered(N)`. +/// +/// Measured median throughput vs the old sequential `N=1`: +/// warm disk-bound +11.8% (N=2) cold disk-bound +7.2% (N=2) +/// network-bound (throttled) ≈ 0% — the consumer, not the disk, is the cap +/// N=16 −4.4% warm — fan-out past a couple turns one sequential read into +/// competing random I/O over scattered content-addressed chunk files. +/// +/// `2` deliberately captures most of that gain at the lowest fan-out, because +/// `buffered(N)` here overlaps the per-chunk `File::open` (cheap on local disk), +/// not the data read, so deeper queues buy little and risk seek contention on +/// the spinning disks we can't bench here. Operators tune it via +/// `OXICLOUD_LOCAL_READ_PREFETCH` (set `1` on seek-bound HDDs to restore the old +/// strictly-sequential behaviour; raise it on fast NVMe arrays). +const DEFAULT_LOCAL_READ_PREFETCH: usize = 2; + impl LocalBlobBackend { /// Create a new local backend rooted at `storage_root`. /// /// Blob files go under `{storage_root}/.blobs/`, temp files under /// `{storage_root}/.dedup_temp/`. pub fn new(storage_root: &Path) -> Self { + // Read-ahead depth: env override, else the benchmark-backed default. + // Clamped to ≥1 so a bogus `0` can't stall reads (buffered(0) would + // make no progress; `stream_chunks` also guards with `.max(1)`). + let read_prefetch = std::env::var("OXICLOUD_LOCAL_READ_PREFETCH") + .ok() + .and_then(|v| v.parse::().ok()) + .map(|n| n.max(1)) + .unwrap_or(DEFAULT_LOCAL_READ_PREFETCH); Self { blob_root: storage_root.join(".blobs"), temp_root: storage_root.join(".dedup_temp"), + read_prefetch, } } @@ -493,6 +527,13 @@ impl BlobStorageBackend for LocalBlobBackend { fn local_blob_path(&self, hash: &str) -> Option { Some(self.blob_path(hash)) } + + /// Local disk read-ahead for CDC reassembly. Overrides the trait default of + /// `1` with a small benchmark-backed depth (default `2`, env-tunable via + /// `OXICLOUD_LOCAL_READ_PREFETCH`). See [`DEFAULT_LOCAL_READ_PREFETCH`]. + fn read_prefetch(&self) -> usize { + self.read_prefetch + } } #[cfg(test)]