perf: cache-stampede coalescing + DB safeguards; ui/i18n fixes

Backend — tail latency & throughput:
- FileContentCache, image transcode, and search now use moka single-flight
  (try_get_with / get_or_load) so N concurrent misses for the same key
  collapse to one disk read / transcode / query instead of a thundering herd.
  Microbenchmark (128 concurrent on one hot key): 128 loads / p99 ~1023ms
  before vs 1 load / p99 ~32ms after.
- DB: configurable per-statement timeout on the primary pool
  (OXICLOUD_DB_STATEMENT_TIMEOUT_SECS, default 30; maintenance pool exempt) so
  a runaway query can't pin a connection and starve the pool.
- DB: background pool-saturation monitor
  (OXICLOUD_DB_POOL_MONITOR_INTERVAL_SECS) that WARNs as the primary pool nears
  exhaustion — the early signal before tail latency cliffs.
- mimalloc: set MIMALLOC_PURGE_DELAY=0 (Dockerfile + compose) so freed pages
  return to the OS and RSS tracks the live working set; benchmarked on
  musl/aarch64 at ~400MB reclaimed vs 0MB with the default.

Frontend — UI / i18n fixes:
- i18n: fix literal "{{count}}" and "{{percentage}}/{{used}}/{{total}}" in the
  selection toolbar and storage line — the call sites passed param names that
  didn't match the locale placeholders; unify on `count` and pass the storage
  template its params. Add es files.selected_count.
- sidebar: hide the drive picker when there's only one drive (the redundant
  "Personal" row); remove the coloured left accent on the active nav item.
- logo: stop clipping the cloud's left bulge — viewBox recentred on the cloud's
  true bbox with proportional SVG size so it keeps the same rendered scale.
- user menu: drop the default <a> underline on the link rows.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
DioCrafts
2026-06-20 14:42:10 +02:00
parent ca18858630
commit b14c4dc911
19 changed files with 848 additions and 310 deletions
@@ -0,0 +1,188 @@
//! Background watchdog that samples primary DB-pool saturation.
//!
//! A runaway query that pins a connection, multiplied across a small pool, ends
//! in pool exhaustion: every new request then blocks on `acquire()` up to the
//! acquire timeout — the correlated tail-latency cliff where one slow query
//! degrades the whole server. `statement_timeout` (see `db.rs`) caps the cause;
//! this monitor surfaces the symptom early by logging a WARN when in-use
//! connections approach the configured maximum, so an operator can raise
//! `OXICLOUD_DB_MAX_CONNECTIONS` or hunt the slow query before users feel it.
//!
//! The loop only reads in-memory pool counters (`size()` / `num_idle()`) — it
//! never issues a query, so it can never itself contend for a connection.
use sqlx::PgPool;
use std::sync::Arc;
use std::sync::atomic::{AtomicU32, Ordering};
use std::time::Duration;
use tracing::{debug, info, warn};
/// WARN once in-use connections reach this fraction of the pool maximum. At
/// ≥90% the pool is one slow query away from forcing `acquire()` waits on
/// every request.
const WARN_UTILIZATION_PCT: u32 = 90;
/// A point-in-time sample of pool occupancy.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct PoolSample {
/// Connections currently checked out (in use).
pub active: u32,
/// Connections sitting idle in the pool.
pub idle: u32,
/// Configured maximum connections.
pub max: u32,
}
impl PoolSample {
/// In-use connections as a percentage of the configured maximum.
/// Saturates rather than dividing by zero on an unconfigured pool.
pub fn utilization_pct(&self) -> u32 {
if self.max == 0 {
return 0;
}
((self.active as u64 * 100) / self.max as u64) as u32
}
/// True once occupancy has reached the warn threshold.
pub fn is_saturated(&self, warn_pct: u32) -> bool {
self.utilization_pct() >= warn_pct
}
}
/// Read a sqlx pool's occupancy. `size()` is total live connections
/// (idle + in-use); `num_idle()` is the idle subset.
fn sample(pool: &PgPool, max: u32) -> PoolSample {
let size = pool.size();
let idle = pool.num_idle() as u32;
PoolSample {
active: size.saturating_sub(idle),
idle,
max,
}
}
/// Background saturation watchdog over the primary (user-facing) pool.
pub struct DbPoolMonitor {
pool: PgPool,
label: &'static str,
max_connections: u32,
interval: Duration,
/// High-water mark of in-use connections since startup (diagnostics).
peak_active: Arc<AtomicU32>,
}
impl DbPoolMonitor {
pub fn new(
pool: PgPool,
label: &'static str,
max_connections: u32,
interval_secs: u64,
) -> Self {
Self {
pool,
label,
max_connections,
// Floor the cadence so a misconfiguration can't busy-loop.
interval: Duration::from_secs(interval_secs.max(1)),
peak_active: Arc::new(AtomicU32::new(0)),
}
}
/// Spawn the sampling loop. Fire-and-forget.
pub fn start(self) {
info!(
"Starting DB pool saturation monitor ({} pool, every {}s, warn ≥{}%)",
self.label,
self.interval.as_secs(),
WARN_UTILIZATION_PCT,
);
tokio::spawn(async move {
let mut ticker = tokio::time::interval(self.interval);
ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
loop {
ticker.tick().await;
let s = sample(&self.pool, self.max_connections);
let peak = self
.peak_active
.fetch_max(s.active, Ordering::Relaxed)
.max(s.active);
if s.is_saturated(WARN_UTILIZATION_PCT) {
warn!(
target: "oxicloud::db",
pool = self.label,
active = s.active,
idle = s.idle,
max = s.max,
utilization_pct = s.utilization_pct(),
peak_active = peak,
"⚠️ DB pool near saturation: {}/{} in use ({}%, peak {}) — requests may \
be queueing on acquire(); raise OXICLOUD_DB_MAX_CONNECTIONS or \
investigate slow queries",
s.active,
s.max,
s.utilization_pct(),
peak,
);
} else {
debug!(
target: "oxicloud::db",
pool = self.label,
active = s.active,
idle = s.idle,
max = s.max,
"DB pool ok: {}/{} in use ({}%)",
s.active,
s.max,
s.utilization_pct(),
);
}
}
});
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn utilization_and_saturation_thresholds() {
// 18/20 in use = 90% → saturated at the 90% threshold, not at 95%.
let near = PoolSample {
active: 18,
idle: 2,
max: 20,
};
assert_eq!(near.utilization_pct(), 90);
assert!(near.is_saturated(WARN_UTILIZATION_PCT));
assert!(!near.is_saturated(95));
// 4/20 in use = 20% → calm.
let calm = PoolSample {
active: 4,
idle: 16,
max: 20,
};
assert_eq!(calm.utilization_pct(), 20);
assert!(!calm.is_saturated(WARN_UTILIZATION_PCT));
// Fully checked out = 100% → saturated.
let full = PoolSample {
active: 20,
idle: 0,
max: 20,
};
assert_eq!(full.utilization_pct(), 100);
assert!(full.is_saturated(WARN_UTILIZATION_PCT));
// Degenerate zero-max pool: no div-by-zero, never saturated.
let zero = PoolSample {
active: 0,
idle: 0,
max: 0,
};
assert_eq!(zero.utilization_pct(), 0);
assert!(!zero.is_saturated(WARN_UTILIZATION_PCT));
}
}
@@ -1,5 +1,7 @@
use crate::common::errors::DomainError;
use bytes::Bytes;
use moka::future::Cache;
use std::future::Future;
use std::sync::Arc;
use std::sync::atomic::{AtomicUsize, Ordering};
use std::time::Duration;
@@ -152,6 +154,57 @@ impl FileContentCache {
debug!("Cached file {} ({} bytes)", file_id, size);
}
/// Get from cache, or load-and-cache with **single-flight coalescing**.
///
/// On a miss, concurrent callers for the same `cache_key` share ONE `load`
/// future (moka `try_get_with`) instead of every caller hitting disk — the
/// classic thundering-herd / cache-stampede fix. With `N` simultaneous
/// requests for the same uncached blob this turns `N` disk reads into `1`
/// read plus `N-1` cheap waits, collapsing tail latency under load.
///
/// Safe because the cache is content-addressed (key = immutable blob hash):
/// the coalesced value is identical for every caller and never goes stale,
/// so there is nothing to invalidate.
///
/// `etag` / `content_type` describe the loaded content and are only used
/// when this call is the one that populates the entry.
pub async fn get_or_load<F>(
&self,
cache_key: String,
etag: Arc<str>,
content_type: Arc<str>,
load: F,
) -> Result<(Bytes, Arc<str>, Arc<str>), DomainError>
where
F: Future<Output = Result<Bytes, DomainError>>,
{
// Fast path: lock-free hit (also keeps hit/miss stats meaningful).
if let Some(hit) = self.get(&cache_key).await {
return Ok(hit);
}
// Slow path: coalesce concurrent misses into a single `load`.
let entry = self
.cache
.try_get_with(cache_key, async move {
let content = load.await?;
Ok::<CacheEntry, DomainError>(CacheEntry {
content,
etag,
content_type,
})
})
.await
// try_get_with hands back `Arc<DomainError>` shared by all waiters;
// DomainError isn't Clone (it carries a boxed source), so rebuild a
// fresh one preserving the kind / entity / message.
.map_err(|shared: Arc<DomainError>| {
DomainError::new(shared.kind, shared.entity_type, shared.message.clone())
})?;
Ok((entry.content, entry.etag, entry.content_type))
}
/// Remove a file from cache (e.g., when file is deleted or modified)
pub async fn invalidate(&self, file_id: &str) {
self.cache.remove(file_id).await;
@@ -302,4 +355,175 @@ mod tests {
assert!(cache.get("file1").await.is_none());
}
/// Correctness of the stampede fix: N concurrent misses for the same key
/// must coalesce into exactly ONE load (moka single-flight).
#[tokio::test]
async fn get_or_load_coalesces_concurrent_misses() {
use std::sync::atomic::AtomicUsize;
let cache = Arc::new(FileContentCache::new(FileContentCacheConfig::default()));
let loads = Arc::new(AtomicUsize::new(0));
let mut handles = Vec::new();
for _ in 0..64 {
let cache = Arc::clone(&cache);
let loads = Arc::clone(&loads);
handles.push(tokio::spawn(async move {
cache
.get_or_load(
"blob-hash".to_string(),
"\"blob-hash\"".into(),
"image/png".into(),
async move {
loads.fetch_add(1, Ordering::SeqCst);
// Slow load so all 64 tasks pile onto the same miss.
tokio::time::sleep(Duration::from_millis(20)).await;
Ok(Bytes::from_static(b"the-blob-bytes"))
},
)
.await
}));
}
for h in handles {
let (bytes, _etag, _ct) = h.await.unwrap().unwrap();
assert_eq!(&bytes[..], b"the-blob-bytes");
}
assert_eq!(
loads.load(Ordering::SeqCst),
1,
"64 concurrent misses must trigger exactly ONE load (single-flight)"
);
}
/// Before/after benchmark for the cache-stampede fix.
///
/// Run with:
/// cargo test --release -p oxicloud bench_stampede -- --ignored --nocapture
///
/// Models a viral hot blob: `K` clients request the same uncached key at
/// once, and each load contends on a bounded resource (the rayon transcode
/// pool / DB pool) with `POOL` permits. Reports work amplification and tail
/// latency for the NAIVE get()+put() pattern vs the COALESCED get_or_load().
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[ignore = "benchmark — run with --ignored --nocapture"]
async fn bench_stampede() {
use std::sync::atomic::AtomicUsize;
use std::time::Instant;
use tokio::sync::Semaphore;
const K: usize = 128; // concurrent clients, all requesting the SAME hot key
const LOAD_MS: u64 = 30; // cost of one expensive load (disk + decode/encode)
const POOL: usize = 4; // bounded resource the loads contend on
// One expensive load: take a permit from the bounded pool, then work.
async fn expensive_load(
sem: Arc<Semaphore>,
loads: Arc<AtomicUsize>,
load_ms: u64,
) -> Bytes {
let _permit = sem.acquire().await.unwrap();
loads.fetch_add(1, Ordering::SeqCst);
tokio::time::sleep(Duration::from_millis(load_ms)).await;
Bytes::from_static(b"blob")
}
fn pct(sorted: &[u128], p: f64) -> u128 {
if sorted.is_empty() {
return 0;
}
let idx = (((sorted.len() - 1) as f64) * p).round() as usize;
sorted[idx]
}
// ── Scenario A: NAIVE get() + put() (today's pattern) ──
let (naive_ms, naive_lats, naive_loads) = {
let cache = Arc::new(FileContentCache::new(FileContentCacheConfig::default()));
let sem = Arc::new(Semaphore::new(POOL));
let loads = Arc::new(AtomicUsize::new(0));
let t0 = Instant::now();
let mut handles = Vec::new();
for _ in 0..K {
let cache = Arc::clone(&cache);
let sem = Arc::clone(&sem);
let loads = Arc::clone(&loads);
handles.push(tokio::spawn(async move {
let r0 = Instant::now();
if cache.get("hot").await.is_some() {
return r0.elapsed().as_millis();
}
let bytes = expensive_load(sem, loads, LOAD_MS).await;
cache
.put("hot".to_string(), bytes, "e".into(), "t".into())
.await;
r0.elapsed().as_millis()
}));
}
let mut lats = Vec::new();
for h in handles {
lats.push(h.await.unwrap());
}
lats.sort_unstable();
(t0.elapsed().as_millis(), lats, loads.load(Ordering::SeqCst))
};
// ── Scenario B: COALESCED get_or_load() (the fix) ──
let (coal_ms, coal_lats, coal_loads) = {
let cache = Arc::new(FileContentCache::new(FileContentCacheConfig::default()));
let sem = Arc::new(Semaphore::new(POOL));
let loads = Arc::new(AtomicUsize::new(0));
let t0 = Instant::now();
let mut handles = Vec::new();
for _ in 0..K {
let cache = Arc::clone(&cache);
let sem = Arc::clone(&sem);
let loads = Arc::clone(&loads);
handles.push(tokio::spawn(async move {
let r0 = Instant::now();
cache
.get_or_load("hot".to_string(), "e".into(), "t".into(), async move {
Ok(expensive_load(sem, loads, LOAD_MS).await)
})
.await
.unwrap();
r0.elapsed().as_millis()
}));
}
let mut lats = Vec::new();
for h in handles {
lats.push(h.await.unwrap());
}
lats.sort_unstable();
(t0.elapsed().as_millis(), lats, loads.load(Ordering::SeqCst))
};
println!(
"\n╔══ Cache stampede: K={K} clients on the same hot key, pool={POOL}, load={LOAD_MS}ms ══"
);
println!("║ pattern │ loads │ p50(ms) │ p99(ms) │ max(ms) │ wall(ms)");
println!(
"║ NAIVE get()+put() │ {naive_loads:>5} │ {:>7} │ {:>7} │ {:>7} │ {naive_ms:>7}",
pct(&naive_lats, 0.50),
pct(&naive_lats, 0.99),
naive_lats.last().copied().unwrap_or(0)
);
println!(
"║ COALESCED get_or_load │ {coal_loads:>5} │ {:>7} │ {:>7} │ {:>7} │ {coal_ms:>7}",
pct(&coal_lats, 0.50),
pct(&coal_lats, 0.99),
coal_lats.last().copied().unwrap_or(0)
);
let amp = naive_loads as f64 / coal_loads.max(1) as f64;
let p99x = pct(&naive_lats, 0.99) as f64 / pct(&coal_lats, 0.99).max(1) as f64;
println!("╚══ {amp:.0}× fewer loads · {p99x:.0}× lower p99 tail latency\n");
// Guard rails so the benchmark also asserts the win.
assert_eq!(coal_loads, 1, "coalesced path must load exactly once");
assert!(
naive_loads > coal_loads * 10,
"naive path should stampede the loader"
);
}
}
@@ -222,7 +222,7 @@ impl ImageTranscodeService {
) -> Result<(Bytes, String, bool), String> {
let cache_key = format!("{}:{}", file_id, target_format.extension());
// ── 1. Check moka memory cache (lock-free read) ──
// ── 1. Fast path: moka memory cache (lock-free read) ──
// An empty-Bytes entry is the negative sentinel: "transcoding this
// file is not beneficial — serve the original". Without it, every
// GET of such an image repeated the full decode + encode just to
@@ -237,18 +237,52 @@ impl ImageTranscodeService {
return Ok((cached, target_format.mime_type().to_string(), true));
}
// ── 2. Check disk cache (async fs) ──
// ── 2. Slow path: single-flight coalescing ──
// A viral image requested as WebP by N clients at once would otherwise
// run N identical disk reads + CPU transcodes, saturating the rayon
// pool and inflating tail latency. `try_get_with` collapses every
// concurrent miss for this key into ONE `compute_transcode`; the other
// callers await its result. The cached value (transcoded bytes, or the
// empty negative sentinel) is what gets stored.
let original_for_loader = original_content.clone(); // O(1) ref-count bump
let cached = self
.memory_cache
.try_get_with(cache_key, async {
self.compute_transcode(file_id, original_for_loader, original_mime, target_format)
.await
})
.await
// try_get_with shares one `Arc<String>` across waiters; DomainError
// here is just a String, so hand callers an owned clone.
.map_err(|shared: Arc<String>| (*shared).clone())?;
if cached.is_empty() {
Ok((original_content, original_mime.to_string(), false))
} else {
Ok((cached, target_format.mime_type().to_string(), true))
}
}
/// Compute the value to cache for `(file_id, target_format)`: either the
/// transcoded WebP bytes, or an **empty `Bytes` negative sentinel** meaning
/// "the result wasn't smaller — serve the original". Runs the disk-cache
/// lookups and the CPU transcode, and is invoked at most once per key,
/// guarded by [`Self::get_transcoded`]'s `try_get_with` single-flight.
async fn compute_transcode(
&self,
file_id: &str,
original_content: Bytes,
original_mime: &str,
target_format: OutputFormat,
) -> Result<Bytes, String> {
// ── Disk cache (async fs) ──
let cache_path = self.get_cache_path(file_id, target_format);
if tokio::fs::try_exists(&cache_path).await.unwrap_or(false) {
match fs::read(&cache_path).await {
Ok(data) => {
let content = Bytes::from(data);
self.memory_cache
.insert(cache_key.clone(), content.clone())
.await;
self.stats.disk_hits.fetch_add(1, Ordering::Relaxed);
tracing::debug!("💾 Transcode disk cache HIT: {}", file_id);
return Ok((content, target_format.mime_type().to_string(), true));
return Ok(Bytes::from(data));
}
Err(e) => {
tracing::warn!("Failed to read cached transcode: {}", e);
@@ -256,16 +290,15 @@ impl ImageTranscodeService {
}
}
// ── 2b. Negative verdict persisted on disk (survives restarts) ──
// ── Negative verdict persisted on disk (survives restarts) ──
let skip_marker = self.get_skip_marker_path(file_id, target_format);
if tokio::fs::try_exists(&skip_marker).await.unwrap_or(false) {
self.memory_cache.insert(cache_key, Bytes::new()).await;
self.stats.disk_hits.fetch_add(1, Ordering::Relaxed);
tracing::debug!("💾 Transcode negative disk marker HIT: {}", file_id);
return Ok((original_content, original_mime.to_string(), false));
return Ok(Bytes::new());
}
// ── 3. Transcode on dedicated rayon pool (never blocks Tokio) ──
// ── Transcode on dedicated rayon pool (never blocks Tokio) ──
let content_for_rayon = original_content.clone(); // O(1) ref-count bump
let mime_owned = original_mime.to_string();
@@ -282,7 +315,7 @@ impl ImageTranscodeService {
let transcoded_bytes = Bytes::from(transcoded);
// ── 4. Evaluate savings ──
// ── Evaluate savings ──
let original_size = original_content.len();
let transcoded_size = transcoded_bytes.len();
@@ -293,11 +326,10 @@ impl ImageTranscodeService {
original_size,
transcoded_size
);
// Remember the negative verdict so the next GET doesn't repeat
// the decode + encode: empty-Bytes sentinel in memory (expires
// with the cache TTL) + zero-byte marker on disk (survives
// restarts; removed by `invalidate` when the file changes).
self.memory_cache.insert(cache_key, Bytes::new()).await;
// Remember the negative verdict so the next GET doesn't repeat the
// decode + encode: the caller caches the empty-Bytes sentinel (TTL)
// and we drop a zero-byte marker on disk (survives restarts;
// removed by `invalidate` when the file changes).
let marker = self.get_skip_marker_path(file_id, target_format);
tokio::spawn(async move {
if let Some(parent) = marker.parent() {
@@ -307,12 +339,12 @@ impl ImageTranscodeService {
tracing::warn!("Failed to persist transcode skip marker: {}", e);
}
});
return Ok((original_content, original_mime.to_string(), false));
return Ok(Bytes::new());
}
let saved = original_size - transcoded_size;
// ── 5. Persist to disk cache (fire-and-forget) ──
// ── Persist to disk cache (fire-and-forget) ──
let cache_path_clone = cache_path.clone();
let transcoded_for_disk = transcoded_bytes.clone();
tokio::spawn(async move {
@@ -324,12 +356,7 @@ impl ImageTranscodeService {
}
});
// ── 6. Store in moka memory cache (lock-free) ──
self.memory_cache
.insert(cache_key, transcoded_bytes.clone())
.await;
// ── 7. Update stats (lock-free atomics) ──
// ── Update stats (lock-free atomics) ──
self.stats.transcodes.fetch_add(1, Ordering::Relaxed);
self.stats
.bytes_saved
@@ -343,11 +370,7 @@ impl ImageTranscodeService {
(1.0 - transcoded_size as f64 / original_size as f64) * 100.0
);
Ok((
transcoded_bytes,
target_format.mime_type().to_string(),
true,
))
Ok(transcoded_bytes)
}
/// Get path for cached transcoded file
+1
View File
@@ -3,6 +3,7 @@ pub mod azure_blob_backend;
pub mod cached_blob_backend;
pub mod chunked_upload_service;
pub mod compression_service;
pub mod db_pool_monitor;
pub mod dedup_service;
pub mod encrypted_blob_backend;
pub mod exif_service;