2026-06-22 08:46:24 +00:00
|
|
|
//! CPU pool concurrency benchmark — thumbnail decode under a CPU quota.
|
|
|
|
|
//!
|
|
|
|
|
//! Measures what `effective_parallelism()` changes for the image pools
|
|
|
|
|
//! (`ThumbnailService::max_concurrent_decodes`, `image_transcode_service`,
|
|
|
|
|
//! `di.rs` video): the number of concurrent CPU-heavy renders permitted. Those
|
|
|
|
|
//! pools used to size from `available_parallelism()`, which ignores the CFS
|
|
|
|
|
//! quota (`--cpus` / cgroup `cpu.max`), so under a container quota they permit
|
|
|
|
|
//! one render per *host* core onto cores the scheduler can't actually give.
|
|
|
|
|
//!
|
|
|
|
|
//! It drives the **real service path** — a `Semaphore(K)` gating
|
|
|
|
|
//! `spawn_blocking(ThumbnailService::bench_render_all)` — with a gallery of
|
|
|
|
|
//! concurrent requests, and sweeps the permit count K. Run pinned to the quota's
|
|
|
|
|
//! cores to reproduce the pathology:
|
|
|
|
|
//! taskset -c 0,1 cargo run --release --features bench --example bench_pool_concurrency
|
|
|
|
|
//! Under `taskset -c 0,1`, `effective_parallelism()` = 2 (the "after"); the
|
|
|
|
|
//! higher K rows are what bare `available_parallelism()` would permit on a
|
|
|
|
|
//! many-core host under a 2-core quota (the "before").
|
|
|
|
|
//!
|
|
|
|
|
//! No Postgres needed. Tunables (env):
|
|
|
|
|
//! BENCH_K_LIST (1,2,4,8,16) BENCH_GALLERY (48) BENCH_SECONDS (4)
|
|
|
|
|
|
|
|
|
|
use std::sync::Arc;
|
2026-06-22 09:00:19 +00:00
|
|
|
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
2026-06-22 08:46:24 +00:00
|
|
|
use std::time::{Duration, Instant};
|
|
|
|
|
|
|
|
|
|
use oxicloud::bench_support;
|
|
|
|
|
use oxicloud::common::runtime::effective_parallelism;
|
|
|
|
|
use oxicloud::infrastructure::services::thumbnail_service::ThumbnailService;
|
|
|
|
|
use tokio::sync::Semaphore;
|
|
|
|
|
|
|
|
|
|
fn env_or<T: std::str::FromStr>(key: &str, default: T) -> T {
|
2026-06-22 10:37:57 -06:00
|
|
|
std::env::var(key)
|
|
|
|
|
.ok()
|
|
|
|
|
.and_then(|v| v.parse().ok())
|
|
|
|
|
.unwrap_or(default)
|
2026-06-22 08:46:24 +00:00
|
|
|
}
|
|
|
|
|
|
2026-06-22 09:00:19 +00:00
|
|
|
#[cfg(target_os = "linux")]
|
|
|
|
|
fn rss_mb() -> u64 {
|
|
|
|
|
std::fs::read_to_string("/proc/self/status")
|
|
|
|
|
.ok()
|
|
|
|
|
.and_then(|s| {
|
|
|
|
|
s.lines()
|
|
|
|
|
.find(|l| l.starts_with("VmRSS:"))
|
|
|
|
|
.and_then(|l| l.split_whitespace().nth(1))
|
|
|
|
|
.and_then(|kb| kb.parse::<u64>().ok())
|
|
|
|
|
})
|
|
|
|
|
.map(|kb| kb / 1024)
|
|
|
|
|
.unwrap_or(0)
|
|
|
|
|
}
|
|
|
|
|
#[cfg(not(target_os = "linux"))]
|
|
|
|
|
fn rss_mb() -> u64 {
|
|
|
|
|
0
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Peak RSS while exactly `k` renders run concurrently (one wave) — the resident
|
|
|
|
|
/// cost of `k` simultaneous decode buffers, i.e. what the permit count bounds.
|
|
|
|
|
fn bench_k_rss(rt: &tokio::runtime::Runtime, img: Arc<Vec<u8>>, k: usize) -> u64 {
|
|
|
|
|
rt.block_on(async move {
|
|
|
|
|
let peak = Arc::new(AtomicU64::new(rss_mb()));
|
|
|
|
|
let stop = Arc::new(AtomicBool::new(false));
|
|
|
|
|
let p = peak.clone();
|
|
|
|
|
let s = stop.clone();
|
|
|
|
|
let sampler = tokio::spawn(async move {
|
|
|
|
|
while !s.load(Ordering::Relaxed) {
|
|
|
|
|
p.fetch_max(rss_mb(), Ordering::Relaxed);
|
|
|
|
|
tokio::time::sleep(Duration::from_millis(2)).await;
|
|
|
|
|
}
|
|
|
|
|
});
|
|
|
|
|
let mut hs = Vec::with_capacity(k);
|
|
|
|
|
for _ in 0..k {
|
|
|
|
|
let img2 = img.clone();
|
|
|
|
|
hs.push(tokio::task::spawn_blocking(move || {
|
|
|
|
|
let _ = ThumbnailService::bench_render_all(&img2);
|
|
|
|
|
}));
|
|
|
|
|
}
|
|
|
|
|
for h in hs {
|
|
|
|
|
let _ = h.await;
|
|
|
|
|
}
|
|
|
|
|
peak.fetch_max(rss_mb(), Ordering::Relaxed);
|
|
|
|
|
stop.store(true, Ordering::Relaxed);
|
|
|
|
|
let _ = sampler.await;
|
|
|
|
|
peak.load(Ordering::Relaxed)
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-22 08:46:24 +00:00
|
|
|
fn percentile(sorted: &[u64], p: f64) -> u64 {
|
|
|
|
|
if sorted.is_empty() {
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
let idx = ((p / 100.0) * (sorted.len() as f64 - 1.0)).round() as usize;
|
|
|
|
|
sorted[idx.min(sorted.len() - 1)]
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// One sweep cell: gallery of `producers` callers, `k` permits, `secs` window.
|
|
|
|
|
/// Returns (renders, renders/s, p50_ms, p99_ms).
|
|
|
|
|
fn bench_k(
|
|
|
|
|
rt: &tokio::runtime::Runtime,
|
|
|
|
|
img: Arc<Vec<u8>>,
|
|
|
|
|
k: usize,
|
|
|
|
|
producers: usize,
|
|
|
|
|
secs: u64,
|
|
|
|
|
) -> (u64, f64, f64, f64) {
|
|
|
|
|
rt.block_on(async move {
|
|
|
|
|
let sem = Arc::new(Semaphore::new(k));
|
|
|
|
|
let deadline = Instant::now() + Duration::from_secs(secs);
|
|
|
|
|
let mut handles = Vec::with_capacity(producers);
|
|
|
|
|
for _ in 0..producers {
|
|
|
|
|
let sem = sem.clone();
|
|
|
|
|
let img = img.clone();
|
|
|
|
|
handles.push(tokio::spawn(async move {
|
|
|
|
|
let mut count = 0u64;
|
|
|
|
|
let mut lats: Vec<u64> = Vec::with_capacity(256);
|
|
|
|
|
while Instant::now() < deadline {
|
|
|
|
|
// Real path: acquire a decode permit, render off-reactor.
|
|
|
|
|
let t = Instant::now();
|
|
|
|
|
let permit = sem.clone().acquire_owned().await.unwrap();
|
|
|
|
|
let img2 = img.clone();
|
|
|
|
|
let _ = tokio::task::spawn_blocking(move || {
|
|
|
|
|
ThumbnailService::bench_render_all(&img2).expect("render_all")
|
|
|
|
|
})
|
|
|
|
|
.await;
|
|
|
|
|
drop(permit);
|
|
|
|
|
lats.push(t.elapsed().as_micros() as u64);
|
|
|
|
|
count += 1;
|
|
|
|
|
}
|
|
|
|
|
(count, lats)
|
|
|
|
|
}));
|
|
|
|
|
}
|
|
|
|
|
let mut total = 0u64;
|
|
|
|
|
let mut all: Vec<u64> = Vec::new();
|
|
|
|
|
for h in handles {
|
|
|
|
|
let (c, l) = h.await.expect("join");
|
|
|
|
|
total += c;
|
|
|
|
|
all.extend_from_slice(&l);
|
|
|
|
|
}
|
|
|
|
|
all.sort_unstable();
|
|
|
|
|
let rps = total as f64 / secs as f64;
|
|
|
|
|
(
|
|
|
|
|
total,
|
|
|
|
|
rps,
|
|
|
|
|
percentile(&all, 50.0) as f64 / 1000.0,
|
|
|
|
|
percentile(&all, 99.0) as f64 / 1000.0,
|
|
|
|
|
)
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
fn main() {
|
|
|
|
|
let k_list: Vec<usize> = std::env::var("BENCH_K_LIST")
|
|
|
|
|
.ok()
|
|
|
|
|
.map(|s| s.split(',').filter_map(|x| x.trim().parse().ok()).collect())
|
|
|
|
|
.filter(|v: &Vec<usize>| !v.is_empty())
|
|
|
|
|
.unwrap_or_else(|| vec![1, 2, 4, 8, 16]);
|
|
|
|
|
let producers: usize = env_or("BENCH_GALLERY", 48);
|
|
|
|
|
let secs: u64 = env_or("BENCH_SECONDS", 4);
|
|
|
|
|
|
|
|
|
|
// Pick the heaviest corpus image — the decode cost the pool gates.
|
|
|
|
|
let corpus = bench_support::load_or_generate();
|
|
|
|
|
let case = corpus
|
|
|
|
|
.iter()
|
|
|
|
|
.max_by(|a, b| a.megapixels().partial_cmp(&b.megapixels()).unwrap())
|
|
|
|
|
.expect("corpus non-empty");
|
|
|
|
|
let img = Arc::new(case.bytes.clone());
|
|
|
|
|
|
|
|
|
|
// The renders run on spawn_blocking; give the blocking pool plenty of room
|
|
|
|
|
// so the Semaphore(K) — not the runtime — is the binding constraint.
|
|
|
|
|
let rt = tokio::runtime::Builder::new_multi_thread()
|
|
|
|
|
.worker_threads(2)
|
|
|
|
|
.max_blocking_threads(64)
|
|
|
|
|
.enable_all()
|
|
|
|
|
.build()
|
|
|
|
|
.expect("runtime");
|
|
|
|
|
|
|
|
|
|
let eff = effective_parallelism();
|
|
|
|
|
println!("\n############################################################");
|
|
|
|
|
println!("# CPU pool concurrency — thumbnail decode under a CPU quota");
|
|
|
|
|
println!(
|
|
|
|
|
"# image: {} ({:.1} MP, {} KiB) gallery: {} concurrent callers window: {}s",
|
|
|
|
|
case.name,
|
|
|
|
|
case.megapixels(),
|
|
|
|
|
case.bytes.len() / 1024,
|
|
|
|
|
producers,
|
|
|
|
|
secs
|
|
|
|
|
);
|
|
|
|
|
println!(
|
|
|
|
|
"# available_parallelism = {} effective_parallelism = {} (= the 'after' permit count)",
|
2026-06-22 10:37:57 -06:00
|
|
|
std::thread::available_parallelism()
|
|
|
|
|
.map(|n| n.get())
|
|
|
|
|
.unwrap_or(0),
|
2026-06-22 08:46:24 +00:00
|
|
|
eff
|
|
|
|
|
);
|
|
|
|
|
println!("# run under `taskset -c 0,1` to model a 2-core quota");
|
|
|
|
|
println!("############################################################\n");
|
|
|
|
|
println!(
|
|
|
|
|
"| {:>8} | {:>9} | {:>10} | {:>9} | {:>9} |",
|
|
|
|
|
"permits", "renders", "renders/s", "p50 ms", "p99 ms"
|
|
|
|
|
);
|
2026-06-22 10:37:57 -06:00
|
|
|
println!(
|
|
|
|
|
"|{:-<10}|{:-<11}|{:-<12}|{:-<11}|{:-<11}|",
|
|
|
|
|
"", "", "", "", ""
|
|
|
|
|
);
|
2026-06-22 08:46:24 +00:00
|
|
|
|
|
|
|
|
// Warm up (also triggers corpus generation / codec init).
|
|
|
|
|
let _ = bench_k(&rt, img.clone(), 2, producers, 1);
|
|
|
|
|
|
|
|
|
|
for &k in &k_list {
|
|
|
|
|
let (renders, rps, p50, p99) = bench_k(&rt, img.clone(), k, producers, secs);
|
|
|
|
|
let tag = if k == eff { " ← effective" } else { "" };
|
|
|
|
|
println!(
|
|
|
|
|
"| {:>8} | {:>9} | {:>10.1} | {:>9.1} | {:>9.1} |{}",
|
|
|
|
|
k, renders, rps, p50, p99, tag
|
|
|
|
|
);
|
|
|
|
|
}
|
2026-06-22 09:00:19 +00:00
|
|
|
|
|
|
|
|
// ── Part B: peak RSS for K concurrent decodes (the real over-permit cost) ──
|
|
|
|
|
println!("\n[B] Peak RSS with K concurrent decodes (one wave)\n");
|
2026-06-22 10:37:57 -06:00
|
|
|
println!(
|
|
|
|
|
"| {:>8} | {:>14} | {:>12} |",
|
|
|
|
|
"permits", "peak RSS MiB", "vs effective"
|
|
|
|
|
);
|
2026-06-22 09:00:19 +00:00
|
|
|
println!("|{:-<10}|{:-<16}|{:-<14}|", "", "", "");
|
|
|
|
|
let mut eff_rss: Option<u64> = None;
|
|
|
|
|
for &k in &k_list {
|
|
|
|
|
let peak = bench_k_rss(&rt, img.clone(), k);
|
|
|
|
|
if k == eff {
|
|
|
|
|
eff_rss = Some(peak);
|
|
|
|
|
}
|
|
|
|
|
let delta = match eff_rss {
|
|
|
|
|
Some(base) if k > eff => format!("+{} MiB", peak.saturating_sub(base)),
|
|
|
|
|
_ => "—".to_string(),
|
|
|
|
|
};
|
|
|
|
|
let tag = if k == eff { " ← effective" } else { "" };
|
|
|
|
|
println!("| {:>8} | {:>14} | {:>12} |{}", k, peak, delta, tag);
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-22 08:46:24 +00:00
|
|
|
println!(
|
2026-06-22 09:00:19 +00:00
|
|
|
"\nThroughput (A) is CPU-bound — flat past the core count, so over-permitting\n\
|
|
|
|
|
buys no throughput. The cost of over-permitting is resident memory (B):\n\
|
|
|
|
|
each concurrent decode holds its buffer, so RSS scales with the permit\n\
|
|
|
|
|
count. Sizing to the *effective* cores (not the host count) is what keeps\n\
|
|
|
|
|
a many-core-host CPU quota from multiplying thumbnail RAM under load.\n"
|
2026-06-22 08:46:24 +00:00
|
|
|
);
|
|
|
|
|
}
|