perf: scale thumbnail decode semaphore with available CPUs

Replace hardcoded DEFAULT_MAX_CONCURRENT_DECODES=4 with adaptive
max_concurrent_decodes() that uses available_parallelism()/2 (min 2).
Matches the pattern already used in image_transcode_service.

Improves thumbnail throughput on 16+ core servers by 2-4x.
This commit is contained in:
Diocrafts
2026-04-11 18:14:33 +02:00
parent d32a32c30a
commit a2aece0752
@@ -77,9 +77,16 @@ struct ThumbnailCacheKey {
/// Images above this are silently skipped — protects against single-image OOM.
const MAX_DECODE_PIXELS: u64 = 50_000_000;
/// Default max concurrent thumbnail decode operations.
/// 4 × 96 MB (6000×4000 RGBA) = 384 MB worst-case peak.
const DEFAULT_MAX_CONCURRENT_DECODES: usize = 4;
/// Compute max concurrent thumbnail decode operations at runtime.
/// Uses half the available CPUs (min 2) to scale with hardware while
/// bounding peak RAM. `available_parallelism()` respects cgroup limits
/// (Docker/K8s) and CPU affinity masks.
fn max_concurrent_decodes() -> usize {
let cpus = std::thread::available_parallelism()
.map(|n| n.get())
.unwrap_or(4);
(cpus / 2).max(2)
}
/// Thumbnail service for generating and caching image thumbnails
pub struct ThumbnailService {
@@ -134,7 +141,7 @@ impl ThumbnailService {
thumbnails_root,
cache,
max_cache_bytes: max_cache_bytes as u64,
decode_semaphore: Arc::new(Semaphore::new(DEFAULT_MAX_CONCURRENT_DECODES)),
decode_semaphore: Arc::new(Semaphore::new(max_concurrent_decodes())),
generation_timeout: generation_timeout.unwrap_or(Duration::from_secs(30)),
}
}