perf(thumbnails): shrink-on-load JPEG decode (1.8-2× faster, 5-15× less RAM)
Decode JPEGs at the smallest DCT scale (1/8·1/4·1/2·1/1) whose long axis is still ≥ the largest needed thumbnail (800px), via jpeg-decoder, instead of a full-resolution decode through the image crate. The full-res bitmap — the dominant time and RAM cost — is never materialised. PNG/GIF/WebP and unusual JPEG colour spaces (CMYK / 16-bit grey) fall back to a full decode. Extracts the shared decode + EXIF-orientation logic into decode_oriented(), removing the duplication that existed between render_thumbnail_from_data and render_all_thumbnails_from_data. Measured on 14 cores (see benches/BASELINE.md): - render_all 1.8-2.0× faster (12MP 111->61ms, 48MP 398->203ms) - peak heap 5.5-14.8× lower, now decoupled from source MP (~18-25MB regardless) - saturated throughput 3-3.6× (parallel efficiency 4.9×->8.5×) - quality SSIM 0.987-0.999 (>=0.98 gate), PSNR 47-55dB Also adds the Phase 0 benchmark harness (gated behind the `bench` feature, zero prod impact): deterministic image corpus (src/bench_support.rs), criterion latency bench (benches/thumbnails.rs), and a peak-RAM/throughput/SSIM harness (examples/bench_thumbnails_mem.rs). Baseline + before/after in benches/BASELINE.md. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
+25
@@ -39,6 +39,10 @@ dotenvy = "0.15.7"
|
||||
moka = { version = "0.12.15", features = ["future", "sync"] }
|
||||
http-range-header = "0.4"
|
||||
image = { version = "0.25.10", default-features = false, features = ["jpeg", "png", "gif", "webp"] }
|
||||
# Shrink-on-load JPEG decode (DCT 1/2·1/4·1/8 scaling) for thumbnails — pure
|
||||
# Rust, no C toolchain. The `image` crate's zune-jpeg backend can't scale during
|
||||
# decode; this can, cutting decode time/RAM ~order-of-magnitude on large photos.
|
||||
jpeg-decoder = "0.3"
|
||||
id3 = "1.17"
|
||||
mp3-duration = "0.1"
|
||||
kamadak-exif = "0.6.1"
|
||||
@@ -103,6 +107,14 @@ load_seed_bin = []
|
||||
# requires OXICLOUD_ENABLE_FACES=true *and* operator-provided ONNX models; without
|
||||
# this feature the People pipeline falls back to the inert NoopFaceAnalyzer.
|
||||
faces-onnx = ["dep:ort", "dep:ndarray"]
|
||||
# Performance benchmark harness (Phase 0). Exposes `bench_support` + thin public
|
||||
# wrappers over the private thumbnail render functions so `benches/` and
|
||||
# `examples/` can measure them. Off by default — adds nothing to prod builds.
|
||||
# Run with: `cargo bench --features bench` / `cargo run --release --features bench --example bench_thumbnails_mem`.
|
||||
bench = []
|
||||
|
||||
[dev-dependencies]
|
||||
criterion = "0.5"
|
||||
|
||||
[lints.rust]
|
||||
unexpected_cfgs = { level = "warn", check-cfg = ['cfg(integration_tests)'] }
|
||||
@@ -123,6 +135,19 @@ path = "src/bin/load-seed.rs"
|
||||
# and load-nightly.yml build it explicitly with --features load_seed_bin.
|
||||
required-features = ["load_seed_bin"]
|
||||
|
||||
# Phase 0 perf harness — Task 0.2 (criterion latency + output-size bench).
|
||||
[[bench]]
|
||||
name = "thumbnails"
|
||||
path = "benches/thumbnails.rs"
|
||||
harness = false
|
||||
required-features = ["bench"]
|
||||
|
||||
# Phase 0 perf harness — Task 0.3 (peak-RAM + saturated-throughput baseline).
|
||||
[[example]]
|
||||
name = "bench_thumbnails_mem"
|
||||
path = "examples/bench_thumbnails_mem.rs"
|
||||
required-features = ["bench"]
|
||||
|
||||
[profile.release]
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
|
||||
Reference in New Issue
Block a user