perf(thumbnails): shrink-on-load JPEG decode (1.8-2× faster, 5-15× less RAM)

Decode JPEGs at the smallest DCT scale (1/8·1/4·1/2·1/1) whose long axis is
still ≥ the largest needed thumbnail (800px), via jpeg-decoder, instead of a
full-resolution decode through the image crate. The full-res bitmap — the
dominant time and RAM cost — is never materialised. PNG/GIF/WebP and unusual
JPEG colour spaces (CMYK / 16-bit grey) fall back to a full decode.

Extracts the shared decode + EXIF-orientation logic into decode_oriented(),
removing the duplication that existed between render_thumbnail_from_data and
render_all_thumbnails_from_data.

Measured on 14 cores (see benches/BASELINE.md):
- render_all 1.8-2.0× faster (12MP 111->61ms, 48MP 398->203ms)
- peak heap 5.5-14.8× lower, now decoupled from source MP (~18-25MB regardless)
- saturated throughput 3-3.6× (parallel efficiency 4.9×->8.5×)
- quality SSIM 0.987-0.999 (>=0.98 gate), PSNR 47-55dB

Also adds the Phase 0 benchmark harness (gated behind the `bench` feature, zero
prod impact): deterministic image corpus (src/bench_support.rs), criterion
latency bench (benches/thumbnails.rs), and a peak-RAM/throughput/SSIM harness
(examples/bench_thumbnails_mem.rs). Baseline + before/after in benches/BASELINE.md.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
DioCrafts
2026-06-21 15:13:03 +02:00
parent 08b36cf0d4
commit fd5808c157
9 changed files with 1415 additions and 39 deletions
+25
View File
@@ -39,6 +39,10 @@ dotenvy = "0.15.7"
moka = { version = "0.12.15", features = ["future", "sync"] }
http-range-header = "0.4"
image = { version = "0.25.10", default-features = false, features = ["jpeg", "png", "gif", "webp"] }
# Shrink-on-load JPEG decode (DCT 1/2·1/4·1/8 scaling) for thumbnails — pure
# Rust, no C toolchain. The `image` crate's zune-jpeg backend can't scale during
# decode; this can, cutting decode time/RAM ~order-of-magnitude on large photos.
jpeg-decoder = "0.3"
id3 = "1.17"
mp3-duration = "0.1"
kamadak-exif = "0.6.1"
@@ -103,6 +107,14 @@ load_seed_bin = []
# requires OXICLOUD_ENABLE_FACES=true *and* operator-provided ONNX models; without
# this feature the People pipeline falls back to the inert NoopFaceAnalyzer.
faces-onnx = ["dep:ort", "dep:ndarray"]
# Performance benchmark harness (Phase 0). Exposes `bench_support` + thin public
# wrappers over the private thumbnail render functions so `benches/` and
# `examples/` can measure them. Off by default — adds nothing to prod builds.
# Run with: `cargo bench --features bench` / `cargo run --release --features bench --example bench_thumbnails_mem`.
bench = []
[dev-dependencies]
criterion = "0.5"
[lints.rust]
unexpected_cfgs = { level = "warn", check-cfg = ['cfg(integration_tests)'] }
@@ -123,6 +135,19 @@ path = "src/bin/load-seed.rs"
# and load-nightly.yml build it explicitly with --features load_seed_bin.
required-features = ["load_seed_bin"]
# Phase 0 perf harness — Task 0.2 (criterion latency + output-size bench).
[[bench]]
name = "thumbnails"
path = "benches/thumbnails.rs"
harness = false
required-features = ["bench"]
# Phase 0 perf harness — Task 0.3 (peak-RAM + saturated-throughput baseline).
[[example]]
name = "bench_thumbnails_mem"
path = "examples/bench_thumbnails_mem.rs"
required-features = ["bench"]
[profile.release]
lto = "thin"
codegen-units = 1