Merge pull request #485 from AtalayaLabs/claude/zealous-faraday-58s1at

feat: Photos evolution — Places (map) & People (faces) + gallery polish
This commit is contained in:
Dionisio Pozo
2026-06-19 14:31:14 +02:00
committed by GitHub
50 changed files with 4780 additions and 127 deletions
@@ -0,0 +1,473 @@
//! Pure geometry + post-processing for the ONNX face pipeline.
//!
//! Everything here is plain Rust (no `ort`, no `ndarray`) so it compiles in the
//! default build and is exercised by `cargo test` — the error-prone numerical
//! parts (SCRFD anchor decode, NMS, 5-point similarity alignment, the affine
//! warp, normalization) are unit-tested in isolation, while the untestable ONNX
//! session calls live behind the `faces-onnx` feature in `onnx_face_analyzer`.
//!
//! The pipeline mirrors InsightFace's reference implementation:
//! SCRFD detector (distance-to-box anchors over strides 8/16/32) → 5-point
//! similarity transform onto the canonical 112×112 ArcFace template → ArcFace
//! embedder → L2-normalized 512-d vector.
use image::RgbImage;
/// One detected face in **detector-input pixel** coordinates (before scaling
/// back to the original image): an axis-aligned box `[x1, y1, x2, y2]`, the
/// five facial landmarks, and the detector confidence.
#[derive(Debug, Clone, Copy)]
pub struct Detection {
pub bbox: [f32; 4],
pub kps: [[f32; 2]; 5],
pub score: f32,
}
/// A 2×3 affine transform mapping an output/template coordinate to a source
/// coordinate: `src = (a·ox + b·oy + tx, c·ox + d·oy + ty)`. Used to sample the
/// source image when warping an aligned face crop.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct Affine {
pub a: f32,
pub b: f32,
pub c: f32,
pub d: f32,
pub tx: f32,
pub ty: f32,
}
/// Canonical ArcFace 5-point template for a 112×112 crop
/// (left eye, right eye, nose, left mouth, right mouth).
pub const ARCFACE_TEMPLATE: [[f32; 2]; 5] = [
[38.2946, 51.6963],
[73.5318, 51.5014],
[56.0252, 71.7366],
[41.5493, 92.3655],
[70.7299, 92.2041],
];
/// Aligned-crop side length expected by the ArcFace embedder.
pub const ALIGN_SIZE: u32 = 112;
/// Letterbox geometry for the detector: the largest scale that fits a
/// `w0 × h0` image into a `det × det` square without distortion, plus the
/// resulting (possibly smaller) dimensions placed at the top-left.
///
/// Returns `(new_w, new_h, scale)` where `scale = min(det/w0, det/h0)` and
/// detector-space coordinates map back to the original by dividing by `scale`.
pub fn letterbox(w0: u32, h0: u32, det: u32) -> (u32, u32, f32) {
if w0 == 0 || h0 == 0 {
return (0, 0, 1.0);
}
let scale = (det as f32 / w0 as f32).min(det as f32 / h0 as f32);
let new_w = ((w0 as f32 * scale).round() as u32).clamp(1, det);
let new_h = ((h0 as f32 * scale).round() as u32).clamp(1, det);
(new_w, new_h, scale)
}
/// `NCHW`, RGB, float input tensor for an ONNX model: `(px − mean) · scale`,
/// channel-major (all R, then all G, then all B). Length is `3 · w · h`.
pub fn chw_normalized(img: &RgbImage, mean: f32, scale: f32) -> Vec<f32> {
let (w, h) = (img.width() as usize, img.height() as usize);
let mut out = vec![0.0f32; 3 * w * h];
let plane = w * h;
for (i, px) in img.pixels().enumerate() {
out[i] = (px[0] as f32 - mean) * scale;
out[plane + i] = (px[1] as f32 - mean) * scale;
out[2 * plane + i] = (px[2] as f32 - mean) * scale;
}
out
}
/// Decode one SCRFD feature-map stride into detections, appending those above
/// `threshold` to `out`. All coordinates are in detector-input pixels.
///
/// `scores` is `[n]`, `bbox` is `[n·4]` (left, top, right, bottom *distances*,
/// already multiplied by `stride`), `kps` (when present) is `[n·10]`
/// (5 × (dx, dy) distances, already multiplied by `stride`), where
/// `n = feat_h · feat_w · num_anchors`. Anchor centers follow InsightFace's
/// row-major `mgrid` order with `num_anchors` consecutive duplicates.
#[allow(clippy::too_many_arguments)]
pub fn decode_stride(
scores: &[f32],
bbox: &[f32],
kps: Option<&[f32]>,
stride: u32,
feat_h: u32,
feat_w: u32,
num_anchors: u32,
threshold: f32,
out: &mut Vec<Detection>,
) {
let stride_f = stride as f32;
let mut idx = 0usize;
for y in 0..feat_h {
for x in 0..feat_w {
let cx = x as f32 * stride_f;
let cy = y as f32 * stride_f;
for _ in 0..num_anchors {
if idx >= scores.len() {
return;
}
let score = scores[idx];
if score >= threshold {
let b = idx * 4;
if b + 3 < bbox.len() {
let det_bbox = [
cx - bbox[b],
cy - bbox[b + 1],
cx + bbox[b + 2],
cy + bbox[b + 3],
];
let mut det_kps = [[0.0f32; 2]; 5];
if let Some(kps) = kps {
let k = idx * 10;
if k + 9 < kps.len() {
for (p, slot) in det_kps.iter_mut().enumerate() {
*slot = [cx + kps[k + p * 2], cy + kps[k + p * 2 + 1]];
}
}
}
out.push(Detection {
bbox: det_bbox,
kps: det_kps,
score,
});
}
}
idx += 1;
}
}
}
}
/// Intersection-over-union of two `[x1, y1, x2, y2]` boxes.
pub fn iou(a: &[f32; 4], b: &[f32; 4]) -> f32 {
let x1 = a[0].max(b[0]);
let y1 = a[1].max(b[1]);
let x2 = a[2].min(b[2]);
let y2 = a[3].min(b[3]);
let iw = (x2 - x1).max(0.0);
let ih = (y2 - y1).max(0.0);
let inter = iw * ih;
let area_a = (a[2] - a[0]).max(0.0) * (a[3] - a[1]).max(0.0);
let area_b = (b[2] - b[0]).max(0.0) * (b[3] - b[1]).max(0.0);
let union = area_a + area_b - inter;
if union <= 0.0 { 0.0 } else { inter / union }
}
/// Greedy non-maximum suppression: keep highest-scoring boxes, drop any whose
/// IoU with an already-kept box exceeds `iou_thresh`. Returns the kept
/// detections, highest score first.
pub fn nms(mut dets: Vec<Detection>, iou_thresh: f32) -> Vec<Detection> {
dets.sort_by(|a, b| b.score.total_cmp(&a.score));
let mut keep: Vec<Detection> = Vec::with_capacity(dets.len());
for d in dets {
if keep.iter().all(|k| iou(&k.bbox, &d.bbox) <= iou_thresh) {
keep.push(d);
}
}
keep
}
/// Least-squares similarity transform (scale + rotation + translation, no
/// shear, no reflection) mapping `src` landmarks onto `dst`, returned as its
/// **inverse** affine (output/template coordinate → source coordinate) ready
/// for backward-warp sampling.
///
/// Solved in closed form via the complex-number formulation: with points as
/// complex numbers, `w = Σ (b'ᵢ · conj(a'ᵢ)) / Σ |a'ᵢ|²` and `t = mean_b −
/// w·mean_a`, which is equivalent to the Umeyama solution InsightFace obtains
/// from `skimage.SimilarityTransform`.
pub fn similarity_transform_inverse(src: &[[f32; 2]; 5], dst: &[[f32; 2]; 5]) -> Affine {
let n = 5.0f32;
let (mut max, mut may, mut mbx, mut mby) = (0.0f32, 0.0f32, 0.0f32, 0.0f32);
for i in 0..5 {
max += src[i][0];
may += src[i][1];
mbx += dst[i][0];
mby += dst[i][1];
}
max /= n;
may /= n;
mbx /= n;
mby /= n;
// num = Σ b'·conj(a') (complex), den = Σ |a'|² (real)
let (mut num_re, mut num_im, mut den) = (0.0f32, 0.0f32, 0.0f32);
for i in 0..5 {
let ax = src[i][0] - max;
let ay = src[i][1] - may;
let bx = dst[i][0] - mbx;
let by = dst[i][1] - mby;
// b' · conj(a') = (bx + i·by)(ax − i·ay)
num_re += bx * ax + by * ay;
num_im += by * ax - bx * ay;
den += ax * ax + ay * ay;
}
let den = if den.abs() < 1e-12 { 1e-12 } else { den };
// w = num/den (forward scale·rotation)
let wr = num_re / den;
let wi = num_im / den;
// t = mean_b − w·mean_a
let tr = mbx - (wr * max - wi * may);
let ti = mby - (wi * max + wr * may);
// Inverse of the similarity: src = Ainv·(out − t), Ainv = [[wr,wi],[−wi,wr]]/|w|²
let det = wr * wr + wi * wi;
let g = if det.abs() < 1e-12 { 0.0 } else { 1.0 / det };
Affine {
a: g * wr,
b: g * wi,
c: -g * wi,
d: g * wr,
tx: -g * (wr * tr + wi * ti),
ty: g * (wi * tr - wr * ti),
}
}
/// Warp `img` into an `ALIGN_SIZE × ALIGN_SIZE` aligned face crop using the
/// inverse affine from [`similarity_transform_inverse`], sampling bilinearly
/// and clamping to the image edge.
pub fn warp_to_aligned(img: &RgbImage, inv: &Affine) -> RgbImage {
let (w, h) = (img.width(), img.height());
let mut out = RgbImage::new(ALIGN_SIZE, ALIGN_SIZE);
for oy in 0..ALIGN_SIZE {
for ox in 0..ALIGN_SIZE {
let sx = inv.a * ox as f32 + inv.b * oy as f32 + inv.tx;
let sy = inv.c * ox as f32 + inv.d * oy as f32 + inv.ty;
let px = bilinear_sample(img, sx, sy, w, h);
out.put_pixel(ox, oy, px);
}
}
out
}
/// Bilinear RGB sample at floating `(x, y)`, clamping out-of-bounds reads to
/// the nearest edge.
fn bilinear_sample(img: &RgbImage, x: f32, y: f32, w: u32, h: u32) -> image::Rgb<u8> {
let x = x.clamp(0.0, (w - 1) as f32);
let y = y.clamp(0.0, (h - 1) as f32);
let x0 = x.floor() as u32;
let y0 = y.floor() as u32;
let x1 = (x0 + 1).min(w - 1);
let y1 = (y0 + 1).min(h - 1);
let dx = x - x0 as f32;
let dy = y - y0 as f32;
let p00 = img.get_pixel(x0, y0);
let p10 = img.get_pixel(x1, y0);
let p01 = img.get_pixel(x0, y1);
let p11 = img.get_pixel(x1, y1);
let mut out = [0u8; 3];
for (ch, slot) in out.iter_mut().enumerate() {
let top = p00[ch] as f32 * (1.0 - dx) + p10[ch] as f32 * dx;
let bot = p01[ch] as f32 * (1.0 - dx) + p11[ch] as f32 * dx;
*slot = (top * (1.0 - dy) + bot * dy).round().clamp(0.0, 255.0) as u8;
}
image::Rgb(out)
}
/// In-place L2 normalization. A zero vector is left unchanged.
pub fn l2_normalize(v: &mut [f32]) {
let norm = v.iter().map(|x| x * x).sum::<f32>().sqrt();
if norm > 1e-12 {
for x in v.iter_mut() {
*x /= norm;
}
}
}
/// Variance of the discrete Laplacian over the luminance of an RGB crop — a
/// cheap focus/sharpness proxy (higher = sharper). Used as a face quality
/// score for cover selection and gating.
pub fn laplacian_variance(img: &RgbImage) -> f32 {
let (w, h) = (img.width() as i64, img.height() as i64);
if w < 3 || h < 3 {
return 0.0;
}
let lum = |x: i64, y: i64| -> f32 {
let p = img.get_pixel(x as u32, y as u32);
0.299 * p[0] as f32 + 0.587 * p[1] as f32 + 0.114 * p[2] as f32
};
let mut vals = Vec::with_capacity(((w - 2) * (h - 2)) as usize);
for y in 1..h - 1 {
for x in 1..w - 1 {
let l = 4.0 * lum(x, y) - lum(x - 1, y) - lum(x + 1, y) - lum(x, y - 1) - lum(x, y + 1);
vals.push(l);
}
}
let n = vals.len() as f32;
if n == 0.0 {
return 0.0;
}
let mean = vals.iter().sum::<f32>() / n;
vals.iter().map(|v| (v - mean) * (v - mean)).sum::<f32>() / n
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn letterbox_fits_and_preserves_aspect() {
// Landscape 1000×500 into 640 → width-bound, scale 0.64.
let (nw, nh, s) = letterbox(1000, 500, 640);
assert_eq!(nw, 640);
assert_eq!(nh, 320);
assert!((s - 0.64).abs() < 1e-6);
// Square fills exactly.
let (nw, nh, s) = letterbox(800, 800, 640);
assert_eq!((nw, nh), (640, 640));
assert!((s - 0.8).abs() < 1e-6);
}
#[test]
fn letterbox_degenerate_is_safe() {
assert_eq!(letterbox(0, 10, 640), (0, 0, 1.0));
}
#[test]
fn chw_layout_and_normalization() {
let mut img = RgbImage::new(2, 1);
img.put_pixel(0, 0, image::Rgb([127, 0, 255]));
img.put_pixel(1, 0, image::Rgb([128, 255, 0]));
let t = chw_normalized(&img, 127.5, 1.0 / 128.0);
// Length = 3 channels × 2 px.
assert_eq!(t.len(), 6);
// R plane first, then G, then B (NCHW).
assert!((t[0] - (127.0 - 127.5) / 128.0).abs() < 1e-6);
assert!((t[1] - (128.0 - 127.5) / 128.0).abs() < 1e-6);
assert!((t[2] - (0.0 - 127.5) / 128.0).abs() < 1e-6); // G of px0
assert!((t[4] - (255.0 - 127.5) / 128.0).abs() < 1e-6); // B of px0
}
#[test]
fn distance_decode_recovers_box_and_kps() {
// 1×2 grid, stride 8, 1 anchor → cell centers (0,0) then (8,0).
let scores = [0.9f32, 0.9];
// distances left/top/right/bottom (already × stride), identical per cell.
let bbox = [2.0, 1.0, 3.0, 4.0, 2.0, 1.0, 3.0, 4.0];
let kps: Vec<f32> = vec![
1.0, 1.0, 2.0, 2.0, 0.0, 0.0, -1.0, 1.0, 1.0, -1.0, // cell 0
1.0, 1.0, 2.0, 2.0, 0.0, 0.0, -1.0, 1.0, 1.0, -1.0, // cell 1
];
let mut out = Vec::new();
decode_stride(&scores, &bbox, Some(&kps), 8, 1, 2, 1, 0.5, &mut out);
assert_eq!(out.len(), 2);
// Cell 0, center (0,0): box = center ± distances, kps = center + offset.
assert_eq!(out[0].bbox, [-2.0, -1.0, 3.0, 4.0]);
assert_eq!(out[0].kps[0], [1.0, 1.0]);
assert_eq!(out[0].kps[1], [2.0, 2.0]);
// Cell 1, center (8,0): anchor center advanced by one stride in x.
assert_eq!(out[1].bbox, [8.0 - 2.0, -1.0, 8.0 + 3.0, 4.0]);
assert_eq!(out[1].kps[0], [9.0, 1.0]);
}
#[test]
fn decode_thresholds_out_low_scores() {
let scores = [0.2f32, 0.8];
let bbox = [0.0, 0.0, 1.0, 1.0, 0.0, 0.0, 1.0, 1.0];
let mut out = Vec::new();
// 1×2 grid, 1 anchor → two cells.
decode_stride(&scores, &bbox, None, 8, 1, 2, 1, 0.5, &mut out);
assert_eq!(out.len(), 1);
assert!((out[0].score - 0.8).abs() < 1e-6);
}
#[test]
fn iou_and_nms() {
let a = [0.0, 0.0, 10.0, 10.0];
let b = [0.0, 0.0, 10.0, 10.0];
assert!((iou(&a, &b) - 1.0).abs() < 1e-6);
let c = [100.0, 100.0, 110.0, 110.0];
assert_eq!(iou(&a, &c), 0.0);
let dets = vec![
Detection {
bbox: a,
kps: [[0.0; 2]; 5],
score: 0.9,
},
Detection {
bbox: b,
kps: [[0.0; 2]; 5],
score: 0.8,
}, // dup of a
Detection {
bbox: c,
kps: [[0.0; 2]; 5],
score: 0.7,
}, // separate
];
let kept = nms(dets, 0.4);
assert_eq!(kept.len(), 2);
assert!((kept[0].score - 0.9).abs() < 1e-6);
}
#[test]
fn similarity_identity() {
let inv = similarity_transform_inverse(&ARCFACE_TEMPLATE, &ARCFACE_TEMPLATE);
assert!((inv.a - 1.0).abs() < 1e-4);
assert!(inv.b.abs() < 1e-4);
assert!(inv.c.abs() < 1e-4);
assert!((inv.d - 1.0).abs() < 1e-4);
assert!(inv.tx.abs() < 1e-3);
assert!(inv.ty.abs() < 1e-3);
}
#[test]
fn similarity_pure_translation() {
// src = dst shifted by (+10, +5); inverse must map out→src by the same shift.
let mut src = ARCFACE_TEMPLATE;
for p in &mut src {
p[0] += 10.0;
p[1] += 5.0;
}
let inv = similarity_transform_inverse(&src, &ARCFACE_TEMPLATE);
assert!((inv.a - 1.0).abs() < 1e-4);
assert!(inv.b.abs() < 1e-4);
assert!((inv.tx - 10.0).abs() < 1e-3);
assert!((inv.ty - 5.0).abs() < 1e-3);
}
#[test]
fn warp_identity_preserves_template_region() {
// A 112×112 gradient warped by identity returns (close to) itself.
let mut img = RgbImage::new(ALIGN_SIZE, ALIGN_SIZE);
for y in 0..ALIGN_SIZE {
for x in 0..ALIGN_SIZE {
img.put_pixel(x, y, image::Rgb([x as u8, y as u8, 128]));
}
}
let inv = similarity_transform_inverse(&ARCFACE_TEMPLATE, &ARCFACE_TEMPLATE);
let out = warp_to_aligned(&img, &inv);
let a = out.get_pixel(40, 60);
assert!((a[0] as i32 - 40).abs() <= 1);
assert!((a[1] as i32 - 60).abs() <= 1);
}
#[test]
fn l2_normalize_unit_length() {
let mut v = vec![3.0f32, 4.0];
l2_normalize(&mut v);
assert!((v[0] - 0.6).abs() < 1e-6);
assert!((v[1] - 0.8).abs() < 1e-6);
let mut z = vec![0.0f32, 0.0];
l2_normalize(&mut z); // unchanged, no NaN
assert_eq!(z, vec![0.0, 0.0]);
}
#[test]
fn laplacian_variance_sharp_vs_flat() {
let flat = RgbImage::from_pixel(8, 8, image::Rgb([100, 100, 100]));
assert!(laplacian_variance(&flat) < 1e-3);
let mut checker = RgbImage::new(8, 8);
for y in 0..8 {
for x in 0..8 {
let v = if (x + y) % 2 == 0 { 0 } else { 255 };
checker.put_pixel(x, y, image::Rgb([v, v, v]));
}
}
assert!(laplacian_variance(&checker) > 1000.0);
}
}
@@ -0,0 +1,191 @@
//! Face indexing as a `FileLifecycleHook`.
//!
//! On image upload it detects + embeds faces (off the request path, in a
//! background task) and stores them. It mirrors `MediaMetadataService`: reads
//! the blob from the local `.blobs` tree, is dedup-aware (identical uploads
//! clone an existing file's faces instead of re-running inference), and is
//! completely inert when no model is configured (`FaceAnalyzerPort::is_ready()
//! == false`) — so the feature compiles and runs with the default no-op
//! analyzer until the operator wires a real ONNX model.
use std::path::{Path, PathBuf};
use std::sync::Arc;
use chrono::Utc;
use sqlx::PgPool;
use uuid::Uuid;
use crate::application::ports::face_ports::{FaceAnalyzerPort, FaceRepository};
use crate::application::ports::file_lifecycle::FileLifecycleHook;
use crate::common::errors::DomainError;
use crate::domain::entities::face::Face;
use crate::infrastructure::repositories::pg::FacePgRepository;
/// Minimum detector confidence for a face to be stored.
const MIN_DET_SCORE: f32 = 0.6;
fn is_image(content_type: &str) -> bool {
content_type.starts_with("image/")
}
pub struct FaceIndexingService {
pool: Arc<PgPool>,
repo: Arc<FacePgRepository>,
analyzer: Arc<dyn FaceAnalyzerPort>,
blob_root: PathBuf,
}
impl FaceIndexingService {
pub fn new(pool: Arc<PgPool>, blob_root: PathBuf, analyzer: Arc<dyn FaceAnalyzerPort>) -> Self {
let repo = Arc::new(FacePgRepository::new(pool.clone()));
Self {
pool,
repo,
analyzer,
blob_root,
}
}
/// Local path of a blob: `.blobs/{prefix}/{hash}.blob`.
fn blob_path(&self, hash: &str) -> PathBuf {
let prefix = if hash.len() >= 2 { &hash[0..2] } else { hash };
self.blob_root.join(prefix).join(format!("{hash}.blob"))
}
/// Spawn a background indexing task. `reuse_dedup` clones faces from an
/// existing file with the same blob hash instead of re-running inference;
/// `delete_first` clears prior faces (used on overwrite).
fn spawn_index(&self, file_id: Uuid, blob_hash: String, reuse_dedup: bool, delete_first: bool) {
let pool = self.pool.clone();
let repo = self.repo.clone();
let analyzer = self.analyzer.clone();
let blob_path = self.blob_path(&blob_hash);
tokio::spawn(async move {
if delete_first {
let _ = repo.delete_faces_for_file(file_id).await;
}
if let Err(e) = index_file(
&pool,
&repo,
analyzer.as_ref(),
file_id,
&blob_path,
&blob_hash,
reuse_dedup,
)
.await
{
tracing::warn!(target: "oxicloud::faces", "face indexing failed for {file_id}: {e}");
}
});
}
}
impl FileLifecycleHook for FaceIndexingService {
fn on_file_created(
&self,
file_id: &str,
blob_hash: &str,
content_type: &str,
is_new_blob: bool,
) {
if !is_image(content_type) || !self.analyzer.is_ready() {
return;
}
if let Ok(fid) = file_id.parse::<Uuid>() {
// Dedup hit (blob already existed) → clone an existing file's faces.
self.spawn_index(fid, blob_hash.to_string(), !is_new_blob, false);
}
}
fn on_file_copied(
&self,
file_id: &str,
blob_hash: &str,
content_type: &str,
_source_file_id: &str,
) {
if !is_image(content_type) || !self.analyzer.is_ready() {
return;
}
if let Ok(fid) = file_id.parse::<Uuid>() {
self.spawn_index(fid, blob_hash.to_string(), true, false);
}
}
fn on_file_updated(&self, file_id: &str, blob_hash: &str, content_type: &str) {
if !is_image(content_type) || !self.analyzer.is_ready() {
return;
}
if let Ok(fid) = file_id.parse::<Uuid>() {
self.spawn_index(fid, blob_hash.to_string(), false, true);
}
}
fn on_file_deleted(&self, _file_id: &str) {
// faces.faces.file_id has ON DELETE CASCADE — the DB cleans up.
}
}
async fn lookup_user(pool: &PgPool, file_id: Uuid) -> Result<Uuid, DomainError> {
let row: (Uuid,) = sqlx::query_as("SELECT user_id FROM storage.files WHERE id = $1")
.bind(file_id)
.fetch_one(pool)
.await
.map_err(|e| DomainError::internal_error("Faces", format!("lookup user: {e}")))?;
Ok(row.0)
}
async fn index_file(
pool: &PgPool,
repo: &FacePgRepository,
analyzer: &dyn FaceAnalyzerPort,
file_id: Uuid,
blob_path: &Path,
blob_hash: &str,
reuse_dedup: bool,
) -> Result<(), DomainError> {
let user_id = lookup_user(pool, file_id).await?;
// Dedup-aware fast path: reuse faces already computed for an identical blob.
if reuse_dedup {
let peers = repo.faces_for_blob(user_id, blob_hash).await?;
let cloned: Vec<Face> = peers
.into_iter()
.filter(|f| f.file_id != file_id)
.map(|f| Face {
id: Uuid::new_v4(),
file_id,
..f
})
.collect();
if !cloned.is_empty() {
repo.save_faces(&cloned).await?;
return Ok(());
}
// No peer found — fall through and analyze.
}
let bytes = tokio::fs::read(blob_path)
.await
.map_err(|e| DomainError::internal_error("Faces", format!("read blob: {e}")))?;
let detected = analyzer.analyze(&bytes).await?;
let faces: Vec<Face> = detected
.into_iter()
.filter(|d| d.det_score >= MIN_DET_SCORE)
.map(|d| Face {
id: Uuid::new_v4(),
file_id,
user_id,
person_id: None,
bbox: d.bbox,
det_score: d.det_score,
quality: d.quality,
embedding: d.embedding,
blob_hash: Some(blob_hash.to_string()),
created_at: Utc::now(),
})
.collect();
repo.save_faces(&faces).await
}
+5
View File
@@ -6,6 +6,8 @@ pub mod compression_service;
pub mod dedup_service;
pub mod encrypted_blob_backend;
pub mod exif_service;
pub mod face_geometry;
pub mod face_indexing_service;
pub mod file_content_cache;
pub mod file_system_i18n_service;
pub mod image_transcode_service;
@@ -17,7 +19,10 @@ pub mod migration_blob_backend;
pub mod migration_job;
pub mod mock_email_sender;
pub mod nextcloud_chunked_upload_service;
pub mod noop_face_analyzer;
pub mod oidc_service;
#[cfg(feature = "faces-onnx")]
pub mod onnx_face_analyzer;
pub mod password_hasher;
pub mod path_resolver_service;
pub mod path_service;
@@ -0,0 +1,26 @@
//! Default no-op face analyzer.
//!
//! Used when no ML model is configured: it reports `is_ready() == false` and
//! returns no faces, so the whole People pipeline compiles and runs inert
//! until a real ONNX-backed analyzer (provided by the operator) replaces it.
use async_trait::async_trait;
use crate::application::ports::face_ports::FaceAnalyzerPort;
use crate::common::errors::DomainError;
use crate::domain::entities::face::DetectedFace;
/// Analyzer that never detects anything.
#[derive(Debug, Default, Clone, Copy)]
pub struct NoopFaceAnalyzer;
#[async_trait]
impl FaceAnalyzerPort for NoopFaceAnalyzer {
fn is_ready(&self) -> bool {
false
}
async fn analyze(&self, _image_bytes: &[u8]) -> Result<Vec<DetectedFace>, DomainError> {
Ok(Vec::new())
}
}
@@ -0,0 +1,340 @@
//! ONNX-backed face analyzer (SCRFD detector + ArcFace embedder).
//!
//! Compiled only with the `faces-onnx` cargo feature. Mirrors the
//! immich/InsightFace pipeline: detect faces + 5-point landmarks (SCRFD),
//! similarity-align each face to the canonical 112×112 template, then embed
//! (ArcFace) into an L2-normalized 512-d vector. All inference runs on a
//! blocking thread (`spawn_blocking`) so it never stalls a Tokio worker, and
//! each ONNX session is serialized behind a `Mutex` (ORT's `run` needs `&mut`).
//!
//! The heavy numerical post-processing lives in [`super::face_geometry`] (plain
//! Rust, unit-tested); this module only wires it to ONNX Runtime.
//!
//! **Models are operator-provided at runtime, never committed.** `load` returns
//! an error (→ caller falls back to the no-op analyzer) if the ONNX Runtime
//! dylib or either model file is missing or incompatible — the server still
//! boots. The dylib is loaded via [`ort::init_from`] (a fallible path) rather
//! than ORT's lazy loader, which would `panic` on a missing library (fatal
//! under `panic = "abort"`).
use std::path::Path;
use std::sync::{Arc, Mutex};
use async_trait::async_trait;
use image::RgbImage;
use ort::session::Session;
use ort::value::Tensor;
use super::face_geometry as geom;
use crate::application::ports::face_ports::FaceAnalyzerPort;
use crate::common::errors::DomainError;
use crate::domain::entities::face::{BoundingBox, DetectedFace, EMBEDDING_DIM};
/// SCRFD pyramid strides for the 3- and 5-level model variants.
const STRIDES_3: [u32; 3] = [8, 16, 32];
const STRIDES_5: [u32; 5] = [8, 16, 32, 64, 128];
/// Discard faces smaller than this (original-image pixels) — embeddings of tiny
/// faces are unreliable.
const MIN_FACE_PX: f32 = 24.0;
/// Hard cap on faces processed per image (bounds work on crowd shots).
const MAX_FACES: usize = 64;
/// Output layout of an InsightFace SCRFD model, inferred from its output count.
#[derive(Clone, Copy)]
struct ScrfdLayout {
/// Feature-map count per output kind (3 for strides 8/16/32, 5 with 64/128).
fmc: usize,
num_anchors: u32,
use_kps: bool,
}
impl ScrfdLayout {
fn from_num_outputs(n: usize) -> Option<Self> {
match n {
6 => Some(Self {
fmc: 3,
num_anchors: 2,
use_kps: false,
}),
9 => Some(Self {
fmc: 3,
num_anchors: 2,
use_kps: true,
}),
10 => Some(Self {
fmc: 5,
num_anchors: 1,
use_kps: false,
}),
15 => Some(Self {
fmc: 5,
num_anchors: 1,
use_kps: true,
}),
_ => None,
}
}
fn strides(&self) -> &'static [u32] {
if self.fmc == 3 {
&STRIDES_3
} else {
&STRIDES_5
}
}
}
/// Where to find the runtime + models, plus detector knobs. Borrowed paths;
/// nothing is retained after [`OnnxFaceAnalyzer::load`].
pub struct OnnxLoadConfig<'a> {
/// Path to `libonnxruntime.{so,dylib,dll}`.
pub dylib: &'a Path,
/// SCRFD detector `.onnx`.
pub detector: &'a Path,
/// ArcFace embedder `.onnx`.
pub embedder: &'a Path,
pub det_size: u32,
pub det_threshold: f32,
pub nms_threshold: f32,
/// ORT intra-op threads (0 = let ONNX Runtime decide).
pub intra_threads: usize,
}
struct Inner {
detector: Mutex<Session>,
embedder: Mutex<Session>,
layout: ScrfdLayout,
det_size: u32,
det_threshold: f32,
nms_threshold: f32,
}
/// Real face analyzer. Cheap to clone (`Arc` inside).
#[derive(Clone)]
pub struct OnnxFaceAnalyzer {
inner: Arc<Inner>,
}
fn dom(e: impl std::fmt::Display) -> DomainError {
DomainError::internal_error("Faces", e.to_string())
}
fn build_session(path: &Path, intra_threads: usize) -> Result<Session, DomainError> {
let mut builder = Session::builder().map_err(dom)?;
if intra_threads > 0 {
builder = builder.with_intra_threads(intra_threads).map_err(dom)?;
}
builder.commit_from_file(path).map_err(dom)
}
impl OnnxFaceAnalyzer {
/// Load the ONNX Runtime dylib and both models. Returns an error (caller
/// falls back to the no-op analyzer) on any missing/incompatible artifact.
pub fn load(cfg: &OnnxLoadConfig<'_>) -> Result<Self, DomainError> {
// Fallible dylib load — populates ORT's global handle so later calls
// never hit the panicking lazy loader.
ort::init_from(cfg.dylib)
.map_err(|e| dom(format!("ONNX Runtime dylib: {e}")))?
.commit();
let detector = build_session(cfg.detector, cfg.intra_threads)?;
let embedder = build_session(cfg.embedder, cfg.intra_threads)?;
let n_out = detector.outputs().len();
let layout = ScrfdLayout::from_num_outputs(n_out).ok_or_else(|| {
dom(format!(
"detector has {n_out} outputs; expected an SCRFD model (6/9/10/15)"
))
})?;
if !layout.use_kps {
tracing::warn!(
target: "oxicloud::faces",
"SCRFD model has no landmark outputs; face alignment will be approximate"
);
}
tracing::info!(
target: "oxicloud::faces",
"ONNX face analyzer ready (detector {} outputs, embedder loaded, det_size={})",
n_out, cfg.det_size
);
Ok(Self {
inner: Arc::new(Inner {
detector: Mutex::new(detector),
embedder: Mutex::new(embedder),
layout,
det_size: cfg.det_size,
det_threshold: cfg.det_threshold,
nms_threshold: cfg.nms_threshold,
}),
})
}
}
impl Inner {
/// Full synchronous pipeline for one encoded image.
fn analyze_blocking(&self, image_bytes: &[u8]) -> Result<Vec<DetectedFace>, DomainError> {
let orig = image::load_from_memory(image_bytes)
.map_err(|e| dom(format!("decode image: {e}")))?
.to_rgb8();
let (w0, h0) = (orig.width(), orig.height());
if w0 == 0 || h0 == 0 {
return Ok(Vec::new());
}
let dets = self.detect(&orig)?;
let mut faces = Vec::new();
for det in dets.into_iter().take(MAX_FACES) {
let fw = det.bbox[2] - det.bbox[0];
let fh = det.bbox[3] - det.bbox[1];
if fw < MIN_FACE_PX || fh < MIN_FACE_PX {
continue;
}
let Some(embedding) = self.embed(&orig, &det)? else {
continue;
};
let aligned_quality = {
let inv = geom::similarity_transform_inverse(&det.kps, &geom::ARCFACE_TEMPLATE);
let aligned = geom::warp_to_aligned(&orig, &inv);
geom::laplacian_variance(&aligned)
};
let x = (det.bbox[0] / w0 as f32).clamp(0.0, 1.0);
let y = (det.bbox[1] / h0 as f32).clamp(0.0, 1.0);
let bw = (fw / w0 as f32).clamp(0.0, 1.0);
let bh = (fh / h0 as f32).clamp(0.0, 1.0);
faces.push(DetectedFace {
bbox: BoundingBox { x, y, w: bw, h: bh },
det_score: det.score,
quality: Some(aligned_quality),
embedding,
});
}
Ok(faces)
}
/// Run SCRFD and return detections in **original-image pixels**.
fn detect(&self, orig: &RgbImage) -> Result<Vec<geom::Detection>, DomainError> {
let det = self.det_size;
let (nw, nh, scale) = geom::letterbox(orig.width(), orig.height(), det);
let resized = image::imageops::resize(orig, nw, nh, image::imageops::FilterType::Triangle);
let mut canvas = RgbImage::new(det, det);
image::imageops::overlay(&mut canvas, &resized, 0, 0);
let input = geom::chw_normalized(&canvas, 127.5, 1.0 / 128.0);
let tensor =
Tensor::from_array(([1_i64, 3, det as i64, det as i64], input)).map_err(dom)?;
let layout = self.layout;
let total = layout.fmc * if layout.use_kps { 3 } else { 2 };
let raw: Vec<Vec<f32>> = {
let mut sess = self
.detector
.lock()
.map_err(|_| dom("detector mutex poisoned"))?;
let outputs = sess.run(ort::inputs![tensor]).map_err(dom)?;
(0..total)
.map(|i| {
outputs[i]
.try_extract_tensor::<f32>()
.map(|(_, data)| data.to_vec())
.map_err(dom)
})
.collect::<Result<_, _>>()?
};
let mut dets = Vec::new();
for (si, &stride) in layout.strides().iter().enumerate() {
let scores = &raw[si];
let bbox: Vec<f32> = raw[layout.fmc + si]
.iter()
.map(|v| v * stride as f32)
.collect();
let kps: Option<Vec<f32>> = if layout.use_kps {
Some(
raw[2 * layout.fmc + si]
.iter()
.map(|v| v * stride as f32)
.collect(),
)
} else {
None
};
let feat = det / stride;
geom::decode_stride(
scores,
&bbox,
kps.as_deref(),
stride,
feat,
feat,
layout.num_anchors,
self.det_threshold,
&mut dets,
);
}
// Scale detector-space coordinates back to the original image.
let inv_scale = if scale.abs() < 1e-9 { 1.0 } else { 1.0 / scale };
for d in &mut dets {
for v in &mut d.bbox {
*v *= inv_scale;
}
for k in &mut d.kps {
k[0] *= inv_scale;
k[1] *= inv_scale;
}
}
Ok(geom::nms(dets, self.nms_threshold))
}
/// Align one detection and run the ArcFace embedder. Returns `None` if the
/// embedder produces an unexpected output length.
fn embed(
&self,
orig: &RgbImage,
det: &geom::Detection,
) -> Result<Option<Vec<f32>>, DomainError> {
let inv = geom::similarity_transform_inverse(&det.kps, &geom::ARCFACE_TEMPLATE);
let aligned = geom::warp_to_aligned(orig, &inv);
let input = geom::chw_normalized(&aligned, 127.5, 1.0 / 127.5);
let size = geom::ALIGN_SIZE as i64;
let tensor = Tensor::from_array(([1_i64, 3, size, size], input)).map_err(dom)?;
let mut embedding: Vec<f32> = {
let mut sess = self
.embedder
.lock()
.map_err(|_| dom("embedder mutex poisoned"))?;
let outputs = sess.run(ort::inputs![tensor]).map_err(dom)?;
let (_, data) = outputs[0].try_extract_tensor::<f32>().map_err(dom)?;
data.to_vec()
};
if embedding.len() != EMBEDDING_DIM {
tracing::warn!(
target: "oxicloud::faces",
"embedder returned {} dims, expected {EMBEDDING_DIM}; skipping face",
embedding.len()
);
return Ok(None);
}
geom::l2_normalize(&mut embedding);
Ok(Some(embedding))
}
}
#[async_trait]
impl FaceAnalyzerPort for OnnxFaceAnalyzer {
fn is_ready(&self) -> bool {
true
}
async fn analyze(&self, image_bytes: &[u8]) -> Result<Vec<DetectedFace>, DomainError> {
let inner = self.inner.clone();
let bytes = image_bytes.to_vec();
tokio::task::spawn_blocking(move || inner.analyze_blocking(&bytes))
.await
.map_err(|e| dom(format!("inference task join: {e}")))?
}
}