Merge pull request #605 from AtalayaLabs/claude/performance-optimization-analysis-sisndc
ROUND4: Row-path allocs, drive cache, CalDAV parse, PROPFIND emit, N+1 hydration, Azure streaming, faces bound
This commit is contained in:
@@ -115,6 +115,11 @@ impl CalendarStoragePort for CalendarStorageAdapter {
|
||||
Ok(CalendarDto::from(calendar))
|
||||
}
|
||||
|
||||
async fn get_calendars_by_ids(&self, ids: &[Uuid]) -> Result<Vec<CalendarDto>, DomainError> {
|
||||
let calendars = self.calendar_repository.find_calendars_by_ids(ids).await?;
|
||||
Ok(calendars.into_iter().map(CalendarDto::from).collect())
|
||||
}
|
||||
|
||||
async fn list_calendars_by_owner(
|
||||
&self,
|
||||
owner_id: Uuid,
|
||||
|
||||
@@ -84,6 +84,15 @@ impl ContactStoragePort for ContactStorageAdapter {
|
||||
.await
|
||||
}
|
||||
|
||||
async fn get_address_books_by_ids(
|
||||
&self,
|
||||
ids: &[Uuid],
|
||||
) -> Result<Vec<AddressBook>, DomainError> {
|
||||
self.address_book_repository
|
||||
.get_address_books_by_ids(ids)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn get_public_address_books(&self) -> Result<Vec<AddressBook>, DomainError> {
|
||||
self.address_book_repository
|
||||
.get_public_address_books()
|
||||
|
||||
@@ -95,6 +95,11 @@ impl MusicStoragePort for MusicStorageAdapter {
|
||||
}
|
||||
}
|
||||
|
||||
async fn get_playlists_by_ids(&self, ids: &[Uuid]) -> Result<Vec<PlaylistDto>, DomainError> {
|
||||
let playlists = self.playlist_repository.find_playlists_by_ids(ids).await?;
|
||||
Ok(playlists.into_iter().map(PlaylistDto::from).collect())
|
||||
}
|
||||
|
||||
async fn list_playlists_by_owner(
|
||||
&self,
|
||||
owner_id: Uuid,
|
||||
|
||||
@@ -110,6 +110,45 @@ impl AddressBookRepository for AddressBookPgRepository {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn get_address_books_by_ids(
|
||||
&self,
|
||||
ids: &[Uuid],
|
||||
) -> AddressBookRepositoryResult<Vec<AddressBook>> {
|
||||
if ids.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let rows = sqlx::query(
|
||||
r#"
|
||||
SELECT id, name, owner_id, description, color, is_public, created_at, updated_at
|
||||
FROM carddav.address_books
|
||||
WHERE id = ANY($1)
|
||||
"#,
|
||||
)
|
||||
.bind(ids)
|
||||
.fetch_all(&*self.pool)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
DomainError::database_error(format!("Failed to get address books by ids: {}", e))
|
||||
})?;
|
||||
|
||||
Ok(rows
|
||||
.iter()
|
||||
.map(|row| {
|
||||
let owner_id: Uuid = row.get("owner_id");
|
||||
AddressBook::from_raw(
|
||||
row.get("id"),
|
||||
row.get("name"),
|
||||
owner_id.to_string(),
|
||||
row.get("description"),
|
||||
row.get("color"),
|
||||
row.get("is_public"),
|
||||
row.get("created_at"),
|
||||
row.get("updated_at"),
|
||||
)
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
|
||||
async fn get_address_book_by_id(
|
||||
&self,
|
||||
id: &Uuid,
|
||||
|
||||
@@ -138,6 +138,42 @@ impl CalendarRepository for CalendarPgRepository {
|
||||
Ok(calendar)
|
||||
}
|
||||
|
||||
async fn find_calendars_by_ids(&self, ids: &[Uuid]) -> CalendarRepositoryResult<Vec<Calendar>> {
|
||||
if ids.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let rows = sqlx::query(
|
||||
r#"
|
||||
SELECT id, name, owner_id, description, color, is_public, created_at, updated_at
|
||||
FROM caldav.calendars
|
||||
WHERE id = ANY($1)
|
||||
"#,
|
||||
)
|
||||
.bind(ids)
|
||||
.fetch_all(&*self.pool)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
DomainError::database_error(format!("Failed to get calendars by ids: {}", e))
|
||||
})?;
|
||||
|
||||
rows.iter()
|
||||
.map(|row| {
|
||||
Calendar::with_id(
|
||||
row.get("id"),
|
||||
row.get("name"),
|
||||
row.get("owner_id"),
|
||||
row.get("description"),
|
||||
row.get("color"),
|
||||
row.get("created_at"),
|
||||
row.get("updated_at"),
|
||||
)
|
||||
.map_err(|e| {
|
||||
DomainError::database_error(format!("Failed to create calendar object: {}", e))
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
async fn list_calendars_by_owner(
|
||||
&self,
|
||||
owner_id: Uuid,
|
||||
|
||||
@@ -41,6 +41,27 @@ pub struct DrivePgRepository {
|
||||
/// provisioning idempotency check (`NotFound` → create) always sees
|
||||
/// the live table.
|
||||
default_drive_cache: Cache<Uuid, DriveWithRootName>,
|
||||
/// caller_id → every drive the caller can read (the full
|
||||
/// role_grants ⋈ drives ⋈ folders join of [`list_readable_by`],
|
||||
/// including the transitive-group expansion).
|
||||
///
|
||||
/// Re-resolved before this cache existed on EVERY native `/webdav`
|
||||
/// request that names an explicit drive selector (all verbs; MOVE
|
||||
/// and COPY twice), plus per-request in search, trash listing and
|
||||
/// the `GET /api/drives` picker — the heaviest per-request query
|
||||
/// left on the DAV path after CHROOT-CACHE. Concurrent misses are
|
||||
/// coalesced (`try_get_with`), errors are never cached.
|
||||
///
|
||||
/// Freshness: every membership/lifecycle mutation that flows
|
||||
/// through this repository or `DriveManagementService` invalidates
|
||||
/// explicitly (per-user when the subject is a User, whole cache for
|
||||
/// Group subjects, whose transitive membership is not resolvable
|
||||
/// here). Residual staleness — a root-folder rename or a grant
|
||||
/// written by a path that can't reach this cache — is bounded by
|
||||
/// the same 30 s TTL the sibling caches accept; actual permission
|
||||
/// enforcement is unaffected (the ACL engine re-checks per
|
||||
/// operation with its own invalidation).
|
||||
readable_cache: Cache<Uuid, Arc<Vec<DriveWithRootName>>>,
|
||||
}
|
||||
|
||||
impl DrivePgRepository {
|
||||
@@ -51,9 +72,27 @@ impl DrivePgRepository {
|
||||
.max_capacity(DEFAULT_DRIVE_CACHE_CAPACITY)
|
||||
.time_to_live(DEFAULT_DRIVE_CACHE_TTL)
|
||||
.build(),
|
||||
readable_cache: Cache::builder()
|
||||
.max_capacity(DEFAULT_DRIVE_CACHE_CAPACITY)
|
||||
.time_to_live(DEFAULT_DRIVE_CACHE_TTL)
|
||||
.build(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop the cached readable-drive list for one user (their grant set
|
||||
/// changed: membership write, personal-drive provisioning, …).
|
||||
pub async fn invalidate_readable_for_user(&self, user_id: Uuid) {
|
||||
self.readable_cache.invalidate(&user_id).await;
|
||||
}
|
||||
|
||||
/// Drop every cached readable-drive list. Used when the affected
|
||||
/// user set is unknown at this layer: group-subject grants, drive
|
||||
/// deletion, policy edits. All are admin-rare; repopulation costs
|
||||
/// one join per active caller.
|
||||
pub fn invalidate_readable_all(&self) {
|
||||
self.readable_cache.invalidate_all();
|
||||
}
|
||||
|
||||
fn map_sqlx_err(context: &'static str, e: sqlx::Error) -> DriveRepositoryError {
|
||||
if let sqlx::Error::Database(ref dberr) = e
|
||||
&& let Some(code) = dberr.code()
|
||||
@@ -112,6 +151,63 @@ impl DrivePgRepository {
|
||||
dwr.caller_role = role_str.as_deref().and_then(Role::parse);
|
||||
Ok(dwr)
|
||||
}
|
||||
|
||||
/// The uncached grants join behind [`DriveRepository::list_readable_by`].
|
||||
///
|
||||
/// Joining role_grants → drives → folders returns every drive the
|
||||
/// caller can read, paired with its display name. Group
|
||||
/// memberships (direct + transitive) are expanded inline by
|
||||
/// `storage.caller_group_ids($caller)` — no Rust-side ceremony.
|
||||
///
|
||||
/// ORDER BY puts default drives first (so the picker UI doesn't
|
||||
/// need a follow-up sort), then alphabetical by name. GROUP BY
|
||||
/// collapses duplicate role_grants on the same drive (direct +
|
||||
/// group-mediated) and sidesteps PostgreSQL's "ORDER BY
|
||||
/// expression must appear in select list" rule that SELECT
|
||||
/// DISTINCT imposes.
|
||||
/// `MIN(g.role)` picks the caller's strongest role on each drive:
|
||||
/// `storage.grant_role` is declared `owner → viewer` (strongest →
|
||||
/// weakest), so MIN returns the strongest. Cast `::text` matches
|
||||
/// the codebase convention for reading enum columns into Rust
|
||||
/// (see `pg_acl_engine.rs`); `Role::parse` handles the trip back.
|
||||
async fn query_readable_by(
|
||||
&self,
|
||||
caller_id: Uuid,
|
||||
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
|
||||
let rows = sqlx::query(
|
||||
r#"
|
||||
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
|
||||
d.quota_bytes, d.used_bytes, d.policies,
|
||||
d.created_at, d.updated_at,
|
||||
f.name AS root_folder_name,
|
||||
MIN(g.role)::text AS caller_role
|
||||
FROM storage.drives d
|
||||
JOIN storage.folders f ON f.id = d.root_folder_id
|
||||
JOIN storage.role_grants g
|
||||
ON g.resource_type = 'drive'
|
||||
AND g.resource_id = d.id
|
||||
WHERE (
|
||||
(g.subject_type = 'user' AND g.subject_id = $1)
|
||||
OR (g.subject_type = 'group' AND g.subject_id IN
|
||||
(SELECT storage.caller_group_ids($1)))
|
||||
)
|
||||
AND (g.expires_at IS NULL OR g.expires_at > NOW())
|
||||
GROUP BY d.id, d.kind, d.default_for_user, d.root_folder_id,
|
||||
d.quota_bytes, d.used_bytes, d.policies,
|
||||
d.created_at, d.updated_at, f.name
|
||||
ORDER BY (d.default_for_user IS NULL) ASC,
|
||||
LOWER(f.name) ASC
|
||||
"#,
|
||||
)
|
||||
.bind(caller_id)
|
||||
.fetch_all(self.pool.as_ref())
|
||||
.await
|
||||
.map_err(|e| Self::map_sqlx_err("list_readable_by", e))?;
|
||||
|
||||
rows.iter()
|
||||
.map(Self::row_to_drive_with_name_and_role)
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
@@ -239,6 +335,8 @@ impl DriveRepository for DrivePgRepository {
|
||||
// Drop any cached default-drive resolution for this user (a stale
|
||||
// NotFound is never cached, but be explicit about the write path).
|
||||
self.default_drive_cache.invalidate(&owner_id).await;
|
||||
// The owner gained a drive — their readable list changed too.
|
||||
self.invalidate_readable_for_user(owner_id).await;
|
||||
|
||||
Self::row_to_drive_with_name(&row)
|
||||
}
|
||||
@@ -347,6 +445,16 @@ impl DriveRepository for DrivePgRepository {
|
||||
.await
|
||||
.map_err(|e| Self::map_sqlx_err("create_shared_drive_atomic.commit", e))?;
|
||||
|
||||
// The owner grant written above changes the grantee's readable
|
||||
// list. User subjects invalidate precisely; Group subjects fall
|
||||
// back to a full clear (transitive members unknown here).
|
||||
match owner_subject {
|
||||
crate::domain::services::authorization::Subject::User(uid) => {
|
||||
self.invalidate_readable_for_user(uid).await;
|
||||
}
|
||||
_ => self.invalidate_readable_all(),
|
||||
}
|
||||
|
||||
Self::row_to_drive_with_name(&row)
|
||||
}
|
||||
|
||||
@@ -425,10 +533,11 @@ impl DriveRepository for DrivePgRepository {
|
||||
tx.commit()
|
||||
.await
|
||||
.map_err(|e| Self::map_sqlx_err("delete_atomic.commit", e))?;
|
||||
// We only have the drive id here; the cache is keyed by user.
|
||||
// Deletion is rare — clearing the whole cache is the simple,
|
||||
// We only have the drive id here; the caches are keyed by user.
|
||||
// Deletion is rare — clearing them whole is the simple,
|
||||
// always-correct move (repopulates at one query per active user).
|
||||
self.default_drive_cache.invalidate_all();
|
||||
self.invalidate_readable_all();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -513,55 +622,21 @@ impl DriveRepository for DrivePgRepository {
|
||||
&self,
|
||||
caller_id: Uuid,
|
||||
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
|
||||
// Joining role_grants → drives → folders returns every drive the
|
||||
// caller can read, paired with its display name. Group
|
||||
// memberships (direct + transitive) are expanded inline by
|
||||
// `storage.caller_group_ids($caller)` — no Rust-side ceremony.
|
||||
//
|
||||
// ORDER BY puts default drives first (so the picker UI doesn't
|
||||
// need a follow-up sort), then alphabetical by name. GROUP BY
|
||||
// collapses duplicate role_grants on the same drive (direct +
|
||||
// group-mediated) and sidesteps PostgreSQL's "ORDER BY
|
||||
// expression must appear in select list" rule that SELECT
|
||||
// DISTINCT imposes.
|
||||
// `MIN(g.role)` picks the caller's strongest role on each drive:
|
||||
// `storage.grant_role` is declared `owner → viewer` (strongest →
|
||||
// weakest), so MIN returns the strongest. Cast `::text` matches
|
||||
// the codebase convention for reading enum columns into Rust
|
||||
// (see `pg_acl_engine.rs`); `Role::parse` handles the trip back.
|
||||
let rows = sqlx::query(
|
||||
r#"
|
||||
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
|
||||
d.quota_bytes, d.used_bytes, d.policies,
|
||||
d.created_at, d.updated_at,
|
||||
f.name AS root_folder_name,
|
||||
MIN(g.role)::text AS caller_role
|
||||
FROM storage.drives d
|
||||
JOIN storage.folders f ON f.id = d.root_folder_id
|
||||
JOIN storage.role_grants g
|
||||
ON g.resource_type = 'drive'
|
||||
AND g.resource_id = d.id
|
||||
WHERE (
|
||||
(g.subject_type = 'user' AND g.subject_id = $1)
|
||||
OR (g.subject_type = 'group' AND g.subject_id IN
|
||||
(SELECT storage.caller_group_ids($1)))
|
||||
)
|
||||
AND (g.expires_at IS NULL OR g.expires_at > NOW())
|
||||
GROUP BY d.id, d.kind, d.default_for_user, d.root_folder_id,
|
||||
d.quota_bytes, d.used_bytes, d.policies,
|
||||
d.created_at, d.updated_at, f.name
|
||||
ORDER BY (d.default_for_user IS NULL) ASC,
|
||||
LOWER(f.name) ASC
|
||||
"#,
|
||||
)
|
||||
.bind(caller_id)
|
||||
.fetch_all(self.pool.as_ref())
|
||||
.await
|
||||
.map_err(|e| Self::map_sqlx_err("list_readable_by", e))?;
|
||||
|
||||
rows.iter()
|
||||
.map(Self::row_to_drive_with_name_and_role)
|
||||
.collect()
|
||||
// Serve from the per-user cache; concurrent misses for the same
|
||||
// caller are coalesced into one join (`try_get_with`), and errors
|
||||
// are never cached. See the `readable_cache` field docs for the
|
||||
// freshness/invalidation contract.
|
||||
let cached = self
|
||||
.readable_cache
|
||||
.try_get_with(caller_id, async move {
|
||||
self.query_readable_by(caller_id).await.map(Arc::new)
|
||||
})
|
||||
.await
|
||||
.map_err(|e: Arc<DriveRepositoryError>| {
|
||||
Arc::try_unwrap(e)
|
||||
.unwrap_or_else(|shared| DriveRepositoryError::StorageError(shared.to_string()))
|
||||
})?;
|
||||
Ok((*cached).clone())
|
||||
}
|
||||
|
||||
async fn list_all(&self) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
|
||||
@@ -717,9 +792,10 @@ impl DriveRepository for DrivePgRepository {
|
||||
.ok_or_else(|| DriveRepositoryError::NotFound(drive_id.to_string()))?
|
||||
.0;
|
||||
// Policy edits must not serve a stale `policies` bag from the
|
||||
// default-drive cache (keyed by user, and we only have the drive
|
||||
// id) — clear it; policy edits are admin-rare.
|
||||
// user-keyed caches (we only have the drive id) — clear both;
|
||||
// policy edits are admin-rare.
|
||||
self.default_drive_cache.invalidate_all();
|
||||
self.invalidate_readable_all();
|
||||
Ok(crate::domain::entities::drive::DrivePolicies::from_value(
|
||||
&raw,
|
||||
))
|
||||
|
||||
@@ -413,14 +413,6 @@ impl FileBlobReadRepository {
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a `StoragePath` from the materialized folder path + file name.
|
||||
fn make_file_path(folder_path: Option<&str>, file_name: &str) -> StoragePath {
|
||||
match folder_path {
|
||||
Some(fp) if !fp.is_empty() => StoragePath::from_string(&format!("{fp}/{file_name}")),
|
||||
_ => StoragePath::from_string(file_name),
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn row_to_file(
|
||||
id: String,
|
||||
@@ -435,11 +427,10 @@ impl FileBlobReadRepository {
|
||||
created_by: Option<Uuid>,
|
||||
updated_by: Option<Uuid>,
|
||||
) -> Result<File, DomainError> {
|
||||
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
|
||||
File::with_timestamps_blob_hash_and_provenance(
|
||||
File::from_materialized_row(
|
||||
id,
|
||||
name,
|
||||
storage_path,
|
||||
folder_path.as_deref(),
|
||||
size as u64,
|
||||
mime_type,
|
||||
folder_id,
|
||||
@@ -930,7 +921,7 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("path: {e}")))?
|
||||
.ok_or_else(|| DomainError::not_found("File", id))?;
|
||||
|
||||
Ok(Self::make_file_path(row.1.as_deref(), &row.0))
|
||||
Ok(StoragePath::from_folder_and_name(row.1.as_deref(), &row.0).0)
|
||||
}
|
||||
|
||||
async fn get_parent_folder_id(
|
||||
|
||||
@@ -17,7 +17,6 @@ use crate::application::dtos::display_helpers::category_order_for;
|
||||
use crate::application::ports::storage_ports::{CopyFolderTreeResult, FileWritePort};
|
||||
use crate::common::errors::DomainError;
|
||||
use crate::domain::entities::file::File;
|
||||
use crate::domain::services::path_service::StoragePath;
|
||||
|
||||
use super::transaction_utils::retry_on_deadlock;
|
||||
use crate::infrastructure::services::dedup_service::DedupService;
|
||||
@@ -61,14 +60,6 @@ impl FileBlobWriteRepository {
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a `StoragePath` from the materialized folder path + file name.
|
||||
fn make_file_path(folder_path: Option<&str>, file_name: &str) -> StoragePath {
|
||||
match folder_path {
|
||||
Some(fp) if !fp.is_empty() => StoragePath::from_string(&format!("{fp}/{file_name}")),
|
||||
_ => StoragePath::from_string(file_name),
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up the materialized folder path. O(1) — no recursive CTE.
|
||||
async fn lookup_folder_path(
|
||||
&self,
|
||||
@@ -108,11 +99,10 @@ impl FileBlobWriteRepository {
|
||||
created_by: Option<Uuid>,
|
||||
updated_by: Option<Uuid>,
|
||||
) -> Result<File, DomainError> {
|
||||
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
|
||||
File::with_timestamps_blob_hash_and_provenance(
|
||||
File::from_materialized_row(
|
||||
id,
|
||||
name,
|
||||
storage_path,
|
||||
folder_path.as_deref(),
|
||||
size as u64,
|
||||
mime_type,
|
||||
folder_id,
|
||||
|
||||
@@ -142,11 +142,10 @@ impl FolderDbRepository {
|
||||
created_by: Option<Uuid>,
|
||||
updated_by: Option<Uuid>,
|
||||
) -> Result<Folder, DomainError> {
|
||||
let storage_path = StoragePath::from_string(&path);
|
||||
Folder::with_timestamps_tree_and_provenance(
|
||||
Folder::from_materialized_row(
|
||||
id,
|
||||
name,
|
||||
storage_path,
|
||||
path,
|
||||
parent_id,
|
||||
drive_id,
|
||||
created_at as u64,
|
||||
|
||||
@@ -180,6 +180,35 @@ impl PlaylistRepository for PlaylistPgRepository {
|
||||
.map_err(|e| DomainError::new(ErrorKind::InternalError, "Playlist", e.to_string()))
|
||||
}
|
||||
|
||||
async fn find_playlists_by_ids(&self, ids: &[Uuid]) -> PlaylistRepositoryResult<Vec<Playlist>> {
|
||||
if ids.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let rows = sqlx::query_as::<_, PlaylistRow>(
|
||||
"SELECT id, name, description, owner_id, is_public, cover_file_id, created_at, updated_at FROM audio.playlists WHERE id = ANY($1)",
|
||||
)
|
||||
.bind(ids)
|
||||
.fetch_all(&*self.pool)
|
||||
.await
|
||||
.map_err(|e| DomainError::database_error(format!("Failed to find playlists: {}", e)))?;
|
||||
|
||||
rows.into_iter()
|
||||
.map(|row| {
|
||||
Playlist::with_id(
|
||||
row.id,
|
||||
row.name,
|
||||
row.description,
|
||||
row.owner_id,
|
||||
row.is_public,
|
||||
row.cover_file_id,
|
||||
row.created_at,
|
||||
row.updated_at,
|
||||
)
|
||||
.map_err(|e| DomainError::new(ErrorKind::InternalError, "Playlist", e.to_string()))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
async fn list_playlists_by_owner(
|
||||
&self,
|
||||
owner_id: Uuid,
|
||||
|
||||
@@ -9,7 +9,7 @@ use std::pin::Pin;
|
||||
use azure_storage::StorageCredentials;
|
||||
use azure_storage_blobs::prelude::*;
|
||||
use bytes::Bytes;
|
||||
use futures::StreamExt;
|
||||
use futures::{StreamExt, TryStreamExt};
|
||||
use tokio::fs;
|
||||
|
||||
use crate::application::ports::blob_storage_ports::{
|
||||
@@ -33,8 +33,21 @@ impl AzureBlobBackend {
|
||||
StorageCredentials::access_key(&config.account_name, config.account_key.clone())
|
||||
};
|
||||
|
||||
let container_client = ClientBuilder::new(&config.account_name, credentials)
|
||||
.container_client(&config.container);
|
||||
// Custom endpoint (Azurite emulator / private deployment /
|
||||
// benches) mirrors S3's `endpoint_url`; default is the public
|
||||
// cloud URL derived from the account name.
|
||||
let container_client = match &config.endpoint_url {
|
||||
Some(uri) => ClientBuilder::with_location(
|
||||
azure_storage::CloudLocation::Custom {
|
||||
account: config.account_name.clone(),
|
||||
uri: uri.trim_end_matches('/').to_string(),
|
||||
},
|
||||
credentials,
|
||||
)
|
||||
.container_client(&config.container),
|
||||
None => ClientBuilder::new(&config.account_name, credentials)
|
||||
.container_client(&config.container),
|
||||
};
|
||||
|
||||
Self {
|
||||
container_client,
|
||||
@@ -169,29 +182,46 @@ impl BlobStorageBackend for AzureBlobBackend {
|
||||
Box::pin(async move {
|
||||
let client = self.blob_client(&hash);
|
||||
|
||||
let mut result_data: Vec<u8> = Vec::new();
|
||||
let mut stream = client.get().into_stream();
|
||||
|
||||
while let Some(response) = stream.next().await {
|
||||
let response = response.map_err(|e| {
|
||||
DomainError::new(
|
||||
// The old implementation drained the ENTIRE blob into one
|
||||
// `Vec<u8>` before yielding a single mega-chunk — whole-blob
|
||||
// RAM residency per reader, and with `read_prefetch() = 8`
|
||||
// up to 8 entire chunk-blobs resident at once during CDC
|
||||
// reassembly. Now the SDK's page/body streams forward
|
||||
// directly. The FIRST page is still awaited eagerly so a
|
||||
// missing blob surfaces as the same up-front NotFound the
|
||||
// old code produced; later pages/chunks map to io::Error
|
||||
// items like every other backend's stream.
|
||||
let mut pages = client.get().into_stream();
|
||||
let first = match pages.next().await {
|
||||
Some(Ok(response)) => response,
|
||||
Some(Err(e)) => {
|
||||
return Err(DomainError::new(
|
||||
ErrorKind::NotFound,
|
||||
"Azure",
|
||||
format!("Failed to get blob {hash}: {e}"),
|
||||
)
|
||||
})?;
|
||||
let mut body = response.data;
|
||||
while let Some(chunk) = body.next().await {
|
||||
let chunk = chunk.map_err(|e| {
|
||||
DomainError::internal_error("Azure", format!("Stream read error: {e}"))
|
||||
})?;
|
||||
result_data.extend_from_slice(&chunk);
|
||||
));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
let empty: BlobStream =
|
||||
Box::pin(futures::stream::once(async move { Ok(Bytes::new()) }));
|
||||
return Ok(empty);
|
||||
}
|
||||
};
|
||||
|
||||
let stream: BlobStream = Box::pin(futures::stream::once(async move {
|
||||
Ok(Bytes::from(result_data))
|
||||
}));
|
||||
let first_body = first.data.map(|chunk| {
|
||||
chunk.map_err(|e| std::io::Error::other(format!("Stream read error: {e}")))
|
||||
});
|
||||
let tail = pages
|
||||
.map(|page| match page {
|
||||
Ok(response) => Ok(response.data.map(|chunk| {
|
||||
chunk.map_err(|e| std::io::Error::other(format!("Stream read error: {e}")))
|
||||
})),
|
||||
Err(e) => Err(std::io::Error::other(format!(
|
||||
"Failed to get blob page: {e}"
|
||||
))),
|
||||
})
|
||||
.try_flatten();
|
||||
let stream: BlobStream = Box::pin(first_body.chain(tail));
|
||||
Ok(stream)
|
||||
})
|
||||
}
|
||||
@@ -212,32 +242,42 @@ impl BlobStorageBackend for AzureBlobBackend {
|
||||
None => azure_core::request_options::Range::new(start, u64::MAX),
|
||||
};
|
||||
|
||||
let mut result_data: Vec<u8> = Vec::new();
|
||||
let mut stream = client.get().range(range).into_stream();
|
||||
|
||||
while let Some(response) = stream.next().await {
|
||||
let response = response.map_err(|e| {
|
||||
DomainError::new(
|
||||
// Same forwarding shape as `get_blob_stream` — a ranged read
|
||||
// doubly so: the caller explicitly asked NOT to pay for the
|
||||
// whole blob, yet the old code buffered the full range.
|
||||
let mut pages = client.get().range(range).into_stream();
|
||||
let first = match pages.next().await {
|
||||
Some(Ok(response)) => response,
|
||||
Some(Err(e)) => {
|
||||
return Err(DomainError::new(
|
||||
ErrorKind::NotFound,
|
||||
"Azure",
|
||||
format!("Failed to get blob range {hash}: {e}"),
|
||||
)
|
||||
})?;
|
||||
let mut body = response.data;
|
||||
while let Some(chunk) = body.next().await {
|
||||
let chunk = chunk.map_err(|e| {
|
||||
DomainError::internal_error(
|
||||
"Azure",
|
||||
format!("Stream range read error: {e}"),
|
||||
)
|
||||
})?;
|
||||
result_data.extend_from_slice(&chunk);
|
||||
));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
let empty: BlobStream =
|
||||
Box::pin(futures::stream::once(async move { Ok(Bytes::new()) }));
|
||||
return Ok(empty);
|
||||
}
|
||||
};
|
||||
|
||||
let stream: BlobStream = Box::pin(futures::stream::once(async move {
|
||||
Ok(Bytes::from(result_data))
|
||||
}));
|
||||
let first_body = first.data.map(|chunk| {
|
||||
chunk.map_err(|e| std::io::Error::other(format!("Stream range read error: {e}")))
|
||||
});
|
||||
let tail = pages
|
||||
.map(|page| match page {
|
||||
Ok(response) => Ok(response.data.map(|chunk| {
|
||||
chunk.map_err(|e| {
|
||||
std::io::Error::other(format!("Stream range read error: {e}"))
|
||||
})
|
||||
})),
|
||||
Err(e) => Err(std::io::Error::other(format!(
|
||||
"Failed to get blob range page: {e}"
|
||||
))),
|
||||
})
|
||||
.try_flatten();
|
||||
let stream: BlobStream = Box::pin(first_body.chain(tail));
|
||||
Ok(stream)
|
||||
})
|
||||
}
|
||||
|
||||
@@ -28,11 +28,35 @@ fn is_image(content_type: &str) -> bool {
|
||||
content_type.starts_with("image/")
|
||||
}
|
||||
|
||||
/// Concurrent index-task budget. Env override
|
||||
/// `OXICLOUD_FACES_INDEX_CONCURRENCY`, else the effective core count —
|
||||
/// each task is a full-image read + decode + ONNX inference, so more
|
||||
/// permits than cores only adds RAM pressure, not throughput.
|
||||
fn max_concurrent_index() -> usize {
|
||||
std::env::var("OXICLOUD_FACES_INDEX_CONCURRENCY")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.filter(|&n: &usize| n > 0)
|
||||
.unwrap_or_else(|| {
|
||||
std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(2)
|
||||
})
|
||||
}
|
||||
|
||||
pub struct FaceIndexingService {
|
||||
pool: Arc<PgPool>,
|
||||
repo: Arc<FacePgRepository>,
|
||||
analyzer: Arc<dyn FaceAnalyzerPort>,
|
||||
blob_root: PathBuf,
|
||||
/// Bounds concurrent indexing tasks. The lifecycle hooks spawn one
|
||||
/// task per uploaded/copied image with no ceiling, so a bulk upload
|
||||
/// used to fan out N simultaneous full-image reads + decodes +
|
||||
/// inferences — peak RSS N × image size plus CPU thrash. Same
|
||||
/// invariant as `ThumbnailService::decode_semaphore`: the permit is
|
||||
/// acquired BEFORE the blob read, so peak memory is
|
||||
/// `permits × image size` regardless of upload concurrency.
|
||||
index_semaphore: Arc<tokio::sync::Semaphore>,
|
||||
}
|
||||
|
||||
impl FaceIndexingService {
|
||||
@@ -43,6 +67,7 @@ impl FaceIndexingService {
|
||||
repo,
|
||||
analyzer,
|
||||
blob_root,
|
||||
index_semaphore: Arc::new(tokio::sync::Semaphore::new(max_concurrent_index())),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -60,7 +85,15 @@ impl FaceIndexingService {
|
||||
let repo = self.repo.clone();
|
||||
let analyzer = self.analyzer.clone();
|
||||
let blob_path = self.blob_path(&blob_hash);
|
||||
let semaphore = self.index_semaphore.clone();
|
||||
tokio::spawn(async move {
|
||||
// Queue behind the concurrency budget BEFORE touching the
|
||||
// blob — excess tasks wait holding only this tiny future,
|
||||
// not a decoded image.
|
||||
let _permit = semaphore
|
||||
.acquire_owned()
|
||||
.await
|
||||
.expect("face index semaphore never closes");
|
||||
if delete_first {
|
||||
let _ = repo.delete_faces_for_file(file_id).await;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user