perf: optimize hot paths — batch concurrency, pagination, sorting, transcoding, folder ops

- batch_operations.rs: replace join_all with buffer_unordered, Arc<str> for shared IDs, remove redundant clones and dead Semaphore
- folder_db_repository.rs: use COUNT(*) OVER() for single-query pagination; UPDATE RETURNING for rename/move (eliminates extra SELECTs)
- folder_service.rs: remove StorageTransaction wrapper from rename/move — direct repo call (4→2 and 5→3 queries)
- search_service.rs: replace sort_by(to_lowercase) with sort_by_cached_key (N vs 2·N·log₂N allocations)
- image_transcode_service.rs: dynamic rayon pool sizing via available_parallelism() instead of hardcoded 2 threads
- Remove dead transactions module (zero consumers after folder_service refactor)
This commit is contained in:
Dionisio
2026-03-02 01:30:34 +01:00
parent 641b6853ad
commit b06206207f
9 changed files with 169 additions and 1311 deletions
@@ -258,6 +258,9 @@ impl FolderRepository for FolderDbRepository {
.collect()
}
/// Paginated folder listing — single query with `COUNT(*) OVER()` window
/// function so the total matching count comes back alongside the data rows,
/// eliminating a separate COUNT round-trip.
async fn list_folders_paginated(
&self,
parent_id: Option<&str>,
@@ -265,34 +268,14 @@ impl FolderRepository for FolderDbRepository {
limit: usize,
include_total: bool,
) -> Result<(Vec<Folder>, Option<usize>), DomainError> {
let total = if include_total {
let count: i64 = if let Some(pid) = parent_id {
sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.folders WHERE parent_id = $1::uuid AND NOT is_trashed",
)
.bind(pid)
.fetch_one(self.pool())
.await
} else {
sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.folders WHERE parent_id IS NULL AND NOT is_trashed",
)
.fetch_one(self.pool())
.await
}
.map_err(|e| DomainError::internal_error("FolderDb", format!("count: {e}")))?;
Some(count as usize)
} else {
None
};
let rows: Vec<(String, String, String, Option<String>, String, i64, i64)> =
let rows: Vec<(String, String, String, Option<String>, String, i64, i64, i64)> =
if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id = $1::uuid AND NOT is_trashed
ORDER BY name
@@ -309,7 +292,8 @@ impl FolderRepository for FolderDbRepository {
r#"
SELECT id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id IS NULL AND NOT is_trashed
ORDER BY name
@@ -323,15 +307,24 @@ impl FolderRepository for FolderDbRepository {
}
.map_err(|e| DomainError::internal_error("FolderDb", format!("paginate: {e}")))?;
// total_count is identical in every row; 0 when the result set is empty.
let total = if include_total {
Some(rows.first().map_or(0, |r| r.7) as usize)
} else {
None
};
let folders: Result<Vec<Folder>, DomainError> = rows
.into_iter()
.map(|(id, name, path, pid, uid, ca, ma)| {
.map(|(id, name, path, pid, uid, ca, ma, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma)
})
.collect();
Ok((folders?, total))
}
/// Paginated folder listing filtered by owner — single query with
/// `COUNT(*) OVER()` to avoid a separate COUNT round-trip.
async fn list_folders_by_owner_paginated(
&self,
parent_id: Option<&str>,
@@ -340,36 +333,14 @@ impl FolderRepository for FolderDbRepository {
limit: usize,
include_total: bool,
) -> Result<(Vec<Folder>, Option<usize>), DomainError> {
let total = if include_total {
let count: i64 = if let Some(pid) = parent_id {
sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.folders WHERE parent_id = $1::uuid AND user_id = $2 AND NOT is_trashed",
)
.bind(pid)
.bind(owner_id)
.fetch_one(self.pool())
.await
} else {
sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.folders WHERE parent_id IS NULL AND user_id = $1 AND NOT is_trashed",
)
.bind(owner_id)
.fetch_one(self.pool())
.await
}
.map_err(|e| DomainError::internal_error("FolderDb", format!("count_by_owner: {e}")))?;
Some(count as usize)
} else {
None
};
let rows: Vec<(String, String, String, Option<String>, String, i64, i64)> =
let rows: Vec<(String, String, String, Option<String>, String, i64, i64, i64)> =
if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id = $1::uuid AND user_id = $2 AND NOT is_trashed
ORDER BY name
@@ -387,7 +358,8 @@ impl FolderRepository for FolderDbRepository {
r#"
SELECT id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id IS NULL AND user_id = $1 AND NOT is_trashed
ORDER BY name
@@ -404,9 +376,15 @@ impl FolderRepository for FolderDbRepository {
DomainError::internal_error("FolderDb", format!("paginate_by_owner: {e}"))
})?;
let total = if include_total {
Some(rows.first().map_or(0, |r| r.7) as usize)
} else {
None
};
let folders: Result<Vec<Folder>, DomainError> = rows
.into_iter()
.map(|(id, name, path, pid, uid, ca, ma)| {
.map(|(id, name, path, pid, uid, ca, ma, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma)
})
.collect();
@@ -417,16 +395,19 @@ impl FolderRepository for FolderDbRepository {
// The BEFORE UPDATE trigger recomputes path/lpath for this row;
// the AFTER UPDATE cascade trigger then batch-updates all
// descendants in a single UPDATE using the GiST lpath index.
sqlx::query(
let row = sqlx::query_as::<_, (String, String, String, Option<String>, String, i64, i64)>(
r#"
UPDATE storage.folders
SET name = $1, updated_at = NOW()
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
"#,
)
.bind(&new_name)
.bind(id)
.execute(self.pool())
.fetch_optional(self.pool())
.await
.map_err(|e| {
if let sqlx::Error::Database(ref db_err) = e
@@ -435,9 +416,10 @@ impl FolderRepository for FolderDbRepository {
return DomainError::already_exists("Folder", format!("{new_name} already exists"));
}
DomainError::internal_error("FolderDb", format!("rename: {e}"))
})?;
})?
.ok_or_else(|| DomainError::not_found("Folder", id))?;
self.get_folder(id).await
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6)
}
async fn move_folder(
@@ -448,20 +430,24 @@ impl FolderRepository for FolderDbRepository {
// The BEFORE UPDATE trigger recomputes path/lpath for this row;
// the AFTER UPDATE cascade trigger then batch-updates all
// descendants in a single UPDATE using the GiST lpath index.
sqlx::query(
let row = sqlx::query_as::<_, (String, String, String, Option<String>, String, i64, i64)>(
r#"
UPDATE storage.folders
SET parent_id = $1::uuid, updated_at = NOW()
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, path, parent_id::text, user_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
"#,
)
.bind(new_parent_id)
.bind(id)
.execute(self.pool())
.fetch_optional(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("move: {e}")))?;
.map_err(|e| DomainError::internal_error("FolderDb", format!("move: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", id))?;
self.get_folder(id).await
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6)
}
async fn delete_folder(&self, id: &str) -> Result<(), DomainError> {
@@ -26,16 +26,28 @@ use crate::domain::errors::{DomainError, ErrorKind};
/// Maximum file size for transcoding (5MB - larger files stream directly)
pub const MAX_TRANSCODE_SIZE: u64 = 5 * 1024 * 1024;
/// Number of threads in the dedicated transcoding pool
const TRANSCODE_POOL_THREADS: usize = 2;
/// Minimum number of threads in the dedicated transcoding pool
const MIN_TRANSCODE_THREADS: usize = 2;
/// Compute the number of transcoding threads: half the available CPUs,
/// with a floor of `MIN_TRANSCODE_THREADS`. `available_parallelism()`
/// respects cgroup limits (Docker/K8s) and CPU affinity masks.
fn transcode_thread_count() -> usize {
let cpus = std::thread::available_parallelism()
.map(|n| n.get())
.unwrap_or(MIN_TRANSCODE_THREADS);
(cpus / 2).max(MIN_TRANSCODE_THREADS)
}
/// Dedicated rayon thread pool for CPU-bound image transcoding.
/// Isolated from Tokio's blocking pool to prevent starvation of other I/O.
/// Thread count scales with available CPUs (half cores, min 2).
fn transcode_pool() -> &'static rayon::ThreadPool {
static POOL: OnceLock<rayon::ThreadPool> = OnceLock::new();
POOL.get_or_init(|| {
let threads = transcode_thread_count();
rayon::ThreadPoolBuilder::new()
.num_threads(TRANSCODE_POOL_THREADS)
.num_threads(threads)
.thread_name(|idx| format!("transcode-{idx}"))
.build()
.expect("Failed to create transcode thread pool")
@@ -171,7 +183,7 @@ impl ImageTranscodeService {
fs::create_dir_all(self.cache_dir.join("webp")).await?;
tracing::info!(
"🖼️ Image transcode service initialized (rayon pool: {} threads, cache dir: {:?})",
TRANSCODE_POOL_THREADS,
transcode_thread_count(),
self.cache_dir
);
Ok(())