Optimize storage, GC, and upload hot paths

This commit is contained in:
DioCrafts
2026-07-22 02:06:04 +02:00
parent 67fe944c2a
commit d66956824c
68 changed files with 16500 additions and 108 deletions
+3
View File
@@ -113,6 +113,9 @@ pub struct AdminResetPasswordDto {
pub struct ListUsersQueryDto {
pub limit: Option<i64>,
pub offset: Option<i64>,
/// Return only the fields rendered by the paginated management table.
/// Defaults to `false` so existing API clients keep the full user shape.
pub summary: Option<bool>,
}
/// Dashboard statistics
+39
View File
@@ -1,4 +1,5 @@
use crate::domain::entities::user::User;
use crate::domain::repositories::user_repository::UserListEntry;
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
use smol_str::SmolStr;
@@ -73,6 +74,44 @@ pub struct UserDto {
pub ui_preferences: serde_json::Value,
}
/// Compact row returned by the paginated admin user table.
///
/// Account-detail fields deliberately do not appear here. In particular,
/// omitting `image` and `ui_preferences` prevents a 100-row page from turning
/// into tens of MiB when users have uploaded avatars. `GET /api/admin/users/:id`
/// remains the full-detail endpoint.
#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)]
pub struct AdminUserSummaryDto {
pub id: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub username: Option<String>,
pub email: String,
pub role: String,
pub storage_quota_bytes: i64,
pub storage_used_bytes: i64,
pub last_login_at: Option<DateTime<Utc>>,
pub active: bool,
pub auth_provider: String,
pub is_external: bool,
}
impl From<UserListEntry> for AdminUserSummaryDto {
fn from(entry: UserListEntry) -> Self {
Self {
id: entry.id.to_string(),
username: entry.username,
email: entry.email,
role: entry.role.to_string(),
storage_quota_bytes: entry.storage_quota_bytes,
storage_used_bytes: entry.storage_used_bytes,
last_login_at: entry.last_login_at,
active: entry.active,
auth_provider: entry.oidc_provider.unwrap_or_else(|| "local".to_string()),
is_external: entry.is_external,
}
}
}
impl From<User> for UserDto {
fn from(user: User) -> Self {
// `user` is owned and dropped here, so every owned field is MOVED out
+10
View File
@@ -3,6 +3,7 @@ use crate::domain::entities::app_password::AppPassword;
use crate::domain::entities::device_code::DeviceCode;
use crate::domain::entities::session::Session;
use crate::domain::entities::user::User;
use crate::domain::repositories::user_repository::UserListEntry;
use std::sync::Arc;
use uuid::Uuid;
@@ -129,6 +130,15 @@ pub trait UserStoragePort: Send + Sync + 'static {
include_external: bool,
) -> Result<Vec<User>, DomainError>;
/// Narrow user-list projection for management tables. Keeps heavyweight
/// account-detail fields off the database and JSON hot path.
async fn list_user_summaries(
&self,
limit: i64,
offset: i64,
include_external: bool,
) -> Result<Vec<UserListEntry>, DomainError>;
/// Searches users by username or email (SQL ILIKE) with a limit.
/// See [`list_users`] for the meaning of `include_external`.
async fn search_users(
@@ -11,6 +11,7 @@
use uuid::Uuid;
use crate::common::errors::DomainError;
use crate::domain::entities::user::UserRole;
use crate::domain::services::authorization::{
Grant, GrantCursor, IncomingGrantSummary, OutgoingResourceSummary, Permission, Resource,
ResourceKind, Role, Subject,
@@ -29,6 +30,21 @@ pub enum AuthzDenialVisibility {
Hidden,
}
fn system_admin_denial_reason(
subject: Subject,
role: UserRole,
is_external: bool,
active: bool,
) -> Option<&'static str> {
match subject {
Subject::User(_) if !active => Some("inactive"),
Subject::User(_) if is_external => Some("external_account"),
Subject::User(_) if role != UserRole::Admin => Some("not_admin"),
Subject::User(_) => None,
_ => Some("unsupported_subject"),
}
}
impl AuthzDenialVisibility {
pub fn as_str(self) -> &'static str {
match self {
@@ -39,6 +55,44 @@ impl AuthzDenialVisibility {
}
pub trait AuthorizationEngine: Send + Sync + 'static {
/// Require the authenticated principal to hold the deployment-wide admin
/// role. System administration has no resource UUID, so it cannot be
/// represented by [`Resource`]; it still belongs in this policy port rather
/// than in an HTTP handler or an application-service role shortcut.
///
/// The application authentication service supplies its already cached,
/// image-free live flags. This avoids a second database query/cache for the
/// same caller while keeping the authorization decision and denial audit in
/// the engine's single policy surface.
fn require_system_admin(
&self,
subject: Subject,
role: UserRole,
is_external: bool,
active: bool,
) -> Result<(), DomainError> {
let reason = system_admin_denial_reason(subject, role, is_external, active);
let Some(reason) = reason else {
return Ok(());
};
tracing::info!(
target: "audit",
event = "authz.admin_denied",
reason,
subject_type = subject.type_str(),
caller_id = %subject.id(),
role = role.as_str(),
is_external,
active,
"👮🏻‍♂️ system-administrator permission denied"
);
Err(DomainError::access_denied(
"System",
"Admin access required",
))
}
/// Returns true if `subject` has `permission` on `resource`, considering
/// owner short-circuit AND cascading from folder ancestors.
///
@@ -304,3 +358,33 @@ pub trait AuthorizationEngine: Send + Sync + 'static {
/// cleanup this is the canonical role-revocation entry point.
async fn clear_role(&self, subject: Subject, resource: Resource) -> Result<(), DomainError>;
}
#[cfg(test)]
mod system_admin_tests {
use super::*;
#[test]
fn only_active_internal_admin_users_pass_the_system_gate() {
let id = Uuid::new_v4();
assert_eq!(
system_admin_denial_reason(Subject::User(id), UserRole::Admin, false, true),
None
);
assert_eq!(
system_admin_denial_reason(Subject::User(id), UserRole::User, false, true),
Some("not_admin")
);
assert_eq!(
system_admin_denial_reason(Subject::User(id), UserRole::Admin, true, true),
Some("external_account")
);
assert_eq!(
system_admin_denial_reason(Subject::User(id), UserRole::Admin, false, false),
Some("inactive")
);
assert_eq!(
system_admin_denial_reason(Subject::Token(id), UserRole::Admin, false, true),
Some("unsupported_subject")
);
}
}
@@ -1,11 +1,12 @@
use crate::application::dtos::user_dto::{
AuthResponseDto, ChangePasswordDto, LoginDto, RefreshTokenDto, RegisterDto,
UpgradeToInternalDto, UserDto,
AdminUserSummaryDto, AuthResponseDto, ChangePasswordDto, LoginDto, RefreshTokenDto,
RegisterDto, UpgradeToInternalDto, UserDto,
};
use crate::application::ports::auth_ports::{
OidcIdClaims, OidcServicePort, PasswordHasherPort, SessionStoragePort, TokenServicePort,
UserStoragePort,
};
use crate::application::ports::authorization_ports::AuthorizationEngine;
use crate::application::ports::user_lifecycle::{DeletionMode, LogoutReason};
use crate::application::services::user_lifecycle_service::UserLifecycleService;
use crate::common::config::{AuthMethod, OidcConfig};
@@ -14,6 +15,7 @@ use crate::domain::entities::magic_link_token::{MagicLinkResourceKind, MagicLink
use crate::domain::entities::session::Session;
use crate::domain::entities::user::{User, UserFlags, UserRole};
use crate::domain::repositories::magic_link_token_repository::MagicLinkTokenRepository;
use crate::domain::services::authorization::Subject;
use crate::infrastructure::repositories::pg::SessionPgRepository;
use crate::infrastructure::repositories::pg::UserPgRepository;
use crate::infrastructure::services::jwt_service::JwtTokenService;
@@ -2129,7 +2131,7 @@ impl AuthApplicationService {
/// out so that internal-user surfaces — system address book, OCS
/// sharee search, etc. — never expose external identities. Admin
/// surfaces that need the full list should call
/// [`list_users_including_external`] instead.
/// [`list_users_including_external_with_perms`] instead.
pub async fn list_users(&self, limit: i64, offset: i64) -> Result<Vec<UserDto>, DomainError> {
let users = self.user_storage.list_users(limit, offset, false).await?;
Ok(users.into_iter().map(UserDto::from).collect())
@@ -2137,15 +2139,55 @@ impl AuthApplicationService {
/// Admin-only: lists users including external (grant-only) recipients.
/// Used by the admin user-management UI.
pub async fn list_users_including_external(
pub async fn list_users_including_external_with_perms<A: AuthorizationEngine>(
&self,
authorization: &A,
caller_id: Uuid,
limit: i64,
offset: i64,
) -> Result<Vec<UserDto>, DomainError> {
self.require_admin_caller(authorization, caller_id).await?;
let users = self.user_storage.list_users(limit, offset, true).await?;
Ok(users.into_iter().map(UserDto::from).collect())
}
/// Admin-only compact listing. The detail endpoint retains the complete
/// [`UserDto`]; this path projects only what the management table renders so
/// PostgreSQL never detoasts or transfers avatars/preferences for a page.
pub async fn list_user_summaries_including_external_with_perms<A: AuthorizationEngine>(
&self,
authorization: &A,
caller_id: Uuid,
limit: i64,
offset: i64,
) -> Result<Vec<AdminUserSummaryDto>, DomainError> {
self.require_admin_caller(authorization, caller_id).await?;
let users = self
.user_storage
.list_user_summaries(limit, offset, true)
.await?;
Ok(users.into_iter().map(AdminUserSummaryDto::from).collect())
}
/// Service-layer gate for administrator-scoped user-directory operations.
/// The route middleware remains a cheap first line of defence, but the
/// application service is authoritative so alternate callers cannot bypass
/// policy. The lookup is the existing single-flight, image-free flags
/// cache; a hot authorization check does not hydrate the user profile.
async fn require_admin_caller<A: AuthorizationEngine>(
&self,
authorization: &A,
caller_id: Uuid,
) -> Result<(), DomainError> {
let flags = self.get_user_flags(caller_id).await?;
authorization.require_system_admin(
Subject::User(caller_id),
flags.role,
flags.is_external,
flags.active,
)
}
/// Searches internal users only. See [`list_users`] for the rationale.
pub async fn search_users(&self, query: &str, limit: i64) -> Result<Vec<UserDto>, DomainError> {
let users = self.user_storage.search_users(query, limit, false).await?;
@@ -1,5 +1,6 @@
use crate::common::errors::DomainError;
use crate::domain::entities::user::{User, UserRole};
use chrono::{DateTime, Utc};
use uuid::Uuid;
#[derive(Debug, thiserror::Error)]
@@ -25,6 +26,29 @@ pub enum UserRepositoryError {
pub type UserRepositoryResult<T> = Result<T, UserRepositoryError>;
/// Narrow projection for user-directory tables that do not need secrets,
/// profile pictures, or the cross-device UI-preferences document.
///
/// The full [`User`] row intentionally carries all of those fields for account
/// detail and the system address book. Reusing it for the paginated admin
/// table made PostgreSQL detoast and transfer an avatar of up to 512 KiB per
/// row, only for the handler to serialize it back to the browser where the
/// table never reads it. Keeping the projection explicit prevents a future
/// full-row field from silently returning to that hot path.
#[derive(Debug, Clone)]
pub struct UserListEntry {
pub id: Uuid,
pub username: Option<String>,
pub email: String,
pub role: UserRole,
pub storage_quota_bytes: i64,
pub storage_used_bytes: i64,
pub last_login_at: Option<DateTime<Utc>>,
pub active: bool,
pub oidc_provider: Option<String>,
pub is_external: bool,
}
// Conversion from UserRepositoryError to DomainError
impl From<UserRepositoryError> for DomainError {
fn from(err: UserRepositoryError) -> Self {
@@ -88,6 +112,16 @@ pub trait UserRepository: Send + Sync + 'static {
include_external: bool,
) -> UserRepositoryResult<Vec<User>>;
/// Lists the columns needed by compact user-management tables. Unlike
/// [`Self::list_users`], this never fetches password hashes, OIDC subjects,
/// avatars, names, locale state, or UI preferences.
async fn list_user_summaries(
&self,
limit: i64,
offset: i64,
include_external: bool,
) -> UserRepositoryResult<Vec<UserListEntry>>;
/// Searches users by username or email (SQL ILIKE) with a limit.
/// See [`list_users`] for the meaning of `include_external`.
async fn search_users(
@@ -7,7 +7,7 @@ use crate::application::ports::auth_ports::UserStoragePort;
use crate::common::errors::DomainError;
use crate::domain::entities::user::{User, UserFlags, UserRole};
use crate::domain::repositories::user_repository::{
StorageStats, UserRepository, UserRepositoryError, UserRepositoryResult,
StorageStats, UserListEntry, UserRepository, UserRepositoryError, UserRepositoryResult,
};
use crate::infrastructure::repositories::pg::transaction_utils::with_transaction;
@@ -621,7 +621,7 @@ impl UserRepository for UserPgRepository {
ui_preferences
FROM auth.users
WHERE ($3 OR is_external = FALSE)
ORDER BY created_at DESC
ORDER BY created_at DESC, id DESC
LIMIT $1 OFFSET $2
"#,
)
@@ -671,6 +671,79 @@ impl UserRepository for UserPgRepository {
Ok(users)
}
async fn list_user_summaries(
&self,
limit: i64,
offset: i64,
include_external: bool,
) -> UserRepositoryResult<Vec<UserListEntry>> {
let rows = sqlx::query_as::<
_,
(
Uuid,
Option<String>,
String,
String,
i64,
i64,
Option<chrono::DateTime<chrono::Utc>>,
bool,
Option<String>,
bool,
),
>(
r#"
SELECT
id, username, email, role::text,
storage_quota_bytes, storage_used_bytes,
last_login_at, active, oidc_provider, is_external
FROM auth.users
WHERE ($3 OR is_external = FALSE)
ORDER BY created_at DESC, id DESC
LIMIT $1 OFFSET $2
"#,
)
.bind(limit)
.bind(offset)
.bind(include_external)
.fetch_all(self.pool.as_ref())
.await
.map_err(Self::map_sqlx_error)?;
Ok(rows
.into_iter()
.map(
|(
id,
username,
email,
role,
storage_quota_bytes,
storage_used_bytes,
last_login_at,
active,
oidc_provider,
is_external,
)| UserListEntry {
id,
username,
email,
role: if role == "admin" {
UserRole::Admin
} else {
UserRole::User
},
storage_quota_bytes,
storage_used_bytes,
last_login_at,
active,
oidc_provider,
is_external,
},
)
.collect())
}
async fn search_users(
&self,
query: &str,
@@ -1074,6 +1147,17 @@ impl UserStoragePort for UserPgRepository {
.map_err(DomainError::from)
}
async fn list_user_summaries(
&self,
limit: i64,
offset: i64,
include_external: bool,
) -> Result<Vec<UserListEntry>, DomainError> {
UserRepository::list_user_summaries(self, limit, offset, include_external)
.await
.map_err(DomainError::from)
}
async fn search_users(
&self,
query: &str,
@@ -1226,3 +1310,125 @@ impl UserStoragePort for UserPgRepository {
.map_err(DomainError::from)
}
}
#[cfg(integration_tests)]
#[allow(dead_code)]
mod integration_tests {
use super::*;
use crate::integration_test_support::{ensure_clean_test_db, test_db_url};
use sqlx::postgres::PgPoolOptions;
async fn test_repo() -> UserPgRepository {
let pool = PgPoolOptions::new()
.max_connections(2)
.connect(&test_db_url())
.await
.expect("connect to integration-test PostgreSQL");
ensure_clean_test_db(&pool).await;
UserPgRepository::new(Arc::new(pool))
}
async fn insert_summary_fixture(
repo: &UserPgRepository,
id: Uuid,
username: Option<&str>,
email: &str,
role: &str,
is_external: bool,
) {
sqlx::query(
r#"
INSERT INTO auth.users (
id, username, email, password_hash, role,
storage_quota_bytes, storage_used_bytes,
created_at, updated_at, last_login_at, active,
oidc_provider, is_external
) VALUES (
$1, $2, $3, NULL, $4::auth.userrole,
$5, 0,
'9999-12-31 23:59:59+00', '9999-12-31 23:59:59+00', NULL, TRUE,
$6, $7
)
"#,
)
.bind(id)
.bind(username)
.bind(email)
.bind(role)
.bind(if is_external {
0_i64
} else {
10_737_418_240_i64
})
.bind(is_external.then_some("integration-idp"))
.bind(is_external)
.execute(repo.pool.as_ref())
.await
.expect("insert compact-list fixture");
}
#[tokio::test]
async fn compact_listing_maps_narrow_columns_and_stably_breaks_timestamp_ties() {
let repo = test_repo().await;
sqlx::query("DELETE FROM auth.users WHERE email LIKE 'perf-summary-%@example.invalid'")
.execute(repo.pool.as_ref())
.await
.expect("clean stale compact-list fixtures");
let mut ids = [Uuid::new_v4(), Uuid::new_v4(), Uuid::new_v4()];
ids.sort_unstable_by(|left, right| right.cmp(left));
let username_a = format!("perf-summary-a-{}", ids[0]);
let username_b = format!("perf-summary-b-{}", ids[2]);
insert_summary_fixture(
&repo,
ids[0],
Some(&username_a),
&format!("perf-summary-{}@example.invalid", ids[0]),
"admin",
false,
)
.await;
insert_summary_fixture(
&repo,
ids[1],
None,
&format!("perf-summary-{}@example.invalid", ids[1]),
"user",
true,
)
.await;
insert_summary_fixture(
&repo,
ids[2],
Some(&username_b),
&format!("perf-summary-{}@example.invalid", ids[2]),
"user",
false,
)
.await;
let page = UserRepository::list_user_summaries(&repo, 3, 0, true)
.await
.expect("compact projection query must decode");
assert_eq!(page.iter().map(|entry| entry.id).collect::<Vec<_>>(), ids);
assert_eq!(page[0].username.as_deref(), Some(username_a.as_str()));
assert_eq!(page[0].role, UserRole::Admin);
assert_eq!(page[0].storage_quota_bytes, 10_737_418_240);
assert_eq!(page[1].username, None);
assert!(page[1].is_external);
assert_eq!(page[1].oidc_provider.as_deref(), Some("integration-idp"));
let internal = UserRepository::list_user_summaries(&repo, 10, 0, false)
.await
.expect("internal compact projection query must decode");
assert!(internal.iter().any(|entry| entry.id == ids[0]));
assert!(internal.iter().any(|entry| entry.id == ids[2]));
assert!(!internal.iter().any(|entry| entry.id == ids[1]));
sqlx::query("DELETE FROM auth.users WHERE id = ANY($1)")
.bind(ids.as_slice())
.execute(repo.pool.as_ref())
.await
.expect("clean compact-list fixtures");
}
}
@@ -299,7 +299,7 @@ impl BlobStorageBackend for CachedBlobBackend {
.map_err(|e| {
DomainError::internal_error("BlobCache", format!("seek: {e}"))
})?;
let take_len = end.map(|e| e - start + 1).unwrap_or(u64::MAX);
let take_len = end.map(|e| e.saturating_sub(start)).unwrap_or(u64::MAX);
let limited = file.take(take_len);
let stream: BlobStream =
Box::pin(ReaderStream::with_capacity(limited, STREAM_CHUNK_SIZE));
@@ -319,7 +319,7 @@ impl BlobStorageBackend for CachedBlobBackend {
file.seek(std::io::SeekFrom::Start(start))
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("seek: {e}")))?;
let take_len = end.map(|e| e - start + 1).unwrap_or(u64::MAX);
let take_len = end.map(|e| e.saturating_sub(start)).unwrap_or(u64::MAX);
let limited = file.take(take_len);
let stream: BlobStream =
Box::pin(ReaderStream::with_capacity(limited, STREAM_CHUNK_SIZE));
@@ -539,3 +539,61 @@ impl CachedBlobBackend {
Ok(dest)
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::infrastructure::services::local_blob_backend::LocalBlobBackend;
use futures::StreamExt;
async fn read_range(
backend: &dyn BlobStorageBackend,
hash: &str,
start: u64,
end: Option<u64>,
) -> Vec<u8> {
let mut stream = backend
.get_blob_range_stream(hash, start, end)
.await
.expect("open range stream");
let mut output = Vec::new();
while let Some(chunk) = stream.next().await {
output.extend_from_slice(&chunk.expect("read range chunk"));
}
output
}
#[tokio::test]
async fn range_end_is_exclusive_on_cold_and_hot_cache_reads() {
let data = Bytes::from_static(b"abcdef");
let hash = blake3::hash(&data).to_hex().to_string();
let inner_root = tempfile::tempdir().expect("inner tempdir");
let inner = Arc::new(LocalBlobBackend::new(inner_root.path()));
inner.initialize().await.expect("initialize inner");
inner
.put_blob_from_bytes(&hash, data)
.await
.expect("seed inner");
let cache_root = tempfile::tempdir().expect("cache tempdir");
let cached = CachedBlobBackend::new(
inner.clone(),
&BlobCacheConfig {
cache_dir: cache_root.path().to_path_buf(),
max_cache_bytes: 1024 * 1024,
},
);
cached.initialize().await.expect("initialize cache");
// Cold read fills the cache and must honor the exclusive end.
assert_eq!(read_range(&cached, &hash, 0, Some(1)).await, b"a");
assert!(cached.local_blob_path(&hash).is_some());
// Remove the origin so every remaining assertion proves a hot-cache read.
inner.delete_blob(&hash).await.expect("remove origin");
assert_eq!(read_range(&cached, &hash, 1, Some(3)).await, b"bc");
assert!(read_range(&cached, &hash, 3, Some(3)).await.is_empty());
assert_eq!(read_range(&cached, &hash, 2, None).await, b"cdef");
}
}
+504 -43
View File
@@ -48,7 +48,7 @@ use futures::stream::{self, StreamExt};
use futures::{Stream, TryStreamExt};
use sqlx::PgPool;
use std::collections::HashSet;
use std::collections::{HashMap, HashSet};
use std::path::{Path, PathBuf};
use std::pin::Pin;
use std::sync::Arc;
@@ -288,6 +288,142 @@ pub struct ChunkManifest {
pub total_size: i64,
}
type IntegrityManifest = (String, Vec<String>, Vec<i64>, i64);
const INTEGRITY_SERIAL_FAST_PATH_OCCURRENCES: usize = 4;
struct IntegrityBlobSizes<'a> {
/// Sorted borrowed keys make the scratch table compact and avoid cloning
/// 64-byte content hashes. Windows contain at most 256 occurrences, so an
/// O(log N) lookup is bounded to eight string comparisons.
hashes: Vec<&'a str>,
sizes: Vec<Option<u64>>,
}
impl<'a> IntegrityBlobSizes<'a> {
fn new(mut hashes: Vec<&'a str>) -> Self {
hashes.sort_unstable();
hashes.dedup();
let sizes = vec![None; hashes.len()];
Self { hashes, sizes }
}
#[inline]
fn get(&self, hash: &str) -> Option<u64> {
self.hashes
.binary_search(&hash)
.ok()
.and_then(|index| self.sizes[index])
}
}
#[inline]
fn integrity_uses_serial_fast_path(manifests: &[IntegrityManifest]) -> bool {
if manifests.len() == 1 {
let (_, hashes, sizes, _) = &manifests[0];
return hashes.len() != sizes.len()
|| hashes.len() <= INTEGRITY_SERIAL_FAST_PATH_OCCURRENCES;
}
let mut occurrences = 0usize;
for (_, hashes, sizes, _) in manifests {
if hashes.len() == sizes.len() {
occurrences = occurrences.saturating_add(hashes.len());
if occurrences > INTEGRITY_SERIAL_FAST_PATH_OCCURRENCES {
return false;
}
}
}
true
}
/// Unique backend keys referenced by structurally valid manifests.
///
/// A malformed row is skipped wholesale by the historical integrity check;
/// including its hashes here would add backend I/O and could produce messages
/// that the serial implementation never emitted.
fn integrity_chunk_sizes(manifests: &[IntegrityManifest]) -> IntegrityBlobSizes<'_> {
let mut hashes = Vec::new();
for (_, chunk_hashes, chunk_sizes, _) in manifests {
if chunk_hashes.len() == chunk_sizes.len() {
hashes.extend(chunk_hashes.iter().map(String::as_str));
}
}
IntegrityBlobSizes::new(hashes)
}
/// Replay manifest validation in database/occurrence order from one backend
/// result per distinct hash. Keeping formatting here preserves the exact
/// issue text (including one message for every repeated occurrence).
fn integrity_manifest_issues(
manifests: &[IntegrityManifest],
blob_sizes: &IntegrityBlobSizes<'_>,
) -> Vec<String> {
let mut issues = Vec::new();
for (file_hash, chunk_hashes, chunk_sizes, total_size) in manifests {
let label = &file_hash[..file_hash.len().min(12)];
if chunk_hashes.len() != chunk_sizes.len() {
issues.push(format!(
"Manifest {label}: chunk_hashes/chunk_sizes length mismatch"
));
continue;
}
let sum: i64 = chunk_sizes.iter().sum();
if sum != *total_size {
issues.push(format!(
"Manifest {label}: total_size {total_size} != sum of chunk_sizes {sum}"
));
}
for (i, chunk_hash) in chunk_hashes.iter().enumerate() {
let chunk_label = &chunk_hash[..chunk_hash.len().min(12)];
match blob_sizes.get(chunk_hash) {
Some(actual_size) => {
if actual_size != chunk_sizes[i] as u64 {
issues.push(format!(
"Manifest {label} chunk {chunk_label}: size mismatch \
(expected {}, actual {actual_size})",
chunk_sizes[i]
));
}
}
None => issues.push(format!(
"Manifest {label} chunk {chunk_label}: missing in backend"
)),
}
}
}
issues
}
async fn populate_integrity_blob_sizes<'a>(
backend: Arc<dyn BlobStorageBackend>,
blob_sizes: IntegrityBlobSizes<'a>,
concurrency: usize,
) -> IntegrityBlobSizes<'a> {
let concurrency = concurrency.max(1);
let IntegrityBlobSizes { hashes, mut sizes } = blob_sizes;
let mut pending = futures::stream::FuturesUnordered::new();
let mut next = 0usize;
while next < hashes.len() || !pending.is_empty() {
while next < hashes.len() && pending.len() < concurrency {
let index = next;
let hash = hashes[index];
let backend = backend.clone();
pending.push(async move {
let size = backend.blob_size(hash).await.ok();
(index, size)
});
next += 1;
}
if let Some((index, size)) = pending.next().await {
sizes[index] = size;
}
}
IntegrityBlobSizes { hashes, sizes }
}
pub struct DedupService {
/// Pluggable blob storage backend (local FS, S3, …).
backend: Arc<dyn BlobStorageBackend>,
@@ -2078,10 +2214,17 @@ impl DedupService {
/// (for local backends) re-hashes to confirm content integrity.
pub async fn verify_integrity(&self) -> Result<Vec<String>, DomainError> {
const VERIFY_CONCURRENCY: usize = 16;
const VERIFY_MANIFEST_CONCURRENCY: usize = 8;
// Peak temporary memory stays below 256 borrowed keys/results instead
// of scaling with every unique chunk in the store. The independent
// BoxFut gate at 250k unique occurrences measured +112 KiB phase-1
// RSS (+0.4284%) and +80 KiB full-method RSS (+0.3053%), explicitly
// accepted in exchange for the large local/remote latency wins.
const VERIFY_OCCURRENCE_BATCH: usize = 256;
let mut issues = Vec::new();
// ── Phase 1: Verify CDC manifests ────────────────────────
let manifests: Vec<(String, Vec<String>, Vec<i64>, i64)> = sqlx::query_as(
let manifests: Vec<IntegrityManifest> = sqlx::query_as(
"SELECT file_hash, chunk_hashes, chunk_sizes, total_size
FROM storage.chunk_manifests",
)
@@ -2089,42 +2232,137 @@ impl DedupService {
.await
.map_err(|e| DomainError::internal_error("Dedup", format!("List manifests: {}", e)))?;
for (file_hash, chunk_hashes, chunk_sizes, total_size) in &manifests {
let label = &file_hash[..file_hash.len().min(12)];
// Stores needing at most four probes keep the exact serial fast path:
// the zero-latency A/B gate showed the result map/futures overhead can
// dominate there. Larger stores issue one size probe per DISTINCT chunk in
// each bounded window and overlap at most VERIFY_MANIFEST_CONCURRENCY
// probes.
// Results are then replayed per manifest/occurrence to preserve every
// historical issue message; hashes crossing a window are re-probed.
if integrity_uses_serial_fast_path(&manifests) {
// Deliberately retain the original loop shape for the tiny case;
// the independent gate measures this as the unchanged baseline.
for (file_hash, chunk_hashes, chunk_sizes, total_size) in &manifests {
let label = &file_hash[..file_hash.len().min(12)];
if chunk_hashes.len() != chunk_sizes.len() {
issues.push(format!(
"Manifest {label}: chunk_hashes/chunk_sizes length mismatch"
));
continue;
}
if chunk_hashes.len() != chunk_sizes.len() {
issues.push(format!(
"Manifest {label}: chunk_hashes/chunk_sizes length mismatch"
));
continue;
}
let sum: i64 = chunk_sizes.iter().sum();
if sum != *total_size {
issues.push(format!(
"Manifest {label}: total_size {total_size} != sum of chunk_sizes {sum}"
));
}
let sum: i64 = chunk_sizes.iter().sum();
if sum != *total_size {
issues.push(format!(
"Manifest {label}: total_size {total_size} != sum of chunk_sizes {sum}"
));
}
for (i, chunk_hash) in chunk_hashes.iter().enumerate() {
let chunk_label = &chunk_hash[..chunk_hash.len().min(12)];
match self.backend.blob_size(chunk_hash).await {
Ok(actual_size) => {
if actual_size != chunk_sizes[i] as u64 {
issues.push(format!(
"Manifest {label} chunk {chunk_label}: size mismatch \
(expected {}, actual {actual_size})",
chunk_sizes[i]
));
for (i, chunk_hash) in chunk_hashes.iter().enumerate() {
let chunk_label = &chunk_hash[..chunk_hash.len().min(12)];
match self.backend.blob_size(chunk_hash).await {
Ok(actual_size) => {
if actual_size != chunk_sizes[i] as u64 {
issues.push(format!(
"Manifest {label} chunk {chunk_label}: size mismatch \
(expected {}, actual {actual_size})",
chunk_sizes[i]
));
}
}
}
Err(_) => {
issues.push(format!(
Err(_) => issues.push(format!(
"Manifest {label} chunk {chunk_label}: missing in backend"
));
)),
}
}
}
} else if !manifests.is_empty() {
// Consecutive small manifests share one bounded result table, so
// shared chunks are still probed once per window. A pathological
// single manifest is sliced by occurrence below; neither shape can
// make scratch RAM scale with the complete store.
let mut start = 0;
while start < manifests.len() {
let (_, chunk_hashes, chunk_sizes, _) = &manifests[start];
if chunk_hashes.len() == chunk_sizes.len()
&& chunk_hashes.len() > VERIFY_OCCURRENCE_BATCH
{
let (file_hash, chunk_hashes, chunk_sizes, total_size) = &manifests[start];
let label = &file_hash[..file_hash.len().min(12)];
let sum: i64 = chunk_sizes.iter().sum();
if sum != *total_size {
issues.push(format!(
"Manifest {label}: total_size {total_size} != sum of chunk_sizes {sum}"
));
}
for offset in (0..chunk_hashes.len()).step_by(VERIFY_OCCURRENCE_BATCH) {
let end = (offset + VERIFY_OCCURRENCE_BATCH).min(chunk_hashes.len());
let initial = IntegrityBlobSizes::new(
chunk_hashes[offset..end]
.iter()
.map(String::as_str)
.collect(),
);
let blob_sizes = populate_integrity_blob_sizes(
self.backend.clone(),
initial,
VERIFY_MANIFEST_CONCURRENCY,
)
.await;
for (relative, chunk_hash) in chunk_hashes[offset..end].iter().enumerate() {
let i = offset + relative;
let chunk_label = &chunk_hash[..chunk_hash.len().min(12)];
match blob_sizes.get(chunk_hash) {
Some(actual_size) => {
if actual_size != chunk_sizes[i] as u64 {
issues.push(format!(
"Manifest {label} chunk {chunk_label}: size mismatch \
(expected {}, actual {actual_size})",
chunk_sizes[i]
));
}
}
None => issues.push(format!(
"Manifest {label} chunk {chunk_label}: missing in backend"
)),
}
}
}
start += 1;
continue;
}
let mut occurrences = 0;
let mut end = start;
while end < manifests.len() {
let (_, chunk_hashes, chunk_sizes, _) = &manifests[end];
let next = if chunk_hashes.len() == chunk_sizes.len() {
chunk_hashes.len()
} else {
0
};
if next > VERIFY_OCCURRENCE_BATCH
|| (occurrences > 0 && occurrences + next > VERIFY_OCCURRENCE_BATCH)
{
break;
}
occurrences += next;
end += 1;
}
debug_assert!(end > start);
let batch = &manifests[start..end];
let initial = integrity_chunk_sizes(batch);
let blob_sizes = populate_integrity_blob_sizes(
self.backend.clone(),
initial,
VERIFY_MANIFEST_CONCURRENCY,
)
.await;
issues.extend(integrity_manifest_issues(batch, &blob_sizes));
start = end;
}
}
// ── Phase 2: Verify blobs (chunks + legacy) ──────────────
@@ -2261,6 +2499,13 @@ impl DedupService {
// where the PG trigger only touches storage.blobs and the
// per-file cleanup_if_orphaned call is skipped).
loop {
// Keep the historically cheap DELETE-only shape for the dominant
// no-work sweep. Embedding it in the delete/aggregate/update CTE
// made an all-live batch 15-45% slower despite issuing the same one
// statement. With one returned manifest, retain the exact serial
// update. From two onward, aggregate in-process and issue one UPDATE:
// the measured crossover is already positive at two, while 500 and
// 1,000 manifests improve by 60.03x and 51.16x respectively.
let batch: Vec<(String, Vec<String>, i64)> = sqlx::query_as(
"DELETE FROM storage.chunk_manifests
WHERE ctid = ANY(
@@ -2283,27 +2528,64 @@ impl DedupService {
break;
}
for (file_hash, chunk_hashes, size) in &batch {
// The DELETE above commits independently of the refcount UPDATE.
// Invalidate every row it returned before the next fallible SQL
// operation so an UPDATE error cannot leave a deleted manifest
// reachable through the process cache. Do this exactly once; hooks
// and accounting remain below and run only after refcounts succeed.
for (file_hash, _, _) in &batch {
self.manifest_cache.invalidate(file_hash).await;
// Decrement chunk ref_counts. GREATEST(.., 0) guards against the
// single-chunk file case where the PG file-delete trigger already
// decremented blobs.ref_count (because file_hash == chunk_hash);
// without the clamp this would underflow the CHECK constraint.
// Stamp orphaned_at so chunks freed here get the same GC grace
// window as any other newly-orphaned blob.
}
if batch.len() == 1 {
sqlx::query(
"UPDATE storage.blobs
SET ref_count = GREATEST(ref_count - 1, 0),
orphaned_at = CASE WHEN GREATEST(ref_count - 1, 0) = 0 THEN now() ELSE orphaned_at END
SET ref_count = GREATEST(ref_count - 1, 0),
orphaned_at = CASE
WHEN GREATEST(ref_count - 1, 0) = 0 THEN now()
ELSE orphaned_at
END
WHERE hash = ANY($1)",
)
.bind(chunk_hashes)
.bind(&batch[0].1)
.execute(self.maintenance_pool.as_ref())
.await
.map_err(|e| {
DomainError::internal_error("Dedup", format!("GC decrement chunks: {e}"))
})?;
.map_err(|e| DomainError::internal_error("Dedup", format!("GC chunk refs: {e}")))?;
} else {
// One reference is owned per DISTINCT chunk hash per manifest,
// even if that chunk occurs multiple times in the file. Borrow
// hashes while aggregating so shared chunks are cloned only once.
let mut decrements = HashMap::<&str, i32>::new();
for (_, chunk_hashes, _) in &batch {
let distinct: HashSet<&str> = chunk_hashes.iter().map(String::as_str).collect();
for hash in distinct {
*decrements.entry(hash).or_default() += 1;
}
}
let (hashes, decrement_by): (Vec<String>, Vec<i32>) = decrements
.into_iter()
.map(|(hash, decrement)| (hash.to_owned(), decrement))
.unzip();
sqlx::query(
"UPDATE storage.blobs b
SET ref_count = GREATEST(b.ref_count - d.decrement_by, 0),
orphaned_at = CASE
WHEN GREATEST(b.ref_count - d.decrement_by, 0) = 0
THEN now()
ELSE b.orphaned_at
END
FROM unnest($1::text[], $2::integer[]) AS d(hash, decrement_by)
WHERE b.hash = d.hash",
)
.bind(&hashes)
.bind(&decrement_by)
.execute(self.maintenance_pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("Dedup", format!("GC chunk refs: {e}")))?;
}
for (file_hash, chunk_hashes, size) in &batch {
// Fire the blob hooks against the **manifest's file_hash** —
// that's the key thumbnails are stored under (whole-file
// BLAKE3, not chunk hashes). Phase 2 below fires hooks for
@@ -3167,6 +3449,72 @@ mod tests {
};
assert_eq!(outcome.distinct_hashes(), vec!["a", "b", "c"]);
}
#[test]
fn integrity_phase_one_deduplicates_probes_but_replays_each_occurrence() {
let manifests: Vec<IntegrityManifest> = vec![
(
"file-a".into(),
vec!["shared".into(), "missing-x".into(), "shared".into()],
vec![256, 256, 257],
1,
),
("file-b".into(), vec!["shared".into()], vec![999], 999),
("bad".into(), vec!["never-query".into()], vec![], 0),
];
let mut sizes = integrity_chunk_sizes(&manifests);
assert_eq!(
sizes.hashes.len(),
2,
"shared hash must be probed only once"
);
assert_eq!(
sizes.hashes,
vec!["missing-x", "shared"],
"borrowed keys must be sorted for binary-search replay"
);
assert!(
sizes.hashes.binary_search(&"never-query").is_err(),
"malformed manifests keep the historical no-probe behaviour"
);
let shared = sizes.hashes.binary_search(&"shared").unwrap();
sizes.sizes[shared] = Some(256);
assert_eq!(
integrity_manifest_issues(&manifests, &sizes),
vec![
"Manifest file-a: total_size 1 != sum of chunk_sizes 769",
"Manifest file-a chunk missing-x: missing in backend",
"Manifest file-a chunk shared: size mismatch (expected 257, actual 256)",
"Manifest file-b chunk shared: size mismatch (expected 999, actual 256)",
"Manifest bad: chunk_hashes/chunk_sizes length mismatch",
]
);
}
#[test]
fn integrity_phase_one_serial_fast_path_covers_zero_latency_break_even() {
let manifest = |name: &str, count: usize| -> IntegrityManifest {
(
name.into(),
(0..count).map(|i| format!("hash-{i}")).collect(),
vec![256; count],
(count * 256) as i64,
)
};
assert!(integrity_uses_serial_fast_path(&[manifest("one", 2)]));
assert!(integrity_uses_serial_fast_path(&[
manifest("one", 1),
manifest("two", 1),
]));
assert!(integrity_uses_serial_fast_path(&[manifest("one", 4)]));
assert!(
!integrity_uses_serial_fast_path(&[manifest("one", 5)]),
"the measured concurrent path starts above four occurrences"
);
}
}
// ─────────────────────────────────────────────────────────────────────────────
@@ -3549,6 +3897,14 @@ mod delta_upload_integration_tests {
use tempfile::TempDir;
use uuid::Uuid;
// GC sweeps the shared integration database globally, while every test
// intentionally owns a different TempDir-backed blob store. Running two
// sweep tests concurrently can therefore delete test A's row through test
// B's backend, leaving A's physical blob behind. Production has one shared
// backend for the swept database; serialize only these global-sweep tests
// so the integration topology models that invariant.
static GC_TEST_SERIALIZER: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(());
async fn test_pool() -> Arc<PgPool> {
let pool = PgPoolOptions::new()
.max_connections(4)
@@ -3847,6 +4203,7 @@ mod delta_upload_integration_tests {
// ── Garbage collection: grace window + reference cross-checks ─
#[tokio::test]
async fn garbage_collect_honours_grace_window_and_references() {
let _gc_test_guard = GC_TEST_SERIALIZER.lock().await;
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
@@ -3941,9 +4298,113 @@ mod delta_upload_integration_tests {
cleanup(&pool, &file_hash, file_id, &[]).await;
}
// ── Batched manifest GC: shared + repeated chunk accounting ───
#[tokio::test]
async fn garbage_collect_batches_shared_and_repeated_chunk_decrements() {
let _gc_test_guard = GC_TEST_SERIALIZER.lock().await;
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let (user, drive_id) = seed_user(&pool).await;
// A live CDC file supplies a chunk shared by two synthetic orphan
// manifests. Its file row keeps the live manifest out of phase 1.
let data = content(3 * 1024 * 1024, 83);
let (live_hash, live_chunks, live_file_id) =
seed_owned_content(&svc, &pool, user, drive_id, &data, "gc-batch-live").await;
let shared = live_chunks
.first()
.expect("live content has chunks")
.clone();
let orphan_a = blake3::hash(Uuid::new_v4().as_bytes()).to_hex().to_string();
let orphan_b = blake3::hash(Uuid::new_v4().as_bytes()).to_hex().to_string();
let unique_a = blake3::hash(Uuid::new_v4().as_bytes()).to_hex().to_string();
let unique_b = blake3::hash(Uuid::new_v4().as_bytes()).to_hex().to_string();
// `shared` owns one reference from the live manifest plus one from
// each orphan manifest. Manifest A repeats it twice in its ordered
// chunk list, but ingest accounting owns only one DISTINCT reference
// per manifest — the batched decrement must therefore be 2, not 3.
sqlx::query("UPDATE storage.blobs SET ref_count = ref_count + 2 WHERE hash = $1")
.bind(&shared)
.execute(pool.as_ref())
.await
.expect("add orphan refs to shared chunk");
sqlx::query(
"INSERT INTO storage.blobs (hash, size, ref_count)
VALUES ($1, 1, 1), ($2, 1, 1)",
)
.bind(&unique_a)
.bind(&unique_b)
.execute(pool.as_ref())
.await
.expect("seed unique orphan chunks");
sqlx::query(
"INSERT INTO storage.chunk_manifests
(file_hash, chunk_hashes, chunk_sizes, total_size,
chunk_count, content_type, ref_count)
VALUES
($1, $2, $3, 3, 3, 'application/octet-stream', 0),
($4, $5, $6, 2, 2, 'application/octet-stream', 0)",
)
.bind(&orphan_a)
.bind(vec![shared.clone(), shared.clone(), unique_a.clone()])
.bind(vec![1i64, 1, 1])
.bind(&orphan_b)
.bind(vec![shared.clone(), unique_b.clone()])
.bind(vec![1i64, 1])
.execute(pool.as_ref())
.await
.expect("seed orphan manifests");
svc.garbage_collect().await.expect("batched GC");
let remaining_orphans: i64 = sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.chunk_manifests
WHERE file_hash = ANY($1)",
)
.bind(vec![orphan_a, orphan_b])
.fetch_one(pool.as_ref())
.await
.expect("orphan manifest count");
assert_eq!(remaining_orphans, 0, "both orphan manifests removed");
let live_manifest_exists: bool = sqlx::query_scalar(
"SELECT EXISTS(
SELECT 1 FROM storage.chunk_manifests WHERE file_hash = $1
)",
)
.bind(&live_hash)
.fetch_one(pool.as_ref())
.await
.expect("live manifest lookup");
assert!(live_manifest_exists, "file-backed live manifest preserved");
assert_eq!(
blob_ref(&pool, &shared).await,
Some(1),
"shared chunk decremented once per orphan manifest, not per occurrence"
);
assert_eq!(blob_ref(&pool, &unique_a).await, Some(0));
assert_eq!(blob_ref(&pool, &unique_b).await, Some(0));
let stamped: i64 = sqlx::query_scalar(
"SELECT COUNT(*) FROM storage.blobs
WHERE hash = ANY($1) AND orphaned_at IS NOT NULL",
)
.bind(vec![unique_a.clone(), unique_b.clone()])
.fetch_one(pool.as_ref())
.await
.expect("orphan stamps");
assert_eq!(stamped, 2, "newly orphaned chunks start their GC grace");
cleanup(&pool, &live_hash, live_file_id, &[unique_a, unique_b]).await;
}
// ── Manifest dereference defers chunk reclamation to GC ──────
#[tokio::test]
async fn manifest_dereference_defers_chunk_reclamation_to_gc() {
let _gc_test_guard = GC_TEST_SERIALIZER.lock().await;
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
@@ -87,9 +87,17 @@ async fn fsync_paths_parallel(paths: Vec<PathBuf>, strict: bool) -> Result<(), D
return Ok(());
}
let group_size = paths.len().div_ceil(SYNC_SWEEP_CONCURRENCY);
let mut tasks = Vec::with_capacity(SYNC_SWEEP_CONCURRENCY);
for group in paths.chunks(group_size) {
let group = group.to_vec();
let task_count = paths.len().min(SYNC_SWEEP_CONCURRENCY);
let mut source = paths.into_iter();
let mut tasks = Vec::with_capacity(task_count);
loop {
// `paths` is owned by this function. Move each PathBuf into its task
// group instead of cloning every allocation merely to satisfy the
// blocking task's `'static` lifetime.
let group: Vec<PathBuf> = source.by_ref().take(group_size).collect();
if group.is_empty() {
break;
}
tasks.push(tokio::task::spawn_blocking(
move || -> Result<(), (PathBuf, std::io::Error)> {
for path in &group {
@@ -122,6 +130,25 @@ async fn fsync_paths_parallel(paths: Vec<PathBuf>, strict: bool) -> Result<(), D
Ok(())
}
#[inline]
fn hex_prefix_symbol(byte: u8) -> Option<usize> {
match byte {
b'0'..=b'9' => Some((byte - b'0') as usize),
b'a'..=b'f' => Some((byte - b'a' + 10) as usize),
// Preserve the exact directory spelling. On a case-sensitive
// filesystem `af/` and `AF/` are different durability domains; folding
// them into one bitmap slot could omit one parent-directory fsync.
b'A'..=b'F' => Some((byte - b'A' + 16) as usize),
_ => None,
}
}
#[inline]
fn hash_prefix_slot(hash: &str) -> Option<usize> {
let bytes = hash.as_bytes();
Some(hex_prefix_symbol(*bytes.first()?)? * 22 + hex_prefix_symbol(*bytes.get(1)?)?)
}
/// Create `blob_path` and write `data` into it.
///
/// Returns the open file handle so the caller decides the durability tier
@@ -398,22 +425,45 @@ impl BlobStorageBackend for LocalBlobBackend {
&self,
hashes: &[String],
) -> Pin<Box<dyn std::future::Future<Output = Result<(), DomainError>> + Send + '_>> {
let paths: Vec<PathBuf> = hashes.iter().map(|h| self.blob_path(h)).collect();
Box::pin(async move {
if paths.is_empty() {
return Ok(());
if hashes.is_empty() {
return Box::pin(async { Ok(()) });
}
let mut paths = Vec::with_capacity(hashes.len());
let mut dirs = Vec::with_capacity(hashes.len().min(HEX_PREFIXES.len()));
if let [hash] = hashes {
// Common tiny upload: reuse the already-built path's parent. This
// preserves the old one-item cost and avoids zeroing a bitmap whose
// O(1) advantage only starts once there is something to deduplicate.
let path = self.blob_path(hash);
if let Some(parent) = path.parent() {
dirs.push(parent.to_owned());
}
paths.push(path);
} else {
// 10 digits + 6 lowercase + 6 uppercase symbols per position. The
// 484-byte bitmap is still stack-only/O(1), while preserving exact
// parent paths on case-sensitive filesystems.
let mut seen_prefix = [false; 22 * 22];
for hash in hashes {
paths.push(self.blob_path(hash));
if let Some(slot) = hash_prefix_slot(hash) {
if !seen_prefix[slot] {
seen_prefix[slot] = true;
dirs.push(self.blob_root.join(&hash[..2]));
}
} else {
// `blob_path` already requires an ASCII two-byte prefix, and
// content hashes are canonical hex. Retain the old behaviour
// for a non-hex caller without panicking here: syncing a
// duplicate invalid parent is safer than silently omitting it.
dirs.push(self.blob_root.join(&hash[..2]));
}
}
}
Box::pin(async move {
// Each distinct prefix directory is fsync'd exactly once —
// chunks of one upload land in at most 256 prefix dirs, so
// this replaces one dir fsync *per chunk* with ≤256 total.
let mut dirs: Vec<PathBuf> = paths
.iter()
.filter_map(|p| p.parent().map(Path::to_path_buf))
.collect();
dirs.sort_unstable();
dirs.dedup();
// Files first (hard requirement), then dirents (best-effort,
// same tier as fsync_parent_dir).
fsync_paths_parallel(paths, true).await?;
@@ -647,4 +697,21 @@ mod tests {
backend.sync_blobs(&[]).await.unwrap();
}
#[test]
fn prefix_slots_cover_lowercase_hex_space_and_preserve_case() {
let mut seen = [false; 22 * 22];
for prefix in HEX_PREFIXES {
let hash = format!("{prefix}{}", "0".repeat(62));
let slot = hash_prefix_slot(&hash).unwrap();
assert!(!seen[slot]);
seen[slot] = true;
}
assert_eq!(seen.into_iter().filter(|value| *value).count(), 256);
assert_ne!(
hash_prefix_slot(&fake_hash("af")),
hash_prefix_slot(&fake_hash("aF"))
);
assert_eq!(hash_prefix_slot("gg"), None);
}
}
+50 -13
View File
@@ -21,7 +21,7 @@ use crate::application::dtos::settings_dto::{
SmtpTestResultDto, StartMigrationDto, TestOidcConnectionDto, TestStorageConnectionDto,
UpdateUserActiveDto, UpdateUserQuotaDto, UpdateUserRoleDto, VerifyMigrationDto,
};
use crate::application::dtos::user_dto::UserDto;
use crate::application::dtos::user_dto::{AdminUserSummaryDto, UserDto};
use crate::application::ports::authorization_ports::AuthorizationEngine;
use crate::application::ports::plugin_ports::{LogQuery, PluginManagementPort, PluginMgmtError};
use crate::application::ports::storage_ports::StorageUsagePort;
@@ -35,6 +35,21 @@ use crate::interfaces::middleware::auth::AuthUser;
use std::sync::Arc;
use uuid::Uuid;
#[derive(serde::Serialize)]
#[serde(untagged)]
enum AdminUsersPayload {
Full(Vec<UserDto>),
Summary(Vec<AdminUserSummaryDto>),
}
#[derive(serde::Serialize)]
struct AdminUsersPageResponse {
users: AdminUsersPayload,
total: i64,
limit: i64,
offset: i64,
}
/// Admin API routes — all require admin role.
pub fn admin_routes() -> Router<Arc<AppState>> {
Router::new()
@@ -747,7 +762,8 @@ pub async fn get_dashboard_stats(
path = "/api/admin/users",
params(
("limit" = Option<i64>, Query, description = "Max users to return (default 100, max 500)"),
("offset" = Option<i64>, Query, description = "Pagination offset")
("offset" = Option<i64>, Query, description = "Pagination offset"),
("summary" = Option<bool>, Query, description = "Return the compact management-table projection")
),
responses(
(status = 200, description = "List of users"),
@@ -759,6 +775,7 @@ pub async fn get_dashboard_stats(
)]
pub async fn list_users(
State(state): State<Arc<AppState>>,
auth_user: AuthUser,
Query(query): Query<ListUsersQueryDto>,
) -> Result<impl IntoResponse, AppError> {
let auth = state
@@ -774,11 +791,31 @@ pub async fn list_users(
// internal-only variant is used by system address book / sharee
// search, where surfacing externals would leak identities. See
// `auth_application_service::list_users` doc for the split.
let users = auth
.auth_application_service
.list_users_including_external(limit, offset)
.await
.map_err(|e| AppError::internal_error(format!("Failed to list users: {}", e)))?;
let users = if query.summary.unwrap_or(false) {
AdminUsersPayload::Summary(
auth.auth_application_service
.list_user_summaries_including_external_with_perms(
state.authorization.as_ref(),
auth_user.id,
limit,
offset,
)
.await
.map_err(AppError::from)?,
)
} else {
AdminUsersPayload::Full(
auth.auth_application_service
.list_users_including_external_with_perms(
state.authorization.as_ref(),
auth_user.id,
limit,
offset,
)
.await
.map_err(AppError::from)?,
)
};
let total = auth
.auth_application_service
@@ -786,12 +823,12 @@ pub async fn list_users(
.await
.unwrap_or(0);
Ok(Json(serde_json::json!({
"users": users,
"total": total,
"limit": limit,
"offset": offset,
})))
Ok(Json(AdminUsersPageResponse {
users,
total,
limit,
offset,
}))
}
/// GET /api/admin/users/:id — get single user