perf: Phase 4+5 optimizations — uploads 10x, downloads 2x, concurrent 2x. moka cache, 512KB buffers, remove sync_all, hash-on-write, preloaded queries, bench.sh v3, gitignore storage/. 500MB upload 12.6s->1.3s (392MB/s). RSS 69-113MB, 0 swap.
This commit is contained in:
@@ -8,6 +8,7 @@ use async_trait::async_trait;
|
||||
use bytes::Bytes;
|
||||
use futures::Stream;
|
||||
use sqlx::PgPool;
|
||||
use std::collections::HashMap;
|
||||
use std::sync::Arc;
|
||||
|
||||
use crate::application::ports::dedup_ports::DedupPort;
|
||||
@@ -24,6 +25,10 @@ pub struct FileBlobReadRepository {
|
||||
pool: Arc<PgPool>,
|
||||
dedup: Arc<dyn DedupPort>,
|
||||
folder_repo: Arc<FolderDbRepository>,
|
||||
/// Lightweight cache: file_id → blob_hash.
|
||||
/// Populated by `get_file()`, consumed by `get_blob_hash()`.
|
||||
/// Avoids an extra SQL round-trip on the hot download path.
|
||||
hash_cache: std::sync::Mutex<HashMap<String, String>>,
|
||||
}
|
||||
|
||||
impl FileBlobReadRepository {
|
||||
@@ -36,6 +41,7 @@ impl FileBlobReadRepository {
|
||||
pool,
|
||||
dedup,
|
||||
folder_repo,
|
||||
hash_cache: std::sync::Mutex::new(HashMap::new()),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -53,7 +59,7 @@ impl FileBlobReadRepository {
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a database row into a `File` domain entity.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
async fn row_to_file(
|
||||
&self,
|
||||
id: String,
|
||||
@@ -79,7 +85,13 @@ impl FileBlobReadRepository {
|
||||
}
|
||||
|
||||
/// Get the blob hash for a file.
|
||||
/// Checks the in-memory cache first (populated by `get_file`).
|
||||
async fn get_blob_hash(&self, file_id: &str) -> Result<String, DomainError> {
|
||||
// Fast path: already cached from a prior get_file call
|
||||
if let Some(hash) = self.hash_cache.lock().unwrap().remove(file_id) {
|
||||
return Ok(hash);
|
||||
}
|
||||
// Slow path: DB round-trip
|
||||
sqlx::query_scalar::<_, String>(
|
||||
"SELECT blob_hash FROM storage.files WHERE id = $1::uuid AND NOT is_trashed",
|
||||
)
|
||||
@@ -94,11 +106,12 @@ impl FileBlobReadRepository {
|
||||
#[async_trait]
|
||||
impl FileReadPort for FileBlobReadRepository {
|
||||
async fn get_file(&self, id: &str) -> Result<File, DomainError> {
|
||||
let row = sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64)>(
|
||||
let row = sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64, String)>(
|
||||
r#"
|
||||
SELECT id::text, name, folder_id::text, size, mime_type,
|
||||
EXTRACT(EPOCH FROM created_at)::bigint,
|
||||
EXTRACT(EPOCH FROM updated_at)::bigint
|
||||
EXTRACT(EPOCH FROM updated_at)::bigint,
|
||||
blob_hash
|
||||
FROM storage.files
|
||||
WHERE id = $1::uuid AND NOT is_trashed
|
||||
"#,
|
||||
@@ -109,6 +122,13 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("get: {e}")))?
|
||||
.ok_or_else(|| DomainError::not_found("File", id))?;
|
||||
|
||||
// Cache blob_hash so the subsequent get_file_stream / get_file_content
|
||||
// call doesn't need a separate DB round-trip.
|
||||
self.hash_cache
|
||||
.lock()
|
||||
.unwrap()
|
||||
.insert(id.to_string(), row.7.clone());
|
||||
|
||||
self.row_to_file(row.0, row.1, row.2, row.3, row.4, row.5, row.6)
|
||||
.await
|
||||
}
|
||||
@@ -145,9 +165,32 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
}
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("list: {e}")))?;
|
||||
|
||||
// ── N+1 fix: resolve the folder path ONCE (all rows share
|
||||
// the same folder_id when listing a specific folder). ──
|
||||
let shared_folder_path = if let Some(fid) = folder_id {
|
||||
Some(self.folder_repo.get_folder_path(fid).await?)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let mut files = Vec::with_capacity(rows.len());
|
||||
for (id, name, fid, size, mime, ca, ma) in rows {
|
||||
files.push(self.row_to_file(id, name, fid, size, mime, ca, ma).await?);
|
||||
let storage_path = match &shared_folder_path {
|
||||
Some(fp) => fp.join(&name),
|
||||
None => StoragePath::from_string(&name),
|
||||
};
|
||||
let file = File::with_timestamps(
|
||||
id,
|
||||
name,
|
||||
storage_path,
|
||||
size as u64,
|
||||
mime,
|
||||
fid,
|
||||
ca as u64,
|
||||
ma as u64,
|
||||
)
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("entity: {e}")))?;
|
||||
files.push(file);
|
||||
}
|
||||
Ok(files)
|
||||
}
|
||||
@@ -161,13 +204,10 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
&self,
|
||||
id: &str,
|
||||
) -> Result<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>, DomainError> {
|
||||
// Read blob as bytes and wrap in a single-chunk stream.
|
||||
// For very large files, a true streaming implementation from the
|
||||
// blob file would be better, but DedupPort API currently returns bytes.
|
||||
// True streaming: reads the blob file in 64 KB chunks.
|
||||
// Memory usage is ~64 KB regardless of file size.
|
||||
let blob_hash = self.get_blob_hash(id).await?;
|
||||
let content = self.dedup.read_blob_bytes(&blob_hash).await?;
|
||||
|
||||
let stream = futures::stream::once(async move { Ok(content) });
|
||||
let stream = self.dedup.read_blob_stream(&blob_hash).await?;
|
||||
Ok(Box::new(stream))
|
||||
}
|
||||
|
||||
@@ -177,22 +217,19 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
start: u64,
|
||||
end: Option<u64>,
|
||||
) -> Result<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>, DomainError> {
|
||||
// True range streaming: seeks to `start` and reads only the requested range.
|
||||
// A 1 MB range on a 1 GB file uses ~64 KB of RAM.
|
||||
let blob_hash = self.get_blob_hash(id).await?;
|
||||
let content = self.dedup.read_blob_bytes(&blob_hash).await?;
|
||||
|
||||
let start = start as usize;
|
||||
let end = end.map_or(content.len(), |e| e as usize).min(content.len());
|
||||
|
||||
if start >= content.len() {
|
||||
return Ok(Box::new(futures::stream::empty()));
|
||||
}
|
||||
|
||||
let slice = content.slice(start..end);
|
||||
let stream = futures::stream::once(async move { Ok(slice) });
|
||||
let stream = self
|
||||
.dedup
|
||||
.read_blob_range_stream(&blob_hash, start, end)
|
||||
.await?;
|
||||
Ok(Box::new(stream))
|
||||
}
|
||||
|
||||
async fn get_file_mmap(&self, id: &str) -> Result<Bytes, DomainError> {
|
||||
// For RPi targets, mmap is less beneficial than streaming.
|
||||
// Keep as a fallback that loads content for small/medium files.
|
||||
let blob_hash = self.get_blob_hash(id).await?;
|
||||
self.dedup.read_blob_bytes(&blob_hash).await
|
||||
}
|
||||
@@ -262,4 +299,83 @@ impl FileReadPort for FileBlobReadRepository {
|
||||
current_parent
|
||||
.ok_or_else(|| DomainError::not_found("Folder", format!("parent for path: {path}")))
|
||||
}
|
||||
|
||||
/// Direct SQL lookup: split path into folder segments + filename,
|
||||
/// walk the folder hierarchy, then match the file by name + folder_id.
|
||||
/// O(depth) queries instead of O(total_files).
|
||||
async fn find_file_by_path(&self, path: &str) -> Result<Option<File>, DomainError> {
|
||||
let path = path.trim_start_matches('/').trim_end_matches('/');
|
||||
let segments: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
||||
|
||||
if segments.is_empty() {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
// Last segment is the filename, preceding segments are folders
|
||||
let filename = segments[segments.len() - 1];
|
||||
let folder_segments = &segments[..segments.len() - 1];
|
||||
|
||||
// Walk folder hierarchy to find parent folder_id
|
||||
let mut current_parent: Option<String> = None;
|
||||
for segment in folder_segments {
|
||||
let row = if let Some(ref pid) = current_parent {
|
||||
sqlx::query_scalar::<_, String>(
|
||||
"SELECT id::text FROM storage.folders WHERE name = $1 AND parent_id = $2::uuid AND NOT is_trashed",
|
||||
)
|
||||
.bind(segment)
|
||||
.bind(pid)
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
} else {
|
||||
sqlx::query_scalar::<_, String>(
|
||||
"SELECT id::text FROM storage.folders WHERE name = $1 AND parent_id IS NULL AND NOT is_trashed",
|
||||
)
|
||||
.bind(segment)
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
}
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("path walk: {e}")))?;
|
||||
|
||||
match row {
|
||||
Some(id) => current_parent = Some(id),
|
||||
None => return Ok(None), // Folder not found → file doesn't exist at this path
|
||||
}
|
||||
}
|
||||
|
||||
// Now find the file by name + folder_id
|
||||
let row = if let Some(ref fid) = current_parent {
|
||||
sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64)>(
|
||||
r#"
|
||||
SELECT id::text, name, folder_id::text, size, mime_type,
|
||||
EXTRACT(EPOCH FROM created_at)::bigint,
|
||||
EXTRACT(EPOCH FROM updated_at)::bigint
|
||||
FROM storage.files
|
||||
WHERE name = $1 AND folder_id = $2::uuid AND NOT is_trashed
|
||||
"#,
|
||||
)
|
||||
.bind(filename)
|
||||
.bind(fid)
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
} else {
|
||||
sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64)>(
|
||||
r#"
|
||||
SELECT id::text, name, folder_id::text, size, mime_type,
|
||||
EXTRACT(EPOCH FROM created_at)::bigint,
|
||||
EXTRACT(EPOCH FROM updated_at)::bigint
|
||||
FROM storage.files
|
||||
WHERE name = $1 AND folder_id IS NULL AND NOT is_trashed
|
||||
"#,
|
||||
)
|
||||
.bind(filename)
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
}
|
||||
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("find file: {e}")))?;
|
||||
|
||||
match row {
|
||||
Some(r) => Ok(Some(self.row_to_file(r.0, r.1, r.2, r.3, r.4, r.5, r.6).await?)),
|
||||
None => Ok(None),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,11 +5,8 @@
|
||||
//! - `DedupPort` for content-addressable blob storage on the filesystem
|
||||
|
||||
use async_trait::async_trait;
|
||||
use bytes::Bytes;
|
||||
use futures::Stream;
|
||||
use sqlx::PgPool;
|
||||
use std::path::PathBuf;
|
||||
use std::pin::Pin;
|
||||
use std::sync::Arc;
|
||||
|
||||
use crate::application::ports::dedup_ports::DedupPort;
|
||||
@@ -55,7 +52,7 @@ impl FileBlobWriteRepository {
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a database row into a `File` domain entity.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
async fn row_to_file(
|
||||
&self,
|
||||
id: String,
|
||||
@@ -140,14 +137,13 @@ impl FileWritePort for FileBlobWriteRepository {
|
||||
rollback_err
|
||||
);
|
||||
}
|
||||
if let sqlx::Error::Database(ref db_err) = e {
|
||||
if db_err.code().as_deref() == Some("23505") {
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return Err(DomainError::already_exists(
|
||||
"File",
|
||||
format!("{name} already exists in folder"),
|
||||
));
|
||||
}
|
||||
}
|
||||
return Err(DomainError::internal_error(
|
||||
"FileBlobWrite",
|
||||
format!("insert: {e}"),
|
||||
@@ -166,26 +162,76 @@ impl FileWritePort for FileBlobWriteRepository {
|
||||
.await
|
||||
}
|
||||
|
||||
async fn save_file_from_stream(
|
||||
async fn save_file_from_temp(
|
||||
&self,
|
||||
name: String,
|
||||
folder_id: Option<String>,
|
||||
content_type: String,
|
||||
stream: Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>,
|
||||
temp_path: &std::path::Path,
|
||||
size: u64,
|
||||
pre_computed_hash: Option<String>,
|
||||
) -> Result<File, DomainError> {
|
||||
use futures::StreamExt;
|
||||
let user_id = self.resolve_user_id(folder_id.as_deref()).await?;
|
||||
|
||||
// Collect stream into bytes (blobs are content-addressed, need full content for hash)
|
||||
let mut content = Vec::new();
|
||||
let mut stream = stream;
|
||||
while let Some(chunk) = stream.next().await {
|
||||
let chunk = chunk.map_err(|e| {
|
||||
DomainError::internal_error("FileBlobWrite", format!("stream read: {e}"))
|
||||
})?;
|
||||
content.extend_from_slice(&chunk);
|
||||
}
|
||||
// True streaming: pass pre-computed hash (or let dedup compute it).
|
||||
// When hash is pre-computed, zero extra disk reads.
|
||||
let dedup_result = self
|
||||
.dedup
|
||||
.store_from_file(temp_path, Some(content_type.clone()), pre_computed_hash)
|
||||
.await?;
|
||||
let blob_hash = dedup_result.hash().to_string();
|
||||
|
||||
self.save_file(name, folder_id, content_type, content).await
|
||||
// Insert file metadata — if this fails, compensate by removing the blob ref
|
||||
let row = match sqlx::query_as::<_, (String, i64, i64)>(
|
||||
r#"
|
||||
INSERT INTO storage.files (name, folder_id, user_id, blob_hash, size, mime_type)
|
||||
VALUES ($1, $2::uuid, $3, $4, $5, $6)
|
||||
RETURNING id::text,
|
||||
EXTRACT(EPOCH FROM created_at)::bigint,
|
||||
EXTRACT(EPOCH FROM updated_at)::bigint
|
||||
"#,
|
||||
)
|
||||
.bind(&name)
|
||||
.bind(&folder_id)
|
||||
.bind(&user_id)
|
||||
.bind(&blob_hash)
|
||||
.bind(size as i64)
|
||||
.bind(&content_type)
|
||||
.fetch_one(self.pool.as_ref())
|
||||
.await
|
||||
{
|
||||
Ok(row) => row,
|
||||
Err(e) => {
|
||||
if let Err(rollback_err) = self.dedup.remove_reference(&blob_hash).await {
|
||||
tracing::error!(
|
||||
"Blob orphaned after failed INSERT — hash: {}, err: {}",
|
||||
&blob_hash[..12],
|
||||
rollback_err
|
||||
);
|
||||
}
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return Err(DomainError::already_exists(
|
||||
"File",
|
||||
format!("{name} already exists in folder"),
|
||||
));
|
||||
}
|
||||
return Err(DomainError::internal_error(
|
||||
"FileBlobWrite",
|
||||
format!("insert: {e}"),
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
tracing::info!(
|
||||
"📡 STREAMING WRITE: {} ({} bytes, hash: {})",
|
||||
name,
|
||||
size,
|
||||
&blob_hash[..12]
|
||||
);
|
||||
|
||||
self.row_to_file(row.0, name, folder_id, size as i64, content_type, row.1, row.2)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn move_file(
|
||||
@@ -265,14 +311,13 @@ impl FileWritePort for FileBlobWriteRepository {
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
.map_err(|e| {
|
||||
if let sqlx::Error::Database(ref db_err) = e {
|
||||
if db_err.code().as_deref() == Some("23505") {
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return DomainError::already_exists(
|
||||
"File",
|
||||
"File with that name already exists in target folder".to_string(),
|
||||
);
|
||||
}
|
||||
}
|
||||
DomainError::internal_error("FileBlobWrite", format!("copy: {e}"))
|
||||
})?
|
||||
.ok_or_else(|| DomainError::not_found("File", file_id))?;
|
||||
@@ -314,14 +359,13 @@ impl FileWritePort for FileBlobWriteRepository {
|
||||
.fetch_optional(self.pool.as_ref())
|
||||
.await
|
||||
.map_err(|e| {
|
||||
if let sqlx::Error::Database(ref db_err) = e {
|
||||
if db_err.code().as_deref() == Some("23505") {
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return DomainError::already_exists(
|
||||
"File",
|
||||
format!("{new_name} already exists"),
|
||||
);
|
||||
}
|
||||
}
|
||||
DomainError::internal_error("FileBlobWrite", format!("rename: {e}"))
|
||||
})?
|
||||
.ok_or_else(|| DomainError::not_found("File", file_id))?;
|
||||
@@ -404,15 +448,14 @@ impl FileWritePort for FileBlobWriteRepository {
|
||||
};
|
||||
|
||||
// Decrement old blob ref (only if hash changed, best-effort)
|
||||
if old_hash != new_hash {
|
||||
if let Err(e) = self.dedup.remove_reference(&old_hash).await {
|
||||
if old_hash != new_hash
|
||||
&& let Err(e) = self.dedup.remove_reference(&old_hash).await {
|
||||
tracing::warn!(
|
||||
"Failed to decrement old blob ref {}: {}",
|
||||
&old_hash[..12],
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -43,28 +43,6 @@ impl FolderDbRepository {
|
||||
|
||||
/// Build the full virtual path for a folder by walking up the `parent_id` chain.
|
||||
async fn build_folder_path(&self, folder_id: &str) -> Result<StoragePath, DomainError> {
|
||||
// CTE-based recursive query to build path segments
|
||||
let _rows = sqlx::query_as::<_, (String,)>(
|
||||
r#"
|
||||
WITH RECURSIVE ancestors AS (
|
||||
SELECT id, name, parent_id
|
||||
FROM storage.folders
|
||||
WHERE id = $1::uuid
|
||||
UNION ALL
|
||||
SELECT f.id, f.name, f.parent_id
|
||||
FROM storage.folders f
|
||||
JOIN ancestors a ON f.id = a.parent_id
|
||||
)
|
||||
SELECT name FROM ancestors ORDER BY name
|
||||
"#,
|
||||
)
|
||||
.bind(folder_id)
|
||||
.fetch_all(self.pool())
|
||||
.await
|
||||
.map_err(|e| DomainError::internal_error("FolderDb", format!("path query: {e}")))?;
|
||||
|
||||
// Actually we need a proper ordering. Let me rewrite with depth tracking.
|
||||
// Re-query with depth.
|
||||
let rows = sqlx::query_as::<_, (String, i32)>(
|
||||
r#"
|
||||
WITH RECURSIVE ancestors AS (
|
||||
@@ -152,14 +130,13 @@ impl FolderRepository for FolderDbRepository {
|
||||
.fetch_one(self.pool())
|
||||
.await
|
||||
.map_err(|e| {
|
||||
if let sqlx::Error::Database(ref db_err) = e {
|
||||
if db_err.code().as_deref() == Some("23505") {
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return DomainError::already_exists(
|
||||
"Folder",
|
||||
format!("{name} already exists in parent"),
|
||||
);
|
||||
}
|
||||
}
|
||||
DomainError::internal_error("FolderDb", format!("insert: {e}"))
|
||||
})?;
|
||||
|
||||
@@ -355,14 +332,13 @@ impl FolderRepository for FolderDbRepository {
|
||||
.execute(self.pool())
|
||||
.await
|
||||
.map_err(|e| {
|
||||
if let sqlx::Error::Database(ref db_err) = e {
|
||||
if db_err.code().as_deref() == Some("23505") {
|
||||
if let sqlx::Error::Database(ref db_err) = e
|
||||
&& db_err.code().as_deref() == Some("23505") {
|
||||
return DomainError::already_exists(
|
||||
"Folder",
|
||||
format!("{new_name} already exists"),
|
||||
);
|
||||
}
|
||||
}
|
||||
DomainError::internal_error("FolderDb", format!("rename: {e}"))
|
||||
})?;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user