perf: Phase 4+5 optimizations — uploads 10x, downloads 2x, concurrent 2x. moka cache, 512KB buffers, remove sync_all, hash-on-write, preloaded queries, bench.sh v3, gitignore storage/. 500MB upload 12.6s->1.3s (392MB/s). RSS 69-113MB, 0 swap.
This commit is contained in:
@@ -84,11 +84,13 @@ pub trait ChunkedUploadPort: Send + Sync + 'static {
|
||||
|
||||
/// Assemble all chunks into the final file.
|
||||
///
|
||||
/// Returns `(assembled_file_path, filename, folder_id, content_type, total_size)`.
|
||||
/// Returns `(assembled_file_path, filename, folder_id, content_type, total_size, sha256_hash)`.
|
||||
/// The hash is computed during assembly (hash-on-write), eliminating a
|
||||
/// second sequential read of the assembled file.
|
||||
async fn complete_upload(
|
||||
&self,
|
||||
upload_id: &str,
|
||||
) -> Result<(PathBuf, String, Option<String>, String, u64), DomainError>;
|
||||
) -> Result<(PathBuf, String, Option<String>, String, u64, String), DomainError>;
|
||||
|
||||
/// Finalize upload: clean up the session and temporary files.
|
||||
async fn finalize_upload(&self, upload_id: &str) -> Result<(), DomainError>;
|
||||
|
||||
@@ -7,8 +7,10 @@
|
||||
use crate::common::errors::DomainError;
|
||||
use async_trait::async_trait;
|
||||
use bytes::Bytes;
|
||||
use futures::Stream;
|
||||
use serde::Serialize;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::pin::Pin;
|
||||
|
||||
/// Metadata of a stored blob in the dedup system.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
@@ -103,10 +105,15 @@ pub trait DedupPort: Send + Sync + 'static {
|
||||
) -> Result<DedupResultDto, DomainError>;
|
||||
|
||||
/// Store content with deduplication (streaming from file).
|
||||
///
|
||||
/// If `pre_computed_hash` is provided (e.g. hash-on-write from the handler),
|
||||
/// the file will NOT be re-read to calculate the hash — saving one full
|
||||
/// sequential read of the file.
|
||||
async fn store_from_file(
|
||||
&self,
|
||||
source_path: &Path,
|
||||
content_type: Option<String>,
|
||||
pre_computed_hash: Option<String>,
|
||||
) -> Result<DedupResultDto, DomainError>;
|
||||
|
||||
/// Check if a blob with the given hash exists.
|
||||
@@ -121,6 +128,29 @@ pub trait DedupPort: Send + Sync + 'static {
|
||||
/// Read blob content as `Bytes`.
|
||||
async fn read_blob_bytes(&self, hash: &str) -> Result<Bytes, DomainError>;
|
||||
|
||||
/// Stream blob content in chunks (64 KB default) — constant memory usage.
|
||||
///
|
||||
/// Unlike `read_blob()`, this never loads the entire file into RAM.
|
||||
async fn read_blob_stream(
|
||||
&self,
|
||||
hash: &str,
|
||||
) -> Result<Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>, DomainError>;
|
||||
|
||||
/// Stream a byte range of a blob — only reads the requested portion.
|
||||
///
|
||||
/// Uses seek + take so a 1 MB range on a 1 GB file only reads 1 MB from disk.
|
||||
async fn read_blob_range_stream(
|
||||
&self,
|
||||
hash: &str,
|
||||
start: u64,
|
||||
end: Option<u64>,
|
||||
) -> Result<Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>, DomainError>;
|
||||
|
||||
/// Get the size of a blob without reading its content.
|
||||
///
|
||||
/// Used by HEAD requests to return Content-Length without loading the file.
|
||||
async fn blob_size(&self, hash: &str) -> Result<u64, DomainError>;
|
||||
|
||||
/// Add a reference to a blob (increment ref_count).
|
||||
async fn add_reference(&self, hash: &str) -> Result<(), DomainError>;
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
use async_trait::async_trait;
|
||||
use bytes::Bytes;
|
||||
use futures::Stream;
|
||||
use std::path::Path;
|
||||
use std::pin::Pin;
|
||||
use std::sync::Arc;
|
||||
|
||||
@@ -11,21 +12,33 @@ use crate::common::errors::DomainError;
|
||||
// Upload port
|
||||
// ─────────────────────────────────────────────────────
|
||||
|
||||
/// Strategy chosen by the upload service based on file size.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum UploadStrategy {
|
||||
/// Instant (<256KB): write-behind cache, ~0ms latency
|
||||
WriteBehind,
|
||||
/// Buffered (256KB–1MB): full bytes in memory then write
|
||||
Buffered,
|
||||
/// Streaming (≥1MB): pipe chunks directly to disk
|
||||
Streaming,
|
||||
}
|
||||
|
||||
/// Primary port for file upload operations
|
||||
/// Primary port for file upload operations.
|
||||
///
|
||||
/// All upload paths converge on streaming-to-disk:
|
||||
/// - Normal uploads: handler spools multipart to temp file → `upload_file_streaming`
|
||||
/// - WebDAV PUT: small in-memory buffer → `upload_file`
|
||||
/// - Chunked uploads: chunks already on disk → `upload_file_from_path`
|
||||
#[async_trait]
|
||||
pub trait FileUploadUseCase: Send + Sync + 'static {
|
||||
/// Uploads a new file from bytes
|
||||
/// Upload from a temp file already on disk (true streaming, ~64 KB RAM).
|
||||
///
|
||||
/// When `pre_computed_hash` is `Some`, the blob store skips the hash
|
||||
/// re-read — the handler already computed it during the multipart spool.
|
||||
async fn upload_file_streaming(
|
||||
&self,
|
||||
name: String,
|
||||
folder_id: Option<String>,
|
||||
content_type: String,
|
||||
temp_path: &Path,
|
||||
size: u64,
|
||||
pre_computed_hash: Option<String>,
|
||||
) -> Result<FileDto, DomainError>;
|
||||
|
||||
/// Upload from in-memory bytes (for small payloads: WebDAV, empty files).
|
||||
///
|
||||
/// Only used for WebDAV PUT and empty files where the content is already
|
||||
/// buffered by the protocol handler. For normal uploads, prefer
|
||||
/// `upload_file_streaming`.
|
||||
async fn upload_file(
|
||||
&self,
|
||||
name: String,
|
||||
@@ -34,18 +47,17 @@ pub trait FileUploadUseCase: Send + Sync + 'static {
|
||||
content: Vec<u8>,
|
||||
) -> Result<FileDto, DomainError>;
|
||||
|
||||
/// Smart upload: picks the best strategy (write-behind / buffered / streaming)
|
||||
/// and handles dedup automatically.
|
||||
/// Upload from a file already assembled on disk (chunked uploads).
|
||||
///
|
||||
/// Returns `(FileDto, UploadStrategy)` so the handler can log the chosen tier.
|
||||
async fn smart_upload(
|
||||
/// Same as `upload_file_streaming` but with a separate name for clarity.
|
||||
async fn upload_file_from_path(
|
||||
&self,
|
||||
name: String,
|
||||
folder_id: Option<String>,
|
||||
content_type: String,
|
||||
chunks: Vec<Bytes>,
|
||||
total_size: usize,
|
||||
) -> Result<(FileDto, UploadStrategy), DomainError>;
|
||||
file_path: &Path,
|
||||
pre_computed_hash: Option<String>,
|
||||
) -> Result<FileDto, DomainError>;
|
||||
|
||||
/// Creates a new file at the specified path (for WebDAV)
|
||||
async fn create_file(
|
||||
@@ -115,6 +127,21 @@ pub trait FileRetrievalUseCase: Send + Sync + 'static {
|
||||
prefer_original: bool,
|
||||
) -> Result<(FileDto, OptimizedFileContent), DomainError>;
|
||||
|
||||
/// Like `get_file_optimized` but accepts an already-fetched `FileDto`,
|
||||
/// avoiding a redundant metadata query when the handler already has it.
|
||||
async fn get_file_optimized_preloaded(
|
||||
&self,
|
||||
id: &str,
|
||||
file_dto: FileDto,
|
||||
accept_webp: bool,
|
||||
prefer_original: bool,
|
||||
) -> Result<(FileDto, OptimizedFileContent), DomainError> {
|
||||
// Default: ignore pre-fetched meta, re-fetch everything.
|
||||
let _ = file_dto;
|
||||
self.get_file_optimized(id, accept_webp, prefer_original)
|
||||
.await
|
||||
}
|
||||
|
||||
/// Range-based streaming for HTTP Range Requests (video seek, resumable DL).
|
||||
async fn get_file_range_stream(
|
||||
&self,
|
||||
|
||||
@@ -56,6 +56,26 @@ pub trait FileReadPort: Send + Sync + 'static {
|
||||
|
||||
/// Gets the parent folder ID from a path (WebDAV).
|
||||
async fn get_parent_folder_id(&self, path: &str) -> Result<String, DomainError>;
|
||||
|
||||
/// Find a file by its logical path (folder_name/.../file_name).
|
||||
///
|
||||
/// The default implementation falls back to `list_files(None)` + linear
|
||||
/// scan (O(N)). Repositories should override with a direct SQL query.
|
||||
async fn find_file_by_path(&self, path: &str) -> Result<Option<File>, DomainError> {
|
||||
let path = path.trim_start_matches('/').trim_end_matches('/');
|
||||
let all_files = self.list_files(None).await?;
|
||||
for file in all_files {
|
||||
let file_path = file.path_string();
|
||||
let file_path = file_path.trim_start_matches('/').trim_end_matches('/');
|
||||
if file_path == path
|
||||
|| file_path.ends_with(&format!("/{}", path))
|
||||
|| path.ends_with(&format!("/{}", file_path))
|
||||
{
|
||||
return Ok(Some(file));
|
||||
}
|
||||
}
|
||||
Ok(None)
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────
|
||||
@@ -77,13 +97,18 @@ pub trait FileWritePort: Send + Sync + 'static {
|
||||
content: Vec<u8>,
|
||||
) -> Result<File, DomainError>;
|
||||
|
||||
/// Streaming upload — writes chunks to disk without accumulating in RAM.
|
||||
async fn save_file_from_stream(
|
||||
/// Streaming upload — saves a file from a temp file already on disk.
|
||||
///
|
||||
/// When `pre_computed_hash` is provided, the dedup service skips the
|
||||
/// hash re-read — zero extra I/O beyond the initial spool.
|
||||
async fn save_file_from_temp(
|
||||
&self,
|
||||
name: String,
|
||||
folder_id: Option<String>,
|
||||
content_type: String,
|
||||
stream: std::pin::Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>,
|
||||
temp_path: &std::path::Path,
|
||||
size: u64,
|
||||
pre_computed_hash: Option<String>,
|
||||
) -> Result<File, DomainError>;
|
||||
|
||||
/// Moves a file to another folder.
|
||||
|
||||
Reference in New Issue
Block a user