2843b3351b
purpose: avoid current scheme:
1. write to disk
2. reopen file to read data and digest it
now: digest is done while writing data to disk
all other implementation than Nextcloud are corrrect
403 lines
16 KiB
Rust
403 lines
16 KiB
Rust
use std::path::PathBuf;
|
||
use tokio::fs;
|
||
use tokio::io::AsyncWriteExt;
|
||
|
||
use crate::common::errors::{DomainError, Result};
|
||
|
||
#[derive(Clone)]
|
||
pub struct NextcloudChunkedUploadService {
|
||
pub base_dir: PathBuf,
|
||
}
|
||
|
||
impl NextcloudChunkedUploadService {
|
||
pub fn new(base_dir: PathBuf) -> Self {
|
||
Self { base_dir }
|
||
}
|
||
|
||
pub fn new_stub() -> Self {
|
||
Self {
|
||
base_dir: PathBuf::from("./storage/.uploads/nextcloud"),
|
||
}
|
||
}
|
||
|
||
/// Validate that a path component contains no traversal characters.
|
||
fn validate_path_component(name: &str, label: &str) -> Result<()> {
|
||
if name.is_empty()
|
||
|| name.contains('/')
|
||
|| name.contains('\\')
|
||
|| name.contains("..")
|
||
|| name == "."
|
||
{
|
||
return Err(DomainError::validation_error(format!(
|
||
"ChunkedUpload: invalid {}: contains path traversal characters",
|
||
label
|
||
)));
|
||
}
|
||
Ok(())
|
||
}
|
||
|
||
/// Build a session directory path and verify it's inside base_dir.
|
||
fn safe_session_dir(&self, user: &str, upload_id: &str) -> Result<PathBuf> {
|
||
Self::validate_path_component(user, "username")?;
|
||
Self::validate_path_component(upload_id, "upload_id")?;
|
||
Ok(self.base_dir.join(user).join(upload_id))
|
||
}
|
||
|
||
/// Create a new upload session directory.
|
||
pub async fn create_session(&self, user: &str, upload_id: &str) -> Result<()> {
|
||
let session_dir = self.safe_session_dir(user, upload_id)?;
|
||
fs::create_dir_all(&session_dir)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
Ok(())
|
||
}
|
||
|
||
/// Resolve and validate the filesystem path for a chunk file.
|
||
///
|
||
/// Public so the interface layer can stream an HTTP body straight into
|
||
/// the chunk file without copying through the service. The service
|
||
/// retains responsibility for path-component validation; the caller
|
||
/// owns the I/O (open, write, fsync, size enforcement, cleanup on
|
||
/// failure). All three `validate_path_component` calls run before the
|
||
/// path is constructed, so a returned `PathBuf` is always inside
|
||
/// `base_dir/{user}/{upload_id}`.
|
||
pub fn safe_chunk_path(
|
||
&self,
|
||
user: &str,
|
||
upload_id: &str,
|
||
chunk_name: &str,
|
||
) -> Result<PathBuf> {
|
||
Self::validate_path_component(chunk_name, "chunk_name")?;
|
||
Ok(self.safe_session_dir(user, upload_id)?.join(chunk_name))
|
||
}
|
||
|
||
/// Store a chunk in the session directory. Buffers `data` in memory —
|
||
/// use [`safe_chunk_path`](Self::safe_chunk_path) + the
|
||
/// `interfaces/upload_spool::stream_body_to_path` helper to stream the
|
||
/// HTTP body directly to disk and avoid materialising the whole chunk
|
||
/// in RAM.
|
||
pub async fn store_chunk(
|
||
&self,
|
||
user: &str,
|
||
upload_id: &str,
|
||
chunk_name: &str,
|
||
data: &[u8],
|
||
) -> Result<()> {
|
||
let chunk_path = self.safe_chunk_path(user, upload_id, chunk_name)?;
|
||
let mut file = fs::File::create(&chunk_path)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
file.write_all(data)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
Ok(())
|
||
}
|
||
|
||
/// Assemble all chunks in numeric order into a temp file, computing
|
||
/// the BLAKE3 of the concatenated stream **during** the same read/
|
||
/// write pass (hash-on-write).
|
||
///
|
||
/// Returns `(temp_path, total_size, blake3_hex)`. The caller passes
|
||
/// the hash to the upload service as `pre_computed_hash` so the
|
||
/// downstream dedup layer never has to re-read the assembled file
|
||
/// to compute it — saving one full file-sized read pass per upload.
|
||
///
|
||
/// The read/hash/write loop runs inside `spawn_blocking` because
|
||
/// BLAKE3 is CPU-bound and would otherwise starve the Tokio worker
|
||
/// running other connections; synchronous I/O is used inside the
|
||
/// blocking thread because the workload is sequential and the
|
||
/// async reactor overhead would only slow it down. For files larger
|
||
/// than ~10 MB BLAKE3's Rayon mode parallelises across cores —
|
||
/// mirrors what `ChunkedUploadService::complete_upload_inner` does
|
||
/// for the REST chunked path.
|
||
pub async fn assemble(&self, user: &str, upload_id: &str) -> Result<(PathBuf, u64, String)> {
|
||
let session_dir = self.safe_session_dir(user, upload_id)?;
|
||
let mut entries: Vec<String> = Vec::new();
|
||
|
||
let mut dir = fs::read_dir(&session_dir)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
|
||
while let Some(entry) = dir
|
||
.next_entry()
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?
|
||
{
|
||
let name = entry.file_name().to_string_lossy().to_string();
|
||
if name == ".file" {
|
||
continue; // Skip the assembly marker.
|
||
}
|
||
entries.push(name);
|
||
}
|
||
|
||
// Sort chunks numerically (Nextcloud sends them as "00001", "00002", ...).
|
||
entries.sort();
|
||
|
||
let temp_path = session_dir.join(".assembled");
|
||
let chunk_paths: Vec<PathBuf> = entries.iter().map(|n| session_dir.join(n)).collect();
|
||
let assembled_for_blocking = temp_path.clone();
|
||
|
||
// Read/hash/write loop runs synchronously on the blocking pool.
|
||
// BLAKE3 is computed in the same pass that copies bytes from chunk
|
||
// files into the assembled file — no second read after the fact.
|
||
let (total_size, hash) =
|
||
tokio::task::spawn_blocking(move || -> std::io::Result<(u64, String)> {
|
||
use std::io::{BufWriter as StdBufWriter, Read, Write};
|
||
|
||
let raw_output = std::fs::OpenOptions::new()
|
||
.create(true)
|
||
.write(true)
|
||
.truncate(true)
|
||
.open(&assembled_for_blocking)?;
|
||
|
||
// 512 KB write buffer — 8× fewer syscalls than 64 KB.
|
||
let mut output = StdBufWriter::with_capacity(524_288, raw_output);
|
||
let mut hasher = blake3::Hasher::new();
|
||
let mut buf = vec![0u8; 524_288];
|
||
let mut total: u64 = 0;
|
||
|
||
// Files >10 MB benefit from BLAKE3's multi-threaded mode.
|
||
// The threshold matches the REST chunked path's heuristic.
|
||
const RAYON_THRESHOLD_PER_FRAME: usize = 128 * 1024;
|
||
|
||
for chunk_path in &chunk_paths {
|
||
let mut chunk_file = std::fs::File::open(chunk_path)?;
|
||
loop {
|
||
let n = chunk_file.read(&mut buf)?;
|
||
if n == 0 {
|
||
break;
|
||
}
|
||
if n >= RAYON_THRESHOLD_PER_FRAME {
|
||
hasher.update_rayon(&buf[..n]);
|
||
} else {
|
||
hasher.update(&buf[..n]);
|
||
}
|
||
output.write_all(&buf[..n])?;
|
||
total += n as u64;
|
||
}
|
||
}
|
||
|
||
output.flush()?;
|
||
// ── Durability boundary ─────────────────────────────────
|
||
// sync_all is the actual fsync; without it, a power loss
|
||
// before the kernel writeback timer (~5 s) loses
|
||
// acknowledged data. Pull the inner File out of the
|
||
// BufWriter so we can sync the underlying handle —
|
||
// dropping the BufWriter wouldn't trigger fsync. macOS
|
||
// caveat: fsync there flushes to the disk controller
|
||
// only; true durability needs F_FULLFSYNC, not exposed
|
||
// by std.
|
||
let raw_output = output
|
||
.into_inner()
|
||
.map_err(|e| std::io::Error::other(format!("into_inner: {e}")))?;
|
||
raw_output.sync_all()?;
|
||
|
||
Ok((total, hasher.finalize().to_hex().to_string()))
|
||
})
|
||
.await
|
||
.map_err(|e| {
|
||
DomainError::internal_error("ChunkedUpload", format!("assemble task: {e}"))
|
||
})?
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
|
||
Ok((temp_path, total_size, hash))
|
||
}
|
||
|
||
/// Delete the upload session directory.
|
||
pub async fn cleanup(&self, user: &str, upload_id: &str) -> Result<()> {
|
||
let session_dir = self.safe_session_dir(user, upload_id)?;
|
||
if session_dir.exists() {
|
||
fs::remove_dir_all(&session_dir)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
}
|
||
Ok(())
|
||
}
|
||
|
||
/// Check if a session directory exists.
|
||
pub async fn session_exists(&self, user: &str, upload_id: &str) -> bool {
|
||
self.safe_session_dir(user, upload_id)
|
||
.map(|p| p.exists())
|
||
.unwrap_or(false)
|
||
}
|
||
|
||
/// Enumerate the chunks already stored in a session, plus the
|
||
/// session directory's own mtime. Used by the PROPFIND handler
|
||
/// to drive NextCloud's resume-upload flow — the Android client
|
||
/// (and several mobile clients) issue PROPFIND on the session
|
||
/// URL to discover which chunks are already uploaded, then only
|
||
/// PUT the missing ones.
|
||
///
|
||
/// Returns `None` when the session directory doesn't exist
|
||
/// (handler maps to 404). The `.file` and `.assembled` markers
|
||
/// are filtered out — they're internal bookkeeping, not real
|
||
/// chunks the client uploaded.
|
||
pub async fn list_chunks(&self, user: &str, upload_id: &str) -> Result<Option<SessionListing>> {
|
||
let session_dir = self.safe_session_dir(user, upload_id)?;
|
||
if !session_dir.exists() {
|
||
return Ok(None);
|
||
}
|
||
|
||
let session_meta = fs::metadata(&session_dir)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
let session_mtime = session_meta
|
||
.modified()
|
||
.ok()
|
||
.and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok())
|
||
.map(|d| d.as_secs())
|
||
.unwrap_or(0);
|
||
|
||
let mut chunks: Vec<ChunkInfo> = Vec::new();
|
||
let mut dir = fs::read_dir(&session_dir)
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
while let Some(entry) = dir
|
||
.next_entry()
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?
|
||
{
|
||
let name = entry.file_name().to_string_lossy().to_string();
|
||
// Filter internal markers — `.file` is the NC-protocol
|
||
// assembly trigger target (it never reaches the disk
|
||
// because MOVE redirects it), `.assembled` is our own
|
||
// staging file from `assemble()`. Surfacing either to
|
||
// the client would confuse its chunk-count check.
|
||
if name == ".file" || name == ".assembled" {
|
||
continue;
|
||
}
|
||
let meta = entry
|
||
.metadata()
|
||
.await
|
||
.map_err(|e| DomainError::internal_error("ChunkedUpload", e.to_string()))?;
|
||
let size = meta.len();
|
||
let mtime = meta
|
||
.modified()
|
||
.ok()
|
||
.and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok())
|
||
.map(|d| d.as_secs())
|
||
.unwrap_or(0);
|
||
chunks.push(ChunkInfo { name, size, mtime });
|
||
}
|
||
// Sort by chunk name so PROPFIND output is deterministic
|
||
// (clients don't strictly require this, but reproducible
|
||
// listings make debugging from logs much easier).
|
||
chunks.sort_by(|a, b| a.name.cmp(&b.name));
|
||
|
||
Ok(Some(SessionListing {
|
||
session_mtime,
|
||
chunks,
|
||
}))
|
||
}
|
||
}
|
||
|
||
/// One chunk file inside an upload session.
|
||
#[derive(Debug, Clone)]
|
||
pub struct ChunkInfo {
|
||
pub name: String,
|
||
pub size: u64,
|
||
pub mtime: u64,
|
||
}
|
||
|
||
/// What `list_chunks` returns: the session's own mtime (for the
|
||
/// collection's `<d:getlastmodified>`) plus the list of stored
|
||
/// chunks.
|
||
#[derive(Debug, Clone)]
|
||
pub struct SessionListing {
|
||
pub session_mtime: u64,
|
||
pub chunks: Vec<ChunkInfo>,
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
|
||
fn test_service() -> (NextcloudChunkedUploadService, tempfile::TempDir) {
|
||
let dir = tempfile::tempdir().expect("create temp dir");
|
||
let svc = NextcloudChunkedUploadService::new(dir.path().to_path_buf());
|
||
(svc, dir)
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_create_session() {
|
||
let (svc, _dir) = test_service();
|
||
svc.create_session("alice", "upload-001").await.unwrap();
|
||
assert!(svc.session_exists("alice", "upload-001").await);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_session_not_exists_before_create() {
|
||
let (svc, _dir) = test_service();
|
||
assert!(!svc.session_exists("alice", "upload-999").await);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_store_and_assemble_chunks() {
|
||
let (svc, _dir) = test_service();
|
||
svc.create_session("alice", "upload-002").await.unwrap();
|
||
|
||
svc.store_chunk("alice", "upload-002", "00001", b"Hello, ")
|
||
.await
|
||
.unwrap();
|
||
svc.store_chunk("alice", "upload-002", "00002", b"World!")
|
||
.await
|
||
.unwrap();
|
||
|
||
let (temp_path, size, hash) = svc.assemble("alice", "upload-002").await.unwrap();
|
||
let assembled = fs::read(&temp_path).await.unwrap();
|
||
assert_eq!(assembled, b"Hello, World!");
|
||
assert_eq!(size, 13);
|
||
// BLAKE3("Hello, World!") — proves hash-on-write happens during
|
||
// the assemble pass, not via a re-read.
|
||
assert_eq!(
|
||
hash,
|
||
"288a86a79f20a3d6dccdca7713beaed178798296bdfa7913fa2a62d9727bf8f8"
|
||
);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_assemble_chunks_in_sorted_order() {
|
||
let (svc, _dir) = test_service();
|
||
svc.create_session("alice", "upload-003").await.unwrap();
|
||
|
||
// Store out of order.
|
||
svc.store_chunk("alice", "upload-003", "00003", b"C")
|
||
.await
|
||
.unwrap();
|
||
svc.store_chunk("alice", "upload-003", "00001", b"A")
|
||
.await
|
||
.unwrap();
|
||
svc.store_chunk("alice", "upload-003", "00002", b"B")
|
||
.await
|
||
.unwrap();
|
||
|
||
let (temp_path, size, hash) = svc.assemble("alice", "upload-003").await.unwrap();
|
||
let assembled = fs::read(&temp_path).await.unwrap();
|
||
assert_eq!(assembled, b"ABC");
|
||
assert_eq!(size, 3);
|
||
// BLAKE3("ABC") — confirms sort happened (chunks were stored in
|
||
// order 3,1,2 but the hash matches "ABC", not "CAB" or "BAC").
|
||
assert_eq!(
|
||
hash,
|
||
"d1717274597cf0289694f75d96d444b992a096f1afd8e7bbfa6ebb1d360fedfc"
|
||
);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_cleanup_removes_session() {
|
||
let (svc, _dir) = test_service();
|
||
svc.create_session("alice", "upload-004").await.unwrap();
|
||
assert!(svc.session_exists("alice", "upload-004").await);
|
||
|
||
svc.cleanup("alice", "upload-004").await.unwrap();
|
||
assert!(!svc.session_exists("alice", "upload-004").await);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_cleanup_nonexistent_session_is_ok() {
|
||
let (svc, _dir) = test_service();
|
||
// Should not error.
|
||
svc.cleanup("alice", "nonexistent").await.unwrap();
|
||
}
|
||
}
|