Merge pull request #479 from EdouardVanbelle/feat/drive-impl

feat/drive impl
This commit is contained in:
Dionisio Pozo
2026-06-19 18:48:32 +02:00
committed by GitHub
102 changed files with 6362 additions and 1078 deletions
+1 -1
View File
@@ -47,7 +47,7 @@ pub async fn create_auth_services(
);
// Wire the user-lifecycle dispatcher. Home-folder provisioning is
// now handled by HomeFolderLifecycleHook (registered on the
// now handled by PersonalDriveLifecycleHook (registered on the
// dispatcher in DI) — AuthApplicationService no longer needs a
// direct FolderService dependency for that path.
auth_app_service = auth_app_service.with_user_lifecycle(user_lifecycle);
+20 -1
View File
@@ -200,6 +200,25 @@ async fn run_migrations(pool: &PgPool) -> Result<()> {
match sqlx::migrate!().run(pool).await {
Ok(()) => Ok(()),
Err(e) => Err(DbError(format!("Migration error: {}", e))),
Err(e) => Err(DbError(format_error_chain("Migration error", &e))),
}
}
/// Format an error and every wrapped `source()` cause on a single line.
///
/// sqlx's `MigrateError::Execute` wraps the underlying `sqlx::Error::Database`
/// which in turn carries the PG `DETAIL` (e.g. `Key (version)=(20260803000000)`
/// for a duplicate-key on `_sqlx_migrations_pkey`). The default `Display`
/// only renders the outermost layer, so the operationally-critical hint
/// gets buried. Walking the chain surfaces it without needing to bump
/// `RUST_LOG` to debug.
fn format_error_chain(prefix: &str, e: &(dyn std::error::Error + 'static)) -> String {
let mut out = format!("{prefix}: {e}");
let mut cur = e.source();
while let Some(c) = cur {
out.push_str(" -> ");
out.push_str(&c.to_string());
cur = c.source();
}
out
}
@@ -0,0 +1,271 @@
//! PostgreSQL implementation of [`DriveRepository`].
//!
//! The repo deals only with the `storage.drives` table itself. Drive
//! membership lives in `storage.role_grants` (`resource_type='drive'`)
//! and is queried through the engine's existing grant paths;
//! `list_for_subjects` below resolves `role_grants` → `storage.drives`
//! via a single join.
//!
//! See `migrations/20260802000000_drives_schema_additive.sql` for the
//! schema and `docs/plan/drive.md` §3 / §15 for the locked design.
use std::sync::Arc;
use sqlx::{PgPool, Row, types::Uuid};
use crate::domain::entities::drive::{Drive, DriveKind};
use crate::domain::repositories::drive_repository::{
DriveRepository, DriveRepositoryError, DriveWithRootName,
};
pub struct DrivePgRepository {
pool: Arc<PgPool>,
}
impl DrivePgRepository {
pub fn new(pool: Arc<PgPool>) -> Self {
Self { pool }
}
fn map_sqlx_err(context: &'static str, e: sqlx::Error) -> DriveRepositoryError {
if let sqlx::Error::Database(ref dberr) = e
&& let Some(code) = dberr.code()
&& code.as_ref() == "23505"
{
// unique_violation. With drives, the only relevant unique is
// the partial index `idx_drives_default_for_user_unique` —
// surface the typed variant so the lifecycle hook can detect
// idempotent re-runs (D0-9 calls create_personal_drive_atomic
// during user provisioning).
return DriveRepositoryError::DefaultDriveAlreadyExists(dberr.to_string());
}
DriveRepositoryError::StorageError(format!("{context}: {e}"))
}
/// Map a row carrying both the drive's columns AND a `root_folder_name`
/// column (sourced via JOIN with `storage.folders`) into the view-model.
fn row_to_drive_with_name(
row: &sqlx::postgres::PgRow,
) -> Result<DriveWithRootName, DriveRepositoryError> {
let kind_str: String = row.get("kind");
let kind = DriveKind::from_sql(&kind_str)?;
let drive = Drive {
id: row.get("id"),
kind,
default_for_user: row.get("default_for_user"),
root_folder_id: row.get("root_folder_id"),
quota_bytes: row.get("quota_bytes"),
used_bytes: row.get("used_bytes"),
policies: row.get("policies"),
created_at: row.get("created_at"),
updated_at: row.get("updated_at"),
};
Ok(DriveWithRootName {
drive,
root_folder_name: row.get("root_folder_name"),
})
}
}
#[async_trait::async_trait]
impl DriveRepository for DrivePgRepository {
async fn create_personal_drive_atomic(
&self,
owner_id: Uuid,
quota_bytes: Option<i64>,
) -> Result<DriveWithRootName, DriveRepositoryError> {
// Four writes wrapped in a single transaction so either all
// commit or none does (docs/plan/drive.md §3). A single CTE
// statement would be cleaner on paper but doesn't work in
// PostgreSQL: CTE sub-statements share an MVCC snapshot, so
// `UPDATE storage.drives WHERE id = …` cannot match a row
// inserted by an earlier CTE branch. We use plain sequential
// statements inside `pool.begin()` instead — each statement
// sees the prior ones' writes (transaction-local visibility),
// and FK constraints are satisfied at insert time because the
// referenced rows already exist.
//
// Rollback semantics: any error before `tx.commit()` (FK
// violation, unique_violation on `default_for_user`, server
// crash) discards every partial write. No orphan drive, no
// folder without a drive, no drive without an owner.
let mut tx = self
.pool
.begin()
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.begin", e))?;
// 1. Drive row (root_folder_id NULL — populated in step 3).
let drive_id: Uuid = sqlx::query_scalar(
r#"
INSERT INTO storage.drives
(kind, default_for_user, quota_bytes, policies)
VALUES ('personal', $1, $2, '{}'::jsonb)
RETURNING id
"#,
)
.bind(owner_id)
.bind(quota_bytes)
.fetch_one(&mut *tx)
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.drive", e))?;
// 2. Root folder. `parent_id IS NULL` makes it a root in the
// drive; `drive_id` closes the FK in this direction.
let folder_id: Uuid = sqlx::query_scalar(
r#"
INSERT INTO storage.folders
(name, parent_id, user_id, drive_id, created_by, updated_by)
VALUES ('Personal', NULL, $1, $2, $1, $1)
RETURNING id
"#,
)
.bind(owner_id)
.bind(drive_id)
.fetch_one(&mut *tx)
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.folder", e))?;
// 3. Close the other side of the circular reference.
sqlx::query(r#"UPDATE storage.drives SET root_folder_id = $1 WHERE id = $2"#)
.bind(folder_id)
.bind(drive_id)
.execute(&mut *tx)
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.wire", e))?;
// 4. Owner role_grant — the caller becomes the drive's sole
// owner (single-user invariant on personal drives, §2).
sqlx::query(
r#"
INSERT INTO storage.role_grants
(subject_type, subject_id, resource_type, resource_id,
role, granted_by)
VALUES ('user', $1, 'drive', $2, 'owner', $1)
"#,
)
.bind(owner_id)
.bind(drive_id)
.execute(&mut *tx)
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.grant", e))?;
// Fetch the row in its final state so the caller gets a
// consistent view (including DB-computed defaults like
// `created_at`, `used_bytes`).
let row = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
WHERE d.id = $1
"#,
)
.bind(drive_id)
.fetch_one(&mut *tx)
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.read", e))?;
tx.commit()
.await
.map_err(|e| Self::map_sqlx_err("create_personal_drive_atomic.commit", e))?;
Self::row_to_drive_with_name(&row)
}
async fn get_by_id(&self, id: Uuid) -> Result<DriveWithRootName, DriveRepositoryError> {
let row = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
WHERE d.id = $1
"#,
)
.bind(id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("get_by_id", e))?
.ok_or_else(|| DriveRepositoryError::NotFound(id.to_string()))?;
Self::row_to_drive_with_name(&row)
}
async fn find_default_for_user(
&self,
user_id: Uuid,
) -> Result<DriveWithRootName, DriveRepositoryError> {
let row = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
WHERE d.default_for_user = $1
"#,
)
.bind(user_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("find_default_for_user", e))?
.ok_or_else(|| DriveRepositoryError::NotFound(user_id.to_string()))?;
Self::row_to_drive_with_name(&row)
}
async fn list_for_subjects(
&self,
subject_types: &[&str],
subject_ids: &[Uuid],
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
// Joining role_grants → drives → folders returns every drive the
// expanded subject set can read, paired with its display name.
// ORDER BY puts default drives first (so the picker UI doesn't
// need a follow-up sort), then alphabetical by name. GROUP BY
// collapses duplicate role_grants on the same drive (direct +
// group-mediated) and sidesteps PostgreSQL's "ORDER BY
// expression must appear in select list" rule that SELECT
// DISTINCT imposes.
let rows = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
JOIN storage.role_grants g
ON g.resource_type = 'drive'
AND g.resource_id = d.id
WHERE g.subject_type = ANY($1)
AND g.subject_id = ANY($2)
AND (g.expires_at IS NULL OR g.expires_at > NOW())
GROUP BY d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at, f.name
ORDER BY (d.default_for_user IS NULL) ASC,
LOWER(f.name) ASC
"#,
)
.bind(
subject_types
.iter()
.map(|s| s.to_string())
.collect::<Vec<_>>(),
)
.bind(subject_ids)
.fetch_all(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("list_for_subjects", e))?;
rows.iter().map(Self::row_to_drive_with_name).collect()
}
}
@@ -19,6 +19,8 @@ type MediaFileRow = (
i64, // updated_at
String, // blob_hash
Option<Uuid>, // user_id
Option<Uuid>, // created_by (§14 provenance)
Option<Uuid>, // updated_by (§14 provenance)
i64, // sort_date
Option<i32>, // width
Option<i32>, // height
@@ -42,7 +44,9 @@ use crate::infrastructure::services::dedup_service::DedupService;
use uuid::Uuid;
/// Type alias for file metadata rows from SQL queries.
/// Fields: id, name, folder_id, folder_path, size, mime_type, created_at, updated_at, blob_hash, user_id
/// Fields: id, name, folder_id, folder_path, size, mime_type,
/// created_at, updated_at, blob_hash, user_id, created_by, updated_by.
/// `created_by` / `updated_by` are the §14 provenance columns.
type FileRow = (
String,
String,
@@ -54,6 +58,8 @@ type FileRow = (
i64,
String,
Option<Uuid>,
Option<Uuid>,
Option<Uuid>,
);
/// Append the optional type/date/size filters from `criteria` to
@@ -228,7 +234,8 @@ impl FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
fi.blob_hash, \
fi.user_id \
fi.user_id, \
fi.created_by, fi.updated_by \
FROM storage.files fi \
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id \
WHERE {where_clause}"
@@ -248,8 +255,10 @@ impl FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect::<Result<Vec<_>, _>>()
@@ -277,7 +286,8 @@ impl FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
fi.blob_hash, \
fi.user_id \
fi.user_id, \
fi.created_by, fi.updated_by \
FROM storage.files fi \
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id \
WHERE fi.id = ANY($1) AND NOT fi.is_trashed",
@@ -291,8 +301,10 @@ impl FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect::<Result<Vec<_>, _>>()
@@ -357,9 +369,11 @@ impl FileBlobReadRepository {
modified_at: i64,
blob_hash: String,
owner_id: Option<Uuid>,
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<File, DomainError> {
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
File::with_timestamps_and_blob_hash(
File::with_timestamps_blob_hash_and_provenance(
id,
name,
storage_path,
@@ -370,6 +384,8 @@ impl FileBlobReadRepository {
modified_at as u64,
owner_id,
blob_hash,
created_by,
updated_by,
)
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("entity: {e}")))
}
@@ -428,6 +444,7 @@ impl FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id,
fi.created_by, fi.updated_by,
EXTRACT(EPOCH FROM fi.media_sort_date)::bigint AS sort_date,
fm.width, fm.height
FROM storage.files fi
@@ -453,9 +470,9 @@ impl FileBlobReadRepository {
let mut sort_dates = Vec::with_capacity(rows.len());
let mut dims = Vec::with_capacity(rows.len());
for (id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, sd, w, h) in rows {
for (id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub, sd, w, h) in rows {
files.push(Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid,
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)?);
sort_dates.push(sd);
dims.push((w, h));
@@ -530,6 +547,8 @@ impl FileReadPort for FileBlobReadRepository {
i64, // updated_at
String, // blob_hash
Option<Uuid>, // user_id (owner)
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
),
>(
r#"
@@ -538,7 +557,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.id = $1::uuid AND NOT fi.is_trashed
@@ -555,7 +575,7 @@ impl FileReadPort for FileBlobReadRepository {
self.hash_cache.insert(id.to_string(), row.8.clone());
Self::row_to_file(
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9,
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9, row.10, row.11,
)
}
@@ -576,6 +596,8 @@ impl FileReadPort for FileBlobReadRepository {
i64,
String,
Option<Uuid>,
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
),
>(
r#"
@@ -584,7 +606,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.id = $1::uuid
@@ -598,7 +621,7 @@ impl FileReadPort for FileBlobReadRepository {
self.hash_cache.insert(id.to_string(), row.8.clone());
Self::row_to_file(
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9,
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9, row.10, row.11,
)
}
@@ -616,6 +639,8 @@ impl FileReadPort for FileBlobReadRepository {
i64, // updated_at
String, // blob_hash
Option<Uuid>, // user_id (owner)
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
),
>(
r#"
@@ -624,7 +649,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.id = $1::uuid
@@ -643,7 +669,7 @@ impl FileReadPort for FileBlobReadRepository {
self.hash_cache.insert(id.to_string(), row.8.clone());
Self::row_to_file(
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9,
row.0, row.1, row.2, row.3, row.4, row.5, row.6, row.7, row.8, row.9, row.10, row.11,
)
}
@@ -657,7 +683,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid AND NOT fi.is_trashed
@@ -675,7 +702,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL AND NOT fi.is_trashed
@@ -689,8 +717,10 @@ impl FileReadPort for FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect()
@@ -711,7 +741,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid AND NOT fi.is_trashed
@@ -731,7 +762,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL AND NOT fi.is_trashed
@@ -747,8 +779,10 @@ impl FileReadPort for FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect()
@@ -777,7 +811,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid AND NOT fi.is_trashed
@@ -798,7 +833,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL AND NOT fi.is_trashed
@@ -815,8 +851,10 @@ impl FileReadPort for FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect()
@@ -839,7 +877,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid AND NOT fi.is_trashed
@@ -862,7 +901,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL AND NOT fi.is_trashed
@@ -883,8 +923,10 @@ impl FileReadPort for FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect()
@@ -935,7 +977,11 @@ impl FileReadPort for FileBlobReadRepository {
Ok(Self::make_file_path(row.1.as_deref(), &row.0))
}
async fn get_parent_folder_id(&self, path: &str) -> Result<String, DomainError> {
async fn get_parent_folder_id(
&self,
path: &str,
drive_id: Uuid,
) -> Result<String, DomainError> {
let path = path.trim_start_matches('/').trim_end_matches('/');
let segments: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
@@ -955,29 +1001,49 @@ impl FileReadPort for FileBlobReadRepository {
));
}
self.get_folder_id_by_path(&folder_path).await
self.get_folder_id_by_path(&folder_path, drive_id).await
}
async fn get_folder_id_by_path(&self, folder_path: &str) -> Result<String, DomainError> {
async fn get_folder_id_by_path(
&self,
folder_path: &str,
drive_id: Uuid,
) -> Result<String, DomainError> {
let folder_path = folder_path.trim_start_matches('/').trim_end_matches('/');
if folder_path.is_empty() {
return Err(DomainError::not_found("Folder", "empty path"));
}
// Post-D0 `storage.folders.path` repeats across drives —
// filter by `drive_id` to scope the lookup.
sqlx::query_scalar::<_, String>(
"SELECT id::text FROM storage.folders WHERE path = $1 AND NOT is_trashed",
"SELECT id::text FROM storage.folders \
WHERE path = $1 AND drive_id = $2 AND NOT is_trashed",
)
.bind(folder_path)
.bind(drive_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("folder lookup: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", format!("path: {folder_path}")))
}
/// Direct SQL lookup using materialized folder paths.
/// Direct SQL lookup using materialized folder paths, scoped to a drive.
/// O(1) query instead of O(depth) folder walk.
async fn find_file_by_path(&self, path: &str) -> Result<Option<File>, DomainError> {
///
/// Post-D0 `storage.folders.path` repeats across drives (each drive
/// has its own root with a name like `"Personal"`). Without the
/// `drive_id` filter the lookup would be non-deterministic. The
/// root-level branch filters on `fi.drive_id`; the nested branch
/// filters on the parent folder's `fo.drive_id` (which closes the
/// leak cleanly and matches the path semantics — see Step 2 of
/// the path-lookup refactor).
async fn find_file_by_path(
&self,
path: &str,
drive_id: Uuid,
) -> Result<Option<File>, DomainError> {
let path = path.trim_start_matches('/').trim_end_matches('/');
let segments: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
@@ -996,7 +1062,8 @@ impl FileReadPort for FileBlobReadRepository {
let folder_path = segments[..segments.len() - 1].join("/");
let row = if folder_path.is_empty() {
// File at root level (no parent folder)
// File at root level (no parent folder) — filter on
// `fi.drive_id` because there's no folder row to join through.
sqlx::query_as::<
_,
(
@@ -1010,6 +1077,8 @@ impl FileReadPort for FileBlobReadRepository {
i64,
String,
Option<Uuid>,
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
),
>(
r#"
@@ -1018,17 +1087,23 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.name = $1 AND fi.folder_id IS NULL AND NOT fi.is_trashed
WHERE fi.name = $1 AND fi.folder_id IS NULL
AND fi.drive_id = $2 AND NOT fi.is_trashed
"#,
)
.bind(filename)
.bind(drive_id)
.fetch_optional(self.pool.as_ref())
.await
} else {
// File inside a folder — look up by folder path + filename
// File inside a folder — look up by folder path + filename,
// filtered by the parent folder's drive_id (path semantics
// are folder-scoped, so this also catches mis-pointed file
// rows during D0/D7's dual-write window).
sqlx::query_as::<
_,
(
@@ -1042,6 +1117,8 @@ impl FileReadPort for FileBlobReadRepository {
i64,
String,
Option<Uuid>,
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
),
>(
r#"
@@ -1050,14 +1127,17 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fo.path = $1 AND fi.name = $2 AND NOT fi.is_trashed
WHERE fo.path = $1 AND fi.name = $2
AND fo.drive_id = $3 AND NOT fi.is_trashed
"#,
)
.bind(&folder_path)
.bind(filename)
.bind(drive_id)
.fetch_optional(self.pool.as_ref())
.await
}
@@ -1065,7 +1145,7 @@ impl FileReadPort for FileBlobReadRepository {
match row {
Some(r) => Ok(Some(Self::row_to_file(
r.0, r.1, r.2, r.3, r.4, r.5, r.6, r.7, r.8, r.9,
r.0, r.1, r.2, r.3, r.4, r.5, r.6, r.7, r.8, r.9, r.10, r.11,
)?)),
None => Ok(None),
}
@@ -1086,6 +1166,7 @@ impl FileReadPort for FileBlobReadRepository {
let mut row_stream = sqlx::query_as::<_, (
String, String, Option<String>, Option<String>,
i64, String, i64, i64, String, Option<Uuid>,
Option<Uuid>, Option<Uuid>,
)>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
@@ -1093,7 +1174,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fo.lpath <@ (SELECT lpath FROM storage.folders WHERE id = $1::uuid)
@@ -1107,9 +1189,9 @@ impl FileReadPort for FileBlobReadRepository {
while let Some(row) = row_stream.try_next().await.map_err(|e| {
DomainError::internal_error("FileBlobRead", format!("subtree stream: {e}"))
})? {
let (id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid) = row;
let (id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub) = row;
let file = FileBlobReadRepository::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid,
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)?;
yield file;
}
@@ -1173,6 +1255,7 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
fi.blob_hash, \
fi.user_id, \
fi.created_by, fi.updated_by, \
COUNT(*) OVER() AS total_count \
FROM storage.files fi \
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id \
@@ -1195,6 +1278,8 @@ impl FileReadPort for FileBlobReadRepository {
i64,
String,
Option<Uuid>,
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
i64,
),
>(&sql)
@@ -1217,13 +1302,15 @@ impl FileReadPort for FileBlobReadRepository {
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("search: {e}")))?;
// total_count is the same in every row; 0 when result set is empty.
let total_count = rows.first().map_or(0, |r| r.10) as usize;
let total_count = rows.first().map_or(0, |r| r.12) as usize;
let files = rows
.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, _total)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub, _total)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect::<Result<Vec<_>, _>>()
@@ -1299,6 +1386,7 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
fi.blob_hash, \
fi.user_id, \
fi.created_by, fi.updated_by, \
COUNT(*) OVER() AS total_count \
FROM storage.files fi \
JOIN storage.folders fo ON fo.id = fi.folder_id \
@@ -1321,6 +1409,8 @@ impl FileReadPort for FileBlobReadRepository {
i64,
String,
Option<Uuid>,
Option<Uuid>, // created_by (§14)
Option<Uuid>, // updated_by (§14)
i64,
),
>(&sql)
@@ -1341,13 +1431,15 @@ impl FileReadPort for FileBlobReadRepository {
DomainError::internal_error("FileBlobRead", format!("subtree search: {e}"))
})?;
let total_count = rows.first().map_or(0, |r| r.10) as usize;
let total_count = rows.first().map_or(0, |r| r.12) as usize;
let files = rows
.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, _total)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub, _total)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect::<Result<Vec<_>, _>>()
@@ -1389,7 +1481,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid
@@ -1418,7 +1511,8 @@ impl FileReadPort for FileBlobReadRepository {
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.user_id
fi.user_id,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL
@@ -1443,8 +1537,10 @@ impl FileReadPort for FileBlobReadRepository {
rows.into_iter()
.map(
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)| {
Self::row_to_file(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid)
|(id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub)| {
Self::row_to_file(
id, name, fid, fpath, size, mime, ca, ma, blob_hash, uid, cb, ub,
)
},
)
.collect()
@@ -27,6 +27,10 @@ use crate::infrastructure::services::dedup_service::DedupService;
pub struct FileBlobWriteRepository {
pool: Arc<PgPool>,
dedup: Arc<DedupService>,
/// Retained on the struct after D0-8 inlined parent-folder lookups
/// directly via SQL; kept for now so D0's diff stays scoped to drive_id
/// + provenance plumbing. Slated for removal in a follow-up cleanup.
#[allow(dead_code)]
folder_repo: Arc<FolderDbRepository>,
/// Shared handle to `FileBlobReadRepository`'s file_id → blob_hash
/// cache. Content swaps and hard deletes invalidate the mapping here
@@ -111,9 +115,11 @@ impl FileBlobWriteRepository {
modified_at: i64,
owner_id: Option<Uuid>,
blob_hash: String,
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<File, DomainError> {
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
File::with_timestamps_and_blob_hash(
File::with_timestamps_blob_hash_and_provenance(
id,
name,
storage_path,
@@ -124,14 +130,33 @@ impl FileBlobWriteRepository {
modified_at as u64,
owner_id,
blob_hash,
created_by,
updated_by,
)
.map_err(|e| DomainError::internal_error("FileBlobWrite", format!("entity: {e}")))
}
/// Derive user_id from the parent folder, or error if folder_id is None.
async fn resolve_user_id(&self, folder_id: Option<&str>) -> Result<Uuid, DomainError> {
/// Derive `(user_id, drive_id)` from the parent folder. Both are
/// needed during the D0 dual-write window: `user_id` for the legacy
/// column (dropped in D7) and `drive_id` for the new owning-drive
/// reference.
async fn resolve_owner_and_drive(
&self,
folder_id: Option<&str>,
) -> Result<(Uuid, Uuid), DomainError> {
match folder_id {
Some(fid) => self.folder_repo.get_folder_user_id(fid).await,
Some(fid) => {
let row: Option<(Uuid, Uuid)> = sqlx::query_as::<_, (Uuid, Uuid)>(
"SELECT user_id, drive_id FROM storage.folders WHERE id = $1::uuid",
)
.bind(fid)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| {
DomainError::internal_error("FileBlobWrite", format!("parent lookup: {e}"))
})?;
row.ok_or_else(|| DomainError::not_found("Folder", fid))
}
None => Err(DomainError::internal_error(
"FileBlobWrite",
"folder_id is required to determine file owner",
@@ -150,12 +175,18 @@ impl FileBlobWriteRepository {
/// `(new_hash, updated_at_epoch)` on success — the effective timestamp
/// is returned so callers can rebuild the fresh entity without
/// re-reading the row.
///
/// §14: `updated_by = $5` (caller_id). The caller mutated this
/// row — not the row's owner. D2 shared drives let non-owners
/// overwrite content; the previous `updated_by = f.user_id` would
/// have silently recorded the wrong principal.
async fn swap_blob_hash(
&self,
file_id: &str,
new_hash: &str,
new_size: i64,
modified_at: Option<i64>,
caller_id: Uuid,
) -> Result<(String, i64), DomainError> {
// Atomic CTE: capture old hash then update in one round-trip, no TOCTOU.
// Deadlock victims (40P01) retry before the compensation below runs —
@@ -168,7 +199,8 @@ impl FileBlobWriteRepository {
)
UPDATE storage.files f
SET blob_hash = $1, size = $2,
updated_at = COALESCE(to_timestamp($4), NOW())
updated_at = COALESCE(to_timestamp($4), NOW()),
updated_by = $5
FROM old
WHERE f.id = old.id
RETURNING old.blob_hash, EXTRACT(EPOCH FROM f.updated_at)::bigint
@@ -178,6 +210,7 @@ impl FileBlobWriteRepository {
.bind(new_size)
.bind(file_id)
.bind(modified_at.map(|t| t as f64))
.bind(caller_id)
.fetch_optional(self.pool.as_ref())
})
.await
@@ -223,6 +256,12 @@ impl FileBlobWriteRepository {
/// Register a file row pointing at a blob already stored in the chunk
/// store (the upload-ingest layer streamed the content in). Consumes the
/// caller's blob reference: any failure releases it before returning.
///
/// §14: `created_by = $7 = updated_by = caller_id` — authorship
/// belongs to the principal performing the upload, not to the parent
/// folder's owner. In D2 shared drives a non-owner member can upload
/// into a folder Alice owns; binding `parent.user_id` would have
/// silently recorded Alice as the author.
async fn save_file_with_blob_impl(
&self,
name: String,
@@ -230,6 +269,7 @@ impl FileBlobWriteRepository {
content_type: String,
blob_hash: &str,
size: u64,
caller_id: Uuid,
) -> Result<File, DomainError> {
// Root files have no parent folder to derive an owner from — keep the
// previous resolve_user_id(None) contract (release the ref, error out).
@@ -260,19 +300,24 @@ impl FileBlobWriteRepository {
// (a retried INSERT can legitimately lose to a concurrent identical
// upload).
let result = retry_on_deadlock("files.insert", || {
sqlx::query_as::<_, (String, Uuid, String, i64, i64)>(
sqlx::query_as::<_, (String, Uuid, String, i64, i64, Option<Uuid>, Option<Uuid>)>(
r#"
WITH parent AS (
SELECT id, user_id, path FROM storage.folders WHERE id = $2::uuid
SELECT id, user_id, drive_id, path FROM storage.folders WHERE id = $2::uuid
)
INSERT INTO storage.files
(name, folder_id, user_id, blob_hash, size, mime_type, category_order)
SELECT $1, parent.id, parent.user_id, $3, $4, $5, $6 FROM parent
(name, folder_id, user_id, drive_id, blob_hash, size,
mime_type, category_order, created_by, updated_by)
SELECT $1, parent.id, parent.user_id, parent.drive_id, $3, $4,
$5, $6, $7, $7
FROM parent
RETURNING id::text,
user_id,
(SELECT path FROM parent),
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
created_by,
updated_by
"#,
)
.bind(&name)
@@ -281,44 +326,46 @@ impl FileBlobWriteRepository {
.bind(size as i64)
.bind(&content_type)
.bind(category_order_for(&name, &content_type))
.bind(caller_id)
.fetch_optional(self.pool.as_ref())
})
.await;
let (id, user_id, folder_path, created_at, updated_at) = match result {
Ok(Some(row)) => row,
Ok(None) => {
if let Err(rollback_err) = self.dedup.remove_reference(blob_hash).await {
tracing::error!(
"Blob orphaned after missing parent folder — hash: {}, err: {}",
&blob_hash[..12],
rollback_err
);
let (id, user_id, folder_path, created_at, updated_at, created_by, updated_by) =
match result {
Ok(Some(row)) => row,
Ok(None) => {
if let Err(rollback_err) = self.dedup.remove_reference(blob_hash).await {
tracing::error!(
"Blob orphaned after missing parent folder — hash: {}, err: {}",
&blob_hash[..12],
rollback_err
);
}
return Err(DomainError::not_found("Folder", fid));
}
return Err(DomainError::not_found("Folder", fid));
}
Err(e) => {
if let Err(rollback_err) = self.dedup.remove_reference(blob_hash).await {
tracing::error!(
"Blob orphaned after failed INSERT — hash: {}, err: {}",
&blob_hash[..12],
rollback_err
);
}
if let sqlx::Error::Database(ref db_err) = e
&& db_err.code().as_deref() == Some("23505")
{
return Err(DomainError::already_exists(
"File",
format!("'{name}' already exists in this folder"),
Err(e) => {
if let Err(rollback_err) = self.dedup.remove_reference(blob_hash).await {
tracing::error!(
"Blob orphaned after failed INSERT — hash: {}, err: {}",
&blob_hash[..12],
rollback_err
);
}
if let sqlx::Error::Database(ref db_err) = e
&& db_err.code().as_deref() == Some("23505")
{
return Err(DomainError::already_exists(
"File",
format!("'{name}' already exists in this folder"),
));
}
return Err(DomainError::internal_error(
"FileBlobWrite",
format!("insert: {e}"),
));
}
return Err(DomainError::internal_error(
"FileBlobWrite",
format!("insert: {e}"),
));
}
};
};
tracing::info!(
"📡 STREAMING WRITE: {} ({} bytes, hash: {})",
@@ -338,6 +385,8 @@ impl FileBlobWriteRepository {
updated_at,
Some(user_id),
blob_hash.to_string(),
created_by,
updated_by,
)
}
}
@@ -350,8 +399,9 @@ impl FileWritePort for FileBlobWriteRepository {
content_type: String,
blob_hash: &str,
size: u64,
caller_id: Uuid,
) -> Result<File, DomainError> {
self.save_file_with_blob_impl(name, folder_id, content_type, blob_hash, size)
self.save_file_with_blob_impl(name, folder_id, content_type, blob_hash, size, caller_id)
.await
}
@@ -359,20 +409,50 @@ impl FileWritePort for FileBlobWriteRepository {
&self,
file_id: &str,
target_folder_id: Option<String>,
caller_id: Uuid,
) -> Result<File, DomainError> {
// If moving to a different folder, get the new user_id (must be same user)
let row = sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64)>(
// If moving to a different folder, get the new user_id (must be same user).
//
// §14: `updated_by = $3` (caller_id) — the caller mutated this
// row. The previous COALESCE derived authorship from the
// destination folder's owner, which is wrong: dest's user_id
// has no claim to authorship of the file's content. D2 shared
// drives surface this most starkly (Alice moves Bob's file
// into Charlie's drive — `updated_by` must be Alice).
let row = sqlx::query_as::<
_,
(
String,
String,
Option<String>,
i64,
String,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
),
>(
r#"
UPDATE storage.files
SET folder_id = $1::uuid, updated_at = NOW()
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, folder_id::text, size, mime_type,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
WITH dest AS (
SELECT user_id, drive_id FROM storage.folders WHERE id = $1::uuid
)
UPDATE storage.files f
SET folder_id = $1::uuid,
user_id = COALESCE((SELECT user_id FROM dest), f.user_id),
drive_id = COALESCE((SELECT drive_id FROM dest), f.drive_id),
updated_at = NOW(),
updated_by = $3
WHERE f.id = $2::uuid AND NOT f.is_trashed
RETURNING f.id::text, f.name, f.folder_id::text, f.size, f.mime_type,
EXTRACT(EPOCH FROM f.created_at)::bigint,
EXTRACT(EPOCH FROM f.updated_at)::bigint,
f.created_by, f.updated_by
"#,
)
.bind(&target_folder_id)
.bind(file_id)
.bind(caller_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("FileBlobWrite", format!("move: {e}")))?
@@ -390,6 +470,8 @@ impl FileWritePort for FileBlobWriteRepository {
row.6,
None,
String::new(),
row.7,
row.8,
)
}
@@ -398,9 +480,16 @@ impl FileWritePort for FileBlobWriteRepository {
file_id: &str,
target_folder_id: Option<String>,
new_name: Option<&str>,
caller_id: Uuid,
) -> Result<File, DomainError> {
// Atomic CTE: read source file → insert new row with same blob_hash → increment ref_count.
// Single round-trip; blob content is NOT copied (dedup makes this zero-copy).
//
// §14: `created_by = $4 = updated_by = caller_id` — the caller
// authored this copy. The previous binding used
// `dest_folder.user_id` which silently recorded the destination
// folder's owner as the author when Adam copied a file into
// Alice's folder.
let target_fid = target_folder_id.clone();
let rename_to = new_name.map(|s| s.to_string());
@@ -416,6 +505,8 @@ impl FileWritePort for FileBlobWriteRepository {
i64,
i64,
String,
Option<Uuid>,
Option<Uuid>,
),
>(
r#"
@@ -424,20 +515,39 @@ impl FileWritePort for FileBlobWriteRepository {
FROM storage.files
WHERE id = $1::uuid AND NOT is_trashed
),
-- The destination folder may differ from the source's
-- folder (when $2 is set); derive drive_id from the
-- DESTINATION so cross-drive copies land in the right
-- drive. Files in personal drives only copy within the
-- same drive today, but the join makes the migration
-- future-proof for D2's cross-drive copy story.
dest_folder AS (
SELECT id, user_id, drive_id
FROM storage.folders
WHERE id = COALESCE($2::uuid,
(SELECT folder_id FROM src))
),
new_file AS (
INSERT INTO storage.files (name, folder_id, user_id, blob_hash, size, mime_type, category_order)
SELECT COALESCE($3::text, name),
COALESCE($2::uuid, folder_id),
user_id,
blob_hash,
size,
mime_type,
category_order
FROM src
INSERT INTO storage.files
(name, folder_id, user_id, drive_id, blob_hash, size,
mime_type, category_order, created_by, updated_by)
SELECT COALESCE($3::text, src.name),
dest_folder.id,
dest_folder.user_id,
dest_folder.drive_id,
src.blob_hash,
src.size,
src.mime_type,
src.category_order,
$4,
$4
FROM src, dest_folder
RETURNING id::text, name, folder_id::text, size, mime_type,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
blob_hash
blob_hash,
created_by,
updated_by
)
SELECT * FROM new_file
"#,
@@ -445,6 +555,7 @@ impl FileWritePort for FileBlobWriteRepository {
.bind(file_id)
.bind(&target_fid)
.bind(&rename_to)
.bind(caller_id)
.fetch_optional(self.pool.as_ref())
})
.await
@@ -490,22 +601,45 @@ impl FileWritePort for FileBlobWriteRepository {
row.6,
None,
row.7,
row.8,
row.9,
)
}
async fn rename_file(&self, file_id: &str, new_name: &str) -> Result<File, DomainError> {
let row = sqlx::query_as::<_, (String, String, Option<String>, i64, String, i64, i64)>(
async fn rename_file(
&self,
file_id: &str,
new_name: &str,
caller_id: Uuid,
) -> Result<File, DomainError> {
// §14: `updated_by = $3` (caller_id), see move_file.
let row = sqlx::query_as::<
_,
(
String,
String,
Option<String>,
i64,
String,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
),
>(
r#"
UPDATE storage.files
SET name = $1, updated_at = NOW()
SET name = $1, updated_at = NOW(), updated_by = $3
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, folder_id::text, size, mime_type,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
created_by, updated_by
"#,
)
.bind(new_name)
.bind(file_id)
.bind(caller_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| {
@@ -530,6 +664,8 @@ impl FileWritePort for FileBlobWriteRepository {
row.6,
None,
String::new(),
row.7,
row.8,
)
}
@@ -559,12 +695,13 @@ impl FileWritePort for FileBlobWriteRepository {
blob_hash: &str,
size: u64,
modified_at: Option<i64>,
caller_id: Uuid,
) -> Result<(String, i64), DomainError> {
// The content was already ingested into the chunk store by the
// upload-ingest layer; swap_blob_hash consumes its reference and
// releases it on failure.
let swapped = self
.swap_blob_hash(file_id, blob_hash, size as i64, modified_at)
.swap_blob_hash(file_id, blob_hash, size as i64, modified_at, caller_id)
.await?;
// The file now maps to a different blob — drop the read-side cache
// entry so streaming downloads cannot serve the previous content
@@ -579,30 +716,41 @@ impl FileWritePort for FileBlobWriteRepository {
folder_id: Option<String>,
content_type: String,
size: u64,
caller_id: Uuid,
) -> Result<(File, PathBuf), DomainError> {
let user_id = self.resolve_user_id(folder_id.as_deref()).await?;
let (user_id, drive_id) = self.resolve_owner_and_drive(folder_id.as_deref()).await?;
// For deferred registration we use a placeholder hash.
// The write-behind cache will call update_file_content later.
let placeholder_hash = "0000000000000000000000000000000000000000000000000000000000000000";
// §14: `created_by = $9 = updated_by = caller_id`. The legacy
// `user_id` column (dropped in D7) stays bound to the parent
// folder's owner; only the two provenance columns flip to the
// caller — see save_file_with_blob_impl.
let row = retry_on_deadlock("files.insert_deferred", || {
sqlx::query_as::<_, (String, i64, i64)>(
sqlx::query_as::<_, (String, i64, i64, Option<Uuid>, Option<Uuid>)>(
r#"
INSERT INTO storage.files (name, folder_id, user_id, blob_hash, size, mime_type, category_order)
VALUES ($1, $2::uuid, $3, $4, $5, $6, $7)
INSERT INTO storage.files
(name, folder_id, user_id, drive_id, blob_hash, size,
mime_type, category_order, created_by, updated_by)
VALUES ($1, $2::uuid, $3, $4, $5, $6, $7, $8, $9, $9)
RETURNING id::text,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint
EXTRACT(EPOCH FROM updated_at)::bigint,
created_by,
updated_by
"#,
)
.bind(&name)
.bind(&folder_id)
.bind(user_id)
.bind(drive_id)
.bind(placeholder_hash)
.bind(size as i64)
.bind(&content_type)
.bind(category_order_for(&name, &content_type))
.bind(caller_id)
.fetch_one(self.pool.as_ref())
})
.await
@@ -620,6 +768,8 @@ impl FileWritePort for FileBlobWriteRepository {
row.2,
Some(user_id),
String::new(),
row.3,
row.4,
)?;
// The target_path is not meaningful for blob storage (content goes to .blobs/)
@@ -631,18 +781,21 @@ impl FileWritePort for FileBlobWriteRepository {
// ── Trash operations ──
async fn move_to_trash(&self, file_id: &str) -> Result<(), DomainError> {
async fn move_to_trash(&self, file_id: &str, caller_id: Uuid) -> Result<(), DomainError> {
// §14: `updated_by = $2` (caller_id), see move_file.
let result = sqlx::query(
r#"
UPDATE storage.files
SET is_trashed = TRUE,
trashed_at = NOW(),
original_folder_id = folder_id,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
WHERE id = $1::uuid AND NOT is_trashed
"#,
)
.bind(file_id)
.bind(caller_id)
.execute(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("FileBlobWrite", format!("trash: {e}")))?;
@@ -657,7 +810,9 @@ impl FileWritePort for FileBlobWriteRepository {
&self,
file_id: &str,
_original_path: &str,
caller_id: Uuid,
) -> Result<(), DomainError> {
// §14: `updated_by = $2` (caller_id), see move_file.
let result = sqlx::query(
r#"
UPDATE storage.files
@@ -665,11 +820,13 @@ impl FileWritePort for FileBlobWriteRepository {
trashed_at = NULL,
folder_id = COALESCE(original_folder_id, folder_id),
original_folder_id = NULL,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
WHERE id = $1::uuid AND is_trashed
"#,
)
.bind(file_id)
.bind(caller_id)
.execute(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("FileBlobWrite", format!("restore: {e}")))?;
@@ -21,36 +21,58 @@ use crate::domain::services::authorization::ResourceKind;
use crate::domain::services::path_service::StoragePath;
/// Type alias for folder metadata rows from SQL queries.
/// Tuple order: id, name, path, parent_id, user_id, created_at,
/// modified_at, tree_modified_at. The trailing `tree_modified_at`
/// feeds [`Folder::etag`] — every SELECT here must include
/// `EXTRACT(EPOCH FROM tree_modified_at)::bigint`.
type FolderRow = (String, String, String, Option<String>, Uuid, i64, i64, i64);
/// Tuple order: id, name, path, parent_id, user_id, drive_id,
/// created_at, modified_at, tree_modified_at, created_by, updated_by.
/// The trailing `tree_modified_at` feeds [`Folder::etag`] — every
/// SELECT here must include `EXTRACT(EPOCH FROM tree_modified_at)::bigint`.
/// `drive_id` is the post-D0 `NOT NULL` scope axis for path-based
/// lookups. `created_by` / `updated_by` are the §14 provenance
/// columns, nullable because the FK is `ON DELETE SET NULL`.
type FolderRow = (
String,
String,
String,
Option<String>,
Uuid,
Uuid,
i64,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
);
/// Type alias for paginated folder rows (includes total_count as
/// the last element after `tree_modified_at`).
/// the last element after the §14 provenance columns).
type FolderRowPaginated = (
String,
String,
String,
Option<String>,
Uuid,
Uuid,
i64,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
i64,
);
/// Type alias for folder rows with optional user_id.
/// Includes the §14 provenance columns `created_by` / `updated_by`.
type FolderRowOptUser = (
String,
String,
String,
Option<String>,
Option<Uuid>,
Uuid,
i64,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
);
/// PostgreSQL-backed folder repository.
@@ -84,7 +106,9 @@ impl FolderDbRepository {
/// Convert a database row into a `Folder` domain entity.
///
/// The `path` comes directly from the materialized `path` column — no
/// extra queries needed.
/// extra queries needed. `created_by` / `updated_by` carry the
/// §14 provenance signal through the entity layer; both are
/// `Option<Uuid>` because the FK is `ON DELETE SET NULL`.
#[allow(clippy::too_many_arguments)]
fn row_to_folder(
id: String,
@@ -92,20 +116,26 @@ impl FolderDbRepository {
path: String,
parent_id: Option<String>,
user_id: Option<Uuid>,
drive_id: Uuid,
created_at: i64,
modified_at: i64,
tree_modified_at: i64,
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<Folder, DomainError> {
let storage_path = StoragePath::from_string(&path);
Folder::with_timestamps_and_tree(
Folder::with_timestamps_tree_and_provenance(
id,
name,
storage_path,
parent_id,
user_id,
drive_id,
created_at as u64,
modified_at as u64,
tree_modified_at as u64,
created_by,
updated_by,
)
.map_err(|e| DomainError::internal_error("FolderDb", format!("entity: {e}")))
}
@@ -123,10 +153,11 @@ impl FolderDbRepository {
let rows = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE id = ANY($1) AND NOT is_trashed
"#,
@@ -137,7 +168,9 @@ impl FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("get_folders_by_ids: {e}")))?;
rows.into_iter()
.map(|r| Self::row_to_folder(r.0, r.1, r.2, r.3, Some(r.4), r.5, r.6, r.7))
.map(|r| {
Self::row_to_folder(r.0, r.1, r.2, r.3, Some(r.4), r.5, r.6, r.7, r.8, r.9, r.10)
})
.collect()
}
}
@@ -147,40 +180,59 @@ impl FolderRepository for FolderDbRepository {
&self,
name: String,
parent_id: Option<String>,
caller_id: Uuid,
) -> Result<Folder, DomainError> {
// Derive user_id from parent folder. Root-level folders require the
// caller to have set up the home folder beforehand (done during user
// registration).
let user_id: Uuid = if let Some(ref pid) = parent_id {
sqlx::query_scalar::<_, Uuid>("SELECT user_id FROM storage.folders WHERE id = $1::uuid")
.bind(pid)
.fetch_optional(self.pool())
.await
.map_err(|e| {
DomainError::internal_error("FolderDb", format!("parent lookup: {e}"))
})?
.ok_or_else(|| DomainError::not_found("Folder", pid))?
// Derive (user_id, drive_id) from parent folder in one round-trip.
// Root-level folders require the caller to have set up the home
// drive beforehand (done during user registration via the
// lifecycle hook).
let (user_id, drive_id): (Uuid, Uuid) = if let Some(ref pid) = parent_id {
sqlx::query_as::<_, (Uuid, Uuid)>(
"SELECT user_id, drive_id FROM storage.folders WHERE id = $1::uuid",
)
.bind(pid)
.fetch_optional(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("parent lookup: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", pid))?
} else {
return Err(DomainError::internal_error(
"FolderDb",
"Cannot create root folder without user_id — use create_home_folder instead",
"Cannot create root folder — root folders are reserved for the \
atomic drive-creation transaction in DrivePgRepository::\
create_personal_drive_atomic (docs/plan/drive.md §3). The \
no-orphan-root-folder trigger enforces this at the DB level.",
));
};
let row = sqlx::query_as::<_, (String, String, i64, i64, i64)>(
// D0 dual-write: drive_id alongside user_id (drops in D7); plus
// §14 provenance — `created_by` / `updated_by` bind to the caller
// ($5), NOT to the parent folder's `user_id`. Pre-D2 they're
// silently equivalent (only the parent's owner can write); the
// distinction matters once shared drives let an Editor mutate
// a folder owned by someone else.
//
// RETURNING also surfaces the two provenance columns so the
// built entity / DTO carries fresh values without a re-read.
let row = sqlx::query_as::<_, (String, String, i64, i64, i64, Option<Uuid>, Option<Uuid>)>(
r#"
INSERT INTO storage.folders (name, parent_id, user_id)
VALUES ($1, $2::uuid, $3)
INSERT INTO storage.folders
(name, parent_id, user_id, drive_id, created_by, updated_by)
VALUES ($1, $2::uuid, $3, $4, $5, $5)
RETURNING id::text,
path,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by,
updated_by
"#,
)
.bind(&name)
.bind(&parent_id)
.bind(user_id)
.bind(drive_id)
.bind(caller_id)
.fetch_one(self.pool())
.await
.map_err(|e| {
@@ -201,19 +253,24 @@ impl FolderRepository for FolderDbRepository {
row.1,
parent_id,
Some(user_id),
drive_id,
row.2,
row.3,
row.4,
// Fresh from RETURNING — caller_id was bound to both columns.
row.5,
row.6,
)
}
async fn get_folder(&self, id: &str) -> Result<Folder, DomainError> {
let row = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE id = $1::uuid AND NOT is_trashed
"#,
@@ -224,10 +281,26 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("get: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", id))?;
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6, row.7)
Self::row_to_folder(
row.0,
row.1,
row.2,
row.3,
Some(row.4),
row.5,
row.6,
row.7,
row.8,
row.9,
row.10,
)
}
async fn get_folder_by_path(&self, storage_path: &StoragePath) -> Result<Folder, DomainError> {
async fn get_folder_by_path(
&self,
storage_path: &StoragePath,
drive_id: Uuid,
) -> Result<Folder, DomainError> {
let path_str = storage_path.to_string();
// Strip leading '/' if present — DB stores "Home - user/Docs", not "/Home - user/Docs"
let lookup = path_str.strip_prefix('/').unwrap_or(&path_str);
@@ -236,23 +309,44 @@ impl FolderRepository for FolderDbRepository {
return Err(DomainError::not_found("Folder", "empty path"));
}
// Scoped by drive_id: post-D0 `storage.folders.path` is unique
// only within a single drive. Root-folder names like
// `"Personal"` repeat across drives, so without the drive_id
// filter the planner returns a non-deterministic row — which
// breaks owner-short-circuit checks and crosses drive
// boundaries (the AuthZ axis that replaces the old per-user
// wrapper scoping post-D0).
let row = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE path = $1 AND NOT is_trashed
WHERE path = $1 AND drive_id = $2 AND NOT is_trashed
"#,
)
.bind(lookup)
.bind(drive_id)
.fetch_optional(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("path lookup: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", lookup))?;
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6, row.7)
Self::row_to_folder(
row.0,
row.1,
row.2,
row.3,
Some(row.4),
row.5,
row.6,
row.7,
row.8,
row.9,
row.10,
)
}
#[allow(clippy::type_complexity)]
@@ -260,10 +354,11 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRow> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id = $1::uuid AND NOT is_trashed
ORDER BY name
@@ -275,10 +370,11 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id IS NULL AND NOT is_trashed
ORDER BY name
@@ -290,8 +386,8 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("list: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, Some(uid), did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -305,10 +401,11 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRow> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id = $1::uuid AND user_id = $2 AND NOT is_trashed
ORDER BY name
@@ -321,10 +418,11 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id IS NULL AND user_id = $1 AND NOT is_trashed
ORDER BY name
@@ -337,8 +435,8 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("list_by_owner: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, Some(uid), did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -357,10 +455,11 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRowPaginated> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id = $1::uuid AND NOT is_trashed
@@ -376,10 +475,11 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id IS NULL AND NOT is_trashed
@@ -396,16 +496,18 @@ impl FolderRepository for FolderDbRepository {
// total_count is identical in every row; 0 when the result set is empty.
let total = if include_total {
Some(rows.first().map_or(0, |r| r.8) as usize)
Some(rows.first().map_or(0, |r| r.11) as usize)
} else {
None
};
let folders: Result<Vec<Folder>, DomainError> = rows
.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma, tma)
})
.map(
|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), did, ca, ma, tma, cb, ub)
},
)
.collect();
Ok((folders?, total))
}
@@ -424,10 +526,11 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRowPaginated> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id = $1::uuid AND user_id = $2 AND NOT is_trashed
@@ -444,10 +547,11 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by,
COUNT(*) OVER() AS total_count
FROM storage.folders
WHERE parent_id IS NULL AND user_id = $1 AND NOT is_trashed
@@ -464,41 +568,55 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("paginate_by_owner: {e}")))?;
let total = if include_total {
Some(rows.first().map_or(0, |r| r.8) as usize)
Some(rows.first().map_or(0, |r| r.11) as usize)
} else {
None
};
let folders: Result<Vec<Folder>, DomainError> = rows
.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma, tma)
})
.map(
|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub, _total)| {
Self::row_to_folder(id, name, path, pid, Some(uid), did, ca, ma, tma, cb, ub)
},
)
.collect();
Ok((folders?, total))
}
async fn rename_folder(&self, id: &str, new_name: String) -> Result<Folder, DomainError> {
async fn rename_folder(
&self,
id: &str,
new_name: String,
caller_id: Uuid,
) -> Result<Folder, DomainError> {
// The BEFORE UPDATE trigger recomputes path/lpath for this row;
// the AFTER UPDATE cascade trigger then batch-updates all
// descendants in a single UPDATE using the GiST lpath index.
// That multi-row rewrite can deadlock against the tree-ETag
// flusher's id-ordered ancestor bump — retry instead of failing
// the user's operation (40P01 only; 23505 still maps below).
//
// §14: `updated_by = $3` (caller_id) — the caller mutated this
// row, not the row's owner. In D2 a shared-drive member can
// rename a row they don't own; the previous `updated_by = user_id`
// would have silently recorded the wrong principal.
let row = retry_on_deadlock("folders.rename", || {
sqlx::query_as::<_, FolderRow>(
r#"
UPDATE storage.folders
SET name = $1, updated_at = NOW()
SET name = $1, updated_at = NOW(), updated_by = $3
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, path, parent_id::text, user_id,
RETURNING id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
"#,
)
.bind(&new_name)
.bind(id)
.bind(caller_id)
.fetch_optional(self.pool())
})
.await
@@ -512,39 +630,68 @@ impl FolderRepository for FolderDbRepository {
})?
.ok_or_else(|| DomainError::not_found("Folder", id))?;
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6, row.7)
Self::row_to_folder(
row.0,
row.1,
row.2,
row.3,
Some(row.4),
row.5,
row.6,
row.7,
row.8,
row.9,
row.10,
)
}
async fn move_folder(
&self,
id: &str,
new_parent_id: Option<&str>,
caller_id: Uuid,
) -> Result<Folder, DomainError> {
// The BEFORE UPDATE trigger recomputes path/lpath for this row;
// the AFTER UPDATE cascade trigger then batch-updates all
// descendants in a single UPDATE using the GiST lpath index.
// Retried on deadlock vs the tree-ETag flusher (see rename_folder).
//
// §14: `updated_by = $3` (caller_id), see rename_folder.
let row = retry_on_deadlock("folders.move", || {
sqlx::query_as::<_, FolderRow>(
r#"
UPDATE storage.folders
SET parent_id = $1::uuid, updated_at = NOW()
SET parent_id = $1::uuid, updated_at = NOW(), updated_by = $3
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, path, parent_id::text, user_id,
RETURNING id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
"#,
)
.bind(new_parent_id)
.bind(id)
.bind(caller_id)
.fetch_optional(self.pool())
})
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("move: {e}")))?
.ok_or_else(|| DomainError::not_found("Folder", id))?;
Self::row_to_folder(row.0, row.1, row.2, row.3, Some(row.4), row.5, row.6, row.7)
Self::row_to_folder(
row.0,
row.1,
row.2,
row.3,
Some(row.4),
row.5,
row.6,
row.7,
row.8,
row.9,
row.10,
)
}
async fn delete_folder(&self, id: &str) -> Result<(), DomainError> {
@@ -582,14 +729,22 @@ impl FolderRepository for FolderDbRepository {
Ok(())
}
async fn folder_exists(&self, storage_path: &StoragePath) -> Result<bool, DomainError> {
async fn folder_exists(
&self,
storage_path: &StoragePath,
drive_id: Uuid,
) -> Result<bool, DomainError> {
let path_str = storage_path.to_string();
let lookup = path_str.strip_prefix('/').unwrap_or(&path_str);
// Post-D0 `storage.folders.path` repeats across drives —
// filter by `drive_id` to scope the existence check.
let exists: bool = sqlx::query_scalar(
"SELECT EXISTS(SELECT 1 FROM storage.folders WHERE path = $1 AND NOT is_trashed)",
"SELECT EXISTS(SELECT 1 FROM storage.folders \
WHERE path = $1 AND drive_id = $2 AND NOT is_trashed)",
)
.bind(lookup)
.bind(drive_id)
.fetch_one(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("exists: {e}")))?;
@@ -611,7 +766,7 @@ impl FolderRepository for FolderDbRepository {
// ── Trash operations ──
async fn move_to_trash(&self, folder_id: &str) -> Result<(), DomainError> {
async fn move_to_trash(&self, folder_id: &str, caller_id: Uuid) -> Result<(), DomainError> {
// Soft-delete the whole subtree in one statement: the root flips
// `is_trashed` and records `original_parent_id` so restore knows
// where to put it back; every descendant (folder or file) that
@@ -626,6 +781,10 @@ impl FolderRepository for FolderDbRepository {
// `/g9-tree/file.txt` still resolved 207 even though the parent
// collection was gone) — a class of data-integrity drift that
// confused desktop-sync tree walks.
//
// §14: all three CTE branches stamp `updated_by = $2`
// (caller_id). The cascade is "the caller trashed this
// subtree", not "each owner trashed their own row".
let result = retry_on_deadlock("folders.trash", || {
sqlx::query_scalar::<_, i64>(
r#"
@@ -634,7 +793,8 @@ impl FolderRepository for FolderDbRepository {
SET is_trashed = TRUE,
trashed_at = NOW(),
original_parent_id = parent_id,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
WHERE id = $1::uuid AND NOT is_trashed
RETURNING id, lpath
),
@@ -642,7 +802,8 @@ impl FolderRepository for FolderDbRepository {
UPDATE storage.folders f
SET is_trashed = TRUE,
trashed_at = NOW(),
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
FROM trash_root tr
WHERE f.lpath <@ tr.lpath
AND f.id != tr.id
@@ -653,7 +814,8 @@ impl FolderRepository for FolderDbRepository {
UPDATE storage.files fi
SET is_trashed = TRUE,
trashed_at = NOW(),
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
FROM trash_root tr
JOIN storage.folders f ON f.lpath <@ tr.lpath
WHERE fi.folder_id = f.id
@@ -664,6 +826,7 @@ impl FolderRepository for FolderDbRepository {
"#,
)
.bind(folder_id)
.bind(caller_id)
.fetch_one(self.pool())
})
.await
@@ -680,6 +843,7 @@ impl FolderRepository for FolderDbRepository {
&self,
folder_id: &str,
_original_path: &str,
caller_id: Uuid,
) -> Result<(), DomainError> {
// Inverse of the cascade in `move_to_trash`: restore the root
// (BEFORE UPDATE trigger recomputes path/lpath via the parent_id
@@ -689,6 +853,10 @@ impl FolderRepository for FolderDbRepository {
// *before* this folder went to trash have `original_*` set, so
// they correctly stay in trash and continue to show up as
// top-level trash entries via `storage.trash_items`.
//
// §14: all three CTE branches stamp `updated_by = $2`
// (caller_id). Restoration is "the caller restored this
// subtree", regardless of who originally owned each row.
let result = retry_on_deadlock("folders.restore", || {
sqlx::query_scalar::<_, i64>(
r#"
@@ -698,7 +866,8 @@ impl FolderRepository for FolderDbRepository {
trashed_at = NULL,
parent_id = COALESCE(original_parent_id, parent_id),
original_parent_id = NULL,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
WHERE id = $1::uuid AND is_trashed
RETURNING id, lpath
),
@@ -706,7 +875,8 @@ impl FolderRepository for FolderDbRepository {
UPDATE storage.folders f
SET is_trashed = FALSE,
trashed_at = NULL,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
FROM restore_root rr
WHERE f.lpath <@ rr.lpath
AND f.id != rr.id
@@ -718,7 +888,8 @@ impl FolderRepository for FolderDbRepository {
UPDATE storage.files fi
SET is_trashed = FALSE,
trashed_at = NULL,
updated_at = NOW()
updated_at = NOW(),
updated_by = $2
FROM restore_root rr
JOIN storage.folders f ON f.lpath <@ rr.lpath
WHERE fi.folder_id = f.id
@@ -730,6 +901,7 @@ impl FolderRepository for FolderDbRepository {
"#,
)
.bind(folder_id)
.bind(caller_id)
.fetch_one(self.pool())
})
.await
@@ -775,61 +947,6 @@ impl FolderRepository for FolderDbRepository {
Ok(())
}
async fn create_home_folder(&self, user_id: Uuid, name: String) -> Result<Folder, DomainError> {
let row = sqlx::query_as::<_, (String, String, i64, i64, i64)>(
r#"
INSERT INTO storage.folders (name, parent_id, user_id)
VALUES ($1, NULL, $2)
ON CONFLICT DO NOTHING
RETURNING id::text,
path,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
"#,
)
.bind(&name)
.bind(user_id)
.fetch_optional(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("home folder: {e}")))?;
match row {
Some((id, path, ca, ma, tma)) => {
Self::row_to_folder(id, name.clone(), path, None, Some(user_id), ca, ma, tma)
}
None => {
// Already exists — fetch it
let existing = sqlx::query_as::<_, (String, String, i64, i64, i64)>(
r#"
SELECT id::text,
path,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
FROM storage.folders
WHERE name = $1 AND user_id = $2 AND parent_id IS NULL
"#,
)
.bind(&name)
.bind(user_id)
.fetch_one(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("home fetch: {e}")))?;
Self::row_to_folder(
existing.0,
name,
existing.1,
None,
Some(user_id),
existing.2,
existing.3,
existing.4,
)
}
}
}
/// Lists every folder in a subtree rooted at `folder_id` (inclusive).
///
/// Single GiST-indexed query: `fo.lpath <@ (root's lpath)`.
@@ -837,10 +954,11 @@ impl FolderRepository for FolderDbRepository {
#[allow(clippy::type_complexity)]
async fn list_subtree_folders(&self, folder_id: &str) -> Result<Vec<Folder>, DomainError> {
let sql = "SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
fo.user_id, \
fo.user_id, fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint, \
fo.created_by, fo.updated_by \
FROM storage.folders fo \
WHERE fo.is_trashed = false \
AND fo.lpath <@ (SELECT lpath FROM storage.folders WHERE id = $1::uuid) \
@@ -855,8 +973,8 @@ impl FolderRepository for FolderDbRepository {
})?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, uid, ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -900,10 +1018,11 @@ impl FolderRepository for FolderDbRepository {
// Recursive, no folder scope → ALL user folders
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
fo.user_id, \
fo.user_id, fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint, \
fo.created_by, fo.updated_by \
FROM storage.folders fo \
WHERE fo.user_id = $1 \
AND fo.is_trashed = false \
@@ -927,8 +1046,8 @@ impl FolderRepository for FolderDbRepository {
return rows
.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, uid, ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)
})
.collect();
}
@@ -937,10 +1056,11 @@ impl FolderRepository for FolderDbRepository {
let sql = if parent_id.is_some() {
format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
fo.user_id, \
fo.user_id, fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint, \
fo.created_by, fo.updated_by \
FROM storage.folders fo \
WHERE fo.parent_id = $1::uuid \
AND fo.user_id = $2 \
@@ -956,10 +1076,11 @@ impl FolderRepository for FolderDbRepository {
};
format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
fo.user_id, \
fo.user_id, fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint, \
fo.created_by, fo.updated_by \
FROM storage.folders fo \
WHERE fo.parent_id IS NULL \
AND fo.user_id = $1 \
@@ -999,8 +1120,8 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("search_folders: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, uid, ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -1025,10 +1146,11 @@ impl FolderRepository for FolderDbRepository {
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
fo.user_id, \
fo.user_id, fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint \
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint, \
fo.created_by, fo.updated_by \
FROM storage.folders fo \
WHERE fo.user_id = $1 \
AND fo.is_trashed = false \
@@ -1055,8 +1177,8 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("descendant search: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, uid, ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -1074,10 +1196,11 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRow> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id = $1::uuid
AND NOT is_trashed
@@ -1100,10 +1223,11 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, user_id,
SELECT id::text, name, path, parent_id::text, user_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id IS NULL
AND NOT is_trashed
@@ -1126,8 +1250,8 @@ impl FolderRepository for FolderDbRepository {
.map_err(|e| DomainError::internal_error("FolderDb", format!("suggest: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, uid, ca, ma, tma)| {
Self::row_to_folder(id, name, path, pid, Some(uid), ca, ma, tma)
.map(|(id, name, path, pid, uid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, Some(uid), did, ca, ma, tma, cb, ub)
})
.collect()
}
@@ -6,6 +6,7 @@ mod contact_group_pg_repository;
mod contact_persistence_dto;
mod contact_pg_repository;
mod device_code_pg_repository;
mod drive_pg_repository;
mod face_pg_repository;
mod favorites_pg_repository;
pub mod file_metadata_repository;
@@ -34,6 +35,7 @@ pub use contact_group_pg_repository::ContactGroupPgRepository;
pub use contact_persistence_dto::*;
pub use contact_pg_repository::ContactPgRepository;
pub use device_code_pg_repository::DeviceCodePgRepository;
pub use drive_pg_repository::DrivePgRepository;
pub use face_pg_repository::FacePgRepository;
pub use favorites_pg_repository::FavoritesPgRepository;
pub use file_blob_read_repository::FileBlobReadRepository;
+49 -28
View File
@@ -1927,7 +1927,7 @@ impl DedupService {
OR b.orphaned_at < now() - ($2::int * interval '1 second'))
AND NOT EXISTS (
SELECT 1 FROM storage.chunk_manifests m
WHERE m.chunk_hashes @> ARRAY[b.hash]
WHERE m.chunk_hashes @> ARRAY[b.hash::text]
)
AND NOT EXISTS (
SELECT 1 FROM storage.files f
@@ -2776,12 +2776,22 @@ mod rechunk_integration_tests {
Arc::new(pool)
}
async fn seed_user(pool: &PgPool) -> Uuid {
sqlx::query("SELECT id FROM auth.users LIMIT 1")
.fetch_one(pool)
.await
.map(|r| r.get::<Uuid, _>("id"))
.expect("auth.users must be seeded (init-test-schema.sh)")
/// Returns `(user_id, drive_id)`. Post-D0 every internal user has a
/// default Personal drive (provisioned by `PersonalDriveLifecycleHook`
/// during init-test-schema.sh's user seeding); the JOIN below picks
/// the user-drive pair atomically so test fixtures can insert into
/// `storage.files` with both `user_id` and `drive_id` populated.
async fn seed_user(pool: &PgPool) -> (Uuid, Uuid) {
sqlx::query(
"SELECT u.id AS user_id, d.id AS drive_id
FROM auth.users u
JOIN storage.drives d ON d.default_for_user = u.id
LIMIT 1",
)
.fetch_one(pool)
.await
.map(|r| (r.get::<Uuid, _>("user_id"), r.get::<Uuid, _>("drive_id")))
.expect("auth.users + storage.drives must be seeded (init-test-schema.sh)")
}
/// Plain local backend in a fresh temp dir.
@@ -2848,7 +2858,7 @@ mod rechunk_integration_tests {
.await
.expect("insert legacy blob row");
let user_id = seed_user(pool).await;
let (user_id, drive_id) = seed_user(pool).await;
let mut file_ids = Vec::new();
for i in 0..n_files {
let name = format!(
@@ -2856,11 +2866,12 @@ mod rechunk_integration_tests {
&Uuid::new_v4().to_string()[..8]
);
let id: Uuid = sqlx::query_scalar(
"INSERT INTO storage.files (name, user_id, blob_hash, size)
VALUES ($1, $2, $3, $4) RETURNING id",
"INSERT INTO storage.files (name, user_id, drive_id, blob_hash, size)
VALUES ($1, $2, $3, $4, $5) RETURNING id",
)
.bind(&name)
.bind(user_id)
.bind(drive_id)
.bind(&hash)
.bind(data.len() as i64)
.fetch_one(pool)
@@ -3121,12 +3132,20 @@ mod delta_upload_integration_tests {
Arc::new(pool)
}
async fn seed_user(pool: &PgPool) -> Uuid {
sqlx::query("SELECT id FROM auth.users LIMIT 1")
.fetch_one(pool)
.await
.map(|r| r.get::<Uuid, _>("id"))
.expect("auth.users must be seeded (init-test-schema.sh)")
/// Returns `(user_id, drive_id)` — same shape as the rechunk tests'
/// `seed_user`. Post-D0 every internal user has a default Personal
/// drive provisioned by `PersonalDriveLifecycleHook`.
async fn seed_user(pool: &PgPool) -> (Uuid, Uuid) {
sqlx::query(
"SELECT u.id AS user_id, d.id AS drive_id
FROM auth.users u
JOIN storage.drives d ON d.default_for_user = u.id
LIMIT 1",
)
.fetch_one(pool)
.await
.map(|r| (r.get::<Uuid, _>("user_id"), r.get::<Uuid, _>("drive_id")))
.expect("auth.users + storage.drives must be seeded (init-test-schema.sh)")
}
async fn local_svc(pool: &Arc<PgPool>, dir: &TempDir) -> DedupService {
@@ -3141,6 +3160,7 @@ mod delta_upload_integration_tests {
svc: &DedupService,
pool: &PgPool,
user_id: Uuid,
drive_id: Uuid,
data: &[u8],
label: &str,
) -> (String, Vec<String>, Uuid) {
@@ -3159,14 +3179,15 @@ mod delta_upload_integration_tests {
.expect("chunks");
let file_id: Uuid = sqlx::query_scalar(
"INSERT INTO storage.files (name, user_id, blob_hash, size)
VALUES ($1, $2, $3, $4) RETURNING id",
"INSERT INTO storage.files (name, user_id, drive_id, blob_hash, size)
VALUES ($1, $2, $3, $4, $5) RETURNING id",
)
.bind(format!(
"rust-test-delta-{label}-{}",
&Uuid::new_v4().to_string()[..8]
))
.bind(user_id)
.bind(drive_id)
.bind(&file_hash)
.bind(data.len() as i64)
.fetch_one(pool)
@@ -3226,13 +3247,13 @@ mod delta_upload_integration_tests {
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let user = seed_user(&pool).await;
let (user, drive_id) = seed_user(&pool).await;
// Owned content (multi-chunk), one foreign chunk (ref 1, no file
// row for this user), one orphan (ref 0), one unknown hash.
let data = content(3 * 1024 * 1024, 21);
let (file_hash, owned_chunks, file_id) =
seed_owned_content(&svc, &pool, user, &data, "claim").await;
seed_owned_content(&svc, &pool, user, drive_id, &data, "claim").await;
assert!(owned_chunks.len() >= 3, "3 MiB must split into ≥3 chunks");
let foreign = blake3::hash(format!("foreign-{}", Uuid::new_v4()).as_bytes())
@@ -3316,12 +3337,12 @@ mod delta_upload_integration_tests {
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let user = seed_user(&pool).await;
let (user, drive_id) = seed_user(&pool).await;
// An owned chunk that the client redundantly re-uploads.
let data = content(100 * 1024, 22);
let (file_hash, owned_chunks, file_id) =
seed_owned_content(&svc, &pool, user, &data, "loose").await;
seed_owned_content(&svc, &pool, user, drive_id, &data, "loose").await;
let owned_chunk_bytes = {
let mut stream = svc.read_blob_stream(&file_hash).await.expect("stream");
let mut out = Vec::new();
@@ -3373,7 +3394,7 @@ mod delta_upload_integration_tests {
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let user = seed_user(&pool).await;
let (user, drive_id) = seed_user(&pool).await;
// (A) An aged orphan (orphaned well past the grace window) with no
// references → must be collected (row + backing file).
@@ -3412,7 +3433,7 @@ mod delta_upload_integration_tests {
// stale ref_count must never delete referenced content.
let data = content(3 * 1024 * 1024, 71);
let (file_hash, owned_chunks, file_id) =
seed_owned_content(&svc, &pool, user, &data, "gc").await;
seed_owned_content(&svc, &pool, user, drive_id, &data, "gc").await;
let referenced = owned_chunks[0].clone();
sqlx::query(
"UPDATE storage.blobs
@@ -3470,12 +3491,12 @@ mod delta_upload_integration_tests {
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let user = seed_user(&pool).await;
let (user, drive_id) = seed_user(&pool).await;
// Single-owner multi-chunk CDC file → its chunks are uniquely owned.
let data = content(3 * 1024 * 1024, 91);
let (file_hash, chunks, file_id) =
seed_owned_content(&svc, &pool, user, &data, "deref").await;
seed_owned_content(&svc, &pool, user, drive_id, &data, "deref").await;
assert!(chunks.len() >= 3, "3 MiB must split into ≥3 chunks");
// The delete_file_permanently sequence: drop the file row (PG trigger)
@@ -3540,11 +3561,11 @@ mod delta_upload_integration_tests {
let pool = test_pool().await;
let dir = TempDir::new().unwrap();
let svc = local_svc(&pool, &dir).await;
let user = seed_user(&pool).await;
let (user, drive_id) = seed_user(&pool).await;
let data = content(2 * 1024 * 1024 + 137, 24);
let (file_hash, _chunks, file_id) =
seed_owned_content(&svc, &pool, user, &data, "verify").await;
seed_owned_content(&svc, &pool, user, drive_id, &data, "verify").await;
let manifest: (Vec<String>, Vec<i64>) = sqlx::query_as(
"SELECT chunk_hashes, chunk_sizes FROM storage.chunk_manifests WHERE file_hash = $1",
@@ -64,6 +64,7 @@ impl PathResolverService {
String, // path
Option<String>, // parent_id
Option<String>, // user_id
Uuid, // drive_id
i64, // created_at
i64, // modified_at
Option<i64>, // size
@@ -72,7 +73,7 @@ impl PathResolverService {
),
>(
r#"
SELECT resource_type, id, name, path, parent_id, user_id,
SELECT resource_type, id, name, path, parent_id, user_id, drive_id,
created_at, modified_at, size, mime_type, folder_id
FROM (
SELECT 'folder'::text AS resource_type,
@@ -81,6 +82,7 @@ impl PathResolverService {
fo.path,
fo.parent_id::text,
fo.user_id::text,
fo.drive_id,
EXTRACT(EPOCH FROM fo.created_at)::bigint AS created_at,
EXTRACT(EPOCH FROM fo.updated_at)::bigint AS modified_at,
NULL::bigint AS size,
@@ -102,6 +104,7 @@ impl PathResolverService {
END AS path,
NULL::text AS parent_id,
fi.user_id::text,
fi.drive_id,
EXTRACT(EPOCH FROM fi.created_at)::bigint AS created_at,
EXTRACT(EPOCH FROM fi.updated_at)::bigint AS modified_at,
fi.size,
@@ -136,6 +139,7 @@ impl PathResolverService {
res_path,
parent_id,
uid,
drive_id,
created_at,
modified_at,
size,
@@ -151,12 +155,19 @@ impl PathResolverService {
path: res_path,
parent_id,
owner_id: uid,
drive_id,
created_at: created_at as u64,
modified_at: modified_at as u64,
is_root: false,
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
// §14 provenance not selected by this resolver path —
// it's used for existence/type discrimination, not
// detailed DTO emission. Callers that need provenance
// reload through the repo.
created_by: None,
updated_by: None,
})),
_ => {
let mime = mime_type.unwrap_or_else(|| "application/octet-stream".to_string());
@@ -183,6 +194,9 @@ impl PathResolverService {
sort_date: None,
content_hash: String::new(),
etag: String::new(),
// §14 provenance not selected by this resolver path
created_by: None,
updated_by: None,
}))
}
}
+68 -1
View File
@@ -259,11 +259,32 @@ impl PgAclEngine {
}
}
/// Public wrapper around `subject_match_set` for callers that need
/// the expanded `(subject_types, subject_ids)` pair without invoking
/// the engine's full `check`/`require` pipeline. Used by
/// `GET /api/drives` (and future drive-aware listing surfaces) to
/// ask the `DriveRepository` for every drive the caller can read,
/// reusing the engine's cached group-expansion logic.
pub async fn expand_subject_for_listing(
&self,
subject: Subject,
) -> Result<(Vec<&'static str>, Vec<Uuid>), DomainError> {
let counters = QueryCounters::default();
self.subject_match_set(subject, &counters).await
}
/// Returns the owner UUID for any resource type.
async fn owner_of(&self, resource: Resource) -> Result<Uuid, DomainError> {
match resource {
Resource::Folder(id) => self.folder_repo.get_folder_user_id(&id.to_string()).await,
Resource::File(id) => self.file_repo.get_file_user_id(&id.to_string()).await,
// Drive owner resolution wires up in D0-6 once `DriveRepository`
// lands (D0-5). Drive entity carries `default_for_user` for
// `kind='personal'`; shared drives resolve through role_grants
// (Owner role). Returning NotFound here means a permission
// check that reached owner_of on a Drive falls through to the
// grant-lookup path — safe default during D0-1.
Resource::Drive(_) => Err(DomainError::not_found("Drive", resource.id().to_string())),
}
}
@@ -389,6 +410,43 @@ impl PgAclEngine {
Ok(exists.is_some())
}
/// Direct grant lookup for a drive — no ltree cascade (drives have
/// no ancestors). Mirrors the cascade helpers above but with a
/// straight `resource_type='drive' AND resource_id=$4` filter.
async fn drive_grant_exists(
&self,
subject_types: &[&str],
subject_ids: &[Uuid],
permission: Permission,
drive_id: Uuid,
counters: &QueryCounters,
) -> Result<bool, DomainError> {
counters.sql_queries.fetch_add(1, Ordering::Relaxed);
let roles = Self::roles_implying_strings(permission);
let exists: Option<i32> = sqlx::query_scalar(
r#"
SELECT 1
FROM storage.role_grants g
WHERE g.subject_type = ANY($1)
AND g.subject_id = ANY($2)
AND g.role = ANY($3::storage.grant_role[])
AND g.resource_type = 'drive'
AND g.resource_id = $4
AND (g.expires_at IS NULL OR g.expires_at > NOW())
LIMIT 1
"#,
)
.bind(subject_types)
.bind(subject_ids)
.bind(&roles)
.bind(drive_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("PgAcl", format!("drive grant: {e}")))?;
Ok(exists.is_some())
}
/// Look up a single role grant by id, returning the actors a revoke /
/// notify handler needs to make a decision without a second round-trip.
/// Returns `(subject, resource, granted_by)` or `None` if no such row.
@@ -488,7 +546,12 @@ impl PgAclEngine {
) -> Result<bool, DomainError> {
// Owner short-circuit (only for User subjects — groups/tokens/external
// are never owners of resources).
if let Subject::User(uid) = subject {
// Owner short-circuit applies to Folder/File only — they carry a
// single-owner `user_id` column in their respective tables. Drives
// model ownership through the `Owner` role in `role_grants`, so
// there's no analogous fast path: the grant lookup below resolves
// a drive owner via the same query that resolves any drive role.
if let (Subject::User(uid), Resource::Folder(_) | Resource::File(_)) = (subject, resource) {
counters.sql_queries.fetch_add(1, Ordering::Relaxed);
match self.owner_of(resource).await {
Ok(owner) if owner == uid => return Ok(true),
@@ -529,6 +592,10 @@ impl PgAclEngine {
)
.await
}
Resource::Drive(id) => {
self.drive_grant_exists(&subject_types, &subject_ids, permission, id, counters)
.await
}
}
}
}
@@ -242,14 +242,15 @@ impl ContentIndexWorker {
// Authoritative state re-read: a queued 'upsert' whose row vanished
// or got trashed in the meantime becomes a delete.
let files: Vec<(Uuid, String, String, String, String, i64)> =
let files: Vec<(Uuid, String, String, String, String, String, i64)> =
if upsert_candidates.is_empty() {
Vec::new()
} else {
sqlx::query_as(
"SELECT fi.id, fi.user_id::text, fi.name, fi.blob_hash, fi.mime_type, fi.size
FROM storage.files fi
WHERE fi.id = ANY($1) AND NOT fi.is_trashed",
"SELECT fi.id, fi.user_id::text, fi.drive_id::text, fi.name,
fi.blob_hash, fi.mime_type, fi.size
FROM storage.files fi
WHERE fi.id = ANY($1) AND NOT fi.is_trashed",
)
.bind(&upsert_candidates)
.fetch_all(self.maintenance_pool.as_ref())
@@ -261,10 +262,10 @@ impl ContentIndexWorker {
// Per-blob text: batch-read the extraction cache, extract misses.
let wanted_hashes: Vec<String> = files
.iter()
.filter(|(_, _, name, _, mime, size)| {
.filter(|(_, _, _, name, _, mime, size)| {
text_extractor::supports(name, mime) && *size as u64 <= self.max_extract_file_bytes
})
.map(|f| f.3.clone())
.map(|f| f.4.clone())
.collect();
let mut text_by_hash: HashMap<String, Option<String>> = HashMap::new();
if !wanted_hashes.is_empty() {
@@ -281,7 +282,7 @@ impl ContentIndexWorker {
}
let mut records = Vec::with_capacity(files.len());
for (file_id, user_id, name, blob_hash, mime, size) in files {
for (file_id, user_id, drive_id, name, blob_hash, mime, size) in files {
let supported = text_extractor::supports(&name, &mime);
let content = if !supported {
None
@@ -301,6 +302,7 @@ impl ContentIndexWorker {
records.push(IndexDocRecord {
file_id: file_id.to_string(),
user_id,
drive_id,
name,
content,
preview,
@@ -37,7 +37,15 @@ use crate::common::errors::DomainError;
/// Bump whenever the Tantivy schema OR the text extractor output changes in a
/// way that requires re-indexing. A mismatch with the on-disk marker wipes the
/// index directory and reseeds the dirty queue with every live file.
pub const INDEX_SCHEMA_VERSION: &str = "1";
///
/// Version history:
/// 1 — initial schema (file_id, user_id, name, content, preview)
/// 2 — D0 added `drive_id` field; query filter pivots from user_id
/// to a `drive_id ∈ accessible_drives` set membership clause. On
/// deploy, every operator's index is wiped and reseeded against
/// the post-D0 schema (the worker drains the dirty queue with
/// drive_id-aware records).
pub const INDEX_SCHEMA_VERSION: &str = "2";
/// Recorded in `storage.blob_extracted_text.extractor`; rows from another
/// version are dropped at worker startup (the reseed re-extracts them).
@@ -73,6 +81,11 @@ const PREFIX_MIN_CHARS: usize = 3;
pub struct IndexDocRecord {
pub file_id: String,
pub user_id: String,
/// Owning drive — written verbatim into the `drive_id` STRING field
/// for set-membership filtering at query time. The user_id field is
/// kept during the D0 dual-write window for rollback safety; the
/// query filter no longer reads it.
pub drive_id: String,
pub name: String,
pub content: Option<String>,
pub preview: Option<String>,
@@ -82,6 +95,7 @@ pub struct IndexDocRecord {
struct IndexFields {
file_id: Field,
user_id: Field,
drive_id: Field,
name: Field,
content: Field,
preview: Field,
@@ -105,6 +119,7 @@ impl TantivyContentIndex {
let fields = IndexFields {
file_id: builder.add_text_field("file_id", STRING | STORED),
user_id: builder.add_text_field("user_id", STRING),
drive_id: builder.add_text_field("drive_id", STRING),
name: builder.add_text_field("name", TEXT),
content: builder.add_text_field("content", TEXT),
preview: builder.add_text_field("preview", STORED),
@@ -197,6 +212,7 @@ impl TantivyContentIndex {
let mut document = doc!(
self.fields.file_id => record.file_id,
self.fields.user_id => record.user_id,
self.fields.drive_id => record.drive_id,
self.fields.name => record.name,
);
if let Some(content) = record.content {
@@ -234,15 +250,26 @@ impl TantivyContentIndex {
/// Build the scored query: every token must match (in name OR content,
/// exact OR fuzzy OR — for the last token — prefix), and the whole thing
/// is `Must`-scoped to the user.
fn build_query(fields: IndexFields, user_id: &str, tokens: &[String]) -> Box<dyn Query> {
let mut clauses: Vec<(Occur, Box<dyn Query>)> = vec![(
Occur::Must,
Box::new(TermQuery::new(
Term::from_field_text(fields.user_id, user_id),
IndexRecordOption::Basic,
)),
)];
/// is `Must`-scoped to the caller's accessible drives.
///
/// The drive filter is expressed as a BoolQuery with `Should` arms —
/// at least one drive_id must match — wrapped under an outer `Must`.
/// Equivalent to a TermSetQuery; this form avoids the API churn of
/// rebuilding the same shape across Tantivy versions.
fn build_query(fields: IndexFields, drive_ids: &[String], tokens: &[String]) -> Box<dyn Query> {
// Drive-membership Must clause: union of Term(drive_id = $each).
let drive_alternatives: Vec<(Occur, Box<dyn Query>)> = drive_ids
.iter()
.map(|d| {
let q: Box<dyn Query> = Box::new(TermQuery::new(
Term::from_field_text(fields.drive_id, d),
IndexRecordOption::Basic,
));
(Occur::Should, q)
})
.collect();
let mut clauses: Vec<(Occur, Box<dyn Query>)> =
vec![(Occur::Must, Box::new(BooleanQuery::new(drive_alternatives)))];
let last = tokens.len().saturating_sub(1);
for (i, token) in tokens.iter().enumerate() {
@@ -306,7 +333,7 @@ impl TantivyContentIndex {
searcher: tantivy::Searcher,
analyzer: TextAnalyzer,
fields: IndexFields,
user_id: &str,
drive_ids: &[String],
raw_query: &str,
limit: usize,
) -> Result<Vec<ContentHitDto>, DomainError> {
@@ -315,7 +342,7 @@ impl TantivyContentIndex {
return Ok(Vec::new());
}
let query = Self::build_query(fields, user_id, &tokens);
let query = Self::build_query(fields, drive_ids, &tokens);
let top_docs = searcher
.search(&query, &TopDocs::with_limit(limit.max(1)).order_by_score())
.map_err(|e| DomainError::internal_error("ContentIndex", format!("search: {e}")))?;
@@ -365,18 +392,25 @@ impl TantivyContentIndex {
impl ContentIndexPort for TantivyContentIndex {
async fn search_content(
&self,
user_id: Uuid,
accessible_drive_ids: &[Uuid],
query: &str,
limit: usize,
) -> Result<Vec<ContentHitDto>, DomainError> {
// No accessible drives → no hits, no Tantivy work. Matches the
// anti-enumeration semantics (empty filter set returns empty
// results without any side channel).
if accessible_drive_ids.is_empty() {
return Ok(Vec::new());
}
let searcher = self.reader.searcher();
let analyzer = self.analyzer.clone();
let fields = self.fields;
let user_id = user_id.to_string();
let drive_ids: Vec<String> = accessible_drive_ids.iter().map(|d| d.to_string()).collect();
let query = query.to_owned();
tokio::task::spawn_blocking(move || {
Self::search_blocking(searcher, analyzer, fields, &user_id, &query, limit)
Self::search_blocking(searcher, analyzer, fields, &drive_ids, &query, limit)
})
.await
.map_err(|e| DomainError::internal_error("ContentIndex", format!("join: {e}")))?
@@ -391,6 +425,10 @@ mod tests {
IndexDocRecord {
file_id: file_id.to_owned(),
user_id: user_id.to_owned(),
// Tests stamp a placeholder drive_id derived from user_id so the
// record satisfies the post-D0 schema. Query-side filtering by
// drive_id is exercised in D0-12's integration tests, not here.
drive_id: format!("{user_id}-drive"),
name: name.to_owned(),
content: content.map(str::to_owned),
preview: content.map(str::to_owned),
@@ -401,11 +439,16 @@ mod tests {
// Force a reader reload — OnCommitWithDelay is asynchronous and tests
// must observe the commit immediately.
index.reader.reload().unwrap();
// Test records derive `drive_id = format!("{user_id}-drive")` —
// the same convention used by `record()`. Filtering by that
// single drive id exercises the same path the production
// search uses.
let drive_ids = vec![format!("{user_id}-drive")];
TantivyContentIndex::search_blocking(
index.reader.searcher(),
index.analyzer.clone(),
index.fields,
user_id,
&drive_ids,
query,
32,
)
@@ -113,13 +113,15 @@ impl TreeEtagFlushService {
FROM storage.tree_etag_dirty
ORDER BY id
LIMIT $1)
RETURNING lpath, folder_id
RETURNING lpath, folder_id, drive_id
),
targets AS (
-- Captured chain: covers target folders deleted or
-- moved away since enqueue (the old location's
-- surviving ancestors still get their bump).
SELECT lpath FROM drained
-- surviving ancestors still get their bump). drive_id
-- comes along so the victims walk can enforce
-- cross-drive isolation (D0-13).
SELECT lpath, drive_id FROM drained
UNION
-- Flush-time resolution: a folder MOVED since
-- enqueue had its subtree's lpaths rewritten, so
@@ -128,19 +130,29 @@ impl TreeEtagFlushService {
-- this, a bump queued just before a move would be
-- silently lost and sync clients would never
-- discover the change.
SELECT fo.lpath
SELECT fo.lpath, fo.drive_id
FROM storage.folders fo
JOIN drained d ON fo.id = d.folder_id
),
victims AS (
-- `lpath @> target` = the target folder itself plus
-- every ancestor (GiST-indexed). Folder rows deleted
-- since enqueue simply don't match. Lock in id order
-- so overlapping closures cannot deadlock.
-- every ancestor (GiST-indexed). The `drive_id`
-- predicate prevents a numerically-overlapping
-- lpath in a SIBLING drive from spuriously matching
-- (D0-13). Rows from old queue entries (pre-M4) have
-- NULL `drive_id` — `IS NOT DISTINCT FROM` falls
-- back to pure lpath matching for those, preserving
-- the rollover semantics for any rows enqueued
-- between this migration committing and the
-- service restart.
SELECT f.id
FROM storage.folders f
WHERE EXISTS (SELECT 1 FROM targets t
WHERE f.lpath @> t.lpath)
WHERE EXISTS (
SELECT 1 FROM targets t
WHERE f.lpath @> t.lpath
AND (t.drive_id IS NULL
OR f.drive_id = t.drive_id)
)
ORDER BY f.id
FOR NO KEY UPDATE
),