From 82a50936d4be729f0489c9f34197022803062757 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Thu, 30 Jul 2026 21:21:26 +0200 Subject: [PATCH 01/17] feat(storage-migration): move storage mig. to recoverable job --- Cargo.toml | 16 +- frontend/src/lib/api/endpoints/admin.ts | 24 +- .../src/routes/admin/[[tab]]/+page.svelte | 210 +++++++- src/application/dtos/settings_dto.rs | 39 +- .../services/storage_settings_service.rs | 356 ++++++++++++- src/common/di.rs | 29 +- .../services/migration_blob_backend.rs | 273 ---------- src/infrastructure/services/migration_job.rs | 240 --------- src/infrastructure/services/mod.rs | 3 +- .../services/storage_migration_service.rs | 486 +++++++++++++++++ src/interfaces/api/handlers/admin_handler.rs | 504 +++++++++++------- src/interfaces/api/mod.rs | 1 - 12 files changed, 1436 insertions(+), 745 deletions(-) delete mode 100644 src/infrastructure/services/migration_blob_backend.rs delete mode 100644 src/infrastructure/services/migration_job.rs create mode 100644 src/infrastructure/services/storage_migration_service.rs diff --git a/Cargo.toml b/Cargo.toml index ce8f9b73..491fbd67 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -877,7 +877,21 @@ strip = true [profile.dev] opt-level = 1 -debug = true +# `line-tables-only` keeps file:line in panic backtraces (what you +# actually need on a long-running server) while dropping the rest of +# DWARF, which is worthless without a debugger. On this crate that +# takes target/debug/deps from ~58 GB to ~15-20 GB. Combined with +# `split-debuginfo = "unpacked"` (macOS-friendly: what little debug +# info remains lands in external .dSYM bundles that the linker +# doesn't embed in every .rlib), a full rebuild fits comfortably. +debug = "line-tables-only" +split-debuginfo = "unpacked" +# Incremental compilation caches per-function IR fingerprints so a +# small edit only recompiles what changed. On a single-crate rebuild +# (oxicloud is one crate) the savings are modest — worth < the ~7 GB +# incremental/ cache costs on disk. Rust-analyzer uses `cargo check`, +# which has its own cache, so LSP responsiveness is unaffected. +incremental = false [profile.bench] lto = "fat" diff --git a/frontend/src/lib/api/endpoints/admin.ts b/frontend/src/lib/api/endpoints/admin.ts index 8d9ca09b..1f773c75 100644 --- a/frontend/src/lib/api/endpoints/admin.ts +++ b/frontend/src/lib/api/endpoints/admin.ts @@ -384,13 +384,28 @@ export interface SmtpTestResult { error?: string; } -/** Result of POST .../settings/storage/test — the S3 connection probe. */ +/** + * Result of POST .../settings/storage/test. Combines reachability + * (`connected` — HEAD bucket / statfs) with a full read/write round- + * trip (`roundtrip_passed` — PUT + GET + verify + DELETE). Overall + * pass = both true. Round-trip fields are absent when the round-trip + * wasn't attempted (typically because reachability already failed). + * `phase_reached` names the last successful round-trip step: + * `initialize` | `put_ok` | `exists_ok` | `get_ok` | `verify_ok` | + * `cleanup_ok`. + */ export interface StorageTestResult { connected?: boolean; success?: boolean; backend_type?: string; available_bytes?: number | null; message?: string; + roundtrip_passed?: boolean; + phase_reached?: string; + bytes_written?: number; + bytes_read?: number; + roundtrip_elapsed_ms?: number; + cleanup_ok?: boolean; } export async function sendSmtpTest(to: string): Promise { @@ -497,7 +512,12 @@ export function getMigration(): Promise { return apiJson('/api/admin/storage/migration', { credentials: 'same-origin' }); } -export function migrationAction(action: 'start' | 'pause' | 'resume' | 'complete'): Promise { +export function migrationAction(action: 'start' | 'pause' | 'resume'): Promise { + // `complete` was retired when the migration became a recoverable + // job — Completed is the terminal `RunSummary.status`; there's + // nothing left to acknowledge. Post-migration cutover now happens + // via .env + restart, prompted by an inline hint on the admin + // storage tab (see `cutoverPending` in +page.svelte). const body = action === 'start' ? { concurrency: 4 } : {}; return mutate(`/api/admin/storage/migration/${action}`, 'POST', body); } diff --git a/frontend/src/routes/admin/[[tab]]/+page.svelte b/frontend/src/routes/admin/[[tab]]/+page.svelte index a8bb88ea..e599fb1f 100644 --- a/frontend/src/routes/admin/[[tab]]/+page.svelte +++ b/frontend/src/routes/admin/[[tab]]/+page.svelte @@ -466,18 +466,29 @@ storageMsg = null; try { const r: StorageTestResult = await testStorage(storageBody()); + // Backend now performs BOTH reachability (health-check) + // and a full read/write round-trip. `connected` gets + // flipped to false by the service if the round-trip + // itself fails, so a single boolean covers the whole + // pass/fail signal. `roundtrip_passed` distinguishes the + // two flavours of failure for the operator. const ok = r.connected ?? r.success ?? false; if (ok) { - let text = t('admin.storage_test_success', 'Connection successful'); + let text = t('admin.storage_test_success', 'Connection + read/write OK'); if (r.backend_type) text += ` (${r.backend_type})`; + if (r.roundtrip_elapsed_ms != null) text += ` — round-trip ${r.roundtrip_elapsed_ms} ms`; if (r.available_bytes != null) - text += ` — ${formatBytes(r.available_bytes)} ${t('admin.available', 'available')}`; + text += ` · ${formatBytes(r.available_bytes)} ${t('admin.available', 'available')}`; + if (r.cleanup_ok === false) + text += ` · ⚠ cleanup DELETE failed — orphan test blob left on backend`; storageMsg = { text, ok: true }; } else { - storageMsg = { - text: `${t('admin.storage_test_failure', 'Connection failed')}: ${r.message ?? ''}`, - ok: false - }; + const phase = r.phase_reached ? ` [phase: ${r.phase_reached}]` : ''; + const label = + r.roundtrip_passed === false + ? t('admin.storage_test_failure', 'Read/write test failed') + : t('admin.storage_test_failure', 'Connection failed'); + storageMsg = { text: `${label}${phase}: ${r.message ?? ''}`, ok: false }; } } catch (e) { storageMsg = { text: errorMessage(e), ok: false }; @@ -508,7 +519,7 @@ stopMigrationPoll(); } } - async function doMigration(action: 'start' | 'pause' | 'resume' | 'complete') { + async function doMigration(action: 'start' | 'pause' | 'resume') { try { await migrationAction(action); await loadMigration(); @@ -517,6 +528,73 @@ } } + // ── Post-migration .env cutover hint ───────────────────────────── + // + // Migration copies blobs to the target backend, but boot-time + // backend selection reads env vars only — never the DB config the + // admin filled in. So the app keeps running on the SOURCE backend + // even after the copy completes. To actually cut over, the + // operator has to add the equivalent env vars to `.env` and + // restart. This hint block spells out those lines with a + // copy-to-clipboard button. + // + // Shown only when: + // - a migration has completed successfully, AND + // - the live backend still differs from the configured target + // (so we're actually pending cutover), AND + // - the backend env var isn't ALREADY overriding (which would + // mean the admin already updated .env or the platform sets it). + const cutoverPending = $derived( + migration?.status === 'completed' && + !!storage && + storage.current_backend != null && + storage.current_backend !== storage.backend && + !(storage.env_overrides ?? []).includes('backend') + ); + + // Env-var lines the admin needs to paste. Credentials are NEVER + // echoed — the storage-settings DTO only returns `_set` booleans + // for access/secret keys (not the values), so we render a + // placeholder line the admin fills in from their own records. + // Local backend still gets a line for completeness, but a Local + // deployment typically has no reason to explicitly set the var + // (default is Local). + const cutoverEnvLines = $derived.by((): string[] => { + if (!storage) return []; + const lines: string[] = []; + switch (storage.backend) { + case 's3': + lines.push('OXICLOUD_STORAGE_BACKEND=s3'); + if (storage.s3_endpoint_url) + lines.push(`OXICLOUD_S3_ENDPOINT_URL=${storage.s3_endpoint_url}`); + if (storage.s3_bucket) lines.push(`OXICLOUD_S3_BUCKET=${storage.s3_bucket}`); + if (storage.s3_region) lines.push(`OXICLOUD_S3_REGION=${storage.s3_region}`); + if (storage.s3_access_key_set) + lines.push('OXICLOUD_S3_ACCESS_KEY='); + if (storage.s3_secret_key_set) + lines.push('OXICLOUD_S3_SECRET_KEY='); + if (storage.s3_force_path_style) lines.push('OXICLOUD_S3_FORCE_PATH_STYLE=true'); + break; + case 'local': + lines.push('OXICLOUD_STORAGE_BACKEND=local'); + break; + // Azure not yet exposed in the admin form; add here when it is. + } + return lines; + }); + + let cutoverCopied = $state(false); + async function copyCutoverEnv() { + try { + await navigator.clipboard.writeText(cutoverEnvLines.join('\n')); + cutoverCopied = true; + setTimeout(() => (cutoverCopied = false), 2000); + } catch { + // Clipboard permission denied — silent; the block is + // selectable so the operator can copy manually. + } + } + // Migration integrity verification (separate result panel). let verifyResult = $state(null); let verifyError = $state(null); @@ -2036,17 +2114,20 @@ data-testid="admin-storage-save-btn" disabled={storageBusy}>{t('common.save', 'Save')} - {#if sForm.backend === 's3'} - - {/if} + +
@@ -2120,7 +2201,10 @@ onclick={() => doMigration('resume')}>{t('admin.mig_resume', 'Resume')} {/if} - + {#if migration.status === 'completed'} {/if} + {#if cutoverPending} + +
+

+ + {t('admin.mig_cutover_title', 'Cutover pending — update .env and restart')} +

+

+ {t( + 'admin.mig_cutover_body', + { target: storage?.backend ?? '?', live: storage?.current_backend ?? '?' }, + 'Blobs are now on {{target}} but the server is still running on {{live}}. To switch, add these lines to your .env and restart the server.' + )} +

+
{cutoverEnvLines.join('\n')}
+
+ +

+ {t( + 'admin.mig_cutover_secret_note', + 'The access key and secret key are placeholders — paste the values you entered when saving these settings. Credentials are never displayed here.' + )} +

+
+
+ {/if} + {#if verifyError}
{verifyError} @@ -4070,6 +4195,43 @@ word-break: break-all; } + .cutover-hint { + margin-top: var(--space-3); + padding: var(--space-3); + border: 1px solid var(--color-warning-border, var(--color-border)); + border-radius: var(--radius-md); + background: var(--color-warning-bg, var(--color-bg-muted)); + } + + .cutover-hint h3 { + margin: 0 0 var(--space-2) 0; + font-size: var(--text-base, 1rem); + } + + .cutover-hint__lines { + margin: var(--space-2) 0; + padding: var(--space-2) var(--space-3); + background: var(--color-bg); + border: 1px solid var(--color-border); + border-radius: var(--radius-sm); + font-size: var(--text-xs, 0.75rem); + white-space: pre; + overflow-x: auto; + } + + .cutover-hint__actions { + display: flex; + align-items: flex-start; + gap: var(--space-3); + flex-wrap: wrap; + } + + .cutover-hint__note { + flex: 1; + min-width: 12rem; + margin: 0; + } + .smtp-test { display: flex; gap: var(--space-2); diff --git a/src/application/dtos/settings_dto.rs b/src/application/dtos/settings_dto.rs index bf0d6e36..05f35b6f 100644 --- a/src/application/dtos/settings_dto.rs +++ b/src/application/dtos/settings_dto.rs @@ -191,13 +191,50 @@ pub struct TestStorageConnectionDto { pub s3_force_path_style: Option, } -/// Result of a storage connection test +/// Result of a storage connection + round-trip test. +/// +/// `connected` is TRUE when the backend was reachable (health-check +/// passed — HEAD bucket / statfs). `roundtrip_passed` is TRUE when +/// the subsequent PUT → GET → verify → DELETE cycle succeeded — it +/// validates the exact permissions the migration job needs +/// (`s3:PutObject` + `s3:GetObject` + `s3:DeleteObject` on S3, disk +/// write permission on Local). All round-trip fields are `None` when +/// reachability failed (we don't attempt the round-trip if we can't +/// even HEAD the bucket). +/// +/// `phase_reached` names the last step that succeeded — on +/// `roundtrip_passed = false` it pinpoints where the failure hit +/// (`put_ok` → wrote but couldn't confirm; `exists_ok` → wrote + +/// confirmed but GET failed; etc.). `cleanup_ok = false` means the +/// backend was readable + writable but the test object may be +/// orphaned on it (~100 B, content-addressed — harmless, admin can +/// reap by hash). #[derive(Debug, Serialize, Deserialize)] pub struct StorageTestResultDto { pub connected: bool, pub message: String, pub backend_type: String, pub available_bytes: Option, + /// Set only when reachability passed AND a round-trip was + /// attempted. `Some(true)` = full write + read + verify success; + /// `Some(false)` = reachability OK, round-trip failed at + /// `phase_reached`; `None` = round-trip not attempted (typically + /// because reachability failed). + #[serde(skip_serializing_if = "Option::is_none")] + pub roundtrip_passed: Option, + /// Last round-trip phase completed successfully — one of + /// `initialize`, `put_ok`, `exists_ok`, `get_ok`, `verify_ok`, + /// `cleanup_ok`. `None` when round-trip wasn't attempted. + #[serde(skip_serializing_if = "Option::is_none")] + pub phase_reached: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub bytes_written: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub bytes_read: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub roundtrip_elapsed_ms: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub cleanup_ok: Option, } // ============================================================================ diff --git a/src/application/services/storage_settings_service.rs b/src/application/services/storage_settings_service.rs index 1d9fbf94..d33322cf 100644 --- a/src/application/services/storage_settings_service.rs +++ b/src/application/services/storage_settings_service.rs @@ -6,11 +6,13 @@ use crate::application::dtos::settings_dto::{ SaveStorageSettingsDto, StorageSettingsDto, StorageTestResultDto, TestStorageConnectionDto, }; use crate::application::ports::blob_storage_ports::BlobStorageBackend; -use crate::common::config::{S3StorageConfig, StorageConfig}; +use crate::common::config::{S3StorageConfig, StorageBackendType, StorageConfig}; use crate::common::errors::{DomainError, ErrorKind}; use crate::domain::repositories::settings_repository::SettingsRepository; use crate::infrastructure::repositories::pg::SettingsPgRepository; +use crate::infrastructure::services::azure_blob_backend::AzureBlobBackend; use crate::infrastructure::services::dedup_service::DedupService; +use crate::infrastructure::services::local_blob_backend::LocalBlobBackend; use crate::infrastructure::services::s3_blob_backend::S3BlobBackend; /// Storage settings service — manages storage backend configuration via the admin panel. @@ -92,6 +94,95 @@ impl StorageSettingsService { } } + /// Physical-storage identity string. Two configs that yield the + /// same `storage_identity` point at the same physical location + /// (same disk directory, same S3 bucket, same Azure container) — + /// used by [`Self::is_source_target_identical`] to detect a no-op + /// migration where source and target are the same backend. + /// + /// Credentials are deliberately excluded: two configs with + /// different access keys pointing at the same bucket ARE the same + /// storage; a migration between them would be a wasted walk. The + /// same principle applies to fields that don't influence which + /// bytes get read/written (chunk sizes, retention days, etc.). + fn storage_identity(config: &StorageConfig) -> String { + match config.backend { + StorageBackendType::Local => format!("local:{}", config.root_dir), + StorageBackendType::S3 => match config.s3.as_ref() { + Some(s3) => format!( + "s3:{}/{}:path_style={}", + s3.endpoint_url.as_deref().unwrap_or("aws"), + s3.bucket, + s3.force_path_style, + ), + None => "s3:".to_string(), + }, + StorageBackendType::Azure => match config.azure.as_ref() { + Some(az) => format!( + "azure:{}/{}", + az.account_name.as_str(), + az.container.as_str(), + ), + None => "azure:".to_string(), + }, + } + } + + /// True iff the *effective* storage config points at the same + /// physical location as the *boot* config — i.e. the migration + /// would be a no-op that walks every blob and skips them all. + /// + /// The migration handler calls this at run start and refuses with + /// `RunOutcome::Failed` if it's true — otherwise a misclick on an + /// S3 deployment would issue one `HEAD` per blob for zero useful + /// work (and real cost). "Legitimate" same-type migrations (e.g. + /// `local:/data` → `local:/newdisk`, or S3 bucket A → S3 bucket B) + /// return false and proceed normally. + pub async fn is_source_target_identical(&self) -> Result { + let effective = self.load_effective_storage_config().await?; + Ok(Self::storage_identity(&self.env_storage_config) == Self::storage_identity(&effective)) + } + + /// Build a `BlobStorageBackend` matching the current *effective* + /// storage config (DB + env-var overrides + defaults). + /// + /// Distinct from `dedup_service.backend()`, which is the LIVE + /// backend the app booted with — this method reflects what the + /// admin has configured *now* and typically resolves to a + /// different backend during a migration (source = live, target = + /// effective). The returned handle is a fresh instance; the caller + /// must `.initialize()` it before first use. + pub async fn build_effective_backend( + &self, + ) -> Result, DomainError> { + let effective = self.load_effective_storage_config().await?; + match effective.backend { + StorageBackendType::Local => Ok(Arc::new(LocalBlobBackend::new(std::path::Path::new( + &effective.root_dir, + )))), + StorageBackendType::S3 => { + let s3 = effective.s3.as_ref().ok_or_else(|| { + DomainError::new( + ErrorKind::InvalidInput, + "Storage", + "S3 backend selected but no S3 configuration is present", + ) + })?; + Ok(Arc::new(S3BlobBackend::new(s3))) + } + StorageBackendType::Azure => { + let az = effective.azure.as_ref().ok_or_else(|| { + DomainError::new( + ErrorKind::InvalidInput, + "Storage", + "Azure backend selected but no Azure configuration is present", + ) + })?; + Ok(Arc::new(AzureBlobBackend::new(az))) + } + } + } + /// Load effective storage config: DB settings + env var overrides + defaults. pub async fn load_effective_storage_config(&self) -> Result { let db: HashMap = self.settings_repo.get_by_category("storage").await?; @@ -245,21 +336,44 @@ impl StorageSettingsService { Ok(()) } - /// Test a storage connection by building a temporary backend and calling health_check(). + /// Test a storage backend: reachability (health-check) followed by + /// a full read/write round-trip (see [`run_backend_roundtrip`]). + /// + /// Two-phase so the operator gets clean diagnosis: if the health- + /// check fails they know it's an auth/endpoint/bucket problem + /// (never even wrote a byte). If it passes but the round-trip + /// fails, they know reachability is fine and the permissions are + /// the gap. Round-trip fields on the result are `None` when we + /// didn't attempt it (health-check failed early). pub async fn test_storage_connection( &self, dto: TestStorageConnectionDto, ) -> Result { match dto.backend.as_str() { "local" => { - // Test local backend health via the current dedup service backend - let status = self.dedup_service.backend().health_check().await?; - Ok(StorageTestResultDto { + // Local: no per-DTO override for the root_dir (the + // form has no such field), so we test the live + // backend the app is running on. health_check() reports + // available_bytes via statfs; the round-trip validates + // disk write + read + delete permissions. + let backend = self.dedup_service.backend().clone(); + let status = backend.health_check().await?; + let mut out = StorageTestResultDto { connected: status.connected, message: status.message, backend_type: "local".to_string(), available_bytes: status.available_bytes, - }) + roundtrip_passed: None, + phase_reached: None, + bytes_written: None, + bytes_read: None, + roundtrip_elapsed_ms: None, + cleanup_ok: None, + }; + if out.connected { + attach_roundtrip(&mut out, backend.as_ref()).await; + } + Ok(out) } "s3" => { let bucket = dto.s3_bucket.as_deref().unwrap_or_default(); @@ -269,11 +383,19 @@ impl StorageSettingsService { message: "S3 bucket name is required".to_string(), backend_type: "s3".to_string(), available_bytes: None, + roundtrip_passed: None, + phase_reached: None, + bytes_written: None, + bytes_read: None, + roundtrip_elapsed_ms: None, + cleanup_ok: None, }); } // Build a temporary S3 backend from the DTO values, - // falling back to existing DB/env config for missing fields. + // falling back to existing DB/env config for missing + // fields — lets the admin test values entered but not + // yet saved (matches the current UX). let effective = self.load_effective_storage_config().await.ok(); let existing_s3 = effective.as_ref().and_then(|c| c.s3.as_ref()); @@ -312,20 +434,36 @@ impl StorageSettingsService { }; let backend = S3BlobBackend::new(&config); - match backend.health_check().await { - Ok(status) => Ok(StorageTestResultDto { + let mut out = match backend.health_check().await { + Ok(status) => StorageTestResultDto { connected: status.connected, message: status.message, backend_type: "s3".to_string(), available_bytes: status.available_bytes, - }), - Err(e) => Ok(StorageTestResultDto { + roundtrip_passed: None, + phase_reached: None, + bytes_written: None, + bytes_read: None, + roundtrip_elapsed_ms: None, + cleanup_ok: None, + }, + Err(e) => StorageTestResultDto { connected: false, message: format!("Connection failed: {}", e), backend_type: "s3".to_string(), available_bytes: None, - }), + roundtrip_passed: None, + phase_reached: None, + bytes_written: None, + bytes_read: None, + roundtrip_elapsed_ms: None, + cleanup_ok: None, + }, + }; + if out.connected { + attach_roundtrip(&mut out, &backend).await; } + Ok(out) } other => Err(DomainError::new( ErrorKind::InvalidInput, @@ -335,3 +473,197 @@ impl StorageSettingsService { } } } + +/// Populate the round-trip fields of `out` by executing a full +/// PUT → EXISTS → GET → VERIFY → DELETE cycle against `backend`. Only +/// called when reachability (`out.connected`) already passed — +/// keeping "wasn't even reachable" and "reachable but round-trip +/// failed" as distinct diagnoses. +/// +/// On round-trip failure, `out.message` is REPLACED with the round- +/// trip diagnosis (the pre-round-trip message was just "connection +/// succeeded" — round-trip failure supersedes it). On round-trip +/// success, `out.message` is REPLACED with the success confirmation +/// so the admin sees the strong claim, not the weaker "reachable". +async fn attach_roundtrip(out: &mut StorageTestResultDto, backend: &dyn BlobStorageBackend) { + let (passed, phase, written, read, elapsed_ms, cleanup_ok, message) = + run_backend_roundtrip(backend).await; + out.roundtrip_passed = Some(passed); + out.phase_reached = Some(phase); + out.bytes_written = Some(written); + out.bytes_read = Some(read); + out.roundtrip_elapsed_ms = Some(elapsed_ms); + out.cleanup_ok = Some(cleanup_ok); + if !passed { + // Reachability was fine — the round-trip is the reason to + // fail this test overall. Flip `connected` to false so the + // UI shows the whole test as failed, and surface the + // round-trip diagnosis in the message. + out.connected = false; + } + out.message = message; +} + +/// Full read/write round-trip against the given backend. PUT a tiny +/// unique object, verify existence, GET it back, check BLAKE3 +/// matches, DELETE it. Validates the exact permissions the migration +/// job needs (`s3:PutObject` + `s3:GetObject` + `s3:DeleteObject` on +/// S3, disk write on Local) — a stronger check than `health_check`. +/// +/// Returns the round-trip fields for [`StorageTestResultDto`]. Errors +/// are folded into the return value (not `Err`) so callers can +/// surface `phase_reached` diagnostics inline. +async fn run_backend_roundtrip( + backend: &dyn BlobStorageBackend, +) -> ( + /* passed */ bool, + /* phase_reached */ String, + /* bytes_written */ u64, + /* bytes_read */ u64, + /* elapsed_ms */ u64, + /* cleanup_ok */ bool, + /* message */ String, +) { + use bytes::Bytes; + use futures::StreamExt; + use std::time::Instant; + + let started = Instant::now(); + + // Content-addressable — hash MUST be BLAKE3 of the payload. UUID + // + timestamp guarantees a fresh key on every test so we never + // collide with a real blob or stale test remnant. + let payload = format!( + "oxicloud-roundtrip-test-{}-{}", + uuid::Uuid::new_v4(), + chrono::Utc::now().timestamp_nanos_opt().unwrap_or(0), + ); + let payload_bytes = payload.into_bytes(); + let hash = blake3::hash(&payload_bytes).to_hex().to_string(); + let bytes_written = payload_bytes.len() as u64; + + if let Err(e) = backend + .put_blob_from_bytes(&hash, Bytes::from(payload_bytes.clone())) + .await + { + return ( + false, + "initialize".to_string(), + 0, + 0, + started.elapsed().as_millis() as u64, + false, + format!("put: {e}"), + ); + } + + match backend.blob_exists(&hash).await { + Ok(true) => {} + Ok(false) => { + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + return ( + false, + "put_ok".to_string(), + bytes_written, + 0, + started.elapsed().as_millis() as u64, + cleanup_ok, + "PUT reported success but blob_exists returned false".to_string(), + ); + } + Err(e) => { + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + return ( + false, + "put_ok".to_string(), + bytes_written, + 0, + started.elapsed().as_millis() as u64, + cleanup_ok, + format!("exists: {e}"), + ); + } + } + + let mut got: Vec = Vec::with_capacity(payload_bytes.len()); + match backend.get_blob_stream(&hash).await { + Ok(stream) => { + let mut stream = std::pin::pin!(stream); + while let Some(chunk) = stream.next().await { + match chunk { + Ok(bytes) => got.extend_from_slice(&bytes), + Err(e) => { + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + return ( + false, + "exists_ok".to_string(), + bytes_written, + got.len() as u64, + started.elapsed().as_millis() as u64, + cleanup_ok, + format!("get stream: {e}"), + ); + } + } + } + } + Err(e) => { + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + return ( + false, + "exists_ok".to_string(), + bytes_written, + 0, + started.elapsed().as_millis() as u64, + cleanup_ok, + format!("get: {e}"), + ); + } + } + let bytes_read = got.len() as u64; + + // VERIFY — recompute BLAKE3 on the received bytes. A hash + // mismatch means the backend returned different bytes than it + // stored (silent corruption in the round-trip). Extremely rare + // but the whole point of doing a byte-level test. + let got_hash = blake3::hash(&got).to_hex().to_string(); + if got_hash != hash { + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + return ( + false, + "get_ok".to_string(), + bytes_written, + bytes_read, + started.elapsed().as_millis() as u64, + cleanup_ok, + format!( + "byte mismatch: wrote {bytes_written} bytes hash {hash}, read back {bytes_read} bytes hash {got_hash}" + ), + ); + } + + // CLEANUP — failure here does NOT flip `passed`. Read/write + // validation succeeded; the backend just left an orphan test + // blob (harmless — content-addressed, ~100 B). + let cleanup_ok = backend.delete_blob(&hash).await.is_ok(); + ( + true, + if cleanup_ok { + "cleanup_ok" + } else { + "verify_ok" + } + .to_string(), + bytes_written, + bytes_read, + started.elapsed().as_millis() as u64, + cleanup_ok, + if cleanup_ok { + "Round-trip OK: write + read + delete all succeeded".to_string() + } else { + format!( + "Round-trip OK (write + read validated), but cleanup DELETE failed — one orphan test blob left at hash {hash}" + ) + }, + ) +} diff --git a/src/common/di.rs b/src/common/di.rs index be862c6f..79412ccc 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -13,7 +13,6 @@ use crate::infrastructure::db::DbPools; use crate::application::services::admin_settings_service::AdminSettingsService; use crate::application::services::auth_application_service::AuthApplicationService; use crate::application::services::storage_settings_service::StorageSettingsService; -use crate::infrastructure::services::migration_blob_backend::MigrationState; use crate::application::ports::file_ports::FileUseCaseFactory; use crate::application::services::favorites_service::FavoritesService; @@ -1862,7 +1861,6 @@ impl AppServiceFactory { admin_settings_service: None, storage_settings_service: None, plugin_management, - migration_state: Arc::new(tokio::sync::RwLock::new(MigrationState::default())), trash_service, share_service, share_browse_service, @@ -2082,9 +2080,33 @@ impl AppServiceFactory { self.config.storage.clone(), app_state.core.dedup_service.clone(), )); - app_state.storage_settings_service = Some(storage_settings_svc); + app_state.storage_settings_service = Some(storage_settings_svc.clone()); tracing::info!("Storage settings service initialized"); + // 9b-1c. Register the storage-backend migration tenant on + // the recoverable-run engine. Must run AFTER the storage + // settings service is built — the tenant resolves the + // *target* backend at each run start by asking the settings + // service for the currently-effective config. Source is + // whatever `dedup_service` booted with; both live on + // `AppState.core`. On-demand only (no periodic tick — an + // operator triggers a copy after switching backend config). + let job_store_provider_dyn: Arc< + dyn crate::infrastructure::scheduler::JobStoreProvider, + > = app_state.core.job_store_provider.clone(); + let _ = Arc::new( + crate::infrastructure::services::storage_migration_service::StorageMigrationService::new( + app_state + .maintenance_pool + .clone() + .expect("maintenance_pool set above"), + app_state.core.blob_backend.clone(), + storage_settings_svc, + ), + ) + .register_recoverable_job(&app_state.core.job_registry, &job_store_provider_dyn) + .await; + // 9b-2. Log whether system needs first-time admin setup if !admin_svc.is_system_initialized().await { tracing::warn!("╔══════════════════════════════════════════════════════════╗"); @@ -2419,7 +2441,6 @@ pub struct AppState { pub plugin_management: Option>, pub storage_settings_service: Option>, - pub migration_state: Arc>, pub trash_service: Option>, pub share_service: Option>, pub share_browse_service: Option>, diff --git a/src/infrastructure/services/migration_blob_backend.rs b/src/infrastructure/services/migration_blob_backend.rs deleted file mode 100644 index dbe32231..00000000 --- a/src/infrastructure/services/migration_blob_backend.rs +++ /dev/null @@ -1,273 +0,0 @@ -//! `MigrationBlobBackend` — decorator that enables zero-downtime migration -//! between blob storage backends. -//! -//! During a migration the decorator writes to the **target** backend and reads -//! from **target-first-then-source** (dual-read). A background job -//! (see `migration_job.rs`) copies remaining blobs in the background. - -use std::future::Future; -use std::path::{Path, PathBuf}; -use std::pin::Pin; -use std::sync::Arc; - -use bytes::Bytes; -use chrono::{DateTime, Utc}; -use serde::Serialize; -use tokio::sync::RwLock; - -use crate::application::ports::blob_storage_ports::{ - BlobStorageBackend, BlobStream, StorageHealthStatus, -}; -use crate::common::errors::DomainError; - -// ── Migration state ──────────────────────────────────────────────── - -/// Progress of an ongoing (or completed) backend migration. -#[derive(Debug, Clone, Serialize)] -pub struct MigrationState { - pub status: MigrationStatus, - pub total_blobs: u64, - pub migrated_blobs: u64, - pub migrated_bytes: u64, - pub failed_blobs: Vec, - pub started_at: Option>, - pub completed_at: Option>, -} - -impl Default for MigrationState { - fn default() -> Self { - Self { - status: MigrationStatus::Idle, - total_blobs: 0, - migrated_blobs: 0, - migrated_bytes: 0, - failed_blobs: Vec::new(), - started_at: None, - completed_at: None, - } - } -} - -/// Status of the migration job. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] -#[serde(rename_all = "snake_case")] -pub enum MigrationStatus { - Idle, - Running, - Paused, - Completed, - Failed, -} - -// ── MigrationBlobBackend ─────────────────────────────────────────── - -/// A `BlobStorageBackend` decorator that proxies requests to a *source* -/// (old) and *target* (new) backend, enabling live migration. -pub struct MigrationBlobBackend { - source: Arc, - target: Arc, - state: Arc>, -} - -impl MigrationBlobBackend { - pub fn new( - source: Arc, - target: Arc, - state: Arc>, - ) -> Self { - Self { - source, - target, - state, - } - } - - pub fn state(&self) -> &Arc> { - &self.state - } - - pub fn source(&self) -> &Arc { - &self.source - } - - pub fn target(&self) -> &Arc { - &self.target - } -} - -/// Boxed future alias (same as in the trait module). -type BoxFut<'a, T> = Pin + Send + 'a>>; - -impl BlobStorageBackend for MigrationBlobBackend { - fn initialize(&self) -> BoxFut<'_, Result<(), DomainError>> { - Box::pin(async move { - self.target.initialize().await?; - // Source is already initialised; call anyway for idempotency. - self.source.initialize().await?; - Ok(()) - }) - } - - /// Writes go to **target** only. - fn put_blob(&self, hash: &str, source_path: &Path) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - let path = source_path.to_path_buf(); - Box::pin(async move { self.target.put_blob(&hash, &path).await }) - } - - /// Writes bytes to **target** only. - fn put_blob_from_bytes(&self, hash: &str, data: Bytes) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { self.target.put_blob_from_bytes(&hash, data).await }) - } - - /// Unsynced writes go to **target** only (same as the synced variant). - fn put_blob_from_bytes_unsynced( - &self, - hash: &str, - data: Bytes, - ) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { self.target.put_blob_from_bytes_unsynced(&hash, data).await }) - } - - /// Durability sweep goes to **target**, where unsynced writes land. - fn sync_blobs(&self, hashes: &[String]) -> BoxFut<'_, Result<(), DomainError>> { - self.target.sync_blobs(hashes) - } - - /// Read from target first; fall back to source. - fn get_blob_stream(&self, hash: &str) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { - match self.target.get_blob_stream(&hash).await { - Ok(stream) => Ok(stream), - Err(_) => self.source.get_blob_stream(&hash).await, - } - }) - } - - fn get_blob_range_stream( - &self, - hash: &str, - start: u64, - end: Option, - ) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { - match self.target.get_blob_range_stream(&hash, start, end).await { - Ok(stream) => Ok(stream), - Err(_) => self.source.get_blob_range_stream(&hash, start, end).await, - } - }) - } - - /// Delete from **both** backends (best-effort on source). - fn delete_blob(&self, hash: &str) -> BoxFut<'_, Result<(), DomainError>> { - let hash = hash.to_string(); - Box::pin(async move { - self.target.delete_blob(&hash).await?; - // Best-effort on source — ignore errors (blob may already be gone). - let _ = self.source.delete_blob(&hash).await; - Ok(()) - }) - } - - /// Exists in either backend. - fn blob_exists(&self, hash: &str) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { - if self.target.blob_exists(&hash).await? { - return Ok(true); - } - self.source.blob_exists(&hash).await - }) - } - - fn blob_size(&self, hash: &str) -> BoxFut<'_, Result> { - let hash = hash.to_string(); - Box::pin(async move { - match self.target.blob_size(&hash).await { - Ok(sz) => Ok(sz), - Err(_) => self.source.blob_size(&hash).await, - } - }) - } - - fn health_check(&self) -> BoxFut<'_, Result> { - Box::pin(async move { - let target_health = self.target.health_check().await?; - let source_health = self.source.health_check().await?; - Ok(StorageHealthStatus { - connected: target_health.connected && source_health.connected, - backend_type: format!( - "migration({} → {})", - source_health.backend_type, target_health.backend_type - ), - message: format!( - "Source: {} | Target: {}", - source_health.message, target_health.message - ), - available_bytes: target_health.available_bytes, - }) - }) - } - - fn backend_type(&self) -> &'static str { - "migration" - } - - /// Reads are served target-first (see `get_blob_stream`), so adopt the - /// target's read-ahead. - fn read_prefetch(&self) -> usize { - self.target.read_prefetch() - } - - fn local_blob_path(&self, hash: &str) -> Option { - // Prefer target, fall back to source. - self.target - .local_blob_path(hash) - .or_else(|| self.source.local_blob_path(hash)) - } - - /// Enumeration during migration is intentionally REFUSED. Both - /// source and target legitimately hold bytes concurrently - /// mid-migration: a blob copied to target but not yet deleted - /// from source would be reported "twice"; a blob in-flight from - /// source to target could be flagged as orphan on whichever - /// side the consistency scan doesn't walk. There's no single - /// authoritative "what's on the backend" answer while a - /// migration is running. - /// - /// Operators wanting to run `backend_consistency` during a - /// migration should either wait for the migration to complete - /// (target becomes authoritative) or cancel it. The - /// `operation_not_supported` error is surfaced by the tenant as - /// a single run-level `backend_unenumerable` finding — no - /// per-blob probes attempted. - fn list_blob_hashes( - &self, - _cursor: Option, - _limit: usize, - ) -> Pin< - Box< - dyn std::future::Future< - Output = Result< - crate::application::ports::blob_storage_ports::BlobListPage, - DomainError, - >, - > + Send - + '_, - >, - > { - Box::pin(async { - Err(DomainError::operation_not_supported( - "list_blob_hashes", - "backend_consistency cannot enumerate while a storage \ - migration is in progress — source and target hold bytes \ - concurrently; wait for migration completion or cancel it \ - before running the scan", - )) - }) - } -} diff --git a/src/infrastructure/services/migration_job.rs b/src/infrastructure/services/migration_job.rs deleted file mode 100644 index c2cd3673..00000000 --- a/src/infrastructure/services/migration_job.rs +++ /dev/null @@ -1,240 +0,0 @@ -//! Background migration job — copies blobs from a source backend to a target -//! backend with configurable concurrency and progress tracking. - -use std::sync::Arc; - -use futures::StreamExt; -use serde::Serialize; -use sqlx::PgPool; -use tokio::sync::RwLock; - -use crate::application::ports::blob_storage_ports::BlobStorageBackend; -use crate::common::errors::DomainError; -use crate::infrastructure::services::migration_blob_backend::{MigrationState, MigrationStatus}; - -/// Run the migration: stream all blob hashes from `storage.blobs` and copy -/// each one from `source` to `target`. -/// -/// * The job respects `Paused` / `Failed` status in `state` — it will stop -/// streaming when the status is no longer `Running`. -/// * Errors on individual blobs are logged and collected in `failed_blobs` -/// but do **not** abort the full run. -/// * `concurrency` controls `buffer_unordered` parallelism (default: 4). -pub async fn run_migration( - source: Arc, - target: Arc, - pool: Arc, - state: Arc>, - concurrency: usize, -) -> Result<(), DomainError> { - // Count total blobs for progress tracking. - let total: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM storage.blobs") - .fetch_one(pool.as_ref()) - .await - .unwrap_or(0); - - { - let mut s = state.write().await; - s.status = MigrationStatus::Running; - s.total_blobs = total as u64; - s.migrated_blobs = 0; - s.migrated_bytes = 0; - s.failed_blobs.clear(); - s.started_at = Some(chrono::Utc::now()); - s.completed_at = None; - } - - // Stream all hashes+sizes with a cursor. - let mut rows = - sqlx::query_as::<_, (String, i64)>("SELECT hash, size FROM storage.blobs ORDER BY hash") - .fetch(pool.as_ref()); - - // Collect all hashes first to avoid holding the cursor across awaits. - let mut work: Vec<(String, i64)> = Vec::with_capacity(total as usize); - while let Some(row) = rows.next().await { - match row { - Ok(r) => work.push(r), - Err(e) => { - tracing::warn!("Error fetching blob row during migration: {}", e); - } - } - } - - // Process in parallel chunks. - let results = futures::stream::iter(work.into_iter().map(|(hash, size)| { - let src = source.clone(); - let tgt = target.clone(); - let st = state.clone(); - async move { - // Check if we should keep running. - { - let s = st.read().await; - if s.status != MigrationStatus::Running { - return; - } - } - - // Skip if already in target. - match tgt.blob_exists(&hash).await { - Ok(true) => { - let mut s = st.write().await; - s.migrated_blobs += 1; - s.migrated_bytes += size as u64; - return; - } - Ok(false) => {} - Err(e) => { - tracing::warn!("blob_exists check failed for {}: {}", hash, e); - } - } - - // Copy: stream from source → temp file → put into target. - if let Err(e) = copy_blob(&src, &tgt, &hash).await { - tracing::warn!("Failed to migrate blob {}: {}", hash, e); - let mut s = st.write().await; - s.failed_blobs.push(hash); - return; - } - - let mut s = st.write().await; - s.migrated_blobs += 1; - s.migrated_bytes += size as u64; - } - })) - .buffer_unordered(concurrency) - .collect::>() - .await; - - drop(results); - - // Finalize state. - let mut s = state.write().await; - if s.status == MigrationStatus::Running { - if s.failed_blobs.is_empty() { - s.status = MigrationStatus::Completed; - } else { - s.status = MigrationStatus::Failed; - } - s.completed_at = Some(chrono::Utc::now()); - } - - tracing::info!( - "Migration finished: {}/{} blobs, {} failures", - s.migrated_blobs, - s.total_blobs, - s.failed_blobs.len() - ); - - Ok(()) -} - -/// Copy a single blob: stream from source → spool to temp file → put_blob into target. -async fn copy_blob( - source: &Arc, - target: &Arc, - hash: &str, -) -> Result<(), DomainError> { - use tokio::io::AsyncWriteExt; - - // Create a temp file to spool content. - let tmp_dir = std::env::temp_dir().join("oxicloud-migration"); - tokio::fs::create_dir_all(&tmp_dir).await.map_err(|e| { - DomainError::internal_error("Migration", format!("Failed to create temp dir: {}", e)) - })?; - - let tmp_path = tmp_dir.join(format!("{}.tmp", hash)); - - // Stream from source. - let stream = source.get_blob_stream(hash).await?; - - // Write to temp file. - let mut file = tokio::fs::File::create(&tmp_path).await.map_err(|e| { - DomainError::internal_error("Migration", format!("Failed to create temp file: {}", e)) - })?; - - let mut stream = std::pin::pin!(stream); - while let Some(chunk) = stream.next().await { - let bytes = chunk.map_err(|e| { - DomainError::internal_error("Migration", format!("Stream error: {}", e)) - })?; - file.write_all(&bytes) - .await - .map_err(|e| DomainError::internal_error("Migration", format!("Write error: {}", e)))?; - } - file.flush() - .await - .map_err(|e| DomainError::internal_error("Migration", format!("Flush error: {}", e)))?; - drop(file); - - // Put into target. - target.put_blob(hash, &tmp_path).await?; - - // Clean up temp file. - let _ = tokio::fs::remove_file(&tmp_path).await; - - Ok(()) -} - -/// Verify migration integrity by comparing blob counts and sampling random hashes. -pub async fn verify_migration( - target: Arc, - pool: Arc, - sample_size: usize, -) -> Result { - // 1. Count blobs in PG. - let pg_count: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM storage.blobs") - .fetch_one(pool.as_ref()) - .await - .unwrap_or(0); - - // 2. Verify sample of blobs exist in target. - let sample_rows: Vec<(String, i64)> = - sqlx::query_as("SELECT hash, size FROM storage.blobs ORDER BY random() LIMIT $1") - .bind(sample_size as i64) - .fetch_all(pool.as_ref()) - .await - .map_err(|e| { - DomainError::internal_error("Migration", format!("Sample query failed: {}", e)) - })?; - - let mut missing = Vec::new(); - let mut size_mismatches = Vec::new(); - - for (hash, expected_size) in &sample_rows { - match target.blob_exists(hash).await { - Ok(false) => missing.push(hash.clone()), - Err(e) => { - tracing::warn!("blob_exists failed for {}: {}", hash, e); - missing.push(hash.clone()); - } - Ok(true) => { - // Verify size matches. - if let Ok(actual_size) = target.blob_size(hash).await - && actual_size != *expected_size as u64 - { - size_mismatches.push(hash.clone()); - } - } - } - } - - let passed = missing.is_empty() && size_mismatches.is_empty(); - - Ok(VerificationResult { - pg_blob_count: pg_count as u64, - sample_checked: sample_rows.len() as u64, - missing_in_target: missing, - size_mismatches, - passed, - }) -} - -/// Result of a post-migration integrity check. -#[derive(Debug, Clone, Serialize, serde::Deserialize)] -pub struct VerificationResult { - pub pg_blob_count: u64, - pub sample_checked: u64, - pub missing_in_target: Vec, - pub size_mismatches: Vec, - pub passed: bool, -} diff --git a/src/infrastructure/services/mod.rs b/src/infrastructure/services/mod.rs index 457e0ef1..1565e65f 100644 --- a/src/infrastructure/services/mod.rs +++ b/src/infrastructure/services/mod.rs @@ -25,8 +25,6 @@ pub mod local_blob_backend; pub mod local_fs_mount_provider; pub mod login_lockout_service; pub mod media_metadata_service; -pub mod migration_blob_backend; -pub mod migration_job; pub mod mock_email_sender; pub mod mount_provider_factory; pub mod nextcloud_chunked_upload_service; @@ -46,6 +44,7 @@ pub mod s3_blob_backend; pub mod search_index; pub mod share_unlock_cookie; pub mod smtp_email_sender; +pub mod storage_migration_service; pub mod thumbnail_service; #[cfg(test)] mod thumbnail_service_test; diff --git a/src/infrastructure/services/storage_migration_service.rs b/src/infrastructure/services/storage_migration_service.rs new file mode 100644 index 00000000..4f8b0979 --- /dev/null +++ b/src/infrastructure/services/storage_migration_service.rs @@ -0,0 +1,486 @@ +//! Storage-backend migration as a recoverable-run tenant (Part 2 engine). +//! +//! Iterates `storage.blobs` and copies each byte payload from the SOURCE +//! backend (whatever the app booted with) to the TARGET backend +//! (whatever the current admin storage-settings config describes). +//! Both legacy whole-file blobs AND CDC chunk blobs are covered by the +//! single walk — they share `storage.blobs` as their physical registry +//! (see memory `project_cdc_dual_storage_registries`). +//! `storage.chunk_manifests` is pure PG state, holds no backend bytes, +//! and needs no migration. +//! +//! Retires the in-memory `Arc>` + one-shot +//! `tokio::spawn` in `migration_job.rs`. The recoverable engine +//! provides cursor persistence, cooperative cancel, boot-time crash +//! recovery, and the uniform `/api/admin/jobs/*` admin surface. +//! +//! ### Restart survival +//! +//! Cursor + per-blob failure findings are persisted after every batch. +//! On restart the boot-time sweep flips any abandoned `Running` row to +//! `Paused`; a subsequent admin trigger resumes from the persisted +//! cursor via `run_or_resume`. At most one batch of already-copied +//! blobs replays, and the `target.blob_exists` short-circuit makes +//! even that replay effectively free. +//! +//! ### Design notes +//! +//! * **Cursor.** UTF-8 hex of the last-processed blob hash (64 chars). +//! Natural lex order matches `ORDER BY hash ASC`. Same encoding +//! `blobs_consistency` uses. +//! * **Target resolution.** Rebuilt at the START of every fresh or +//! resumed run via `StorageSettingsService::build_effective_backend`. +//! Held for the duration of the run; a mid-run settings change is +//! ignored until the next run. On resume the admin may have paused +//! *specifically* to fix a broken target config, so we re-derive +//! rather than pin. +//! * **Per-blob failures don't fail the run.** Each failure records a +//! `migration_failed` finding (severity `data_loss` — the bytes +//! didn't cross) and the walk continues. A run that completes with +//! zero findings is proof the target has every blob. +//! * **Skip already-present blobs.** `target.blob_exists(hash)` before +//! the copy — makes cheap re-runs safe and lets a paused run resume +//! without redoing bytes. +//! * **No per-batch concurrency knob.** The old code buffered N copies +//! in parallel. Sequential is easier to reason about with cooperative +//! cancel + cursor discipline; the batch loop is I/O-bound anyway. +//! Add concurrency later if a real throughput need appears. + +use std::path::Path; +use std::sync::Arc; + +use async_trait::async_trait; +use futures::StreamExt; +use sqlx::PgPool; + +use crate::application::ports::blob_storage_ports::BlobStorageBackend; +use crate::application::services::storage_settings_service::StorageSettingsService; +use crate::common::errors::DomainError; +use crate::infrastructure::scheduler::{ + JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome, + RunStatus, record_or_log, +}; + +pub const STORAGE_MIGRATION_JOB_NAME: &str = "storage_migration"; + +/// Rows per batch. Copies are I/O-bound (source read + target write); +/// larger batches amortise fewer SQL round-trips but the checkpoint +/// / cancel-poll cadence lengthens. 100 balances the two — one +/// checkpoint per ~hundred blobs is fine, and the cancel-poll comes +/// every 100 rows too. Match `blobs_consistency` for consistency. +const BATCH_SIZE: i64 = 100; + +pub struct StorageMigrationService { + pool: Arc, + source: Arc, + storage_settings: Arc, +} + +impl StorageMigrationService { + pub fn new( + pool: Arc, + source: Arc, + storage_settings: Arc, + ) -> Self { + Self { + pool, + source, + storage_settings, + } + } + + /// Chainable self-registration — mirrors the `*_consistency` + /// tenants. On-demand only (no periodic tick). + pub async fn register_recoverable_job( + self: Arc, + registry: &JobRegistry, + provider: &Arc, + ) -> Arc { + registry + .register_recoverable_job(self.clone(), provider.clone(), None) + .await; + self + } +} + +#[async_trait] +impl RecoverableJobHandler for StorageMigrationService { + fn name(&self) -> &str { + STORAGE_MIGRATION_JOB_NAME + } + + /// Definitive count — one row per blob. `SELECT COUNT(*) FROM + /// storage.blobs` on a modern PG is a sub-second index-only scan + /// even at millions of rows. + async fn count_total(&self) -> Option { + let row: Result<(i64,), sqlx::Error> = sqlx::query_as("SELECT COUNT(*) FROM storage.blobs") + .fetch_one(self.pool.as_ref()) + .await; + match row { + Ok((n,)) => Some(n.max(0) as u64), + Err(e) => { + tracing::debug!( + target: "oxicloud::migration", + event = "storage_migration.count_total_failed", + error = %e, + "count_total failed — run will not surface a progress bar" + ); + None + } + } + } + + async fn run_resumable( + &self, + store: &dyn JobStore, + _args: &JobRunArgs, + resume_cursor: Option>, + ) -> RunOutcome { + // No-op guard — refuse when the effective (target) config + // points at the same physical storage as the source (boot + // config). Without this, a misclick on an S3 deployment + // issues one HEAD per blob for zero useful work — cheap on + // local, expensive on remote. Same-type-different-location + // migrations (local dir change, S3 bucket change) pass this + // check and proceed normally. + match self.storage_settings.is_source_target_identical().await { + Ok(true) => { + tracing::warn!( + target: "audit", + event = "storage_migration.refused_noop", + run_id = %store.run_id(), + "storage_migration refused: source and target point at the same storage" + ); + return RunOutcome::Failed { + message: + "target equals source; change storage settings before triggering a migration" + .to_string(), + }; + } + Ok(false) => {} + Err(e) => { + return RunOutcome::Failed { + message: format!("identity check: {e}"), + }; + } + } + + // Resolve target at run start. + let target = match self.storage_settings.build_effective_backend().await { + Ok(t) => t, + Err(e) => { + return RunOutcome::Failed { + message: format!("resolve target backend: {e}"), + }; + } + }; + if let Err(e) = target.initialize().await { + return RunOutcome::Failed { + message: format!("target backend init: {e}"), + }; + } + + let source_kind = self.source.backend_type(); + let target_kind = target.backend_type(); + tracing::info!( + target: "audit", + event = "storage_migration.run_started", + run_id = %store.run_id(), + source = source_kind, + target = target_kind, + resuming = resume_cursor.is_some(), + "storage_migration starting {source_kind} → {target_kind}" + ); + + // Cursor = the last-visited blob hash, UTF-8-encoded. On resume + // walk `WHERE hash > $cursor`. `None` / empty = start from the + // smallest hash. Same shape `blobs_consistency` uses. + let mut cursor: Option = match resume_cursor { + None => None, + Some(bytes) if bytes.is_empty() => None, + Some(bytes) => match String::from_utf8(bytes) { + Ok(s) => Some(s), + Err(e) => { + return RunOutcome::Failed { + message: format!("invalid cursor: not valid UTF-8: {e}"), + }; + } + }, + }; + + let mut copied_count = 0u64; + let mut skipped_count = 0u64; + let mut failed_count = 0u64; + let mut source_missing_count = 0u64; + + loop { + // Cooperative cancel poll between batches. + match store.status().await { + Ok(RunStatus::CancelRequested) => { + tracing::info!( + target: "oxicloud::migration", + event = "storage_migration.cancelled", + run_id = %store.run_id(), + copied = copied_count, + skipped = skipped_count, + failed = failed_count, + source_missing = source_missing_count, + "storage_migration cancelled cooperatively, pausing" + ); + return RunOutcome::Paused { + cursor: cursor + .as_ref() + .map(|s| s.as_bytes().to_vec()) + .unwrap_or_default(), + }; + } + Ok(_) => {} + Err(e) => { + return RunOutcome::Failed { + message: format!("status poll: {e}"), + }; + } + } + + // Fetch the next batch. `hash > $1` keyset pagination on + // the PK; index-only scan. + let rows: Vec<(String, i64)> = match sqlx::query_as( + r#" + SELECT hash, size + FROM storage.blobs + WHERE ($1::text IS NULL OR hash > $1) + ORDER BY hash + LIMIT $2 + "#, + ) + .bind(cursor.as_deref()) + .bind(BATCH_SIZE) + .fetch_all(self.pool.as_ref()) + .await + { + Ok(r) => r, + Err(e) => { + return RunOutcome::Failed { + message: format!("batch fetch: {e}"), + }; + } + }; + + if rows.is_empty() { + tracing::info!( + target: "oxicloud::migration", + event = "storage_migration.completed", + run_id = %store.run_id(), + copied = copied_count, + skipped = skipped_count, + failed = failed_count, + source_missing = source_missing_count, + "storage_migration completed" + ); + return RunOutcome::Completed; + } + + for (hash, size) in &rows { + // Probe SOURCE first — without this a run would + // silently "succeed" against a source that's missing + // blobs the DB expects, and the audit intent of the + // walk is lost (relevant on any post-migration state + // where target may already have every blob). A + // missing-on-source blob is a real data-loss + // condition; record it and move on — we never + // "copy" from nothing. + match self.source.blob_exists(hash).await { + Ok(true) => {} + Ok(false) => { + source_missing_count += 1; + tracing::warn!( + target: "oxicloud::migration", + event = "storage_migration.source_missing", + run_id = %store.run_id(), + hash = %hash, + source = source_kind, + "blob absent from source; recording data-loss finding, no copy" + ); + record_or_log( + store, + STORAGE_MIGRATION_JOB_NAME, + "source_missing", + "data_loss", + None, + serde_json::json!({ + "hash": hash, + "size": size, + "source": source_kind, + "target": target_kind, + }), + ) + .await; + continue; + } + Err(e) => { + // Transient probe failure on source is NOT a + // finding — treat like a network blip. + // Skipping this row on this run; a re-run + // will re-probe. If the failure is + // persistent, `blobs_consistency` catches + // it. + tracing::warn!( + target: "oxicloud::migration", + event = "storage_migration.source_probe_error", + run_id = %store.run_id(), + hash = %hash, + error = %e, + "source blob_exists probe failed; skipping this row" + ); + continue; + } + } + + // Skip when the target already has it — supports + // idempotent resume and cheap re-runs against a + // partially-migrated target. + match target.blob_exists(hash).await { + Ok(true) => { + skipped_count += 1; + continue; + } + Ok(false) => {} + Err(e) => { + tracing::warn!( + target: "oxicloud::migration", + event = "storage_migration.blob_exists_error", + run_id = %store.run_id(), + hash = %hash, + error = %e, + "blob_exists probe on target failed; attempting copy anyway" + ); + } + } + + match copy_blob(self.source.as_ref(), target.as_ref(), hash).await { + Ok(()) => { + copied_count += 1; + } + Err(e) => { + failed_count += 1; + tracing::warn!( + target: "oxicloud::migration", + event = "storage_migration.blob_failed", + run_id = %store.run_id(), + hash = %hash, + error = %e, + "failed to migrate blob; recording finding, continuing" + ); + // resource_id stays None — blob hash isn't a + // UUID. Real identifier lives in `detail.hash` + // where the admin UI reads it. + record_or_log( + store, + STORAGE_MIGRATION_JOB_NAME, + "migration_failed", + "data_loss", + None, + serde_json::json!({ + "hash": hash, + "size": size, + "source": source_kind, + "target": target_kind, + "error": e.to_string(), + }), + ) + .await; + } + } + } + + // Advance cursor + checkpoint. `delta_count` counts WORK + // ATTEMPTED (copied + skipped + failed), not successful + // copies alone — otherwise the progress bar stalls whenever + // a batch is dominated by already-present blobs, which is + // exactly the case on a resume. + let last_hash = rows.last().map(|(h, _)| h.clone()).expect("non-empty rows"); + cursor = Some(last_hash.clone()); + let batch_len = rows.len() as u64; + if let Err(e) = store.checkpoint(last_hash.into_bytes(), batch_len).await { + return RunOutcome::Failed { + message: format!("checkpoint: {e}"), + }; + } + + if (rows.len() as i64) < BATCH_SIZE { + tracing::info!( + target: "oxicloud::migration", + event = "storage_migration.completed", + run_id = %store.run_id(), + copied = copied_count, + skipped = skipped_count, + failed = failed_count, + source_missing = source_missing_count, + "storage_migration completed" + ); + return RunOutcome::Completed; + } + } + } +} + +/// Copy one blob: stream source bytes to a temp file, then hand the +/// path to `target.put_blob`. The spool-through-disk shape matches +/// what the old `migration_job::copy_blob` did — some backends' +/// `put_blob` want a path they can `rename(2)` or multi-part upload +/// from, not an in-memory buffer. The temp file lives in +/// `std::env::temp_dir()/oxicloud-migration/{hash}.tmp` and is +/// removed on success (best-effort on the failure paths — the OS +/// cleans up on reboot). +async fn copy_blob( + source: &dyn BlobStorageBackend, + target: &dyn BlobStorageBackend, + hash: &str, +) -> Result<(), DomainError> { + let tmp_dir = std::env::temp_dir().join("oxicloud-migration"); + tokio::fs::create_dir_all(&tmp_dir).await.map_err(|e| { + DomainError::internal_error( + "StorageMigration", + format!("create temp dir {}: {e}", tmp_dir.display()), + ) + })?; + let tmp_path = tmp_dir.join(format!("{hash}.tmp")); + + if let Err(e) = write_source_to_tmp(source, hash, &tmp_path).await { + let _ = tokio::fs::remove_file(&tmp_path).await; + return Err(e); + } + + let put_result = target.put_blob(hash, &tmp_path).await; + let _ = tokio::fs::remove_file(&tmp_path).await; + put_result.map(|_bytes_written| ()) +} + +async fn write_source_to_tmp( + source: &dyn BlobStorageBackend, + hash: &str, + tmp_path: &Path, +) -> Result<(), DomainError> { + use tokio::io::AsyncWriteExt; + + let stream = source.get_blob_stream(hash).await?; + let mut file = tokio::fs::File::create(tmp_path).await.map_err(|e| { + DomainError::internal_error( + "StorageMigration", + format!("create temp file {}: {e}", tmp_path.display()), + ) + })?; + let mut stream = std::pin::pin!(stream); + while let Some(chunk) = stream.next().await { + let bytes = chunk.map_err(|e| { + DomainError::internal_error("StorageMigration", format!("source stream read: {e}")) + })?; + file.write_all(&bytes).await.map_err(|e| { + DomainError::internal_error("StorageMigration", format!("temp file write: {e}")) + })?; + } + file.flush() + .await + .map_err(|e| DomainError::internal_error("StorageMigration", format!("temp flush: {e}")))?; + Ok(()) +} diff --git a/src/interfaces/api/handlers/admin_handler.rs b/src/interfaces/api/handlers/admin_handler.rs index 77f8e7f5..eb172bda 100644 --- a/src/interfaces/api/handlers/admin_handler.rs +++ b/src/interfaces/api/handlers/admin_handler.rs @@ -24,9 +24,13 @@ use crate::application::dtos::settings_dto::{ use crate::application::dtos::user_dto::{AdminUserSummaryDto, UserDto}; use crate::application::ports::authorization_ports::AuthorizationEngine; use crate::application::ports::plugin_ports::{LogQuery, PluginManagementPort, PluginMgmtError}; +// JobStoreProvider is used only by the storage-migration shims below, +// but the compiler needs the trait in scope for method resolution on +// the concrete `PgJobStoreProvider` that lives on `AppState`. use crate::common::di::AppState; use crate::domain::repositories::drive_repository::DriveRepository; use crate::domain::services::authorization::{Resource, Subject}; +use crate::infrastructure::scheduler::JobStoreProvider; use crate::interfaces::api::handlers::dedup_handler::{get_stats, recalculate_stats}; use crate::interfaces::api::handlers::search_handler::clear_search_cache; use crate::interfaces::errors::AppError; @@ -70,12 +74,17 @@ pub fn admin_routes() -> Router> { .route("/settings/storage", get(get_storage_settings)) .route("/settings/storage", put(save_storage_settings)) .route("/settings/storage/test", post(test_storage_connection)) - // Storage migration + // Storage migration — thin shims over the recoverable-run + // engine (job_name = "storage_migration"). Retained under + // /storage/migration/* until the admin UI is rewired to + // /api/admin/jobs/storage_migration/*; both paths route to + // the same underlying JobRegistry dispatch. The old /complete + // endpoint is retired — a finished run is a Completed row, + // there's nothing to acknowledge. .route("/storage/migration", get(get_migration_status)) .route("/storage/migration/start", post(start_migration)) .route("/storage/migration/pause", post(pause_migration)) .route("/storage/migration/resume", post(resume_migration)) - .route("/storage/migration/complete", post(complete_migration)) .route("/storage/migration/verify", post(verify_migration)) // Encryption key generation .route( @@ -361,7 +370,15 @@ async fn test_storage_connection( // Storage migration handlers // ───────────────────────────────────────────────────── -/// GET /api/admin/storage/migration — current migration progress +/// GET /api/admin/storage/migration — current migration progress. +/// +/// Shim over the recoverable-run engine: reads the latest +/// `storage_migration` run from `jobs.recoverable_runs` (via the +/// `JobStoreProvider`) and projects it into the legacy +/// `MigrationStateDto` shape the admin storage tab expects. When no +/// run has ever been triggered the response is an empty "idle" DTO — +/// same behaviour the old in-memory `MigrationState::default()` +/// produced. #[utoipa::path( get, path = "/api/admin/storage/migration", @@ -376,17 +393,59 @@ async fn test_storage_connection( pub async fn get_migration_status( State(state): State>, ) -> Result { - let s = state.migration_state.read().await; - Ok(Json(migration_state_to_dto(&s))) + use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME; + + let provider = state.core.job_store_provider.clone(); + let latest = provider + .list_runs(STORAGE_MIGRATION_JOB_NAME, 1) + .await + .map_err(AppError::from)? + .into_iter() + .next(); + + let Some(run) = latest else { + return Ok(Json(idle_migration_dto())); + }; + + // Failed blobs are stored as findings, kind = "migration_failed". + // Pull up to a reasonable ceiling — the DTO ships the full list, + // and the admin UI truncates its own display. + let findings = provider + .list_findings(run.id, 500, 0) + .await + .map_err(AppError::from)?; + let failed_blobs: Vec = findings + .into_iter() + .filter(|f| f.kind == "migration_failed") + .filter_map(|f| { + f.detail + .get("hash") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()) + }) + .collect(); + + Ok(Json(run_to_migration_dto(&run, failed_blobs))) } -/// POST /api/admin/storage/migration/start — begin background migration +/// POST /api/admin/storage/migration/start — begin background migration. +/// +/// Shim that forwards to `JobRegistry::trigger("storage_migration", +/// ...)`. `run_or_resume` (the RecoverableAdapter's inner dispatch) +/// resumes a Paused run or starts a fresh one — one endpoint covers +/// both. Exclusivity is enforced at the DB layer (the partial unique +/// index on `jobs.recoverable_runs`), so a second concurrent trigger +/// is a no-op that returns the existing run. +/// +/// `StartMigrationDto.concurrency` is currently ignored — the +/// recoverable copy loop runs sequentially. Kept in the DTO for +/// wire-compat with the admin UI; will be honoured if a concurrency +/// knob is added later. #[utoipa::path( post, path = "/api/admin/storage/migration/start", responses( (status = 200, description = "Migration started"), - (status = 400, description = "Migration already running"), (status = 401, description = "Unauthorized"), (status = 403, description = "Admin required") ), @@ -395,73 +454,23 @@ pub async fn get_migration_status( )] pub async fn start_migration( State(state): State>, - Json(dto): Json, + Json(_dto): Json, ) -> Result { - use crate::infrastructure::services::migration_blob_backend::MigrationStatus; - - // Check not already running. - { - let s = state.migration_state.read().await; - if s.status == MigrationStatus::Running { - return Err(AppError::bad_request("A migration is already running")); - } - } - - let pool = state - .db_pool - .clone() - .ok_or_else(|| AppError::internal_error("Database not available"))?; - - let source = state.core.dedup_service.backend().clone(); - let svc = state - .storage_settings_service - .as_ref() - .ok_or_else(|| AppError::internal_error("Storage settings service not available"))?; - - // Build target backend from saved settings. - let effective = svc - .load_effective_storage_config() - .await - .map_err(|e| AppError::internal_error(format!("Failed to load storage config: {}", e)))?; - - let target = build_backend_from_config(&effective) - .map_err(|e| AppError::internal_error(format!("Failed to build target backend: {}", e)))?; - target - .initialize() - .await - .map_err(|e| AppError::internal_error(format!("Target backend init failed: {}", e)))?; - - let concurrency = dto.concurrency.unwrap_or(4).clamp(1, 16); - let migration_state = state.migration_state.clone(); - - // Spawn the background migration job. - tokio::spawn(async move { - if let Err(e) = crate::infrastructure::services::migration_job::run_migration( - source, - target, - pool, - migration_state, - concurrency, - ) - .await - { - tracing::error!("Migration job error: {}", e); - } - }); - - Ok(( - StatusCode::OK, - Json(serde_json::json!({ "message": "Migration started" })), - )) + trigger_storage_migration(state).await } -/// POST /api/admin/storage/migration/pause — pause running migration +/// POST /api/admin/storage/migration/pause — pause a running migration. +/// +/// Shim over cooperative cancel: flips the run row's status to +/// `CancelRequested`; the recoverable handler polls between batches +/// and returns `Paused` at the next boundary. If nothing is running, +/// returns 200 with `paused: false` — matches the "no-op is fine" +/// contract of `/api/admin/jobs/{name}/cancel`. #[utoipa::path( post, path = "/api/admin/storage/migration/pause", responses( - (status = 200, description = "Migration paused"), - (status = 400, description = "No running migration"), + (status = 200, description = "Pause signalled (or no-op)"), (status = 401, description = "Unauthorized"), (status = 403, description = "Admin required") ), @@ -471,26 +480,45 @@ pub async fn start_migration( pub async fn pause_migration( State(state): State>, ) -> Result { - use crate::infrastructure::services::migration_blob_backend::MigrationStatus; + use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME; + + tracing::info!( + target: "audit", + event = "storage_migration.pause_requested", + "👮🏻‍♂️ Admin requested storage_migration pause" + ); + + let flipped = state + .core + .job_store_provider + .request_cancel(STORAGE_MIGRATION_JOB_NAME) + .await + .map_err(AppError::from)?; - let mut s = state.migration_state.write().await; - if s.status != MigrationStatus::Running { - return Err(AppError::bad_request("No running migration to pause")); - } - s.status = MigrationStatus::Paused; Ok(( StatusCode::OK, - Json(serde_json::json!({ "message": "Migration paused" })), + Json(serde_json::json!({ + "paused": flipped.is_some(), + "run_id": flipped, + "message": if flipped.is_some() { + "Pause requested — handler will yield at the next batch boundary" + } else { + "No running migration to pause" + }, + })), )) } -/// POST /api/admin/storage/migration/resume — resume paused migration +/// POST /api/admin/storage/migration/resume — resume a paused migration. +/// +/// Same underlying trigger as `/start`: `run_or_resume` inspects the +/// latest row and picks Fresh / Resume / AlreadyActive at dispatch +/// time. Kept as a distinct endpoint for wire-compat. #[utoipa::path( post, path = "/api/admin/storage/migration/resume", responses( - (status = 200, description = "Migration resumed"), - (status = 400, description = "No paused migration"), + (status = 200, description = "Migration resumed (or already running)"), (status = 401, description = "Unauthorized"), (status = 403, description = "Admin required") ), @@ -500,59 +528,15 @@ pub async fn pause_migration( pub async fn resume_migration( State(state): State>, ) -> Result { - use crate::infrastructure::services::migration_blob_backend::MigrationStatus; - - // Set status back to Running — the background task checks on each blob. - let mut s = state.migration_state.write().await; - if s.status != MigrationStatus::Paused { - return Err(AppError::bad_request("No paused migration to resume")); - } - s.status = MigrationStatus::Running; - Ok(( - StatusCode::OK, - Json(serde_json::json!({ "message": "Migration resumed" })), - )) + trigger_storage_migration(state).await } -/// POST /api/admin/storage/migration/complete — finalize migration -#[utoipa::path( - post, - path = "/api/admin/storage/migration/complete", - responses( - (status = 200, description = "Migration finalized"), - (status = 400, description = "Migration not completed"), - (status = 401, description = "Unauthorized"), - (status = 403, description = "Admin required") - ), - security(("bearerAuth" = [])), - tag = "admin" -)] -pub async fn complete_migration( - State(state): State>, -) -> Result { - use crate::infrastructure::services::migration_blob_backend::MigrationStatus; - - let s = state.migration_state.read().await; - if s.status != MigrationStatus::Completed { - return Err(AppError::bad_request( - "Migration must be completed (100%) before finalizing", - )); - } - drop(s); - - // Mark as idle — the admin has acknowledged completion. - let mut s = state.migration_state.write().await; - s.status = MigrationStatus::Idle; - - Ok(( - StatusCode::OK, - Json( - serde_json::json!({ "message": "Migration finalized. Restart the server to use the new backend." }), - ), - )) -} - -/// POST /api/admin/storage/migration/verify — run integrity check +/// POST /api/admin/storage/migration/verify — post-migration integrity check. +/// +/// Independent of the copy job: samples `sample_size` random blobs +/// from `storage.blobs` and probes the currently-effective target +/// backend for their existence + declared size. Passes iff no +/// samples are missing and no sizes disagree. #[utoipa::path( post, path = "/api/admin/storage/migration/verify", @@ -579,12 +563,9 @@ pub async fn verify_migration( .as_ref() .ok_or_else(|| AppError::internal_error("Storage settings service not available"))?; - let effective = svc - .load_effective_storage_config() + let target = svc + .build_effective_backend() .await - .map_err(|e| AppError::internal_error(format!("Failed to load storage config: {}", e)))?; - - let target = build_backend_from_config(&effective) .map_err(|e| AppError::internal_error(format!("Failed to build target backend: {}", e)))?; target .initialize() @@ -593,38 +574,176 @@ pub async fn verify_migration( let sample_size = dto.sample_size.unwrap_or(100).clamp(1, 1000); - let result = - crate::infrastructure::services::migration_job::verify_migration(target, pool, sample_size) - .await - .map_err(|e| AppError::internal_error(format!("Verification failed: {}", e)))?; + let result = verify_backend_sample(target.as_ref(), pool.as_ref(), sample_size) + .await + .map_err(|e| AppError::internal_error(format!("Verification failed: {}", e)))?; Ok(Json(result)) } -/// Helper: convert MigrationState to DTO for JSON serialization. -fn migration_state_to_dto( - s: &crate::infrastructure::services::migration_blob_backend::MigrationState, -) -> MigrationStateDto { - let throughput = match (s.started_at, s.migrated_bytes) { - (Some(start), bytes) if bytes > 0 => { - let elapsed = chrono::Utc::now() - .signed_duration_since(start) - .num_seconds() - .max(1) as f64; - Some(bytes as f64 / elapsed) +/// Shared body for `start` / `resume` — both funnel through +/// `run_or_resume` via `JobRegistry::trigger`. Detaches into a +/// `tokio::spawn` so the HTTP response returns immediately — same +/// rationale as `trigger_job` above (browser timeout mid-await would +/// desync `current_run_start` from the actually-running task). The +/// admin UI polls `GET /storage/migration` for progress; the trigger +/// itself is fire-and-forget. +async fn trigger_storage_migration( + state: Arc, +) -> Result { + use crate::infrastructure::scheduler::JobRunArgs; + use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME; + + tracing::info!( + target: "audit", + event = "storage_migration.trigger_requested", + "👮🏻‍♂️ Admin triggered storage_migration" + ); + + let registry = state.core.job_registry.clone(); + tokio::spawn(async move { + registry + .trigger(STORAGE_MIGRATION_JOB_NAME, &JobRunArgs::default()) + .await; + }); + + Ok(( + StatusCode::ACCEPTED, + Json(serde_json::json!({ + "message": "Migration dispatched — poll GET /api/admin/storage/migration for status", + "detached": true, + })), + ) + .into_response()) +} + +/// Verify a random sample of blobs against the given target backend. +/// Inlined from the retired `migration_job::verify_migration` — same +/// query, same result shape; the recoverable-run engine has no reason +/// to own an integrity check. +async fn verify_backend_sample( + target: &dyn crate::application::ports::blob_storage_ports::BlobStorageBackend, + pool: &sqlx::PgPool, + sample_size: usize, +) -> Result { + use crate::common::errors::DomainError; + + let pg_count: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM storage.blobs") + .fetch_one(pool) + .await + .unwrap_or(0); + + let sample_rows: Vec<(String, i64)> = + sqlx::query_as("SELECT hash, size FROM storage.blobs ORDER BY random() LIMIT $1") + .bind(sample_size as i64) + .fetch_all(pool) + .await + .map_err(|e| { + DomainError::internal_error("Migration", format!("Sample query failed: {}", e)) + })?; + + let mut missing = Vec::new(); + let mut size_mismatches = Vec::new(); + + for (hash, expected_size) in &sample_rows { + match target.blob_exists(hash).await { + Ok(false) => missing.push(hash.clone()), + Err(e) => { + tracing::warn!("blob_exists failed for {}: {}", hash, e); + missing.push(hash.clone()); + } + Ok(true) => { + if let Ok(actual_size) = target.blob_size(hash).await + && actual_size != *expected_size as u64 + { + size_mismatches.push(hash.clone()); + } + } } - _ => None, + } + + let passed = missing.is_empty() && size_mismatches.is_empty(); + Ok(MigrationVerifyResult { + pg_blob_count: pg_count as u64, + sample_checked: sample_rows.len() as u64, + missing_in_target: missing, + size_mismatches, + passed, + }) +} + +/// Post-migration verification result — same shape as the retired +/// `migration_job::VerificationResult` (kept identical so the admin +/// UI's `MigrationVerifyResult` decoder needs no change). +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct MigrationVerifyResult { + pub pg_blob_count: u64, + pub sample_checked: u64, + pub missing_in_target: Vec, + pub size_mismatches: Vec, + pub passed: bool, +} + +/// Idle-state DTO — no run has been triggered yet. +fn idle_migration_dto() -> MigrationStateDto { + MigrationStateDto { + status: "idle".to_string(), + total_blobs: 0, + migrated_blobs: 0, + // `migrated_bytes` and `throughput_bytes_per_sec` are no + // longer tracked — the recoverable engine bumps + // `stats.scanned_count` (a blob-count aggregator), not a + // bytes counter. The admin UI keeps these fields for + // wire-compat; they read 0 / null. + migrated_bytes: 0, + failed_blobs: Vec::new(), + started_at: None, + completed_at: None, + throughput_bytes_per_sec: None, + } +} + +/// Project a recoverable `RunSummary` into the admin UI's +/// `MigrationStateDto`. Byte-counter fields are always 0 / None — +/// see `idle_migration_dto`'s comment. +fn run_to_migration_dto( + run: &crate::infrastructure::scheduler::RunSummary, + failed_blobs: Vec, +) -> MigrationStateDto { + use crate::infrastructure::scheduler::RunStatus; + + // Fold CancelRequested into "paused" — from the admin UI's + // point of view a cancel-in-flight is the "waiting for the + // handler to yield" state. Same visual affordance as Paused. + let status = match run.status { + RunStatus::Running => "running", + RunStatus::Paused => "paused", + RunStatus::CancelRequested => "paused", + RunStatus::Completed => "completed", + RunStatus::Failed => "failed", + } + .to_string(); + + let (total_blobs, migrated_blobs) = match run.progress.as_ref() { + Some(p) => (p.total, p.scanned), + None => ( + 0, + run.stats + .get("scanned_count") + .and_then(|v| v.as_u64()) + .unwrap_or(0), + ), }; MigrationStateDto { - status: format!("{:?}", s.status).to_lowercase(), - total_blobs: s.total_blobs, - migrated_blobs: s.migrated_blobs, - migrated_bytes: s.migrated_bytes, - failed_blobs: s.failed_blobs.clone(), - started_at: s.started_at.map(|d| d.to_rfc3339()), - completed_at: s.completed_at.map(|d| d.to_rfc3339()), - throughput_bytes_per_sec: throughput, + status, + total_blobs, + migrated_blobs, + migrated_bytes: 0, + failed_blobs, + started_at: Some(run.started_at.to_rfc3339()), + completed_at: run.completed_at.map(|d| d.to_rfc3339()), + throughput_bytes_per_sec: None, } } @@ -652,34 +771,6 @@ pub async fn generate_encryption_key() -> Result { }))) } -/// Helper: build a BlobStorageBackend from StorageConfig. -fn build_backend_from_config( - config: &crate::common::config::StorageConfig, -) -> Result< - std::sync::Arc, - String, -> { - match config.backend { - crate::common::config::StorageBackendType::Local => Ok(std::sync::Arc::new( - crate::infrastructure::services::local_blob_backend::LocalBlobBackend::new( - std::path::Path::new(&config.root_dir), - ), - )), - crate::common::config::StorageBackendType::S3 => { - let s3 = config.s3.as_ref().ok_or("S3 config missing")?; - Ok(std::sync::Arc::new( - crate::infrastructure::services::s3_blob_backend::S3BlobBackend::new(s3), - )) - } - crate::common::config::StorageBackendType::Azure => { - let az = config.azure.as_ref().ok_or("Azure config missing")?; - Ok(std::sync::Arc::new( - crate::infrastructure::services::azure_blob_backend::AzureBlobBackend::new(az), - )) - } - } -} - // ============================================================================ // Dashboard / Stats // ============================================================================ @@ -2159,6 +2250,40 @@ pub async fn trigger_job( force: query.force, deep: query.deep, }; + + // Jobs that can run for hours (storage_migration, future + // reextract_*) are detached: `tokio::spawn` the trigger so the + // HTTP request returns immediately. Without this, browser HTTP + // timeouts drop the request future mid-await → the SemaphorePermit + // gets released while the spawned handler task keeps running → + // `current_run_start` goes stale → a second click enters the + // "already_running" short-circuit and CLEARS the in-memory state + // even though the original task is still copying blobs → the + // Cancel button hides because `job.running = false`. Detaching + // keeps the permit + `current_run_start` scoped to the actual + // handler-task lifetime. + // + // Fast-completing jobs (consistency checks, batch coordinator) + // stay inline so the operator sees the outcome envelope. + if is_detached_job(&name) { + let name_clone = name.clone(); + let registry = state.core.job_registry.clone(); + tokio::spawn(async move { + registry.trigger(&name_clone, &args).await; + }); + return ( + StatusCode::ACCEPTED, + Json(serde_json::json!({ + "ok": true, + "dispatched": true, + "detached": true, + "name": name, + "message": "Dispatched — poll /runs for status", + })), + ) + .into_response(); + } + match state.core.job_registry.trigger(&name, &args).await { Some(outcome) => ( StatusCode::OK, @@ -2176,6 +2301,15 @@ pub async fn trigger_job( } } +/// Jobs that MUST be dispatched with `tokio::spawn` because they run +/// long enough to outlast an HTTP request timeout. Kept as a small +/// hardcoded allowlist (rather than a flag on `JobEntry`) until +/// there's a second long-running tenant that justifies the plumbing. +/// See the comment in `trigger_job` for why detach matters. +fn is_detached_job(name: &str) -> bool { + matches!(name, "storage_migration") +} + /// `POST /api/admin/jobs/{name}/cancel` — cooperative cancel of the /// currently-running recoverable run for `{name}`. /// diff --git a/src/interfaces/api/mod.rs b/src/interfaces/api/mod.rs index 64d7308d..e9c510d7 100644 --- a/src/interfaces/api/mod.rs +++ b/src/interfaces/api/mod.rs @@ -225,7 +225,6 @@ use crate::interfaces::api::handlers::file_handler::MoveFilePayload; handlers::admin_handler::start_migration, handlers::admin_handler::pause_migration, handlers::admin_handler::resume_migration, - handlers::admin_handler::complete_migration, handlers::admin_handler::verify_migration, handlers::admin_handler::generate_encryption_key, // JobRegistry admin surface — production, always-on, From 2de5abc6cacfb0ca0161a3b78edb5b6047c983da Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 11:52:00 +0200 Subject: [PATCH 02/17] plan(storage-multi-entry): simplify the storage migration MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two chronic problems fall out: 1. **Split-brain config.** Admin edits DB via the panel; app boot ignores DB. Migration completes; live backend hasn't moved. Admin has to remember to copy env vars into `.env` and restart. Two sources of truth for the same setting. Cutover is a manual multi-step flow; users routinely get it wrong. 2. **Migration data-loss window on concurrent writes.** The copy walks `storage.blobs` in hash order. A blob whose hash is lex-lower than the current cursor, written to source AFTER migration passed it, is never copied to target. `passed=true, findings=0` completion does NOT guarantee target has every blob. Silent. 3. **Migration target selection is fragile.** DTO passes the whole S3 config at trigger time; secrets sit plaintext in `admin_settings`. Any future pluggable-storage story compounds this (Azure, GCS, WebDAV-as-source, …). This plan replaces the split-brain model with a single-source-of-truth architecture: - `.env` declares **N named storage entries** (immutable per-deploy). - `admin_settings.storage.active_backend_name` holds ONE row — which named entry the app currently runs on. That's the whole runtime config. - Migration is the atomic transition from one active entry to another. Server is put in read-only mode for the copy window; on completion, the active pointer flips; a restart cuts over. --- docs/plan/storage-multi-entry.md | 477 +++++++++++++++++++++++++++++++ 1 file changed, 477 insertions(+) create mode 100644 docs/plan/storage-multi-entry.md diff --git a/docs/plan/storage-multi-entry.md b/docs/plan/storage-multi-entry.md new file mode 100644 index 00000000..584aab50 --- /dev/null +++ b/docs/plan/storage-multi-entry.md @@ -0,0 +1,477 @@ +# Plan — Multi-entry storage config + name-selected active backend + +## Context + +Today OxiCloud has a single storage backend, configured via a flat set of env +vars (`OXICLOUD_STORAGE_BACKEND`, `OXICLOUD_S3_*`, `OXICLOUD_STORAGE_ENCRYPTION_KEY`, +…). The admin panel has a second, parallel storage-config surface backed by +`admin_settings.storage.*` rows, used by the migration tool to pick a target. +Priority is `env > DB > defaults`, so the DB config effectively acts as a +staging area for "what the migration should copy INTO" but never wins at boot. + +Two chronic problems fall out: + +1. **Split-brain config.** Admin edits DB via the panel; app boot ignores DB. + Migration completes; live backend hasn't moved. Admin has to remember to + copy env vars into `.env` and restart. Two sources of truth for the same + setting. Cutover is a manual multi-step flow; users routinely get it wrong. + +2. **Migration data-loss window on concurrent writes.** The copy walks + `storage.blobs` in hash order. A blob whose hash is lex-lower than the + current cursor, written to source AFTER migration passed it, is never + copied to target. `passed=true, findings=0` completion does NOT guarantee + target has every blob. Silent. + +3. **Migration target selection is fragile.** DTO passes the whole S3 config + at trigger time; secrets sit plaintext in `admin_settings`. Any future + pluggable-storage story compounds this (Azure, GCS, WebDAV-as-source, …). + +This plan replaces the split-brain model with a single-source-of-truth +architecture: + +- `.env` declares **N named storage entries** (immutable per-deploy). +- `admin_settings.storage.active_backend_name` holds ONE row — which named + entry the app currently runs on. That's the whole runtime config. +- Migration is the atomic transition from one active entry to another. Server + is put in read-only mode for the copy window; on completion, the active + pointer flips; a restart cuts over. + +Named entries also solve two adjacent problems Ed flagged during design: +- **Per-entry encryption keys** enable `local (raw) → s3 (encrypted K1)` + moves AND `s3 (K1) → s3-new-bucket (K2)` key-rotation moves. +- **`?storage=` on `blobs_consistency`/`backend_consistency`** lets + operators audit any registered entry (target verification, pre-decommission + check, etc.), replacing the sample-based `verify_migration` endpoint with a + full-walk audit. + +## Design decisions + +### Named entries in `.env` — explicit allowlist + +``` +OXICLOUD_STORAGE_ENTRIES=local_main,s3_prod + +OXICLOUD_STORAGE_local_main_BACKEND=local +OXICLOUD_STORAGE_local_main_ROOT_DIR=/data + +OXICLOUD_STORAGE_s3_prod_BACKEND=s3 +OXICLOUD_STORAGE_s3_prod_S3_BUCKET=my-bucket +OXICLOUD_STORAGE_s3_prod_S3_ENDPOINT_URL=https://s3.example.com +OXICLOUD_STORAGE_s3_prod_S3_REGION=us-east-1 +OXICLOUD_STORAGE_s3_prod_S3_ACCESS_KEY=... +OXICLOUD_STORAGE_s3_prod_S3_SECRET_KEY=... +OXICLOUD_STORAGE_s3_prod_S3_FORCE_PATH_STYLE=true +OXICLOUD_STORAGE_s3_prod_ENCRYPTION_KEY= +``` + +- **Explicit `_ENTRIES` allowlist** — order-independent, admin-authored names. + Env-pattern-matching was considered and rejected: too fragile + (someone typos `OXICLOUD_STORAGE_s3_prud_BUCKET`, gets a silently-registered + ghost entry). Explicit list forces a declaration. +- **Names** are admin-chosen strings matching `[a-z0-9_-]{1,32}`, unique + within the list. Parsed at boot; unparseable → fail-fast with the offending + name in the error. +- **Order does not matter** — DB says which is active. Reordering entries in + `.env` never changes runtime behaviour. + +### Legacy flat-var interaction — three states, one is fail-fast + +Multi-entry lives alongside the pre-existing single-backend flat vars +(`OXICLOUD_STORAGE_BACKEND`, `OXICLOUD_S3_*`, `OXICLOUD_AZURE_*`, +`OXICLOUD_STORAGE_ENCRYPTION_KEY`, `_ENABLED`). The parser resolves the +interaction as follows: + +| `_ENTRIES` | Legacy storage-backend vars present | Behaviour | +|---|---|---| +| Unset / empty | Absent | `storage_entries = []`. Boot uses framework defaults (Local at `storage/`). | +| Unset / empty | Present | **Synthesize** a single entry named `default` from the legacy vars. Preserves upgrade path — existing deployments keep working without touching `.env`. | +| Set (e.g. `foo,bar`) | Absent | Parse each named entry from `_STORAGE__*` vars. Fail-fast if any declared entry is missing its required per-name fields. | +| Set (e.g. `foo,bar`) | **Present** | **FAIL FAST**. Boot aborts with an error listing every legacy var found. Admin must remove them or migrate them into per-entry `_STORAGE__*` form. | + +**Why fail-fast on the "both set" case:** without it, admin state is +ambiguous — someone edits `OXICLOUD_S3_BUCKET` expecting it to matter, +but it's silently ignored because `_ENTRIES` won. The subtle-bug cost +is much higher than the one-time cleanup cost. Refusing to boot forces +the conversion once, with a clear message naming the exact vars to +remove. + +**Set of "legacy storage-backend vars"** counted for the conflict check: +`OXICLOUD_STORAGE_BACKEND`, all seven `OXICLOUD_S3_*`, all five +`OXICLOUD_AZURE_*`, `OXICLOUD_STORAGE_ENCRYPTION_ENABLED`, and +`OXICLOUD_STORAGE_ENCRYPTION_KEY`. `OXICLOUD_STORAGE_PATH` is NOT in +this set — it drives multiple non-backend things (chunk dir default, +etc.) and remains a valid ambient path; per-entry `_ROOT_DIR` falls +back to it for Local entries when unset. + +### One DB row: `active_backend_name` + +Single setting in `admin_settings`: + +``` +storage.active_backend_name = "local_main" +``` + +- **Boot logic**: + 1. Parse `AppConfig.storage_entries` from env. + 2. Read `active_backend_name` from `admin_settings`. + 3. If unset (fresh install): use the FIRST name in `_ENTRIES`. Log which + one, don't fail. + 4. Look up the entry by name. Build `blob_backend` from it. + 5. If the name doesn't exist in current env (deploy drift — someone removed + an entry): **fail-fast at boot** with a clear error naming the missing + entry AND listing the available ones. Admin fixes env or overrides the + setting via a fallback CLI (see §Fallback). +- Everything the admin panel currently writes about storage + (`s3.bucket`, `s3.access_key`, …) is **removed** from DB. `save_storage_settings` + is deleted along with those rows. + +### Encryption is per-entry + +`OXICLOUD_STORAGE__ENCRYPTION_KEY` (base64 of exactly 32 bytes) on an +entry → that entry's backend is wrapped in `EncryptedBlobBackend` at build +time. Absent → raw backend. + +- **Presence-implies-enabled.** No separate `_ENCRYPTION_ENABLED` toggle — + one env var per entry is enough. +- **Fail-fast on invalid key**: bad base64, wrong decoded length → boot + aborts with the entry name in the error message. A bad key is a real + deployment error, must be caught at boot, never silently disabled. +- **Legacy `OXICLOUD_STORAGE_ENCRYPTION_KEY`** remains honoured for the + synthesized `default` entry when `_ENTRIES` is empty (upgrade path). + +Cross-entry encryption combinations work out of the box because +`EncryptedBlobBackend` is a decorator and `copy_blob` in the migration +handler always spools plaintext to a tmp file: + +| Source | Target | Migration behaviour | +|---|---|---| +| Raw local | Encrypted S3 (K1) | Read plaintext → write encrypts with K1 | +| Encrypted S3 (K1) | Raw local | Read decrypts with K1 → write plaintext | +| Encrypted S3 (K1) | Encrypted S3 (K2) on different bucket | Read decrypts K1 → write encrypts K2 (rotation via new bucket) | +| Encrypted S3 (K1) | Encrypted S3 (K2) on SAME bucket | **REFUSED** — see below | + +**In-place key rotation is refused.** Same physical bucket + different +encryption key would have the migration overwrite `.blob` with K2 +ciphertext while the LIVE backend is still K1-configured → readers get +K1-decrypt-of-K2-bytes → 500. Silent data-loss window. The +`is_source_target_identical` guard (see §Migration flow) catches this because +`storage_identity` deliberately excludes the encryption key. The refusal +message spells the case out and recommends the two-step workaround (rotate +via a temp bucket). + +**Proper in-place rotation (deferred future slice)** + +The safe implementation is to namespace object keys by encryption +generation — `.k2.blob` or `k2/.blob` (backend-specific +convention). Then: + +- Both key generations coexist in the same bucket during rotation. +- Reads continue against K1 (LIVE backend) — old keys untouched. +- Target writes go to K2 keys. +- Cutover restarts with K2 as live; K1 objects can be reaped async by + a follow-up sweep. +- The `is_source_target_identical` guard's `storage_identity` string + changes to INCLUDE the encryption generation — same-bucket + different-generation then correctly registers as a legitimate + migration, no longer refused. + +Schema changes required: +- `EncryptedBlobBackend` gains a `generation: u32` field. +- Object keys become `.k.blob` (or backend-specific + equivalent — S3 key naming, Azure blob naming). +- Read path tries current-gen first, falls back to prior-gen for the + rotation window (bounded by cutover-completion + async-reap + duration). + +Not built now — real key-rotation demand is rare enough that the +two-step workaround via temp bucket is acceptable. File this as +a follow-up slice AFTER multi-entry lands; the named-entry +infrastructure is a prerequisite (generation-versioned keys only +make sense when there's an entry model to hold "which generation +is active" as configuration). + +### Read-only mode reuses the existing AuthZ short-circuit + +`DrivePolicies.read_only` already gates writes at +`PgAclEngine::check_inner` — short-circuits `Create|Update|Delete|Share` +Permission checks on any resource in a read-only drive. We extend that clause +with a global check: + +```rust +// Inside PgAclEngine::check_inner, before per-drive read_only check: +if permission.is_write() && self.migration_readonly.load(Ordering::Relaxed) { + return AclDecision::Denied("server in migration read-only mode"); +} +``` + +- The flag is a `AtomicBool` on `AppState`, backed by + `admin_settings.storage.migration_readonly` so it **survives restart** + (server crashes mid-migration → boots read-only → admin retriggers → still + safe). +- **Admin operations bypass** as they already do — admin can still exit + read-only, cancel migration, restart the server. +- **Reads are unaffected**. Users can still browse and download during + migration. +- **Boot-time clearing**: if boot detects `migration_readonly=true` AND no + in-flight `storage_migration` row (no `Running`/`Paused`) AND + `active_backend_name` matches the entry the app booted onto → assume + successful cutover completed on prior boot, clear the flag. Otherwise leave + it set; admin knows they still need to finish something. + +### Migration flow — atomic + restart-proof + +``` +1. Admin picks target entry from a dropdown → clicks "Migrate to s3_prod" + +2. Backend: + - Verify target_name exists in AppConfig.storage_entries + - Verify target_name != active_backend_name (no-op guard; existing + `is_source_target_identical` refactored to compare NAMES not identity + strings — but the identity check still runs as a second-line defence + against the encryption-in-place case) + - Write admin_settings.storage.migration_readonly = true + - Trigger `storage_migration` recoverable job with + params = { source_name: "local_main", target_name: "s3_prod" } + +3. Migration runs — target resolved fresh each batch by NAME lookup, so + the run's params carries only the name. No secrets in params. If the + process restarts mid-run, the resume path re-resolves the entry from + env by the persisted name. Env is the source of truth for credentials. + +4. On Completed: + - Write admin_settings.storage.active_backend_name = "s3_prod" + - Log "Cutover complete — restart the server to switch to the new backend" + - LEAVE read-only mode on. The app is still running with local_main as + the live backend; if we lifted read-only now, writes would go to + local_main even though the DB pointer says s3_prod. Forces the operator + restart, which resolves the ambiguity. + +5. Admin restarts the server: + - Boot reads active_backend_name = "s3_prod" → live backend is now S3 + - Boot's read-only-clear rule fires (no in-flight migration + active + matches booted entry) → migration_readonly cleared + - Server writable, on the new backend. Cutover complete. +``` + +**Handling in-progress restart (server dies while migration is running)**: +- Boot sweep flips `Running` → `Paused` on the migration row (existing Part + 2 machinery). +- `active_backend_name` unchanged. Boots on old backend. +- `migration_readonly` stays true (in-flight migration → boot-time clear + rule DOESN'T fire). +- Admin retriggers migration → run_or_resume reads params.target_name → + resumes from cursor with the same target. + +**No secret ever reaches the DB.** Params holds only the entry names. +Credentials stay in env; migration resolves them at each batch by name. + +### Concurrent-write safety — read-only for the copy window + +The full-quiesce trade-off: users can browse/download during migration but +cannot upload, rename, delete, or share. For a multi-hour migration this is +noticeable; the alternative (dual-write decorator) is genuinely a week of +work (runtime backend swapping, failure-mode reconciliation, `MigrationBlobBackend` +rebuild) and only justified if migration is a routine op. Read-only is the +honest v1 answer — pick a low-traffic window, run the migration, restart. + +Read-only is engaged at trigger time and cleared at boot after cutover. +Nothing more elaborate. Dual-write is filed as a future upgrade if operator +demand appears. + +### `?storage=` for consistency audits + +Once entries are named, `?storage=` becomes the generic "probe any +registered entry" knob on the two tenants that touch a backend: + +- `blobs_consistency?storage=` — DB → backend probe. +- `backend_consistency?storage=` — backend → DB probe. + +Unspecified → falls through to the active backend (today's behaviour, +preserved). + +**Use cases this unlocks**: +- Pre-cutover verification: `blobs_consistency?storage=s3_prod` after + migration completes — full walk (not a sample) proving target has every + blob before the .env flip + restart. **Retires `verify_migration`** — + sample-check becomes redundant when full audit is one click away. +- Pre-decommission verification: `?storage=local_main` after cutover — check + the old backend still has every blob DB expects before `rm -rf` the local + `.blobs/`. +- Ad-hoc audit of any registered entry, backup-restore verification, etc. + +**Plumbing**: +- `JobRunArgs` gains `storage: Option`. +- `TriggerJobQuery` on the admin trigger endpoint parses `?storage=`. +- `BlobsConsistencyCheck` and `BackendConsistencyCheck` constructors take + `Arc` (or a smaller `EntryResolver` port); at run + start they use `args.storage` to pick the backend, falling back to the + injected active-backend Arc. +- **Unknown entry**: fail-fast at HTTP layer (`400` before the run ever + starts) with `known: [local_main, s3_prod]` in the response. Cheaper than + burning a run row. +- **Combined with `?deep=true`**: legit — "full byte-level integrity audit + of a named entry, not the live one". Audit log records both. +- Params on the run row records `probed_storage: ` (or "active") for + post-hoc diagnosis. + +### Fallback for the "boot fails on missing entry" case + +If admin renames an entry in `.env` (or removes one that DB still points +at), boot fails fast with a clear error. Operator has two ways out: + +1. Fix `.env` — add the missing entry back OR update `_ENTRIES` to include + an alternative that DOES exist, plus flip `active_backend_name` before + restart. +2. **CLI repair flag on the `oxicloud` binary itself**: + ``` + oxicloud --select-storage + ``` + Behaviour: parse `.env`, verify `` exists in `_ENTRIES` (fail-fast + with the available names listed if not), connect to DB, UPDATE + `admin_settings.storage.active_backend_name`, print confirmation, exit 0. + Does NOT continue to boot the server — one-shot repair; admin starts + the server normally afterwards. + +The bare-flag on the shipped binary is chosen over a separate `just` +recipe or auxiliary bin because: +- **Docker-friendly**: `docker exec oxicloud oxicloud --select-storage foo` + — no need to install extra tooling in the container. +- **Systemd-friendly**: can be run as a `ExecStartPre=` one-shot before the + main service unit. +- **No dep on `just`** being installed (dev-machine tool, not typical prod). +- **Same binary, same env-parse code path**: the repair uses THE SAME + `.env` parser the server does, so "verified present" means the server + will succeed on next boot too. No parser drift possible. + +The boot-time error message points at this flag explicitly, with the +exact command line filled in. + +### Interaction with existing surfaces + +- **`/admin/storage` tab** rewritten: + - Read-only listing of entries (name, backend type, encryption on/off, is + active). + - "Migrate to X" dropdown (choose target entry). + - Migration progress + verify/cancel (existing). + - Read-only banner when `migration_readonly` is on. + - The Save form, S3-field editors, .env cutover hint — all deleted (no + settings edit here anymore). +- **`test_storage_connection`** loses the S3-fields DTO; becomes a + round-trip probe against a named entry: `POST .../test?storage=`. +- **`verify_migration`** deleted; the "Verify integrity" button on the + storage tab is rewired to trigger `blobs_consistency?storage=` + against the migration target (or dropped in favour of the standard + admin/jobs Run button — TBD in slice 6). + +## Slice breakdown + +Single PR is too big; splitting into a coherent sequence. Slices 1-2 are +foundational; the rest layer on top independently within reason. + +| # | Slice | Depends on | Rough size | +|---|---|---|---| +| 1 | Config parser: `OXICLOUD_STORAGE_ENTRIES` + per-entry vars + `_ENCRYPTION_KEY` → `Vec` on `AppConfig`. Legacy synthesis for empty `_ENTRIES` (default entry from legacy flat vars). | — | 1 day | +| 2 | Boot: read `active_backend_name` from `admin_settings`; look up entry; build `blob_backend` via a shared `build_entry_backend(&NamedStorageEntry)` factory (wraps encryption decorator when key present). Fail-fast on missing entry with actionable message. | 1 | ~half day | +| 3 | Migration handler rewrite: params carry `target_name` only. Handler resolves target by name from `storage_settings.build_entry_backend(name)`. `is_source_target_identical` guard refactored to compare NAMES first; keeps physical-identity check as second-line refusal (with encryption-differs-specific message). Retire the migration DTO S3-field body. | 1, 2 | ~half day | +| 4 | Global `migration_readonly` flag on `AppState`, backed by `admin_settings.storage.migration_readonly`. One clause added to `PgAclEngine::check_inner`. Boot-time clear rule (no in-flight + active matches booted → clear). | 2 | ~half day | +| 5 | Cutover state machine: on migration `Completed`, write `active_backend_name = target_name`, keep read-only on. Boot on new backend after operator restart. | 4 | ~half day | +| 6 | Admin storage tab rewrite: list entries, show active, migrate dropdown, read-only banner. Delete Save form + S3 field editors + .env cutover hint. | 1, 3, 4 | 1 day | +| 7 | `?storage=` on `blobs_consistency` + `backend_consistency`. `JobRunArgs.storage` plumbing, `TriggerJobQuery.storage`, entry-resolver at run start, params records probed name. Retire `verify_migration` + its DTO + its route + its handler. | 1, 3 | 1 day | +| 8 | `oxicloud --select-storage ` bare-flag repair command on the main binary. Parses `.env`, verifies entry exists, UPDATEs DB, exits. Boot-time missing-entry error message points at it. See §Fallback. | 2 | ~quarter day | + +**Total: ~5-6 days end to end.** Slices 6 and 7 can proceed in parallel with +each other once 1-5 land. Slice 8 is an ops nicety, could ship whenever. + +## Verification + +Per slice, plus these end-to-end scenarios in Hurl: + +1. **Fresh install, no `_ENTRIES`**: boot uses synthesized `default` entry from + legacy vars, `active_backend_name` unset. `GET /admin/settings/storage` + returns one entry, active. No cutover UI. +2. **Two entries, no active set**: boot picks first in `_ENTRIES`, logs it, + proceeds. Admin panel shows both entries with the picked one marked active. +3. **Change active without migration** (rarely useful but must be safe): + `POST /admin/settings/storage/active-name` (or however the pointer is + exposed) updates the row; next restart boots on the new entry. Existing + blobs on the new entry NOT verified — operator's problem, but + `blobs_consistency` catches the drift on next run. +4. **Migration happy path**: two entries, entry A active, migrate to B. + Read-only comes on. Copy runs. `active_backend_name` flips to B on + completion. Read-only stays on until restart. After restart, live is B, + read-only cleared. +5. **Restart mid-migration**: kill server after checkpoint N. Boot: row Paused, + `active` still A, `migration_readonly` still true. Admin retriggers → + resumes from checkpoint N against target B (resolved by name from params). + Complete → pointer flips → restart → live is B. +6. **In-place encryption rotation refused**: two entries, same S3 bucket, + different encryption keys. Trigger migration → refuses with the specific + error message pointing at the encryption case and the two-step workaround. +7. **`?storage=` on blobs_consistency**: run against `s3_prod` before + cutover. Full walk, `probed_storage` in run row. Then cutover, then rerun + against `local_main` — verifies old backend still has everything. +8. **Unknown storage name**: `POST /admin/jobs/blobs_consistency/trigger?storage=nope` + → 400 with known-names list. No run row created. +9. **Missing entry at boot**: `active_backend_name = "gone"` but `_ENTRIES` + doesn't include it → boot aborts with the specific message pointing at + `oxicloud --select-storage ` (with the available names filled in). + Re-run the binary with `--select-storage local_main` → verifies + updates + DB + exits 0. Restart the server → boots cleanly on `local_main`. +10. **Encryption key invalid**: `OXICLOUD_STORAGE__ENCRYPTION_KEY=badbase64` + → boot aborts with entry name + reason (not valid base64 / wrong length). +11. **Legacy vars alongside `_ENTRIES`**: set `_ENTRIES=foo` AND leave a + stale `OXICLOUD_S3_BUCKET=...` in `.env`. Boot aborts with the full + list of legacy vars detected, tells the admin to remove them (or move + them into `OXICLOUD_STORAGE__S3_BUCKET` form). Removing the + legacy var → next boot succeeds. +12. **Legacy synthesis path**: `_ENTRIES` unset, `OXICLOUD_STORAGE_BACKEND=s3` + + `OXICLOUD_S3_BUCKET=...` set. Boot synthesizes one entry named + `default` from those vars. `storage_entries.len() == 1`, name is + `default`, backend is S3, config carries the flat-var values. + +## Out of scope + +- **Dual-write decorator / zero-downtime migration**: read-only is v1; + dual-write is filed under "if operator demand appears". Would rebuild + `MigrationBlobBackend` (deleted in the recoverable-migration PR — retained + in git for the same reason). +- **In-place encryption key rotation** (same-bucket, different key): refused + by the identity guard; workaround is two-step via temp bucket. Proper fix + (per-generation object naming) is documented inline in the Encryption + section above; deferred until real demand appears. +- **Runtime backend hot-swap without restart**: not attempted. Requires + `Arc>>` indirection at every call site + plus per-request coordination; huge blast radius. Read-only + restart is + the honest answer. +- **Per-user or per-drive storage backends**: everything in this plan is + server-scoped. If per-drive storage becomes a real need, the named-entry + registry is the right substrate but `blob_backend` on AppState becomes a + `EntryResolver` and every call site changes. Not now. +- **DB storage config UI**: retired. Admin panel is a controller (list + + test + migrate + audit), never a persister. If persisted per-entry knobs + become a need (retention days per entry, quota per entry, …), those live + in DB rows keyed by entry name — a small extension, not a return to the + old model. +- **Secret encryption in DB**: `admin_settings.storage.*` secret rows are + deleted along with `save_storage_settings`. If the migration params + approach ever grows to store secrets (e.g., a future "supply the target + creds inline for one-off migrations"), those get encrypted at rest with + a KMS-provided key. Not needed for this plan — params holds only names. + +## Related memory notes + +- `feedback_no_abbreviated_env_vars` — full-word env var names + (`OXICLOUD_STORAGE_local_main_S3_ENDPOINT_URL`, not + `OXICLOUD_STORAGE_local_main_S3_EP`). +- `feedback_config_file_overrides_shell` — `dotenvy` behaviour on explicit + config path. Matters when `_ENTRIES` values are loaded from `--config` vs + shell. +- `project_admin_middleware_layer` — admin ops bypass read-only via existing + middleware; the AuthZ short-circuit added in slice 4 doesn't need any + new admin carve-out. +- `bug_drive_rename_editor_can_do_it` — reminder that + `PgAclEngine::check_inner` is where write-permission short-circuits live; + the new global read-only clause lands next to the per-drive one. +- `docs/plan/job-registry.md` Part 2 — recoverable-run engine that + `storage_migration` runs on; `params` field, resume semantics, boot + sweep. From 7534427dc2e4b74279d4373ef4ea9542a21c73c0 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 12:27:20 +0200 Subject: [PATCH 03/17] fix(oidc): apply clippy recos on PR 652 --- .../services/auth_application_service.rs | 30 +++++++++---------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/src/application/services/auth_application_service.rs b/src/application/services/auth_application_service.rs index a7076270..1ba1f762 100644 --- a/src/application/services/auth_application_service.rs +++ b/src/application/services/auth_application_service.rs @@ -2774,21 +2774,21 @@ impl AuthApplicationService { let provider_name = oidc.provider_name().to_string(); // Check email_verified - only if email is present in claims, and email verification is required. - if self.require_verified_email() { - if let Some(email) = &claims.email { - let verified = claims.email_verified.unwrap_or(false); - if !verified { - tracing::warn!( - "OIDC login rejected: email not verified (provider: {}, email: {})", - provider_name, - email - ); - return Err(DomainError::new( - ErrorKind::AccessDenied, - "OIDC", - "Email verification required. Please verify your email at the identity provider.", - )); - } + if self.require_verified_email() + && let Some(email) = &claims.email + { + let verified = claims.email_verified.unwrap_or(false); + if !verified { + tracing::warn!( + "OIDC login rejected: email not verified (provider: {}, email: {})", + provider_name, + email + ); + return Err(DomainError::new( + ErrorKind::AccessDenied, + "OIDC", + "Email verification required. Please verify your email at the identity provider.", + )); } } From 354e0581141019d4948ca6e98d4858fd764fc450 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 12:32:04 +0200 Subject: [PATCH 04/17] feat(storage): add multi entry in config --- src/common/config.rs | 750 ++++++++++++++++++- src/common/di.rs | 161 +++- src/infrastructure/services/entry_backend.rs | 180 +++++ src/infrastructure/services/mod.rs | 1 + 4 files changed, 1052 insertions(+), 40 deletions(-) create mode 100644 src/infrastructure/services/entry_backend.rs diff --git a/src/common/config.rs b/src/common/config.rs index 6d5bacfd..7dd1dc22 100644 --- a/src/common/config.rs +++ b/src/common/config.rs @@ -383,6 +383,392 @@ impl Default for RetryConfig { } } +/// One named storage entry declared in `.env`. +/// +/// See `docs/plan/storage-multi-entry.md`. Each entry is a fully-realised +/// backend configuration that the admin can point the runtime at via +/// `admin_settings.storage.active_backend_name`. Migrations move blobs +/// between two entries; consistency audits can be scoped to any registered +/// entry via `?storage=`. +/// +/// Entries are parsed from env at boot and held on `AppConfig.storage_entries`. +/// The set is immutable per-deploy — adding/removing entries requires a +/// server restart. The DB pointer `active_backend_name` is the ONLY mutable +/// runtime storage-selection surface. +#[derive(Debug, Clone)] +pub struct NamedStorageEntry { + /// Stable admin-authored identifier, `[a-z0-9_-]{1,32}`. Unique + /// within the entry list. Referenced from + /// `admin_settings.storage.active_backend_name` and from + /// `?storage=` on the audit APIs. + pub name: String, + /// Which backend type this entry uses. + pub backend: StorageBackendType, + /// Root directory for `Local`. `None` for `S3`/`Azure`. + /// Defaults to `"storage"` when the entry is Local and no + /// `_ROOT_DIR` is set (matches today's flat-var default). + pub root_dir: Option, + /// S3 configuration for `S3`. `None` for other backends. + pub s3: Option, + /// Azure configuration for `Azure`. `None` for other backends. + pub azure: Option, + /// Per-entry AES-256-GCM key (base64, exactly 32 bytes decoded). + /// Presence implies encryption is enabled on this entry — no + /// separate `_ENCRYPTION_ENABLED` toggle. Absence = raw backend. + /// Validated at parse time; boot aborts on invalid key. + pub encryption_key_base64: Option, +} + +/// Validation for a `NamedStorageEntry.name`. Restricts to a safe subset +/// so that entry names embed cleanly in env-var suffixes without +/// escaping (`OXICLOUD_STORAGE__BACKEND=...`) and in query params +/// (`?storage=`) without URL-encoding surprises. +/// +/// Rules: +/// - 1 to 32 chars. +/// - Lowercase ASCII letters, digits, `_`, `-` only. +/// +/// Deliberately no uppercase — the env-var expansion is +/// `OXICLOUD_STORAGE__...`, and mixing case would create confusing +/// same-looking-different-behaving variants (`_S3_` vs `_s3_`). Lowercase- +/// only keeps the surface unambiguous. +pub fn is_valid_entry_name(name: &str) -> bool { + let len = name.len(); + if !(1..=32).contains(&len) { + return false; + } + name.chars() + .all(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || c == '_' || c == '-') +} + +/// Env-var names counted as "legacy flat storage-backend vars" — set that +/// triggers the conflict-with-`_ENTRIES` fail-fast. Kept in one place so +/// docs, error messages, and tests reference the same list. +/// +/// **Not included**: `OXICLOUD_STORAGE_PATH` — that variable drives +/// multiple non-backend things (chunk-dir default, ambient storage path) +/// and remains valid alongside `_ENTRIES`. Per-entry `_ROOT_DIR` falls +/// back to it for Local entries when unset. +pub const LEGACY_STORAGE_BACKEND_VARS: &[&str] = &[ + "OXICLOUD_STORAGE_BACKEND", + "OXICLOUD_S3_ENDPOINT_URL", + "OXICLOUD_S3_BUCKET", + "OXICLOUD_S3_REGION", + "OXICLOUD_S3_ACCESS_KEY", + "OXICLOUD_S3_SECRET_KEY", + "OXICLOUD_S3_FORCE_PATH_STYLE", + "OXICLOUD_AZURE_ACCOUNT_NAME", + "OXICLOUD_AZURE_ACCOUNT_KEY", + "OXICLOUD_AZURE_CONTAINER", + "OXICLOUD_AZURE_SAS_TOKEN", + "OXICLOUD_AZURE_ENDPOINT_URL", + "OXICLOUD_STORAGE_ENCRYPTION_ENABLED", + "OXICLOUD_STORAGE_ENCRYPTION_KEY", +]; + +/// Parse the multi-entry storage config from env vars. +/// +/// See `docs/plan/storage-multi-entry.md` for the full model. This +/// function encodes the four-cell decision matrix documented there: +/// +/// | `_ENTRIES` set? | legacy vars present? | Return | +/// |---|---|---| +/// | No | No | `Ok(vec![])` — no entries; caller uses framework defaults. | +/// | No | Yes | `Ok(vec![synthesized_default])` — upgrade path. | +/// | Yes | No | `Ok(vec![entry_1, entry_2, …])` — the declared entries. | +/// | Yes | Yes | `Err("legacy vars alongside _ENTRIES: …")` — fail-fast. | +/// +/// All fatal shapes return `Err`. Callers in the boot path +/// (`AppConfig::from_env`) `.expect(...)` on the result — an invalid +/// storage config must abort at startup, not silently degrade. +pub fn parse_storage_entries() -> Result, String> { + let entries_raw = env::var("OXICLOUD_STORAGE_ENTRIES").unwrap_or_default(); + let entries_raw = entries_raw.trim(); + let legacy_present: Vec<&&str> = LEGACY_STORAGE_BACKEND_VARS + .iter() + .filter(|v| env::var(v).is_ok()) + .collect(); + + if entries_raw.is_empty() { + // Cell (No, No) OR (No, Yes). + if legacy_present.is_empty() { + return Ok(Vec::new()); + } + // Legacy synthesis: build one `default` entry from flat vars. + // Same field-extraction logic AppConfig::from_env uses today + // for `storage.backend` / `storage.s3` / `storage.azure` / + // `storage.encryption`; centralised here so the new entry + // model and the legacy path stay bit-identical. + return Ok(vec![synthesize_default_from_legacy_vars()?]); + } + + // `_ENTRIES` is set. + if !legacy_present.is_empty() { + // Cell (Yes, Yes) — fail-fast. Name every legacy var found so + // the operator sees the exact cleanup list, not a generic + // "conflict" message. + let names: Vec = legacy_present.iter().map(|v| (**v).to_string()).collect(); + return Err(format!( + "OXICLOUD_STORAGE_ENTRIES is set, but legacy storage-backend env vars are also \ + present: [{}]. In multi-entry mode these flat vars are ignored — remove them \ + from your environment / .env, or migrate their values into per-entry \ + OXICLOUD_STORAGE__* form. See docs/plan/storage-multi-entry.md \ + §'Legacy flat-var interaction'.", + names.join(", ") + )); + } + + // Cell (Yes, No): validate the name list structure first (empty + // names, invalid chars, duplicates), THEN parse each entry's + // fields. Two-phase separation guarantees the operator sees the + // structural problem — "you have a typo in `_ENTRIES` itself" — + // before any per-entry field-missing message, which would be + // more confusing than helpful when the underlying issue is the + // list itself. + let raw_names: Vec<&str> = entries_raw.split(',').map(str::trim).collect(); + let mut names_seen: std::collections::HashSet = std::collections::HashSet::new(); + for name in &raw_names { + if name.is_empty() { + return Err(format!( + "OXICLOUD_STORAGE_ENTRIES contains an empty name (from `{entries_raw}`) — \ + commas must separate non-empty names." + )); + } + if !is_valid_entry_name(name) { + return Err(format!( + "OXICLOUD_STORAGE_ENTRIES: invalid entry name `{name}` — allowed characters \ + are lowercase ASCII letters, digits, `_`, `-`; length 1-32." + )); + } + if !names_seen.insert((*name).to_string()) { + return Err(format!( + "OXICLOUD_STORAGE_ENTRIES: duplicate name `{name}` — entry names must be unique." + )); + } + } + let mut out: Vec = Vec::with_capacity(raw_names.len()); + for name in raw_names { + out.push(parse_named_entry(name)?); + } + Ok(out) +} + +/// Read the per-entry env vars for `name` and build a +/// [`NamedStorageEntry`]. Called by [`parse_storage_entries`] for each +/// declared name. +fn parse_named_entry(name: &str) -> Result { + let backend_raw = env::var(format!("OXICLOUD_STORAGE_{name}_BACKEND")).map_err(|_| { + format!( + "OXICLOUD_STORAGE_{name}_BACKEND is missing — every entry declared in \ + OXICLOUD_STORAGE_ENTRIES must specify its backend type (local / s3 / azure)." + ) + })?; + let backend = match backend_raw.to_lowercase().as_str() { + "local" => StorageBackendType::Local, + "s3" => StorageBackendType::S3, + "azure" => StorageBackendType::Azure, + other => { + return Err(format!( + "OXICLOUD_STORAGE_{name}_BACKEND=`{other}` is not a known backend type — \ + expected one of: local, s3, azure." + )); + } + }; + + let mut root_dir = None; + let mut s3 = None; + let mut azure = None; + + match backend { + StorageBackendType::Local => { + // `_ROOT_DIR` optional; falls back to `OXICLOUD_STORAGE_PATH`, + // then to the framework default at build time. The fallback + // means a Local entry can be declared with just `_BACKEND=local` + // in .env — matches the friction level of the flat-var world. + root_dir = env::var(format!("OXICLOUD_STORAGE_{name}_ROOT_DIR")) + .ok() + .or_else(|| env::var("OXICLOUD_STORAGE_PATH").ok()); + } + StorageBackendType::S3 => { + let bucket = env::var(format!("OXICLOUD_STORAGE_{name}_S3_BUCKET")) + .map_err(|_| { + format!( + "OXICLOUD_STORAGE_{name}_S3_BUCKET is required when \ + OXICLOUD_STORAGE_{name}_BACKEND=s3." + ) + })? + .trim() + .to_string(); + if bucket.is_empty() { + return Err(format!( + "OXICLOUD_STORAGE_{name}_S3_BUCKET is empty — bucket name is required for S3 \ + backend entries." + )); + } + s3 = Some(S3StorageConfig { + endpoint_url: env::var(format!("OXICLOUD_STORAGE_{name}_S3_ENDPOINT_URL")).ok(), + bucket, + region: env::var(format!("OXICLOUD_STORAGE_{name}_S3_REGION")) + .unwrap_or_else(|_| "us-east-1".to_string()), + access_key: env::var(format!("OXICLOUD_STORAGE_{name}_S3_ACCESS_KEY")) + .unwrap_or_default(), + secret_key: env::var(format!("OXICLOUD_STORAGE_{name}_S3_SECRET_KEY")) + .unwrap_or_default(), + force_path_style: env::var(format!("OXICLOUD_STORAGE_{name}_S3_FORCE_PATH_STYLE")) + .map(|v| v.parse::().unwrap_or(false)) + .unwrap_or(false), + }); + } + StorageBackendType::Azure => { + let container = env::var(format!("OXICLOUD_STORAGE_{name}_AZURE_CONTAINER")) + .map_err(|_| { + format!( + "OXICLOUD_STORAGE_{name}_AZURE_CONTAINER is required when \ + OXICLOUD_STORAGE_{name}_BACKEND=azure." + ) + })? + .trim() + .to_string(); + if container.is_empty() { + return Err(format!( + "OXICLOUD_STORAGE_{name}_AZURE_CONTAINER is empty — container name is required \ + for Azure backend entries." + )); + } + azure = Some(AzureStorageConfig { + account_name: env::var(format!("OXICLOUD_STORAGE_{name}_AZURE_ACCOUNT_NAME")) + .unwrap_or_default(), + account_key: env::var(format!("OXICLOUD_STORAGE_{name}_AZURE_ACCOUNT_KEY")) + .unwrap_or_default(), + container, + sas_token: env::var(format!("OXICLOUD_STORAGE_{name}_AZURE_SAS_TOKEN")).ok(), + endpoint_url: env::var(format!("OXICLOUD_STORAGE_{name}_AZURE_ENDPOINT_URL")).ok(), + }); + } + } + + // Encryption key — presence implies enabled. Validate now (base64 + + // decoded length) so we fail at boot, not at first blob write. + let encryption_key_base64 = match env::var(format!("OXICLOUD_STORAGE_{name}_ENCRYPTION_KEY")) { + Ok(k) if !k.is_empty() => { + validate_encryption_key(name, &k)?; + Some(k) + } + _ => None, + }; + + Ok(NamedStorageEntry { + name: name.to_string(), + backend, + root_dir, + s3, + azure, + encryption_key_base64, + }) +} + +/// Build the synthesized `default` entry from the pre-multi-entry flat +/// vars. Called only when `_ENTRIES` is unset AND at least one legacy +/// storage-backend var is present. Preserves the exact field-population +/// logic `AppConfig::from_env` used to run inline. +fn synthesize_default_from_legacy_vars() -> Result { + let backend = match env::var("OXICLOUD_STORAGE_BACKEND") + .unwrap_or_default() + .to_lowercase() + .as_str() + { + "s3" => StorageBackendType::S3, + "azure" => StorageBackendType::Azure, + _ => StorageBackendType::Local, + }; + + let mut root_dir = None; + let mut s3 = None; + let mut azure = None; + + match backend { + StorageBackendType::Local => { + root_dir = env::var("OXICLOUD_STORAGE_PATH").ok(); + } + StorageBackendType::S3 => { + let bucket = env::var("OXICLOUD_S3_BUCKET").unwrap_or_default(); + if bucket.is_empty() { + return Err( + "OXICLOUD_STORAGE_BACKEND=s3 but OXICLOUD_S3_BUCKET is not set — legacy \ + synthesis requires the bucket name." + .to_string(), + ); + } + s3 = Some(S3StorageConfig { + endpoint_url: env::var("OXICLOUD_S3_ENDPOINT_URL").ok(), + bucket, + region: env::var("OXICLOUD_S3_REGION").unwrap_or_else(|_| "us-east-1".to_string()), + access_key: env::var("OXICLOUD_S3_ACCESS_KEY").unwrap_or_default(), + secret_key: env::var("OXICLOUD_S3_SECRET_KEY").unwrap_or_default(), + force_path_style: env::var("OXICLOUD_S3_FORCE_PATH_STYLE") + .map(|v| v.parse::().unwrap_or(false)) + .unwrap_or(false), + }); + } + StorageBackendType::Azure => { + let container = env::var("OXICLOUD_AZURE_CONTAINER").unwrap_or_default(); + if container.is_empty() { + return Err( + "OXICLOUD_STORAGE_BACKEND=azure but OXICLOUD_AZURE_CONTAINER is not set — \ + legacy synthesis requires the container name." + .to_string(), + ); + } + azure = Some(AzureStorageConfig { + account_name: env::var("OXICLOUD_AZURE_ACCOUNT_NAME").unwrap_or_default(), + account_key: env::var("OXICLOUD_AZURE_ACCOUNT_KEY").unwrap_or_default(), + container, + sas_token: env::var("OXICLOUD_AZURE_SAS_TOKEN").ok(), + endpoint_url: env::var("OXICLOUD_AZURE_ENDPOINT_URL").ok(), + }); + } + } + + let encryption_key_base64 = match env::var("OXICLOUD_STORAGE_ENCRYPTION_KEY") { + Ok(k) if !k.is_empty() => { + validate_encryption_key("default", &k)?; + Some(k) + } + _ => None, + }; + + Ok(NamedStorageEntry { + name: "default".to_string(), + backend, + root_dir, + s3, + azure, + encryption_key_base64, + }) +} + +/// Validate a base64-encoded 32-byte encryption key. Called at parse +/// time (not first-use) so a bad key aborts boot with a clear +/// per-entry message instead of failing much later inside +/// `EncryptedBlobBackend::new`. +fn validate_encryption_key(entry_name: &str, key_b64: &str) -> Result<(), String> { + use base64::Engine; + let decoded = base64::engine::general_purpose::STANDARD + .decode(key_b64) + .map_err(|e| { + format!("OXICLOUD_STORAGE_{entry_name}_ENCRYPTION_KEY is not valid base64: {e}") + })?; + if decoded.len() != 32 { + return Err(format!( + "OXICLOUD_STORAGE_{entry_name}_ENCRYPTION_KEY decodes to {} bytes; must be exactly \ + 32 bytes (AES-256). Generate a fresh key via \ + `POST /api/admin/settings/storage/generate-key`.", + decoded.len() + )); + } + Ok(()) +} + impl Default for StorageConfig { fn default() -> Self { // Architecture-appropriate max upload size to avoid overflow on 32-bit systems @@ -1378,8 +1764,37 @@ pub struct AppConfig { pub resources: ResourceConfig, /// Concurrency configuration pub concurrency: ConcurrencyConfig, - /// Storage configuration + /// Storage configuration. + /// + /// **Legacy flat-var surface.** Populated from `OXICLOUD_STORAGE_*`, + /// `OXICLOUD_S3_*`, `OXICLOUD_AZURE_*`, `OXICLOUD_STORAGE_ENCRYPTION_*`. + /// Represents the single-backend model that predates + /// `docs/plan/storage-multi-entry.md`. When that plan lands, + /// runtime boot reads `active_backend_name` from DB and picks + /// an entry from [`Self::storage_entries`] instead — but existing + /// deployments and code paths that still consult this field keep + /// working via the legacy-synthesis fallback (if `_ENTRIES` is + /// empty, one entry named `default` is synthesized from these + /// flat vars and mirrored here). pub storage: StorageConfig, + /// Named storage entries declared in `.env` via + /// `OXICLOUD_STORAGE_ENTRIES=name1,name2,...` plus per-entry + /// `OXICLOUD_STORAGE__*` env vars. See + /// `docs/plan/storage-multi-entry.md`. + /// + /// - Empty when neither `_ENTRIES` nor any legacy flat storage + /// var is set (fresh install without any explicit storage + /// config — the default is used, matching today's behaviour). + /// - Exactly one synthesized `default` entry when `_ENTRIES` is + /// unset/empty but legacy flat vars are present (upgrade + /// path — existing deployments keep working without touching + /// `.env`). + /// - N entries when `_ENTRIES` is set — one per name. + /// + /// Boot / migration / consistency-audit code looks entries up + /// by name in this vec. Order is preserved from `_ENTRIES` for + /// the "no active pointer yet, pick the first one" fallback. + pub storage_entries: Vec, /// Database configuration pub database: DatabaseConfig, /// Authentication configuration @@ -1448,6 +1863,7 @@ impl Default for AppConfig { resources: ResourceConfig::default(), concurrency: ConcurrencyConfig::default(), storage: StorageConfig::default(), + storage_entries: Vec::new(), database: DatabaseConfig::default(), auth: AuthConfig::default(), features: FeaturesConfig::default(), @@ -2095,6 +2511,20 @@ impl AppConfig { enabled.eq_ignore_ascii_case("true") || enabled == "1"; } + // Multi-entry storage config (docs/plan/storage-multi-entry.md). + // Parsed here so a bad config aborts boot with a clear error + // message before any downstream service tries to use it. + // Legacy synthesis is folded into the same call — when + // `_ENTRIES` is unset AND legacy flat vars are present, we get + // back a one-element vec named `default`. The flat-var block + // below still populates the legacy `config.storage.*` fields + // for any code path that hasn't migrated to reading from + // `storage_entries` yet — the two live side by side until + // slice 2 flips the boot backend to read from entries. + config.storage_entries = parse_storage_entries().unwrap_or_else(|e| { + panic!("Invalid storage configuration in environment: {e}"); + }); + // Storage backend selection if let Ok(backend) = env::var("OXICLOUD_STORAGE_BACKEND") { match backend.to_lowercase().as_str() { @@ -2499,4 +2929,322 @@ mod tests { assert!(!cfg.is_email_allowed("not-an-email")); assert!(!cfg.is_email_allowed("")); } + + // ── Multi-entry storage config parser tests + // + // These tests mutate process env. Cargo runs tests in parallel by + // default, so a shared Mutex serialises the env-touching sections. + // Every test acquires the guard, wipes the entire storage-related + // env surface, applies its own fixture, calls parse_storage_entries, + // and drops out. State is scrubbed by the pre-test wipe rather than + // any post-test cleanup — safer against a panic mid-test. + mod storage_entry_parser { + use super::*; + use std::sync::Mutex; + + static ENV_GUARD: Mutex<()> = Mutex::new(()); + + /// Every env var the parser reads. Wiped before each test so + /// leftover state from another test (or from the developer's + /// shell) can't influence the result. + const ALL_PARSER_VARS: &[&str] = &[ + "OXICLOUD_STORAGE_ENTRIES", + "OXICLOUD_STORAGE_PATH", + // Legacy flat vars + "OXICLOUD_STORAGE_BACKEND", + "OXICLOUD_S3_ENDPOINT_URL", + "OXICLOUD_S3_BUCKET", + "OXICLOUD_S3_REGION", + "OXICLOUD_S3_ACCESS_KEY", + "OXICLOUD_S3_SECRET_KEY", + "OXICLOUD_S3_FORCE_PATH_STYLE", + "OXICLOUD_AZURE_ACCOUNT_NAME", + "OXICLOUD_AZURE_ACCOUNT_KEY", + "OXICLOUD_AZURE_CONTAINER", + "OXICLOUD_AZURE_SAS_TOKEN", + "OXICLOUD_AZURE_ENDPOINT_URL", + "OXICLOUD_STORAGE_ENCRYPTION_ENABLED", + "OXICLOUD_STORAGE_ENCRYPTION_KEY", + // Common per-entry vars used in these tests + "OXICLOUD_STORAGE_local_main_BACKEND", + "OXICLOUD_STORAGE_local_main_ROOT_DIR", + "OXICLOUD_STORAGE_local_main_ENCRYPTION_KEY", + "OXICLOUD_STORAGE_s3_prod_BACKEND", + "OXICLOUD_STORAGE_s3_prod_S3_BUCKET", + "OXICLOUD_STORAGE_s3_prod_S3_ENDPOINT_URL", + "OXICLOUD_STORAGE_s3_prod_S3_REGION", + "OXICLOUD_STORAGE_s3_prod_S3_ACCESS_KEY", + "OXICLOUD_STORAGE_s3_prod_S3_SECRET_KEY", + "OXICLOUD_STORAGE_s3_prod_S3_FORCE_PATH_STYLE", + "OXICLOUD_STORAGE_s3_prod_ENCRYPTION_KEY", + "OXICLOUD_STORAGE_foo_BACKEND", + "OXICLOUD_STORAGE_bar_BACKEND", + "OXICLOUD_STORAGE_bar_S3_BUCKET", + ]; + + fn wipe_env() { + for k in ALL_PARSER_VARS { + // SAFETY: the ENV_GUARD mutex serialises all env + // mutation inside this module; the outer test harness + // makes no other calls to `set_var`/`remove_var` from + // concurrent threads for these keys. + unsafe { env::remove_var(k) }; + } + } + + fn set(k: &str, v: &str) { + unsafe { env::set_var(k, v) }; + } + + /// Base64 of 32 zero bytes — a syntactically-valid AES-256 key. + const VALID_KEY_B64: &str = "AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA="; + + // ── Cell (No, No) — no entries, no legacy vars + + #[test] + fn empty_no_legacy_returns_empty_vec() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + assert_eq!(parse_storage_entries().unwrap(), vec![]); + } + + // ── Cell (No, Yes) — legacy synthesis + + #[test] + fn legacy_local_synthesizes_default_entry() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_BACKEND", "local"); + set("OXICLOUD_STORAGE_PATH", "/data"); + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].name, "default"); + assert_eq!(entries[0].backend, StorageBackendType::Local); + assert_eq!(entries[0].root_dir.as_deref(), Some("/data")); + assert!(entries[0].s3.is_none()); + assert!(entries[0].encryption_key_base64.is_none()); + } + + #[test] + fn legacy_s3_synthesizes_default_entry() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_BACKEND", "s3"); + set("OXICLOUD_S3_BUCKET", "my-bucket"); + set("OXICLOUD_S3_REGION", "eu-west-1"); + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].name, "default"); + assert_eq!(entries[0].backend, StorageBackendType::S3); + let s3 = entries[0].s3.as_ref().unwrap(); + assert_eq!(s3.bucket, "my-bucket"); + assert_eq!(s3.region, "eu-west-1"); + } + + #[test] + fn legacy_encryption_key_carries_into_synthesized_default() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + // Only encryption var set — counts as legacy present, so + // synthesis fires. Backend defaults to Local. + set("OXICLOUD_STORAGE_ENCRYPTION_KEY", VALID_KEY_B64); + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!( + entries[0].encryption_key_base64.as_deref(), + Some(VALID_KEY_B64) + ); + } + + #[test] + fn legacy_s3_without_bucket_fails() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_BACKEND", "s3"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("OXICLOUD_S3_BUCKET"), "err was: {err}"); + } + + // ── Cell (Yes, No) — declared entries + + #[test] + fn two_entries_local_and_s3() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "local_main,s3_prod"); + set("OXICLOUD_STORAGE_local_main_BACKEND", "local"); + set("OXICLOUD_STORAGE_local_main_ROOT_DIR", "/srv/oxicloud"); + set("OXICLOUD_STORAGE_s3_prod_BACKEND", "s3"); + set("OXICLOUD_STORAGE_s3_prod_S3_BUCKET", "prod-bucket"); + + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0].name, "local_main"); + assert_eq!(entries[0].backend, StorageBackendType::Local); + assert_eq!(entries[0].root_dir.as_deref(), Some("/srv/oxicloud")); + assert_eq!(entries[1].name, "s3_prod"); + assert_eq!(entries[1].backend, StorageBackendType::S3); + assert_eq!(entries[1].s3.as_ref().unwrap().bucket, "prod-bucket"); + } + + #[test] + fn entries_order_preserved_from_env() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "bar,foo"); + set("OXICLOUD_STORAGE_bar_BACKEND", "s3"); + set("OXICLOUD_STORAGE_bar_S3_BUCKET", "b"); + set("OXICLOUD_STORAGE_foo_BACKEND", "local"); + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries[0].name, "bar"); + assert_eq!(entries[1].name, "foo"); + } + + #[test] + fn per_entry_encryption_key_parsed_and_kept() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "local_main"); + set("OXICLOUD_STORAGE_local_main_BACKEND", "local"); + set("OXICLOUD_STORAGE_local_main_ENCRYPTION_KEY", VALID_KEY_B64); + let entries = parse_storage_entries().unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!( + entries[0].encryption_key_base64.as_deref(), + Some(VALID_KEY_B64) + ); + } + + // ── Cell (Yes, Yes) — fail fast on conflict + + #[test] + fn entries_plus_legacy_vars_fails_and_lists_conflicts() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "local_main"); + set("OXICLOUD_STORAGE_local_main_BACKEND", "local"); + set("OXICLOUD_STORAGE_BACKEND", "s3"); + set("OXICLOUD_S3_BUCKET", "stale-bucket"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("OXICLOUD_STORAGE_ENTRIES"), "err was: {err}"); + assert!(err.contains("OXICLOUD_STORAGE_BACKEND"), "err was: {err}"); + assert!(err.contains("OXICLOUD_S3_BUCKET"), "err was: {err}"); + } + + // ── Name validation + + #[test] + fn uppercase_name_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "S3prod"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("S3prod"), "err was: {err}"); + } + + #[test] + fn empty_name_between_commas_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "foo,,bar"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("empty name"), "err was: {err}"); + } + + #[test] + fn over_length_name_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + let too_long = "a".repeat(33); + set("OXICLOUD_STORAGE_ENTRIES", &too_long); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains(&too_long), "err was: {err}"); + } + + #[test] + fn duplicate_names_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "foo,foo"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("duplicate name"), "err was: {err}"); + assert!(err.contains("foo"), "err was: {err}"); + } + + // ── Per-entry field validation + + #[test] + fn missing_backend_for_declared_entry_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "foo"); + let err = parse_storage_entries().unwrap_err(); + assert!( + err.contains("OXICLOUD_STORAGE_foo_BACKEND"), + "err was: {err}" + ); + } + + #[test] + fn s3_entry_missing_bucket_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "s3_prod"); + set("OXICLOUD_STORAGE_s3_prod_BACKEND", "s3"); + let err = parse_storage_entries().unwrap_err(); + assert!( + err.contains("OXICLOUD_STORAGE_s3_prod_S3_BUCKET"), + "err was: {err}" + ); + } + + #[test] + fn unknown_backend_type_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "foo"); + set("OXICLOUD_STORAGE_foo_BACKEND", "gcs"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("gcs"), "err was: {err}"); + assert!(err.contains("local, s3, azure"), "err was: {err}"); + } + + // ── Encryption key validation + + #[test] + fn invalid_base64_encryption_key_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "local_main"); + set("OXICLOUD_STORAGE_local_main_BACKEND", "local"); + set( + "OXICLOUD_STORAGE_local_main_ENCRYPTION_KEY", + "!!!not-base64!!!", + ); + let err = parse_storage_entries().unwrap_err(); + assert!( + err.contains("OXICLOUD_STORAGE_local_main_ENCRYPTION_KEY"), + "err was: {err}" + ); + assert!(err.contains("base64"), "err was: {err}"); + } + + #[test] + fn wrong_length_encryption_key_rejected() { + let _g = ENV_GUARD.lock().unwrap_or_else(|p| p.into_inner()); + wipe_env(); + set("OXICLOUD_STORAGE_ENTRIES", "local_main"); + set("OXICLOUD_STORAGE_local_main_BACKEND", "local"); + // "AAAA" decodes to 3 bytes — valid base64, wrong length. + set("OXICLOUD_STORAGE_local_main_ENCRYPTION_KEY", "AAAA"); + let err = parse_storage_entries().unwrap_err(); + assert!(err.contains("32 bytes"), "err was: {err}"); + } + } + + impl PartialEq for NamedStorageEntry { + fn eq(&self, other: &Self) -> bool { + self.name == other.name && self.backend == other.backend + } + } } diff --git a/src/common/di.rs b/src/common/di.rs index 79412ccc..9edecebb 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -215,46 +215,122 @@ impl AppServiceFactory { ); image_transcode_service.initialize().await?; - // Build blob storage backend based on configuration - let base_backend: Arc = match self.config.storage.backend { - StorageBackendType::S3 => { - let s3_config = self - .config - .storage - .s3 - .as_ref() - .expect("S3 config required when OXICLOUD_STORAGE_BACKEND=s3"); - Arc::new( - crate::infrastructure::services::s3_blob_backend::S3BlobBackend::new(s3_config), - ) - } - StorageBackendType::Azure => { - let az_config = self - .config - .storage - .azure - .as_ref() - .expect("Azure config required when OXICLOUD_STORAGE_BACKEND=azure"); - Arc::new( - crate::infrastructure::services::azure_blob_backend::AzureBlobBackend::new( - az_config, + // Build blob storage backend. + // + // Two paths, chosen by whether `_ENTRIES` (or the legacy + // synthesis) populated `storage_entries` at parse time: + // + // * `storage_entries` non-empty — multi-entry mode + // (`docs/plan/storage-multi-entry.md`). Look up the active + // entry name in `auth.admin_settings.storage.active_backend_name`, + // fall back to the first entry when unset (fresh install, + // no admin has picked one yet), fail-fast when the DB + // points at a name that isn't declared. Build via the + // shared `entry_backend::build_entry_backend` factory so + // the encryption decorator is applied uniformly here and + // in the migration handler (slice 3). + // + // * `storage_entries` empty — no explicit storage config at + // all (fresh install without env vars). Use the framework + // defaults captured on `config.storage` — matches today's + // behaviour so a bare `cargo run` in a dev workspace keeps + // working. Encryption never applies here (legacy synthesis + // would have created an entry if any legacy var was set). + let active_backend_kind: StorageBackendType; + let base_backend: Arc = if self.config.storage_entries.is_empty() { + active_backend_kind = self.config.storage.backend.clone(); + tracing::info!( + "Storage: no OXICLOUD_STORAGE_ENTRIES declared and no legacy vars — using \ + framework default (backend={:?}, path={:?})", + active_backend_kind, + self.storage_path, + ); + match active_backend_kind { + StorageBackendType::S3 => { + let s3_config = self + .config + .storage + .s3 + .as_ref() + .expect("S3 config required when OXICLOUD_STORAGE_BACKEND=s3"); + Arc::new( + crate::infrastructure::services::s3_blob_backend::S3BlobBackend::new( + s3_config, + ), + ) + } + StorageBackendType::Azure => { + let az_config = self + .config + .storage + .azure + .as_ref() + .expect("Azure config required when OXICLOUD_STORAGE_BACKEND=azure"); + Arc::new( + crate::infrastructure::services::azure_blob_backend::AzureBlobBackend::new( + az_config, + ), + ) + } + StorageBackendType::Local => Arc::new( + crate::infrastructure::services::local_blob_backend::LocalBlobBackend::new( + &self.storage_path, ), - ) - } - StorageBackendType::Local => Arc::new( - crate::infrastructure::services::local_blob_backend::LocalBlobBackend::new( - &self.storage_path, ), - ), + } + } else { + use crate::infrastructure::services::entry_backend::{ + ActiveEntry, build_entry_backend, resolve_active_entry, + }; + let active = resolve_active_entry(db_pool, &self.config.storage_entries) + .await + .unwrap_or_else(|e| { + panic!("Storage boot failed: {e}"); + }); + let entry = match active { + ActiveEntry::Explicit(e) => { + tracing::info!( + "Storage: booting on entry `{}` (from auth.admin_settings.storage.active_backend_name)", + e.name, + ); + e + } + ActiveEntry::Unset => { + // No admin pick yet. Fall back to the first entry + // in `_ENTRIES` order. `storage_entries` is + // guaranteed non-empty in this branch, so [0] is + // safe. Loud info-level log so operators see + // which entry was chosen for them. + let first = &self.config.storage_entries[0]; + tracing::info!( + "Storage: no active_backend_name set in DB — defaulting to first entry \ + `{}` (declared first in OXICLOUD_STORAGE_ENTRIES). Set explicitly via \ + the admin storage tab or `oxicloud --select-storage ` to pin.", + first.name, + ); + first + } + }; + active_backend_kind = entry.backend.clone(); + build_entry_backend(entry, &self.storage_path) }; - // Stack decorators: retry → encryption → cache (inner-to-outer) + // Stack decorators: retry → encryption → cache (inner-to-outer). + // + // Encryption is applied INSIDE build_entry_backend (per-entry + // key), so it's already on the base returned above when the + // entry declared one. The legacy-vars-no-entries branch skips + // encryption (that path exists only for zero-storage-config + // installs); if legacy synthesis fired it produced an entry + // and we're on the entry branch instead. + // + // Retry + cache are app-level (config.storage.retry/cache), not + // per-entry, so they still apply here. `active_backend_kind` + // gates the "remote-only" decorators the same as before. let mut blob_backend: Arc = base_backend; // Retry decorator (for remote backends) - if self.config.storage.retry.enabled - && self.config.storage.backend != StorageBackendType::Local - { + if self.config.storage.retry.enabled && active_backend_kind != StorageBackendType::Local { use crate::infrastructure::services::retry_blob_backend::{ RetryBlobBackend, RetryPolicy, }; @@ -272,8 +348,17 @@ impl AppServiceFactory { tracing::info!("Blob storage retry decorator enabled"); } - // Encryption decorator - if self.config.storage.encryption.enabled { + // Encryption decorator — legacy path only. + // + // When `storage_entries` is non-empty, encryption is already + // applied inside `build_entry_backend` from the entry's own + // `encryption_key_base64` (per-entry key). This block is the + // pre-multi-entry fallback that reads the flat + // `OXICLOUD_STORAGE_ENCRYPTION_*` vars — reachable only for + // fresh installs with no explicit storage config at all + // (legacy synthesis would have created an entry if any legacy + // var, including the encryption ones, was present). + if self.config.storage_entries.is_empty() && self.config.storage.encryption.enabled { use crate::infrastructure::services::encrypted_blob_backend::EncryptedBlobBackend; let key_b64 = self .config @@ -289,13 +374,11 @@ impl AppServiceFactory { "OXICLOUD_STORAGE_ENCRYPTION_KEY must be exactly 32 bytes (base64 of 32 bytes)", ); blob_backend = Arc::new(EncryptedBlobBackend::new(blob_backend, &key)); - tracing::info!("Blob storage encryption decorator enabled (AES-256-GCM)"); + tracing::info!("Blob storage encryption decorator enabled (AES-256-GCM) — legacy path"); } // Cache decorator (for remote backends only) - if self.config.storage.cache.enabled - && self.config.storage.backend != StorageBackendType::Local - { + if self.config.storage.cache.enabled && active_backend_kind != StorageBackendType::Local { use crate::infrastructure::services::cached_blob_backend::{ BlobCacheConfig as CacheCfg, CachedBlobBackend, }; diff --git a/src/infrastructure/services/entry_backend.rs b/src/infrastructure/services/entry_backend.rs new file mode 100644 index 00000000..f62af05e --- /dev/null +++ b/src/infrastructure/services/entry_backend.rs @@ -0,0 +1,180 @@ +//! Factory that builds a `BlobStorageBackend` from a `NamedStorageEntry`. +//! +//! Central to `docs/plan/storage-multi-entry.md`: the same function is +//! called by the boot path (to build the LIVE backend for the active +//! entry) and by the migration handler (to build a target backend for +//! any named entry). Keeping one factory means the encryption-decorator +//! wrapping decision is expressed exactly once — no chance of the +//! migration copy path silently omitting encryption while boot applies +//! it (or vice versa). +//! +//! What this factory does NOT do: +//! - Retry decorator — applied per-app-instance in `common/di.rs` +//! because policy comes from `AppConfig.storage.retry`, not the +//! entry. If per-entry retry becomes a need, add a +//! `RetryConfig` field to `NamedStorageEntry` and move the +//! wrapping in here. +//! - Cache decorator — same story: cache path/size are ambient +//! `AppConfig.storage.cache` settings, not per-entry. +//! +//! So the returned backend is `base [+ encryption]` — the two layers +//! whose choice is tied to the entry itself. The caller stacks any +//! remaining decorators. + +use std::path::{Path, PathBuf}; +use std::sync::Arc; + +use sqlx::PgPool; + +use crate::application::ports::blob_storage_ports::BlobStorageBackend; +use crate::common::config::{NamedStorageEntry, StorageBackendType}; + +/// Key in `auth.admin_settings` that holds the currently-active +/// storage entry's name. Single source of truth for runtime backend +/// selection (see `docs/plan/storage-multi-entry.md` §"One DB row"). +pub const ACTIVE_BACKEND_NAME_KEY: &str = "storage.active_backend_name"; + +/// Result of [`resolve_active_entry`]. +pub enum ActiveEntry<'a> { + /// DB has an `active_backend_name` set AND that name matches an + /// entry declared in the current env. Boot uses this entry. + Explicit(&'a NamedStorageEntry), + /// DB has NO `active_backend_name` set (fresh install, or the row + /// was intentionally cleared). Caller falls back to a sensible + /// default — typically the first entry in `_ENTRIES` order. + Unset, +} + +/// Look up the entry the app should boot with. +/// +/// Returns: +/// - `Ok(ActiveEntry::Explicit(entry))` when DB has a value AND that +/// value names an entry in `entries`. +/// - `Ok(ActiveEntry::Unset)` when the DB row is absent (never +/// written). Caller decides the fallback. +/// - `Err(msg)` when the DB row IS set but the named entry is missing +/// from the current env (deploy drift — someone removed an entry +/// from `.env` or renamed it). The error message names the missing +/// entry, lists the available ones, and points at the +/// `oxicloud --select-storage ` repair flag. Boot must abort +/// on this — silently falling back to a different entry would move +/// the app's live backend without operator consent. +pub async fn resolve_active_entry<'a>( + pool: &PgPool, + entries: &'a [NamedStorageEntry], +) -> Result, String> { + let stored: Option = + sqlx::query_scalar("SELECT value FROM auth.admin_settings WHERE key = $1") + .bind(ACTIVE_BACKEND_NAME_KEY) + .fetch_optional(pool) + .await + .map_err(|e| { + format!("reading `{ACTIVE_BACKEND_NAME_KEY}` from auth.admin_settings failed: {e}") + })?; + + match stored { + None => Ok(ActiveEntry::Unset), + Some(name) => match entries.iter().find(|e| e.name == name) { + Some(entry) => Ok(ActiveEntry::Explicit(entry)), + None => { + let available = if entries.is_empty() { + "(none — no OXICLOUD_STORAGE_ENTRIES declared)".to_string() + } else { + entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", ") + }; + Err(format!( + "auth.admin_settings.storage.active_backend_name = `{name}`, but no entry \ + with that name is declared in OXICLOUD_STORAGE_ENTRIES. Available: [{available}]. \ + Either add `{name}` back to your .env, or repair the DB pointer with:\n \ + oxicloud --select-storage " + )) + } + }, + } +} + +/// Build a `BlobStorageBackend` matching the given entry, with the +/// encryption decorator applied when the entry declares a key. +/// +/// `local_storage_path_fallback` is the ambient `AppConfig.storage_path` +/// — used for a Local entry when `entry.root_dir` is `None`. Matches +/// the fallback rule documented in +/// `docs/plan/storage-multi-entry.md` §Legacy: per-entry `_ROOT_DIR` +/// falls back to `OXICLOUD_STORAGE_PATH` for Local entries when unset. +/// +/// Panics with a targeted message on the two configuration errors that +/// slip past env-parse-time validation: +/// - S3 entry with `entry.s3 == None` — parser invariant violated. +/// - Encryption key that fails base64 / length validation — the parser +/// validates at env time, so hitting this means the entry was +/// constructed programmatically without going through +/// `parse_storage_entries`. +/// +/// Both are boot-fatal and indicate a code (not config) bug, so +/// panic is the honest response. +pub fn build_entry_backend( + entry: &NamedStorageEntry, + local_storage_path_fallback: &Path, +) -> Arc { + let base: Arc = match entry.backend { + StorageBackendType::Local => { + let path = entry + .root_dir + .as_ref() + .map(PathBuf::from) + .unwrap_or_else(|| local_storage_path_fallback.to_path_buf()); + Arc::new( + crate::infrastructure::services::local_blob_backend::LocalBlobBackend::new(&path), + ) + } + StorageBackendType::S3 => { + let s3 = entry.s3.as_ref().unwrap_or_else(|| { + panic!( + "entry `{}` has backend=s3 but no s3 config — parser invariant violated", + entry.name + ) + }); + Arc::new(crate::infrastructure::services::s3_blob_backend::S3BlobBackend::new(s3)) + } + StorageBackendType::Azure => { + let az = entry.azure.as_ref().unwrap_or_else(|| { + panic!( + "entry `{}` has backend=azure but no azure config — parser invariant violated", + entry.name + ) + }); + Arc::new(crate::infrastructure::services::azure_blob_backend::AzureBlobBackend::new(az)) + } + }; + + // Encryption decorator — presence-implies-enabled, per plan §Encryption. + let Some(key_b64) = entry.encryption_key_base64.as_ref() else { + return base; + }; + use crate::infrastructure::services::encrypted_blob_backend::EncryptedBlobBackend; + let key_bytes = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, key_b64) + .unwrap_or_else(|e| { + panic!( + "entry `{}` encryption key is not valid base64: {e} — parser was supposed to \ + catch this at boot", + entry.name + ) + }); + let key: [u8; 32] = key_bytes.try_into().unwrap_or_else(|v: Vec| { + panic!( + "entry `{}` encryption key decoded to {} bytes; must be 32 — parser was supposed to \ + catch this at boot", + entry.name, + v.len() + ) + }); + tracing::info!( + "Storage entry `{}` encrypted with AES-256-GCM (key from env)", + entry.name + ); + Arc::new(EncryptedBlobBackend::new(base, &key)) +} diff --git a/src/infrastructure/services/mod.rs b/src/infrastructure/services/mod.rs index 1565e65f..a9a782d6 100644 --- a/src/infrastructure/services/mod.rs +++ b/src/infrastructure/services/mod.rs @@ -10,6 +10,7 @@ pub mod db_pool_monitor; pub mod dedup_service; pub mod drives_consistency_service; pub mod encrypted_blob_backend; +pub mod entry_backend; pub mod exif_service; pub mod face_geometry; pub mod face_indexing_service; From 6b7bb67500070503a07ae69557d5a4f834ebbe5b Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 12:59:35 +0200 Subject: [PATCH 05/17] feat(storage): jobs can choose storage to migrate/scan --- src/application/dtos/settings_dto.rs | 17 ++ src/common/di.rs | 36 ++- src/infrastructure/scheduler/pg_job_store.rs | 31 +++ src/infrastructure/scheduler/recoverable.rs | 32 +++ src/infrastructure/scheduler/types.rs | 12 + .../services/storage_migration_service.rs | 248 +++++++++++++++--- src/interfaces/api/handlers/admin_handler.rs | 58 +++- 7 files changed, 385 insertions(+), 49 deletions(-) diff --git a/src/application/dtos/settings_dto.rs b/src/application/dtos/settings_dto.rs index 05f35b6f..fe04d4b2 100644 --- a/src/application/dtos/settings_dto.rs +++ b/src/application/dtos/settings_dto.rs @@ -257,9 +257,26 @@ pub struct MigrationStateDto { } /// Request body for `POST /api/admin/storage/migration/start`. +/// +/// **Multi-entry contract** (see `docs/plan/storage-multi-entry.md`): +/// `target_name` is REQUIRED — it names the storage entry the copy +/// job will move blobs INTO. The admin picks it from the entries +/// declared in `OXICLOUD_STORAGE_ENTRIES`. The trigger endpoint +/// rejects the request when the name doesn't exist or equals the +/// currently-active entry (no-op guard). #[derive(Debug, Serialize, Deserialize, ToSchema)] pub struct StartMigrationDto { + /// Name of the storage entry to migrate blobs INTO. Must be + /// present in `OXICLOUD_STORAGE_ENTRIES` and must differ from + /// the currently-active entry. + pub target_name: String, /// How many blobs to copy in parallel (default: 4). + /// + /// **Currently ignored** — the recoverable copy loop is + /// sequential (one blob at a time within the batch). Kept in + /// the DTO for wire-compat with the admin UI form; will be + /// honoured once per-batch fan-out lands (dual-write / + /// concurrent-copy future slice). pub concurrency: Option, } diff --git a/src/common/di.rs b/src/common/di.rs index 9edecebb..540c8119 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -237,8 +237,16 @@ impl AppServiceFactory { // working. Encryption never applies here (legacy synthesis // would have created an entry if any legacy var was set). let active_backend_kind: StorageBackendType; + // Track which named entry the LIVE backend was built from so + // the migration handler can name-compare `target != active` + // without re-reading the DB. `"legacy"` sentinel for the + // no-entries branch — the migration handler refuses that + // target name anyway (no entry exists), which is the correct + // behaviour for the zero-config path. + let active_backend_name: String; let base_backend: Arc = if self.config.storage_entries.is_empty() { active_backend_kind = self.config.storage.backend.clone(); + active_backend_name = "legacy".to_string(); tracing::info!( "Storage: no OXICLOUD_STORAGE_ENTRIES declared and no legacy vars — using \ framework default (backend={:?}, path={:?})", @@ -312,6 +320,7 @@ impl AppServiceFactory { } }; active_backend_kind = entry.backend.clone(); + active_backend_name = entry.name.clone(); build_entry_backend(entry, &self.storage_path) }; @@ -549,6 +558,7 @@ impl AppServiceFactory { job_registry, job_store_provider, blob_backend: blob_backend_for_consistency, + active_backend_name, }) } @@ -2167,13 +2177,11 @@ impl AppServiceFactory { tracing::info!("Storage settings service initialized"); // 9b-1c. Register the storage-backend migration tenant on - // the recoverable-run engine. Must run AFTER the storage - // settings service is built — the tenant resolves the - // *target* backend at each run start by asking the settings - // service for the currently-effective config. Source is - // whatever `dedup_service` booted with; both live on - // `AppState.core`. On-demand only (no periodic tick — an - // operator triggers a copy after switching backend config). + // the recoverable-run engine. Target is resolved by NAME + // from `params.target_name` on each run — plumbed from + // the trigger endpoint. Constructor takes the ambient + // entries snapshot + active-name + storage_path fallback + // so no DB read is needed per run for target lookup. let job_store_provider_dyn: Arc< dyn crate::infrastructure::scheduler::JobStoreProvider, > = app_state.core.job_store_provider.clone(); @@ -2184,7 +2192,9 @@ impl AppServiceFactory { .clone() .expect("maintenance_pool set above"), app_state.core.blob_backend.clone(), - storage_settings_svc, + app_state.core.active_backend_name.clone(), + app_state.core.config.storage_entries.clone(), + self.storage_path.clone(), ), ) .register_recoverable_job(&app_state.core.job_registry, &job_store_provider_dyn) @@ -2436,6 +2446,16 @@ pub struct CoreServices { /// `build_app_state` — can probe `blob_exists()` / re-hash bytes /// through the same stack DedupService uses. pub blob_backend: Arc, + /// Name of the storage entry the LIVE `blob_backend` was built + /// from. Populated at boot: either from + /// `admin_settings.storage.active_backend_name` when set, or the + /// first entry in `OXICLOUD_STORAGE_ENTRIES` when unset. For the + /// no-entries legacy path this is `"default"` (the synthesized + /// name) or `"legacy"` (framework-defaults case with zero storage + /// config at all). Migration handler consumes this to enforce the + /// "target != active" no-op guard by name; without needing to + /// re-read DB on every trigger. + pub active_backend_name: String, } /// Container for repository services diff --git a/src/infrastructure/scheduler/pg_job_store.rs b/src/infrastructure/scheduler/pg_job_store.rs index 64762458..a8a3436c 100644 --- a/src/infrastructure/scheduler/pg_job_store.rs +++ b/src/infrastructure/scheduler/pg_job_store.rs @@ -198,6 +198,37 @@ impl JobStore for PgJobStore { Ok(()) } + async fn set_string_param(&self, key: &str, value: &str) -> Result<(), DomainError> { + sqlx::query( + r#" + UPDATE jobs.recoverable_runs + SET params = jsonb_set(params, ARRAY[$2], to_jsonb($3::text)) + WHERE id = $1 + "#, + ) + .bind(self.run_id) + .bind(key) + .bind(value) + .execute(self.pool.as_ref()) + .await + .map_err(|e| map_sqlx_err("set_string_param", e))?; + Ok(()) + } + + async fn get_string_param(&self, key: &str) -> Result, DomainError> { + // `params -> $2` returns JSONB; `->>` returns text (null when + // key absent OR value isn't a string). Handler treats absence + // and null-value identically — either way it's "not set." + let row: Option<(Option,)> = + sqlx::query_as("SELECT params ->> $2 FROM jobs.recoverable_runs WHERE id = $1") + .bind(self.run_id) + .bind(key) + .fetch_optional(self.pool.as_ref()) + .await + .map_err(|e| map_sqlx_err("get_string_param", e))?; + Ok(row.and_then(|(v,)| v)) + } + async fn mark_completed(&self) -> Result<(), DomainError> { sqlx::query( r#" diff --git a/src/infrastructure/scheduler/recoverable.rs b/src/infrastructure/scheduler/recoverable.rs index eef85e55..dd1a44c5 100644 --- a/src/infrastructure/scheduler/recoverable.rs +++ b/src/infrastructure/scheduler/recoverable.rs @@ -246,6 +246,24 @@ pub trait JobStore: Send + Sync { async fn seed_progress_params(&self, total: u64, kind: ProgressKind) -> Result<(), DomainError>; + /// Set an arbitrary string field on `params` (JSONB). Used by + /// handlers on a Fresh run to persist per-run configuration that + /// must survive a mid-run restart — e.g. `storage_migration` + /// stamping `params.target_name` at run start so a resume can + /// pick up the same target without the admin re-specifying it. + /// + /// Handler-callable (unlike `seed_progress_params`, which is + /// engine-only). Idempotent: re-writing the same value is a + /// no-op UPDATE. + async fn set_string_param(&self, key: &str, value: &str) -> Result<(), DomainError>; + + /// Read a string field from `params` (JSONB). Returns `None` when + /// the key is absent or its value isn't a JSON string. Paired + /// with [`Self::set_string_param`] — handlers on a Resumed run + /// use this to recover per-run config that a prior Fresh open + /// stamped. + async fn get_string_param(&self, key: &str) -> Result, DomainError>; + /// Persist one finding to `jobs.run_findings` and bump /// `stats.finding_count` on the parent run. Consistency handlers /// call this in place of the transitional @@ -838,6 +856,7 @@ mod tests { findings: Vec, progress_total: Option, progress_kind: Option, + string_params: std::collections::HashMap, } #[async_trait] @@ -886,6 +905,17 @@ mod tests { s.progress_kind = Some(kind); Ok(()) } + async fn set_string_param(&self, key: &str, value: &str) -> Result<(), DomainError> { + self.state + .lock() + .unwrap() + .string_params + .insert(key.to_string(), value.to_string()); + Ok(()) + } + async fn get_string_param(&self, key: &str) -> Result, DomainError> { + Ok(self.state.lock().unwrap().string_params.get(key).cloned()) + } async fn mark_completed(&self) -> Result<(), DomainError> { self.state.lock().unwrap().status = RunStatus::Completed; Ok(()) @@ -937,6 +967,7 @@ mod tests { findings: Vec::new(), progress_total: None, progress_kind: None, + string_params: std::collections::HashMap::new(), }), }); let id = store.run_id; @@ -995,6 +1026,7 @@ mod tests { findings: Vec::new(), progress_total: None, progress_kind: None, + string_params: std::collections::HashMap::new(), }), }); stores.push(store.clone()); diff --git a/src/infrastructure/scheduler/types.rs b/src/infrastructure/scheduler/types.rs index 396686c3..7a3c8a9a 100644 --- a/src/infrastructure/scheduler/types.rs +++ b/src/infrastructure/scheduler/types.rs @@ -34,10 +34,22 @@ use serde::{Deserialize, Serialize}; /// - `storage_consistency` (future) — enables per-blob re-BLAKE3 (bitrot /// detection) + mime sniff alongside the fast orphan check. /// - Others — ignored. +/// +/// Semantics of `storage`, per job (added for the multi-entry storage +/// design — see `docs/plan/storage-multi-entry.md`): +/// - `storage_migration` — the NAME of the target storage entry to +/// copy blobs INTO. Required on a Fresh run (handler refuses +/// without it); ignored on a Resumed run (target read from the +/// persisted `params.target_name`). +/// - `blobs_consistency` / `backend_consistency` (slice 7) — the NAME +/// of the entry to probe instead of the currently-active backend. +/// `None` falls through to the live backend (today's behaviour). +/// - Others — ignored. #[derive(Debug, Clone, Default)] pub struct JobRunArgs { pub force: bool, pub deep: bool, + pub storage: Option, } /// Uniform outcome the supervisor logs and stores for every job dispatch. diff --git a/src/infrastructure/services/storage_migration_service.rs b/src/infrastructure/services/storage_migration_service.rs index 4f8b0979..eb5f6461 100644 --- a/src/infrastructure/services/storage_migration_service.rs +++ b/src/infrastructure/services/storage_migration_service.rs @@ -46,7 +46,7 @@ //! cancel + cursor discipline; the batch loop is I/O-bound anyway. //! Add concurrency later if a real throughput need appears. -use std::path::Path; +use std::path::{Path, PathBuf}; use std::sync::Arc; use async_trait::async_trait; @@ -54,15 +54,24 @@ use futures::StreamExt; use sqlx::PgPool; use crate::application::ports::blob_storage_ports::BlobStorageBackend; -use crate::application::services::storage_settings_service::StorageSettingsService; +use crate::common::config::NamedStorageEntry; use crate::common::errors::DomainError; use crate::infrastructure::scheduler::{ JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome, RunStatus, record_or_log, }; +use crate::infrastructure::services::entry_backend::build_entry_backend; pub const STORAGE_MIGRATION_JOB_NAME: &str = "storage_migration"; +/// The `params` JSONB key under which the run's target entry name is +/// stashed at Fresh-open time via `JobStore::set_string_param`. +/// Handlers re-read it on Resume so a paused run survives a restart +/// without the admin re-specifying the target. Exposed publicly so +/// the trigger endpoint's audit lines and the admin UI's run-detail +/// projections read the same constant. +pub const TARGET_NAME_PARAM: &str = "target_name"; + /// Rows per batch. Copies are I/O-bound (source read + target write); /// larger batches amortise fewer SQL round-trips but the checkpoint /// / cancel-poll cadence lengthens. 100 balances the two — one @@ -72,20 +81,39 @@ const BATCH_SIZE: i64 = 100; pub struct StorageMigrationService { pool: Arc, + /// Backend the running app is bound to — the migration COPIES + /// FROM this. Set once at boot and never changes for the + /// process's lifetime (cutover requires a restart, per plan). source: Arc, - storage_settings: Arc, + /// Name of the currently-active entry (i.e. the one `source` + /// corresponds to). Used to refuse a same-name target at run + /// start. Same reasoning as `source` — locked at boot. + active_backend_name: String, + /// All entries declared in env, held as a snapshot for name + /// lookup during migration. Immutable per-deploy — matches + /// `AppConfig.storage_entries`. + storage_entries: Vec, + /// Ambient `AppConfig.storage_path` used as the `root_dir` + /// fallback for a Local target entry that doesn't declare its + /// own `_ROOT_DIR`. Same fallback rule as boot + /// (`build_entry_backend`). + storage_path_fallback: PathBuf, } impl StorageMigrationService { pub fn new( pool: Arc, source: Arc, - storage_settings: Arc, + active_backend_name: String, + storage_entries: Vec, + storage_path_fallback: PathBuf, ) -> Self { Self { pool, source, - storage_settings, + active_backend_name, + storage_entries, + storage_path_fallback, } } @@ -133,47 +161,152 @@ impl RecoverableJobHandler for StorageMigrationService { async fn run_resumable( &self, store: &dyn JobStore, - _args: &JobRunArgs, + args: &JobRunArgs, resume_cursor: Option>, ) -> RunOutcome { - // No-op guard — refuse when the effective (target) config - // points at the same physical storage as the source (boot - // config). Without this, a misclick on an S3 deployment - // issues one HEAD per blob for zero useful work — cheap on - // local, expensive on remote. Same-type-different-location - // migrations (local dir change, S3 bucket change) pass this - // check and proceed normally. - match self.storage_settings.is_source_target_identical().await { - Ok(true) => { - tracing::warn!( - target: "audit", - event = "storage_migration.refused_noop", - run_id = %store.run_id(), - "storage_migration refused: source and target point at the same storage" - ); + // Resolve the target entry NAME. Two paths: + // + // * Fresh run — `args.storage` MUST be Some (the trigger + // endpoint enforces this at the HTTP layer). Handler + // stamps it into `params.target_name` so a mid-run restart + // can resume without re-input. + // * Resumed run — `args.storage` is typically None (admin + // just clicked Run on a Paused row). Handler reads the + // target from `params.target_name` written on the + // original Fresh open. + // + // A Fresh run without `args.storage` is a client bug — refuse + // rather than default to something and quietly copy blobs + // into the wrong entry. + let is_fresh = resume_cursor.is_none(); + let target_name = if is_fresh { + let Some(name) = args.storage.clone() else { return RunOutcome::Failed { message: - "target equals source; change storage settings before triggering a migration" + "storage_migration requires `target_name` on a fresh run — trigger via \ + POST /api/admin/storage/migration/start with `{\"target_name\": \"\"}`." .to_string(), }; - } - Ok(false) => {} - Err(e) => { + }; + if let Err(e) = store.set_string_param(TARGET_NAME_PARAM, &name).await { return RunOutcome::Failed { - message: format!("identity check: {e}"), + message: format!("failed to persist target_name to params: {e}"), }; } + name + } else { + match store.get_string_param(TARGET_NAME_PARAM).await { + Ok(Some(name)) => name, + Ok(None) => { + return RunOutcome::Failed { + message: format!( + "resumed run has no {TARGET_NAME_PARAM} in params — cannot infer \ + target. Likely a Paused row from before the multi-entry migration \ + refactor; cancel + trigger fresh." + ), + }; + } + Err(e) => { + return RunOutcome::Failed { + message: format!("read {TARGET_NAME_PARAM} from params: {e}"), + }; + } + } + }; + + // First-line guard: target name equals the currently-active + // entry. Silent no-op if we let it through — the app would + // walk every blob and skip because `target.blob_exists` is + // trivially true (target = live source). Even on the same + // local disk that's a lot of syscalls for no reason; on S3 + // it costs one HEAD per blob for zero copies. + if target_name == self.active_backend_name { + tracing::warn!( + target: "audit", + event = "storage_migration.refused_noop", + run_id = %store.run_id(), + target_name = %target_name, + active = %self.active_backend_name, + "storage_migration refused: target equals the currently-active entry" + ); + return RunOutcome::Failed { + message: format!( + "target entry `{target_name}` is the currently-active entry — nothing to \ + migrate. Pick a different target." + ), + }; } - // Resolve target at run start. - let target = match self.storage_settings.build_effective_backend().await { - Ok(t) => t, - Err(e) => { + // Look up the target entry by name. + let target_entry = match self.storage_entries.iter().find(|e| e.name == target_name) { + Some(e) => e, + None => { + let available = if self.storage_entries.is_empty() { + "(none)".to_string() + } else { + self.storage_entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", ") + }; return RunOutcome::Failed { - message: format!("resolve target backend: {e}"), + message: format!( + "target entry `{target_name}` not found in OXICLOUD_STORAGE_ENTRIES. \ + Available: [{available}]. If the entry was removed from .env since this \ + run started, restore it or cancel this run." + ), }; } }; + let source_entry = self + .storage_entries + .iter() + .find(|e| e.name == self.active_backend_name); + + // Second-line guard: physical-identity check for the + // encryption-differs case. Two entries with different names + // can still point at the same physical bucket (only their + // encryption key differs). That's the "in-place encryption + // rotation" case — refused because reads during migration + // would fail (LIVE backend uses K1, storage is being + // overwritten with K2). See plan §Encryption "Proper + // in-place rotation" for the deferred fix. The compare uses + // `entry_identity` (backend + physical location, EXCLUDING + // encryption key). + if let Some(source) = source_entry + && entry_identity(source) == entry_identity(target_entry) + { + let key_differs = source.encryption_key_base64 != target_entry.encryption_key_base64; + let hint = if key_differs { + " (encryption key differs → this looks like an in-place key rotation; \ + create a new entry pointing at a DIFFERENT bucket / dir, migrate to it, \ + then move back if desired)" + } else { + "" + }; + tracing::warn!( + target: "audit", + event = "storage_migration.refused_same_physical_storage", + run_id = %store.run_id(), + target_name = %target_name, + source_name = %self.active_backend_name, + encryption_differs = key_differs, + "storage_migration refused: named target differs from source but physical storage matches" + ); + return RunOutcome::Failed { + message: format!( + "target entry `{target_name}` names a different entry than the active \ + `{}`, but they point at the same physical storage{hint}.", + self.active_backend_name, + ), + }; + } + + // Build target backend via the shared factory — same code + // path boot uses, so the encryption decorator wrapping is + // uniform. + let target = build_entry_backend(target_entry, &self.storage_path_fallback); if let Err(e) = target.initialize().await { return RunOutcome::Failed { message: format!("target backend init: {e}"), @@ -186,10 +319,13 @@ impl RecoverableJobHandler for StorageMigrationService { target: "audit", event = "storage_migration.run_started", run_id = %store.run_id(), - source = source_kind, - target = target_kind, - resuming = resume_cursor.is_some(), - "storage_migration starting {source_kind} → {target_kind}" + source_name = %self.active_backend_name, + target_name = %target_name, + source_kind = source_kind, + target_kind = target_kind, + resuming = !is_fresh, + "storage_migration starting {} ({source_kind}) → {target_name} ({target_kind})", + self.active_backend_name, ); // Cursor = the last-visited blob hash, UTF-8-encoded. On resume @@ -424,6 +560,48 @@ impl RecoverableJobHandler for StorageMigrationService { } } +/// Physical-storage identity string for a `NamedStorageEntry`. Two +/// entries with the same identity point at the same physical +/// location (same disk dir, same S3 bucket, same Azure container) +/// regardless of encryption key or credentials. Used by the second- +/// line refusal in `run_resumable` to catch in-place encryption +/// rotation attempts (same physical storage, K1 → K2 → reads-during- +/// migration break). See `docs/plan/storage-multi-entry.md` §Encryption. +/// +/// Deliberately excludes: +/// - Encryption key — otherwise same-bucket-different-key would look +/// like a legit migration, hiding the corruption. +/// - Credentials — two entries with different access keys pointing +/// at the same bucket ARE the same physical storage. +/// - Region for S3 — the bucket URI is the primary key; region is a +/// routing hint (though endpoint_url is included since it changes +/// the actual host bytes land on). +fn entry_identity(entry: &NamedStorageEntry) -> String { + use crate::common::config::StorageBackendType; + match entry.backend { + StorageBackendType::Local => { + format!("local:{}", entry.root_dir.as_deref().unwrap_or("")) + } + StorageBackendType::S3 => match entry.s3.as_ref() { + Some(s3) => format!( + "s3:{}/{}:path_style={}", + s3.endpoint_url.as_deref().unwrap_or("aws"), + s3.bucket, + s3.force_path_style, + ), + None => "s3:".to_string(), + }, + StorageBackendType::Azure => match entry.azure.as_ref() { + Some(az) => format!( + "azure:{}/{}", + az.account_name.as_str(), + az.container.as_str() + ), + None => "azure:".to_string(), + }, + } +} + /// Copy one blob: stream source bytes to a temp file, then hand the /// path to `target.put_blob`. The spool-through-disk shape matches /// what the old `migration_job::copy_blob` did — some backends' diff --git a/src/interfaces/api/handlers/admin_handler.rs b/src/interfaces/api/handlers/admin_handler.rs index eb172bda..a0925a97 100644 --- a/src/interfaces/api/handlers/admin_handler.rs +++ b/src/interfaces/api/handlers/admin_handler.rs @@ -454,9 +454,36 @@ pub async fn get_migration_status( )] pub async fn start_migration( State(state): State>, - Json(_dto): Json, + Json(dto): Json, ) -> Result { - trigger_storage_migration(state).await + // Validate at the HTTP layer (before spawning) so unknown / no-op + // targets get a synchronous 400 response instead of burning a + // failed run row. The handler's own checks are second-line + // defence for the resume path where args aren't repeated. + let entries = &state.core.config.storage_entries; + let active = &state.core.active_backend_name; + if entries.iter().all(|e| e.name != dto.target_name) { + let available = if entries.is_empty() { + "(none)".to_string() + } else { + entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", ") + }; + return Err(AppError::bad_request(format!( + "unknown target entry `{}`. Available: [{available}]", + dto.target_name + ))); + } + if dto.target_name == *active { + return Err(AppError::bad_request(format!( + "target `{}` is the currently-active entry — pick a different entry to migrate to", + dto.target_name + ))); + } + trigger_storage_migration(state, Some(dto.target_name)).await } /// POST /api/admin/storage/migration/pause — pause a running migration. @@ -528,7 +555,11 @@ pub async fn pause_migration( pub async fn resume_migration( State(state): State>, ) -> Result { - trigger_storage_migration(state).await + // Resume path — no target_name in the body. The handler reads + // it from `params.target_name` stamped on the original Fresh + // open. Refuses gracefully via RunOutcome::Failed if there is + // no Paused row to resume. + trigger_storage_migration(state, None).await } /// POST /api/admin/storage/migration/verify — post-migration integrity check. @@ -590,6 +621,7 @@ pub async fn verify_migration( /// itself is fire-and-forget. async fn trigger_storage_migration( state: Arc, + target_name: Option, ) -> Result { use crate::infrastructure::scheduler::JobRunArgs; use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME; @@ -597,14 +629,17 @@ async fn trigger_storage_migration( tracing::info!( target: "audit", event = "storage_migration.trigger_requested", + target_name = target_name.as_deref().unwrap_or(""), "👮🏻‍♂️ Admin triggered storage_migration" ); let registry = state.core.job_registry.clone(); + let args = JobRunArgs { + storage: target_name, + ..JobRunArgs::default() + }; tokio::spawn(async move { - registry - .trigger(STORAGE_MIGRATION_JOB_NAME, &JobRunArgs::default()) - .await; + registry.trigger(STORAGE_MIGRATION_JOB_NAME, &args).await; }); Ok(( @@ -2203,6 +2238,16 @@ pub struct TriggerJobQuery { pub force: bool, #[serde(default)] pub deep: bool, + /// Optional named storage entry to scope the run against — used by + /// tenants that respect `JobRunArgs.storage` (currently + /// `storage_migration` for its target; `blobs_consistency` / + /// `backend_consistency` will pick this up in slice 7 to probe a + /// non-active entry). Ignored by tenants that don't declare a + /// semantic for it. Unknown-name validation is per-tenant — the + /// generic trigger endpoint doesn't cross-check against + /// `AppConfig.storage_entries`. + #[serde(default)] + pub storage: Option, } /// `POST /api/admin/jobs/{name}/trigger` — dispatch one run off-schedule. @@ -2249,6 +2294,7 @@ pub async fn trigger_job( let args = JobRunArgs { force: query.force, deep: query.deep, + storage: query.storage.clone(), }; // Jobs that can run for hours (storage_migration, future From 2de71b6d9a2ce10935e95e01c4daaae9d423aaca Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 13:27:49 +0200 Subject: [PATCH 06/17] feat(storage): add readonly during storage migration --- examples/bench_favorites_authz.rs | 1 + examples/bench_range_seek_authz.rs | 1 + examples/bench_round12_queries.rs | 1 + examples/bench_round24_zip_authz.rs | 1 + examples/bench_thumbnail_cascade_cache.rs | 1 + src/application/services/folder_service.rs | 9 +- src/common/di.rs | 143 +++++++++++++++++- src/infrastructure/services/entry_backend.rs | 78 ++++++++++ src/infrastructure/services/pg_acl_engine.rs | 47 ++++++ .../services/storage_migration_service.rs | 143 +++++++++++++++--- 10 files changed, 400 insertions(+), 25 deletions(-) diff --git a/examples/bench_favorites_authz.rs b/examples/bench_favorites_authz.rs index c97cce8f..335bb1bc 100644 --- a/examples/bench_favorites_authz.rs +++ b/examples/bench_favorites_authz.rs @@ -186,6 +186,7 @@ fn fresh_engine(pool: &Arc) -> Arc { folder_repo, file_repo, group_repo, + Arc::new(std::sync::atomic::AtomicBool::new(false)), )) } diff --git a/examples/bench_range_seek_authz.rs b/examples/bench_range_seek_authz.rs index 50d5b700..82dfd2eb 100644 --- a/examples/bench_range_seek_authz.rs +++ b/examples/bench_range_seek_authz.rs @@ -176,6 +176,7 @@ fn fresh_engine(pool: &Arc) -> Arc { folder_repo, file_repo, group_repo, + Arc::new(std::sync::atomic::AtomicBool::new(false)), )) } diff --git a/examples/bench_round12_queries.rs b/examples/bench_round12_queries.rs index 3f3d8b53..cb7f25c5 100644 --- a/examples/bench_round12_queries.rs +++ b/examples/bench_round12_queries.rs @@ -769,6 +769,7 @@ fn wopi_engine(pool: &Arc) -> (Arc, Arc) -> (Arc, Arc) -> Arc { folder_repo, file_repo, group_repo, + Arc::new(std::sync::atomic::AtomicBool::new(false)), )) } diff --git a/src/application/services/folder_service.rs b/src/application/services/folder_service.rs index c2db8125..d8365127 100644 --- a/src/application/services/folder_service.rs +++ b/src/application/services/folder_service.rs @@ -1810,6 +1810,7 @@ mod mount_authz_integration { Arc::new(FolderDbRepository::new(pool.clone())), Arc::new(FileBlobReadRepository::new_stub()), Arc::new(SubjectGroupPgRepository::new(pool.clone())), + Arc::new(std::sync::atomic::AtomicBool::new(false)), )) } @@ -2377,7 +2378,13 @@ mod cascade_hook_integration_tests { folder_repo.clone(), )); let group_repo = Arc::new(SubjectGroupPgRepository::new(pool.clone())); - Arc::new(PgAclEngine::new(pool, folder_repo, file_repo, group_repo)) + Arc::new(PgAclEngine::new( + pool, + folder_repo, + file_repo, + group_repo, + Arc::new(std::sync::atomic::AtomicBool::new(false)), + )) } /// Seed a file row under `folder_id`. `blob_hash` is just a string — diff --git a/src/common/di.rs b/src/common/di.rs index 540c8119..7903da7b 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -1487,11 +1487,34 @@ impl AppServiceFactory { let subject_group_repo = Arc::new( crate::infrastructure::repositories::pg::SubjectGroupPgRepository::new(pool.clone()), ); + // Migration-readonly atomic. Seeded from + // `admin_settings.storage.migration_readonly` so the flag + // survives restart (an operator won't see writes accidentally + // re-enabled between a crash mid-migration and the retrigger). + // Shared with the AuthZ engine so it can short-circuit write + // permissions without a per-check DB round-trip. The boot + // clear rule (§Read-only mode) runs after this seeding, after + // the boot recovery sweep — enough for the runtime state + // machine to decide whether to keep or clear. + let migration_readonly = Arc::new(std::sync::atomic::AtomicBool::new( + crate::infrastructure::services::entry_backend::load_migration_readonly(&pool).await, + )); + if migration_readonly.load(std::sync::atomic::Ordering::Relaxed) { + tracing::warn!( + target: "oxicloud::scheduler", + event = "storage.migration_readonly.loaded_true_at_boot", + "Server booted with migration_readonly=true — writes will be refused by AuthZ \ + until the flag is cleared (either by the boot-clear rule or via the admin \ + storage tab)." + ); + } + let authorization = build_authorization_engine( pool.clone(), repos.folder_repository.clone(), repos.file_read_repository.clone(), subject_group_repo.clone(), + migration_readonly.clone(), ); // Recent service + recording hook are built up-front so the @@ -1979,6 +2002,7 @@ impl AppServiceFactory { webdav_dead_props: crate::infrastructure::services::webdav_dead_property_store::create_dead_property_store(pool.clone()), authorization: authorization.clone(), + migration_readonly: migration_readonly.clone(), drive_repo: drive_repo.clone(), drive_management_service: Arc::new( crate::application::services::drive_management_service::DriveManagementService::new( @@ -2195,6 +2219,7 @@ impl AppServiceFactory { app_state.core.active_backend_name.clone(), app_state.core.config.storage_entries.clone(), self.storage_path.clone(), + app_state.migration_readonly.clone(), ), ) .register_recoverable_job(&app_state.core.job_registry, &job_store_provider_dyn) @@ -2393,6 +2418,105 @@ impl AppServiceFactory { ), } + // Migration-readonly boot-clear rule. See + // `docs/plan/storage-multi-entry.md` §"Read-only mode". + // + // If the flag was set true at boot AND no storage_migration + // run is currently non-terminal AND active_backend_name + // matches the entry the app actually booted onto — that means + // the cutover completed on a prior boot (the run reached + // Completed, the pointer flipped, the operator restarted). + // Safe to clear now: no in-flight migration means no one + // still needs writes-off, and matching active_backend_name + // means we're already on the target the run was pointing at. + // + // If ANY of those conditions fails (flag was false at boot; + // there's still a Paused/Running/CancelRequested run in the + // way; active doesn't match booted — mismatch means someone + // manually edited the pointer while readonly was on) we + // leave the flag alone. Operator has to decide. + if app_state + .migration_readonly + .load(std::sync::atomic::Ordering::Relaxed) + { + use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME; + let has_in_flight = match app_state + .core + .job_store_provider + .list_runs(STORAGE_MIGRATION_JOB_NAME, 5) + .await + { + Ok(runs) => runs.iter().any(|r| { + matches!( + r.status, + crate::infrastructure::scheduler::RunStatus::Running + | crate::infrastructure::scheduler::RunStatus::Paused + | crate::infrastructure::scheduler::RunStatus::CancelRequested + ) + }), + Err(e) => { + tracing::warn!( + target: "oxicloud::scheduler", + event = "storage.migration_readonly.clear_check_failed", + error = %e, + "failed to list storage_migration runs during readonly-clear check; \ + leaving migration_readonly flag as-is" + ); + // Play it safe: assume in-flight to avoid clearing prematurely. + true + } + }; + + // Look up the DB pointer to compare against the booted + // active_backend_name. Absence (Unset) is treated as "no + // mismatch to complain about" — the boot fallback already + // picked the first entry. + let db_active_matches = { + use crate::infrastructure::services::entry_backend::{ + ActiveEntry, resolve_active_entry, + }; + match resolve_active_entry(&pool, &app_state.core.config.storage_entries).await { + Ok(ActiveEntry::Explicit(e)) => e.name == app_state.core.active_backend_name, + Ok(ActiveEntry::Unset) => true, + Err(_) => false, + } + }; + + if !has_in_flight && db_active_matches { + use crate::infrastructure::services::entry_backend::persist_migration_readonly; + match persist_migration_readonly(&pool, false).await { + Ok(()) => { + app_state + .migration_readonly + .store(false, std::sync::atomic::Ordering::Relaxed); + tracing::info!( + target: "audit", + event = "storage.migration_readonly.cleared_at_boot", + active = %app_state.core.active_backend_name, + "🧊 migration_readonly cleared at boot: no in-flight migration + \ + active_backend_name matches booted entry (cutover complete on prior boot)" + ); + } + Err(e) => tracing::warn!( + target: "oxicloud::scheduler", + event = "storage.migration_readonly.clear_persist_failed", + error = %e, + "cleared migration_readonly in memory would have been safe, but the DB \ + write failed — leaving the DB row alone; will re-check next boot" + ), + } + } else { + tracing::info!( + target: "oxicloud::scheduler", + event = "storage.migration_readonly.retained_at_boot", + has_in_flight = has_in_flight, + db_active_matches = db_active_matches, + "migration_readonly retained at boot (in-flight run and/or active-name \ + mismatch prevents auto-clear)" + ); + } + } + // Start the periodic-job scheduler AFTER every native service has // finished registering its jobs on `core.job_registry`. Starting // it earlier would race the first tick against late registrations. @@ -2589,6 +2713,16 @@ pub struct AppState { /// an enum dispatcher or `Arc` (with /// `async_trait` boxing). pub authorization: Arc, + /// Global "server is in migration read-only mode" flag, shared + /// with [`Self::authorization`] so it can short-circuit write + /// permissions. Backed by + /// `admin_settings.storage.migration_readonly` for restart + /// survival. Slice 5's cutover state machine flips this atomic + /// (via `Ordering::Relaxed`) and calls + /// `entry_backend::persist_migration_readonly` to keep DB and + /// memory in sync. See `docs/plan/storage-multi-entry.md` + /// §"Read-only mode". + pub migration_readonly: Arc, /// Drive entity repository — `GET /api/drives`, the personal-drive /// lifecycle hook, and (post-D2) shared-drive creation flow all read /// through this. Backing table is `storage.drives`; membership is @@ -2731,6 +2865,7 @@ fn build_authorization_engine( crate::infrastructure::repositories::pg::file_blob_read_repository::FileBlobReadRepository, >, group_repo: Arc, + migration_readonly: Arc, ) -> Arc { use crate::infrastructure::services::pg_acl_engine::PgAclEngine; @@ -2742,7 +2877,13 @@ fn build_authorization_engine( "OXICLOUD_AUTHZ_ENGINE={other:?} is not yet supported. Only 'postgres' is implemented; leave the variable unset to use the default." ); } - Arc::new(PgAclEngine::new(pool, folder_repo, file_repo, group_repo)) + Arc::new(PgAclEngine::new( + pool, + folder_repo, + file_repo, + group_repo, + migration_readonly, + )) } /// Pair returned by [`build_email_sender`] when wiring DI: the diff --git a/src/infrastructure/services/entry_backend.rs b/src/infrastructure/services/entry_backend.rs index f62af05e..fa427f21 100644 --- a/src/infrastructure/services/entry_backend.rs +++ b/src/infrastructure/services/entry_backend.rs @@ -34,6 +34,84 @@ use crate::common::config::{NamedStorageEntry, StorageBackendType}; /// selection (see `docs/plan/storage-multi-entry.md` §"One DB row"). pub const ACTIVE_BACKEND_NAME_KEY: &str = "storage.active_backend_name"; +/// Key in `auth.admin_settings` that holds the persistent-across-restart +/// migration-readonly flag. See +/// `docs/plan/storage-multi-entry.md` §"Read-only mode reuses the +/// existing AuthZ short-circuit". Value is `"true"` or `"false"` +/// (plain text; the settings table stores strings). +pub const MIGRATION_READONLY_KEY: &str = "storage.migration_readonly"; + +/// Read the persisted `migration_readonly` flag from `admin_settings`. +/// Absent row / parse failure / DB error all resolve to `false` — the +/// safer default when we can't determine the intent, since a false +/// value only means "writes allowed by AuthZ" not "migration is +/// running." Called once at boot to seed the in-memory `AtomicBool`. +pub async fn load_migration_readonly(pool: &PgPool) -> bool { + let row: Result,)>, sqlx::Error> = + sqlx::query_as("SELECT value FROM auth.admin_settings WHERE key = $1") + .bind(MIGRATION_READONLY_KEY) + .fetch_optional(pool) + .await; + match row { + Ok(Some((Some(v),))) => matches!(v.to_lowercase().as_str(), "true" | "1"), + Ok(_) => false, + Err(e) => { + tracing::warn!( + target: "oxicloud::scheduler", + event = "storage.migration_readonly.load_failed", + error = %e, + "failed to read {MIGRATION_READONLY_KEY} at boot; defaulting to false" + ); + false + } + } +} + +/// Persist the `migration_readonly` flag. Idempotent — upserts the +/// `admin_settings` row. Called by the cutover state machine (slice 5) +/// when a migration starts (set true) or completes cleanly across a +/// restart (set false via the boot clear rule). Handler / trigger +/// callers should also update the in-memory `AtomicBool` alongside +/// this call to keep the two in sync. +pub async fn persist_migration_readonly(pool: &PgPool, value: bool) -> Result<(), sqlx::Error> { + sqlx::query( + r#" + INSERT INTO auth.admin_settings (key, value, category, is_secret) + VALUES ($1, $2, 'storage', FALSE) + ON CONFLICT (key) + DO UPDATE SET value = EXCLUDED.value, updated_at = NOW() + "#, + ) + .bind(MIGRATION_READONLY_KEY) + .bind(if value { "true" } else { "false" }) + .execute(pool) + .await?; + Ok(()) +} + +/// Persist the `active_backend_name` pointer. Called by the migration +/// handler on `RunOutcome::Completed` to flip the runtime backend to +/// the just-migrated target entry. The next boot reads this via +/// `resolve_active_entry` and picks the new entry for the LIVE +/// backend; before the restart the process is still on the OLD +/// backend (that's what the `migration_readonly` gate is protecting). +/// Idempotent UPSERT. +pub async fn persist_active_backend_name(pool: &PgPool, name: &str) -> Result<(), sqlx::Error> { + sqlx::query( + r#" + INSERT INTO auth.admin_settings (key, value, category, is_secret) + VALUES ($1, $2, 'storage', FALSE) + ON CONFLICT (key) + DO UPDATE SET value = EXCLUDED.value, updated_at = NOW() + "#, + ) + .bind(ACTIVE_BACKEND_NAME_KEY) + .bind(name) + .execute(pool) + .await?; + Ok(()) +} + /// Result of [`resolve_active_entry`]. pub enum ActiveEntry<'a> { /// DB has an `active_backend_name` set AND that name matches an diff --git a/src/infrastructure/services/pg_acl_engine.rs b/src/infrastructure/services/pg_acl_engine.rs index f48d785c..2e55431e 100644 --- a/src/infrastructure/services/pg_acl_engine.rs +++ b/src/infrastructure/services/pg_acl_engine.rs @@ -253,6 +253,18 @@ pub struct PgAclEngine { /// Total parent-resolution queries actually issued (point + batches) — /// exposed via [`Self::parent_query_count`] for benches/operators. parent_queries: Arc, + /// Global "server is in migration read-only mode" flag. When + /// `true`, `check_inner` short-circuits every write-adjacent + /// permission (`Create`/`Update`/`Delete`/`Share`/`Comment`/`Manage`) + /// with a `Denied` decision — same reason as the per-drive + /// `read_only` gate below, but scoped to the whole process rather + /// than a specific drive. Backed by + /// `admin_settings.storage.migration_readonly` so it survives + /// restart (see `docs/plan/storage-multi-entry.md` §"Read-only + /// mode"). Shared as `Arc` with `AppState` so the + /// cutover state machine (slice 5) can flip it without needing + /// to reach into the engine. + migration_readonly: Arc, } /// One parked parent-resolution request: file id + reply slot. A dropped @@ -297,12 +309,14 @@ impl PgAclEngine { folder_repo: Arc, file_repo: Arc, group_repo: Arc, + migration_readonly: Arc, ) -> Self { Self { pool, folder_repo, file_repo, group_repo: Some(group_repo), + migration_readonly, user_groups_cache: Cache::builder() .max_capacity(50_000) .time_to_live(Duration::from_secs(30)) @@ -424,6 +438,7 @@ impl PgAclEngine { .build(), parent_batch: Arc::new(std::sync::Mutex::new(None)), parent_queries: Arc::new(AtomicU64::new(0)), + migration_readonly: Arc::new(std::sync::atomic::AtomicBool::new(false)), } } @@ -1461,6 +1476,38 @@ impl PgAclEngine { resource: Resource, counters: &QueryCounters, ) -> Result { + // Global migration-readonly short-circuit. Applies to every + // resource type — no drive lookup, no per-resource state. When + // the server is in migration read-only mode, every mutating + // permission is refused with an audit line naming the specific + // `migration_readonly` reason so operators filtering the audit + // stream can distinguish it from per-drive freezes. Reads pass + // (browsers, downloads, PROPFIND all keep working — same as the + // per-drive gate). Admin operations don't reach `check_inner` + // — they go through `admin_guard` middleware which bypasses + // authz entirely, so the admin can still exit the mode, cancel + // the migration, restart the server, etc. + // + // See `docs/plan/storage-multi-entry.md` §"Read-only mode". + if Self::read_only_gate_applies(permission) + && self + .migration_readonly + .load(std::sync::atomic::Ordering::Relaxed) + { + tracing::info!( + target: "audit", + event = "authz.denied", + reason = "migration_readonly", + subject_type = subject.type_str(), + subject_id = %subject.id(), + permission = permission.as_str(), + resource_type = resource.type_str(), + resource_id = %resource.id(), + "🚧 mutation refused: server is in storage-migration read-only mode", + ); + return Ok(false); + } + // Drive-membership precheck for File/Folder. A role on the resource's // drive is the baseline floor (`drive.md §5`): the caller passes any // permission check the role bundle covers. Replaces the legacy diff --git a/src/infrastructure/services/storage_migration_service.rs b/src/infrastructure/services/storage_migration_service.rs index eb5f6461..e0be47ab 100644 --- a/src/infrastructure/services/storage_migration_service.rs +++ b/src/infrastructure/services/storage_migration_service.rs @@ -48,6 +48,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; use async_trait::async_trait; use futures::StreamExt; @@ -60,7 +61,9 @@ use crate::infrastructure::scheduler::{ JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome, RunStatus, record_or_log, }; -use crate::infrastructure::services::entry_backend::build_entry_backend; +use crate::infrastructure::services::entry_backend::{ + build_entry_backend, persist_active_backend_name, persist_migration_readonly, +}; pub const STORAGE_MIGRATION_JOB_NAME: &str = "storage_migration"; @@ -98,15 +101,26 @@ pub struct StorageMigrationService { /// own `_ROOT_DIR`. Same fallback rule as boot /// (`build_entry_backend`). storage_path_fallback: PathBuf, + /// Shared `AppState.migration_readonly` handle. Handler flips + /// this atomic (and persists to DB) at run start once all + /// guards pass, so writes across the whole app get refused by + /// the AuthZ short-circuit for the duration of the copy. Kept + /// ON when Completed — the boot-clear rule (slice 4) resets it + /// on the next restart after cutover, so operators can't + /// accidentally re-enable writes on the OLD backend while the + /// pointer already says the NEW one is active. + migration_readonly: Arc, } impl StorageMigrationService { + #[allow(clippy::too_many_arguments)] pub fn new( pool: Arc, source: Arc, active_backend_name: String, storage_entries: Vec, storage_path_fallback: PathBuf, + migration_readonly: Arc, ) -> Self { Self { pool, @@ -114,6 +128,7 @@ impl StorageMigrationService { active_backend_name, storage_entries, storage_path_fallback, + migration_readonly, } } @@ -313,6 +328,37 @@ impl RecoverableJobHandler for StorageMigrationService { }; } + // All guards passed. Engage server-wide read-only mode for + // the duration of the copy so new writes can't create blobs + // the migration walk has already stepped past. Both DB and + // in-memory atomic get flipped in lock-step. Idempotent under + // resume — the row is already `true` from the original open + // (survived a restart via slice 4's boot seed), but rewriting + // it doesn't hurt. + // + // A DB persist failure aborts before any copy — we won't + // silently proceed with writes-allowed. If the atomic write + // succeeded but DB failed we'd still have writes-off in this + // process, but a restart mid-migration would lose it. Fail + // early instead so operators see the actual DB problem. + if let Err(e) = persist_migration_readonly(self.pool.as_ref(), true).await { + return RunOutcome::Failed { + message: format!( + "engage migration_readonly (persist): {e} — refusing to copy without the \ + write freeze in place" + ), + }; + } + self.migration_readonly.store(true, Ordering::Relaxed); + tracing::info!( + target: "audit", + event = "storage.migration_readonly.engaged", + run_id = %store.run_id(), + target_name = %target_name, + "🚧 migration_readonly engaged: writes across the whole app are refused until \ + cutover completes and the operator restarts" + ); + let source_kind = self.source.backend_type(); let target_kind = target.backend_type(); tracing::info!( @@ -403,17 +449,16 @@ impl RecoverableJobHandler for StorageMigrationService { }; if rows.is_empty() { - tracing::info!( - target: "oxicloud::migration", - event = "storage_migration.completed", - run_id = %store.run_id(), - copied = copied_count, - skipped = skipped_count, - failed = failed_count, - source_missing = source_missing_count, - "storage_migration completed" - ); - return RunOutcome::Completed; + return self + .finish_completed( + store, + &target_name, + copied_count, + skipped_count, + failed_count, + source_missing_count, + ) + .await; } for (hash, size) in &rows { @@ -544,22 +589,74 @@ impl RecoverableJobHandler for StorageMigrationService { } if (rows.len() as i64) < BATCH_SIZE { - tracing::info!( - target: "oxicloud::migration", - event = "storage_migration.completed", - run_id = %store.run_id(), - copied = copied_count, - skipped = skipped_count, - failed = failed_count, - source_missing = source_missing_count, - "storage_migration completed" - ); - return RunOutcome::Completed; + return self + .finish_completed( + store, + &target_name, + copied_count, + skipped_count, + failed_count, + source_missing_count, + ) + .await; } } } } +impl StorageMigrationService { + /// Terminal successful path — reached from both Completed sites + /// in the batch loop (empty-first-batch and short-batch). Flips + /// the runtime `active_backend_name` pointer to the target entry + /// so the NEXT boot picks it up. Leaves `migration_readonly` ON + /// — the boot-clear rule (slice 4) drops it after the operator + /// restart when no in-flight run remains AND the DB pointer + /// matches the entry the app booted onto. + /// + /// Pointer-write failure is FATAL to the outcome. Reporting + /// `Completed` while the DB still says the old entry is active + /// would strand the migrated bytes: the next boot would come up + /// on the OLD backend (writes to old!), while the operator + /// thinks cutover is done. `Failed` keeps the situation legible: + /// admin sees the error, can retry the pointer write, then + /// restart. + #[allow(clippy::too_many_arguments)] + async fn finish_completed( + &self, + store: &dyn JobStore, + target_name: &str, + copied: u64, + skipped: u64, + failed: u64, + source_missing: u64, + ) -> RunOutcome { + if let Err(e) = persist_active_backend_name(self.pool.as_ref(), target_name).await { + return RunOutcome::Failed { + message: format!( + "copy finished but writing active_backend_name = `{target_name}` to \ + admin_settings failed: {e}. Bytes are on the target; retrigger the run \ + once the DB is reachable and it will short-circuit on already-present \ + blobs and re-attempt the pointer flip." + ), + }; + } + tracing::info!( + target: "audit", + event = "storage_migration.completed", + run_id = %store.run_id(), + active_backend_name = target_name, + previous_active = %self.active_backend_name, + copied = copied, + skipped = skipped, + failed = failed, + source_missing = source_missing, + "✅ storage_migration completed — active_backend_name = `{target_name}`. Restart the \ + server to switch the live backend (migration_readonly stays ON until then)." + ); + RunOutcome::Completed + } +} + /// Physical-storage identity string for a `NamedStorageEntry`. Two /// entries with the same identity point at the same physical /// location (same disk dir, same S3 bucket, same Azure container) From d7c19570a506a403f6cdc806dc4ed76722613a68 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 13:54:00 +0200 Subject: [PATCH 07/17] feat(storage): wire choice of storage --- frontend/src/lib/api/endpoints/admin.test.ts | 11 +- frontend/src/lib/api/endpoints/admin.ts | 75 +++-- .../src/routes/admin/[[tab]]/+page.svelte | 266 ++++++++++++------ src/application/dtos/settings_dto.rs | 65 ++++- .../services/storage_settings_service.rs | 68 ++++- src/common/di.rs | 14 +- .../services/backend_consistency_service.rs | 113 +++++++- .../services/blobs_consistency_service.rs | 114 +++++++- src/interfaces/api/handlers/admin_handler.rs | 131 +-------- src/interfaces/api/mod.rs | 3 +- 10 files changed, 574 insertions(+), 286 deletions(-) diff --git a/frontend/src/lib/api/endpoints/admin.test.ts b/frontend/src/lib/api/endpoints/admin.test.ts index 1df88227..5c4a6977 100644 --- a/frontend/src/lib/api/endpoints/admin.test.ts +++ b/frontend/src/lib/api/endpoints/admin.test.ts @@ -104,15 +104,8 @@ describe('admin test/probe endpoints', () => { await expect(admin.testStorage({ backend: 's3' })).resolves.toMatchObject({ connected: true }); }); - it('verifyMigration fills defaults and throws on error', async () => { - fetchMock.mockResolvedValue(okRes({ passed: true })); - await expect(admin.verifyMigration(10)).resolves.toMatchObject({ - passed: true, - sample_checked: 0 - }); - fetchMock.mockResolvedValue(errRes(500, {})); - await expect(admin.verifyMigration()).rejects.toThrow(/verify failed/); - }); + // verifyMigration retired in slice 7 of docs/plan/storage-multi-entry.md. + // Superseded by `POST /api/admin/jobs/blobs_consistency/trigger?storage=`. it('installPlugin posts a FormData bundle', async () => { fetchMock.mockResolvedValue(okRes({ id: 'com.example.hello' })); diff --git a/frontend/src/lib/api/endpoints/admin.ts b/frontend/src/lib/api/endpoints/admin.ts index 1f773c75..b15c5bae 100644 --- a/frontend/src/lib/api/endpoints/admin.ts +++ b/frontend/src/lib/api/endpoints/admin.ts @@ -466,6 +466,15 @@ export function saveOidc(body: Record): Promise { // ── Storage settings + migration ─────────────────────────────────────────── +export interface StorageEntrySummary { + name: string; + backend: string; + is_active: boolean; + encryption_enabled: boolean; + /** Human-readable physical hint (root_dir / bucket / container). */ + location_hint?: string | null; +} + export interface StorageSettings { backend: string; s3_endpoint_url?: string | null; @@ -479,6 +488,11 @@ export interface StorageSettings { total_blobs?: number; total_bytes_stored?: number; dedup_ratio?: number; + // Multi-entry view (slice 6 of docs/plan/storage-multi-entry.md). + // `entries` is empty for the legacy zero-entries path. + entries?: StorageEntrySummary[]; + active_entry_name?: string; + migration_readonly?: boolean; } export function getStorageSettings(): Promise { @@ -512,50 +526,33 @@ export function getMigration(): Promise { return apiJson('/api/admin/storage/migration', { credentials: 'same-origin' }); } -export function migrationAction(action: 'start' | 'pause' | 'resume'): Promise { +export function migrationAction( + action: 'start' | 'pause' | 'resume', + targetName?: string +): Promise { // `complete` was retired when the migration became a recoverable // job — Completed is the terminal `RunSummary.status`; there's - // nothing left to acknowledge. Post-migration cutover now happens - // via .env + restart, prompted by an inline hint on the admin - // storage tab (see `cutoverPending` in +page.svelte). - const body = action === 'start' ? { concurrency: 4 } : {}; + // nothing left to acknowledge. Post-migration cutover happens on + // operator restart (server re-boots on active_backend_name = the + // new entry; the boot-clear rule drops migration_readonly). + // + // `start` REQUIRES `targetName` in multi-entry mode — the backend + // rejects an unnamed start with 400 (see StartMigrationDto). + // `pause` and `resume` take no body (resume reads target_name + // from the paused run's params). + const body: Record = + action === 'start' ? { target_name: targetName ?? '', concurrency: 4 } : {}; return mutate(`/api/admin/storage/migration/${action}`, 'POST', body); } -/** Result of a `verify` integrity check (POST .../migration/verify). */ -export interface MigrationVerifyResult { - passed: boolean; - sample_checked: number; - pg_blob_count: number; - missing_in_target: string[]; - size_mismatches: string[]; -} - -/** - * Run an integrity verification pass over a sample of migrated blobs. Unlike - * the other migration actions this returns a structured result that the caller - * renders (passed / sample-checked / missing / size-mismatch counts). - */ -export async function verifyMigration(sampleSize = 100): Promise { - const res = await apiFetch('/api/admin/storage/migration/verify', { - method: 'POST', - credentials: 'same-origin', - headers: { ...JSON_HEADERS, ...getCsrfHeaders() }, - body: JSON.stringify({ sample_size: sampleSize }) - }); - if (!res.ok) { - const e = (await res.json().catch(() => ({}))) as { message?: string }; - throw new Error(e.message || `verify failed: ${res.status}`); - } - const r = (await res.json()) as Partial; - return { - passed: r.passed ?? false, - sample_checked: r.sample_checked ?? 0, - pg_blob_count: r.pg_blob_count ?? 0, - missing_in_target: r.missing_in_target ?? [], - size_mismatches: r.size_mismatches ?? [] - }; -} +// verifyMigration + MigrationVerifyResult retired in slice 7 of +// docs/plan/storage-multi-entry.md — the corresponding backend +// endpoint's sample-based check is superseded by +// `POST /api/admin/jobs/blobs_consistency/trigger?storage=`, +// which does a full walk against any named entry and integrates +// with the standard runs / findings admin surface. Trigger from +// the Jobs tab; the Storage tab drops the "Verify integrity" +// button. // ── Plugins ───────────────────────────────────────────────────────────── diff --git a/frontend/src/routes/admin/[[tab]]/+page.svelte b/frontend/src/routes/admin/[[tab]]/+page.svelte index e599fb1f..8b28a792 100644 --- a/frontend/src/routes/admin/[[tab]]/+page.svelte +++ b/frontend/src/routes/admin/[[tab]]/+page.svelte @@ -35,7 +35,6 @@ setUserRole, testOidc, testStorage, - verifyMigration, createExternalMount, deleteExternalMount, listExternalMounts, @@ -44,7 +43,6 @@ type AdminDashboard, type GeneratedKey, type MigrationStatus, - type MigrationVerifyResult, type OidcSettings, type OidcTestResult, type PluginInfo, @@ -519,15 +517,39 @@ stopMigrationPoll(); } } - async function doMigration(action: 'start' | 'pause' | 'resume') { + async function doMigration(action: 'start' | 'pause' | 'resume', targetName?: string) { try { - await migrationAction(action); + await migrationAction(action, targetName); await loadMigration(); } catch (e) { reportError(e); } } + // ── Multi-entry migration target picker (slice 6) ──────────────── + // + // Multi-entry mode requires the admin to name the target entry + // before starting a migration. Backend rejects an unnamed start + // with 400. Dropdown shows every non-active entry; picking one + // enables the Start button. + let migrationTarget = $state(''); + const availableTargets = $derived( + (storage?.entries ?? []).filter((e) => !e.is_active).map((e) => e.name) + ); + // Sync target when the entries list first appears — pick the first + // non-active entry by default so the operator can just click Start + // on a simple two-entry setup. + $effect(() => { + if (!migrationTarget && availableTargets.length > 0) { + migrationTarget = availableTargets[0]; + } + // Also unset when the previously-chosen target became active + // (cutover completed under our feet). + if (migrationTarget && !availableTargets.includes(migrationTarget)) { + migrationTarget = availableTargets[0] ?? ''; + } + }); + // ── Post-migration .env cutover hint ───────────────────────────── // // Migration copies blobs to the target backend, but boot-time @@ -595,22 +617,11 @@ } } - // Migration integrity verification (separate result panel). - let verifyResult = $state(null); - let verifyError = $state(null); - let verifying = $state(false); - async function doVerify() { - verifying = true; - verifyResult = null; - verifyError = null; - try { - verifyResult = await verifyMigration(100); - } catch (e) { - verifyError = errorMessage(e); - } finally { - verifying = false; - } - } + // Migration integrity verification retired in slice 7 — the + // sample-based /storage/migration/verify endpoint is replaced by + // `POST /api/admin/jobs/blobs_consistency/trigger?storage=`, + // a full walk. Operators trigger it from the Jobs tab. + const migrationPct = $derived( migration && migration.total_blobs > 0 ? Math.round((migration.migrated_blobs / migration.total_blobs) * 100) @@ -2147,6 +2158,62 @@

{t('admin.migration', 'Storage migration')}

+ + {#if storage?.entries && storage.entries.length > 0} + {#if storage.migration_readonly} +
+

+ + {t('admin.mig_readonly_title', 'Server in migration read-only mode')} +

+

+ {t( + 'admin.mig_readonly_body', + 'All writes (upload, rename, delete, share) are refused by AuthZ until the migration completes and you restart the server. Reads (browse, download) are unaffected.' + )} +

+
+ {/if} + + + + + + + + + + + + {#each storage.entries as entry (entry.name)} + + + + + + + + {/each} + +
{t('admin.entry_name', 'Entry')}{t('admin.entry_backend', 'Backend')}{t('admin.entry_location', 'Location')}{t('admin.entry_encryption', 'Encryption')}{t('admin.entry_status', 'Status')}
{entry.name}{entry.backend}{entry.location_hint ?? '—'} + {#if entry.encryption_enabled} + AES-256 + {:else} + — + {/if} + + {#if entry.is_active} + {t('admin.entry_active', 'active')} + {:else} + {t('admin.entry_inactive', 'available')} + {/if} +
+ {/if} {#if !migration}

{t('common.loading', 'Loading…')}

{:else} @@ -2179,13 +2246,46 @@ {/if}
- + {#if migration.status !== 'running' && migration.status !== 'paused' && migration.status !== 'completed'} - + {#if storage?.entries && storage.entries.length > 0} + + + {:else} + + {/if} {/if} {#if migration.status === 'running'} {/if} - - {#if migration.status === 'completed'} - - {/if} +
{#if cutoverPending} @@ -2265,55 +2353,6 @@
{/if} - - {#if verifyError} -
- {verifyError} -
- {:else if verifyResult} -
- - - {verifyResult.passed - ? t('admin.mig_verify_passed', 'Verification passed') - : t('admin.mig_verify_failed', 'Verification failed')} - - {#if verifyResult.passed} -

- {t( - 'admin.mig_verify_summary', - { checked: verifyResult.sample_checked, total: verifyResult.pg_blob_count }, - '{{checked}} blobs checked, {{total}} total in database' - )} -

- {:else} -

- {[ - verifyResult.missing_in_target.length - ? t( - 'admin.mig_verify_missing', - { n: verifyResult.missing_in_target.length }, - '{{n}} missing' - ) - : '', - verifyResult.size_mismatches.length - ? t( - 'admin.mig_verify_mismatch', - { n: verifyResult.size_mismatches.length }, - '{{n}} size mismatches' - ) - : '' - ] - .filter(Boolean) - .join(', ')} -

- {/if} -
- {/if} {/if} @@ -4226,6 +4265,49 @@ flex-wrap: wrap; } + .cutover-hint--readonly { + border-color: var(--color-danger-border, var(--color-border)); + background: var(--color-danger-bg, var(--color-bg-muted)); + } + + .entries-table { + width: 100%; + margin-bottom: var(--space-3); + border-collapse: collapse; + font-size: var(--text-sm, 0.875rem); + } + + .entries-table th, + .entries-table td { + padding: var(--space-2) var(--space-3); + border-bottom: 1px solid var(--color-border); + text-align: left; + } + + .entries-table th { + font-weight: 600; + color: var(--color-text-muted); + } + + .entries-table tr.entry-active { + background: var(--color-bg-muted); + } + + .migration-target-picker { + display: inline-flex; + align-items: center; + gap: var(--space-2); + margin-right: var(--space-2); + } + + .migration-target-picker select { + padding: var(--space-1) var(--space-2); + border: 1px solid var(--color-border); + border-radius: var(--radius-sm); + background: var(--color-bg); + color: var(--color-text); + } + .cutover-hint__note { flex: 1; min-width: 12rem; diff --git a/src/application/dtos/settings_dto.rs b/src/application/dtos/settings_dto.rs index fe04d4b2..ab90e1b0 100644 --- a/src/application/dtos/settings_dto.rs +++ b/src/application/dtos/settings_dto.rs @@ -143,10 +143,19 @@ pub struct DashboardStatsDto { // Storage Settings DTOs (Admin Panel) // ============================================================================ -/// Current storage settings returned to admin UI (secrets masked) +/// Current storage settings returned to admin UI (secrets masked). +/// +/// Post-multi-entry (`docs/plan/storage-multi-entry.md`) this carries +/// two shapes side-by-side: the legacy flat storage-config fields +/// (for the pre-multi-entry admin UI, until slice 6 completes the +/// form retirement), plus the new `entries` / `active_entry_name` / +/// `migration_readonly` view the multi-entry UI drives its entries-list, +/// migration-target-dropdown, and readonly-banner from. #[derive(Debug, Serialize, Deserialize)] pub struct StorageSettingsDto { - /// Active backend type: "local" or "s3" + /// Active backend type: "local" or "s3". Legacy flat-config field + /// — mirrors `entries[i where is_active].backend` for the active + /// entry when multi-entry is in use. pub backend: String, pub s3_endpoint_url: Option, pub s3_bucket: Option, @@ -163,6 +172,47 @@ pub struct StorageSettingsDto { pub total_blobs: u64, pub total_bytes_stored: u64, pub dedup_ratio: f64, + // ── Multi-entry view (slice 6) ── + /// All named storage entries declared in env. Empty when running + /// in legacy single-backend mode (`OXICLOUD_STORAGE_ENTRIES` + /// unset AND no legacy synthesis happened). Order matches + /// `_ENTRIES`. + pub entries: Vec, + /// Name of the entry the LIVE backend is currently bound to. + /// Populated as the boot-selected name (per + /// `CoreServices.active_backend_name`). Empty string for the + /// zero-entries legacy path (`"legacy"` sentinel). + pub active_entry_name: String, + /// Global read-only flag — when true, all write-adjacent + /// AuthZ checks refuse. Set by the migration handler at run + /// start; cleared by the boot-clear rule after operator + /// restart. Frontend renders a banner on the storage tab when + /// true. + pub migration_readonly: bool, +} + +/// Per-entry summary emitted in `StorageSettingsDto.entries`. Never +/// carries credentials — those live in env vars only. `is_active` +/// marks which entry the LIVE backend uses right now (matches +/// `active_entry_name` on the parent DTO). +#[derive(Debug, Serialize, Deserialize)] +pub struct StorageEntrySummaryDto { + pub name: String, + /// Backend type — "local" / "s3" / "azure". + pub backend: String, + /// True for exactly one entry (the entry the LIVE backend is on). + /// Frontend uses this to badge the active row and to exclude it + /// from the migration-target dropdown. + pub is_active: bool, + /// True when the entry has a per-entry encryption key. UI shows + /// a lock icon. Presence-only — the key bytes never leave the + /// server. + pub encryption_enabled: bool, + /// Human-readable physical location hint, if the backend surfaces + /// one (`root_dir` for Local, `bucket` for S3, `container` for + /// Azure). Cosmetic — helps the admin distinguish two Local + /// entries pointing at different disks. + pub location_hint: Option, } /// Request body for saving storage settings from the admin panel @@ -280,12 +330,11 @@ pub struct StartMigrationDto { pub concurrency: Option, } -/// Request body (empty) for `POST /api/admin/storage/migration/verify`. -#[derive(Debug, Serialize, Deserialize, ToSchema)] -pub struct VerifyMigrationDto { - /// Number of random blobs to sample-check (default: 100). - pub sample_size: Option, -} +// VerifyMigrationDto retired in slice 7 of +// docs/plan/storage-multi-entry.md — the corresponding endpoint's +// sample-based check is superseded by +// `blobs_consistency?storage=`, a full walk that emits +// structured findings per mismatch. // ============================================================================ // SMTP Settings DTOs (Admin Panel) diff --git a/src/application/services/storage_settings_service.rs b/src/application/services/storage_settings_service.rs index d33322cf..7d955e8c 100644 --- a/src/application/services/storage_settings_service.rs +++ b/src/application/services/storage_settings_service.rs @@ -2,11 +2,16 @@ use std::collections::HashMap; use std::sync::Arc; use uuid::Uuid; +use std::sync::atomic::{AtomicBool, Ordering}; + use crate::application::dtos::settings_dto::{ - SaveStorageSettingsDto, StorageSettingsDto, StorageTestResultDto, TestStorageConnectionDto, + SaveStorageSettingsDto, StorageEntrySummaryDto, StorageSettingsDto, StorageTestResultDto, + TestStorageConnectionDto, }; use crate::application::ports::blob_storage_ports::BlobStorageBackend; -use crate::common::config::{S3StorageConfig, StorageBackendType, StorageConfig}; +use crate::common::config::{ + NamedStorageEntry, S3StorageConfig, StorageBackendType, StorageConfig, +}; use crate::common::errors::{DomainError, ErrorKind}; use crate::domain::repositories::settings_repository::SettingsRepository; use crate::infrastructure::repositories::pg::SettingsPgRepository; @@ -22,18 +27,39 @@ pub struct StorageSettingsService { settings_repo: Arc, env_storage_config: StorageConfig, dedup_service: Arc, + /// Multi-entry snapshot from `AppConfig.storage_entries`. Populated + /// at DI time; immutable per-process (env can only change on + /// restart, per `docs/plan/storage-multi-entry.md`). Empty when + /// running in the pre-multi-entry legacy path. + storage_entries: Vec, + /// Name of the entry the LIVE backend is bound to (matches + /// `CoreServices.active_backend_name`). Empty string / "legacy" + /// for the zero-entries path. + active_entry_name: String, + /// Shared readonly flag — read into the admin DTO so the UI can + /// render a "server in migration read-only mode" banner. Same + /// atomic as `AppState.migration_readonly`; changes made by the + /// migration handler are visible without a DB round-trip. + migration_readonly: Arc, } impl StorageSettingsService { + #[allow(clippy::too_many_arguments)] pub fn new( settings_repo: Arc, env_storage_config: StorageConfig, dedup_service: Arc, + storage_entries: Vec, + active_entry_name: String, + migration_readonly: Arc, ) -> Self { Self { settings_repo, env_storage_config, dedup_service, + storage_entries, + active_entry_name, + migration_readonly, } } @@ -262,6 +288,27 @@ impl StorageSettingsService { crate::common::config::StorageBackendType::Azure => "azure", }; + // Project the multi-entry view. `is_active` is name-compared + // against the boot-selected `active_entry_name` (matches + // exactly one entry when we're in multi-entry mode; matches + // nothing when running the zero-entries legacy path, which + // is expected — the frontend hides the entries table then). + let entries: Vec = self + .storage_entries + .iter() + .map(|e| StorageEntrySummaryDto { + name: e.name.clone(), + backend: match e.backend { + StorageBackendType::Local => "local".to_string(), + StorageBackendType::S3 => "s3".to_string(), + StorageBackendType::Azure => "azure".to_string(), + }, + is_active: e.name == self.active_entry_name, + encryption_enabled: e.encryption_key_base64.is_some(), + location_hint: entry_location_hint(e), + }) + .collect(); + Ok(StorageSettingsDto { backend: backend_str.to_string(), s3_endpoint_url: effective.s3.as_ref().and_then(|s| s.endpoint_url.clone()), @@ -275,6 +322,9 @@ impl StorageSettingsService { total_blobs: stats.total_blobs, total_bytes_stored: stats.total_bytes_stored, dedup_ratio: stats.dedup_ratio, + entries, + active_entry_name: self.active_entry_name.clone(), + migration_readonly: self.migration_readonly.load(Ordering::Relaxed), }) } @@ -667,3 +717,17 @@ async fn run_backend_roundtrip( }, ) } + +/// Cosmetic human-readable identifier for an entry — the physical +/// location piece an admin uses to disambiguate two entries of the +/// same backend type. Never carries credentials. `None` when the +/// entry doesn't have a natural short label (S3 without a bucket, +/// which shouldn't happen because the parser rejects that shape at +/// boot). +fn entry_location_hint(entry: &NamedStorageEntry) -> Option { + match entry.backend { + StorageBackendType::Local => entry.root_dir.clone(), + StorageBackendType::S3 => entry.s3.as_ref().map(|s3| s3.bucket.clone()), + StorageBackendType::Azure => entry.azure.as_ref().map(|az| az.container.clone()), + } +} diff --git a/src/common/di.rs b/src/common/di.rs index 7903da7b..459537a2 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -1435,6 +1435,8 @@ impl AppServiceFactory { crate::infrastructure::services::blobs_consistency_service::BlobsConsistencyCheck::new( maintenance_pool.clone(), core.blob_backend.clone(), + core.config.storage_entries.clone(), + self.storage_path.clone(), ), ) .register_recoverable_job(&core.job_registry, &job_store_provider_dyn) @@ -1453,6 +1455,8 @@ impl AppServiceFactory { crate::infrastructure::services::backend_consistency_service::BackendConsistencyCheck::new( maintenance_pool.clone(), core.blob_backend.clone(), + core.config.storage_entries.clone(), + self.storage_path.clone(), ), ) .register_recoverable_job(&core.job_registry, &job_store_provider_dyn) @@ -2191,11 +2195,19 @@ impl AppServiceFactory { app_state.admin_settings_service = Some(admin_svc.clone()); - // 9b-1b. Wire storage settings service (reuses same settings_repo) + // 9b-1b. Wire storage settings service (reuses same settings_repo). + // Multi-entry view fields (entries + active_entry_name + + // migration_readonly) are populated from the same sources + // the migration handler and AuthZ engine read from — one + // snapshot at DI, shared atomic for the readonly flag so + // changes are visible without a DB round-trip. let storage_settings_svc = Arc::new(StorageSettingsService::new( settings_repo.clone(), self.config.storage.clone(), app_state.core.dedup_service.clone(), + app_state.core.config.storage_entries.clone(), + app_state.core.active_backend_name.clone(), + app_state.migration_readonly.clone(), )); app_state.storage_settings_service = Some(storage_settings_svc.clone()); tracing::info!("Storage settings service initialized"); diff --git a/src/infrastructure/services/backend_consistency_service.rs b/src/infrastructure/services/backend_consistency_service.rs index 71d5d8a7..67f9b656 100644 --- a/src/infrastructure/services/backend_consistency_service.rs +++ b/src/infrastructure/services/backend_consistency_service.rs @@ -57,6 +57,12 @@ use crate::infrastructure::scheduler::{ pub const BACKEND_CONSISTENCY_JOB_NAME: &str = "backend_consistency"; +/// Same `params` JSONB key `blobs_consistency` uses — kept identical +/// so operators grepping run rows see the same convention across +/// both storage-audit tenants. +pub const PROBED_STORAGE_PARAM: &str = + crate::infrastructure::services::blobs_consistency_service::PROBED_STORAGE_PARAM; + /// Batch size for backend enumeration + DB probe. 500 is enough to /// amortise the DB round-trip while keeping the cancel-poll cadence /// sub-second (each batch = one backend list + one DB probe + Rust @@ -78,12 +84,32 @@ const _MAX_EXAMPLES: usize = 5; pub struct BackendConsistencyCheck { pool: Arc, + /// Default backend to enumerate when `args.storage` is `None` — + /// the live LIVE backend, injected at DI. `?storage=` + /// swaps in a fresh backend for the named entry (via + /// [`build_entry_backend`]). backend: Arc, + /// Snapshot of `AppConfig.storage_entries` for `?storage=` + /// resolution. Same rule blobs_consistency uses. + storage_entries: Vec, + /// `OXICLOUD_STORAGE_PATH` fallback for Local entries with no + /// `_ROOT_DIR`. Same fallback rule as boot. + storage_path_fallback: std::path::PathBuf, } impl BackendConsistencyCheck { - pub fn new(pool: Arc, backend: Arc) -> Self { - Self { pool, backend } + pub fn new( + pool: Arc, + backend: Arc, + storage_entries: Vec, + storage_path_fallback: std::path::PathBuf, + ) -> Self { + Self { + pool, + backend, + storage_entries, + storage_path_fallback, + } } pub async fn register_recoverable_job( @@ -138,9 +164,76 @@ impl RecoverableJobHandler for BackendConsistencyCheck { async fn run_resumable( &self, store: &dyn JobStore, - _args: &JobRunArgs, + args: &JobRunArgs, resume_cursor: Option>, ) -> RunOutcome { + // Resolve the backend to probe. Mirrors the shape + // `blobs_consistency` uses — Fresh + args.storage=Some stamps + // probed_storage into params; Resumed reads it back so a + // mid-audit restart re-uses the same target. + let is_fresh = resume_cursor.is_none(); + let probed_storage: Option = if is_fresh { + let name = args.storage.clone(); + if let Some(n) = &name + && let Err(e) = store.set_string_param(PROBED_STORAGE_PARAM, n).await + { + return RunOutcome::Failed { + message: format!("persist {PROBED_STORAGE_PARAM} to params: {e}"), + }; + } + name + } else { + match store.get_string_param(PROBED_STORAGE_PARAM).await { + Ok(v) => v, + Err(e) => { + return RunOutcome::Failed { + message: format!("read {PROBED_STORAGE_PARAM} from params: {e}"), + }; + } + } + }; + let backend: Arc = match &probed_storage { + None => self.backend.clone(), + Some(name) => match self.storage_entries.iter().find(|e| &e.name == name) { + Some(entry) => crate::infrastructure::services::entry_backend::build_entry_backend( + entry, + &self.storage_path_fallback, + ), + None => { + let available = if self.storage_entries.is_empty() { + "(none)".to_string() + } else { + self.storage_entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", ") + }; + return RunOutcome::Failed { + message: format!( + "storage entry `{name}` not found in OXICLOUD_STORAGE_ENTRIES. \ + Available: [{available}]" + ), + }; + } + }, + }; + if let Err(e) = backend.initialize().await { + return RunOutcome::Failed { + message: format!("probed backend init: {e}"), + }; + } + if let Some(name) = &probed_storage { + tracing::info!( + target: "audit", + event = "backend_consistency.probe_scoped", + run_id = %store.run_id(), + probed_storage = %name, + "backend_consistency enumerating entry `{name}` (via ?storage=) instead \ + of live backend" + ); + } + // Cursor = opaque backend continuation token, UTF-8-encoded. // Each backend defines its own format (local = shard/hash, // S3 = ListObjectsV2 continuation token, Azure = list @@ -190,11 +283,7 @@ impl RecoverableJobHandler for BackendConsistencyCheck { // splits canonical blobs (checked for orphan) from // "unknown" entries (sidecar files, foreign namespaces — // emitted as informational notices). - let page = match self - .backend - .list_blob_hashes(cursor.clone(), BATCH_SIZE) - .await - { + let page = match backend.list_blob_hashes(cursor.clone(), BATCH_SIZE).await { Ok(v) => v, Err(e) => { // Backend refuses / can't enumerate. First-batch @@ -219,7 +308,7 @@ impl RecoverableJobHandler for BackendConsistencyCheck { "anomaly", None, serde_json::json!({ - "backend": self.backend.backend_type(), + "backend": backend.backend_type(), "error": format!("{e}"), "note": "backend refused enumeration; no per-blob orphan probes attempted", }), @@ -229,7 +318,7 @@ impl RecoverableJobHandler for BackendConsistencyCheck { target: "oxicloud::consistency", event = "backend_consistency.unenumerable", run_id = %store.run_id(), - backend = self.backend.backend_type(), + backend = backend.backend_type(), "backend refused enumeration (typical during migration or on backends without list support)" ); return RunOutcome::Completed; @@ -265,7 +354,7 @@ impl RecoverableJobHandler for BackendConsistencyCheck { serde_json::json!({ "path": unknown.path, "mtime": unknown.mtime.map(|t| t.to_rfc3339()), - "backend": self.backend.backend_type(), + "backend": backend.backend_type(), "note": "non-canonical file in blob namespace (sidecar / wrong extension); not managed by dedup", }), ) @@ -331,7 +420,7 @@ impl RecoverableJobHandler for BackendConsistencyCheck { serde_json::json!({ "hash": entry.hash, "mtime": entry.mtime.map(|t| t.to_rfc3339()), - "backend": self.backend.backend_type(), + "backend": backend.backend_type(), }), ) .await; diff --git a/src/infrastructure/services/blobs_consistency_service.rs b/src/infrastructure/services/blobs_consistency_service.rs index f210e645..9112352b 100644 --- a/src/infrastructure/services/blobs_consistency_service.rs +++ b/src/infrastructure/services/blobs_consistency_service.rs @@ -47,6 +47,7 @@ //! pointing at reaped chunks) — already covered by //! `files_consistency::chunk_missing`. +use std::path::PathBuf; use std::sync::Arc; use async_trait::async_trait; @@ -54,13 +55,21 @@ use chrono::{DateTime, Duration, Utc}; use sqlx::PgPool; use crate::application::ports::blob_storage_ports::BlobStorageBackend; +use crate::common::config::NamedStorageEntry; use crate::infrastructure::scheduler::{ JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome, RunStatus, record_or_log, }; +use crate::infrastructure::services::entry_backend::build_entry_backend; pub const BLOBS_CONSISTENCY_JOB_NAME: &str = "blobs_consistency"; +/// `params` JSONB key under which the entry name being probed is +/// stashed on a Fresh run (matches `TARGET_NAME_PARAM` on +/// `storage_migration`). Resumed runs re-read it so a paused audit +/// survives restart without the admin re-specifying the target. +pub const PROBED_STORAGE_PARAM: &str = "probed_storage"; + /// Rows per batch. Blobs are numerous (millions on a busy install) /// but per-row work is one indexed backend probe + one indexed SQL /// ref-count query. 200 balances cancel-poll cadence against @@ -82,12 +91,35 @@ const AFFECTED_FILES_SAMPLE: i64 = 5; pub struct BlobsConsistencyCheck { pool: Arc, + /// The default backend to probe when `args.storage` is `None` — + /// the currently-active LIVE backend, injected at DI time. Runs + /// with `?storage=` build a fresh backend for the named + /// entry instead (via [`build_entry_backend`]). backend: Arc, + /// Snapshot of `AppConfig.storage_entries` used to resolve + /// `args.storage` to a `NamedStorageEntry`. Empty for the + /// legacy zero-entries path — `?storage=` runs then + /// fail-fast with a clear "no entries declared" message. + storage_entries: Vec, + /// Ambient `AppConfig.storage_path` — used as the `root_dir` + /// fallback for a Local target entry with no `_ROOT_DIR`. Same + /// fallback rule the boot path uses. + storage_path_fallback: PathBuf, } impl BlobsConsistencyCheck { - pub fn new(pool: Arc, backend: Arc) -> Self { - Self { pool, backend } + pub fn new( + pool: Arc, + backend: Arc, + storage_entries: Vec, + storage_path_fallback: PathBuf, + ) -> Self { + Self { + pool, + backend, + storage_entries, + storage_path_fallback, + } } pub async fn register_recoverable_job( @@ -147,6 +179,80 @@ impl RecoverableJobHandler for BlobsConsistencyCheck { args: &JobRunArgs, resume_cursor: Option>, ) -> RunOutcome { + // Resolve the backend to probe. Two paths, mirroring the + // Fresh/Resumed split the storage_migration handler uses: + // + // * Fresh + args.storage=Some — probe that named entry + // instead of the live backend. Stamp probed_storage in + // params so a mid-audit restart resumes against the same + // entry without re-input. + // * Fresh + args.storage=None — probe the live backend + // (today's default; audit of what the app is actually + // using). + // * Resumed — read probed_storage from params; None means + // the original run was against the live backend. + let is_fresh = resume_cursor.is_none(); + let probed_storage: Option = if is_fresh { + let name = args.storage.clone(); + if let Some(n) = &name + && let Err(e) = store.set_string_param(PROBED_STORAGE_PARAM, n).await + { + return RunOutcome::Failed { + message: format!( + "persist {PROBED_STORAGE_PARAM} to params: {e}" + ), + }; + } + name + } else { + match store.get_string_param(PROBED_STORAGE_PARAM).await { + Ok(v) => v, + Err(e) => { + return RunOutcome::Failed { + message: format!("read {PROBED_STORAGE_PARAM} from params: {e}"), + }; + } + } + }; + let backend: Arc = match &probed_storage { + None => self.backend.clone(), + Some(name) => match self.storage_entries.iter().find(|e| &e.name == name) { + Some(entry) => build_entry_backend(entry, &self.storage_path_fallback), + None => { + let available = if self.storage_entries.is_empty() { + "(none)".to_string() + } else { + self.storage_entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", ") + }; + return RunOutcome::Failed { + message: format!( + "storage entry `{name}` not found in OXICLOUD_STORAGE_ENTRIES. \ + Available: [{available}]" + ), + }; + } + }, + }; + if let Err(e) = backend.initialize().await { + return RunOutcome::Failed { + message: format!("probed backend init: {e}"), + }; + } + if let Some(name) = &probed_storage { + tracing::info!( + target: "audit", + event = "blobs_consistency.probe_scoped", + run_id = %store.run_id(), + probed_storage = %name, + "blobs_consistency probing entry `{name}` (via ?storage=) instead of \ + live backend" + ); + } + // Cursor = the last-visited `hash` string, UTF-8-encoded. On // resume, we walk `WHERE hash > $cursor` in ASC order. First // batch: NULL cursor → start from the smallest hash. @@ -318,7 +424,7 @@ impl RecoverableJobHandler for BlobsConsistencyCheck { // error (log + skip): a transient S3 network blip // shouldn't produce a flood of false data_loss // findings. - let exists = match self.backend.blob_exists(&row.hash).await { + let exists = match backend.blob_exists(&row.hash).await { Ok(v) => v, Err(e) => { tracing::warn!( @@ -369,7 +475,7 @@ impl RecoverableJobHandler for BlobsConsistencyCheck { // avoid mistaking it for "the hash we expect to // see on disk (i.e. what will fix this)". if args.deep { - match recompute_hash(self.backend.as_ref(), &row.hash).await { + match recompute_hash(backend.as_ref(), &row.hash).await { Ok(computed_hash) if computed_hash == row.hash => {} Ok(computed_hash) => { finding_count += 1; diff --git a/src/interfaces/api/handlers/admin_handler.rs b/src/interfaces/api/handlers/admin_handler.rs index a0925a97..2d1f91b1 100644 --- a/src/interfaces/api/handlers/admin_handler.rs +++ b/src/interfaces/api/handlers/admin_handler.rs @@ -19,7 +19,7 @@ use crate::application::dtos::settings_dto::{ AdminCreateUserDto, AdminResetPasswordDto, DashboardStatsDto, ListUsersQueryDto, MigrationStateDto, SaveOidcSettingsDto, SaveStorageSettingsDto, SendSmtpTestDto, SmtpInfoDto, SmtpTestResultDto, StartMigrationDto, TestOidcConnectionDto, TestStorageConnectionDto, - UpdateUserActiveDto, UpdateUserQuotaDto, UpdateUserRoleDto, VerifyMigrationDto, + UpdateUserActiveDto, UpdateUserQuotaDto, UpdateUserRoleDto, }; use crate::application::dtos::user_dto::{AdminUserSummaryDto, UserDto}; use crate::application::ports::authorization_ports::AuthorizationEngine; @@ -85,7 +85,9 @@ pub fn admin_routes() -> Router> { .route("/storage/migration/start", post(start_migration)) .route("/storage/migration/pause", post(pause_migration)) .route("/storage/migration/resume", post(resume_migration)) - .route("/storage/migration/verify", post(verify_migration)) + // NOTE: /storage/migration/verify retired in slice 7 (see the + // comment near where `verify_migration` used to live). Use + // `POST /api/admin/jobs/blobs_consistency/trigger?storage=`. // Encryption key generation .route( "/settings/storage/generate-key", @@ -562,55 +564,15 @@ pub async fn resume_migration( trigger_storage_migration(state, None).await } -/// POST /api/admin/storage/migration/verify — post-migration integrity check. -/// -/// Independent of the copy job: samples `sample_size` random blobs -/// from `storage.blobs` and probes the currently-effective target -/// backend for their existence + declared size. Passes iff no -/// samples are missing and no sizes disagree. -#[utoipa::path( - post, - path = "/api/admin/storage/migration/verify", - responses( - (status = 200, description = "Verification result"), - (status = 401, description = "Unauthorized"), - (status = 403, description = "Admin required"), - (status = 500, description = "Verification failed") - ), - security(("bearerAuth" = [])), - tag = "admin" -)] -pub async fn verify_migration( - State(state): State>, - Json(dto): Json, -) -> Result { - let pool = state - .db_pool - .clone() - .ok_or_else(|| AppError::internal_error("Database not available"))?; - - let svc = state - .storage_settings_service - .as_ref() - .ok_or_else(|| AppError::internal_error("Storage settings service not available"))?; - - let target = svc - .build_effective_backend() - .await - .map_err(|e| AppError::internal_error(format!("Failed to build target backend: {}", e)))?; - target - .initialize() - .await - .map_err(|e| AppError::internal_error(format!("Target backend init failed: {}", e)))?; - - let sample_size = dto.sample_size.unwrap_or(100).clamp(1, 1000); - - let result = verify_backend_sample(target.as_ref(), pool.as_ref(), sample_size) - .await - .map_err(|e| AppError::internal_error(format!("Verification failed: {}", e)))?; - - Ok(Json(result)) -} +// verify_migration endpoint retired (slice 7 of +// docs/plan/storage-multi-entry.md). It was a sample-based sanity +// check against the currently-effective target backend; superseded +// by `POST /api/admin/jobs/blobs_consistency/trigger?storage=` +// which does a full walk against ANY named entry (not just the +// migration target), records structured findings per mismatch, and +// integrates with the standard runs / cancel / findings admin +// surface. Frontend "Verify integrity" button removed in the same +// slice. /// Shared body for `start` / `resume` — both funnel through /// `run_or_resume` via `JobRegistry::trigger`. Detaches into a @@ -652,73 +614,6 @@ async fn trigger_storage_migration( .into_response()) } -/// Verify a random sample of blobs against the given target backend. -/// Inlined from the retired `migration_job::verify_migration` — same -/// query, same result shape; the recoverable-run engine has no reason -/// to own an integrity check. -async fn verify_backend_sample( - target: &dyn crate::application::ports::blob_storage_ports::BlobStorageBackend, - pool: &sqlx::PgPool, - sample_size: usize, -) -> Result { - use crate::common::errors::DomainError; - - let pg_count: i64 = sqlx::query_scalar("SELECT COUNT(*) FROM storage.blobs") - .fetch_one(pool) - .await - .unwrap_or(0); - - let sample_rows: Vec<(String, i64)> = - sqlx::query_as("SELECT hash, size FROM storage.blobs ORDER BY random() LIMIT $1") - .bind(sample_size as i64) - .fetch_all(pool) - .await - .map_err(|e| { - DomainError::internal_error("Migration", format!("Sample query failed: {}", e)) - })?; - - let mut missing = Vec::new(); - let mut size_mismatches = Vec::new(); - - for (hash, expected_size) in &sample_rows { - match target.blob_exists(hash).await { - Ok(false) => missing.push(hash.clone()), - Err(e) => { - tracing::warn!("blob_exists failed for {}: {}", hash, e); - missing.push(hash.clone()); - } - Ok(true) => { - if let Ok(actual_size) = target.blob_size(hash).await - && actual_size != *expected_size as u64 - { - size_mismatches.push(hash.clone()); - } - } - } - } - - let passed = missing.is_empty() && size_mismatches.is_empty(); - Ok(MigrationVerifyResult { - pg_blob_count: pg_count as u64, - sample_checked: sample_rows.len() as u64, - missing_in_target: missing, - size_mismatches, - passed, - }) -} - -/// Post-migration verification result — same shape as the retired -/// `migration_job::VerificationResult` (kept identical so the admin -/// UI's `MigrationVerifyResult` decoder needs no change). -#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] -pub struct MigrationVerifyResult { - pub pg_blob_count: u64, - pub sample_checked: u64, - pub missing_in_target: Vec, - pub size_mismatches: Vec, - pub passed: bool, -} - /// Idle-state DTO — no run has been triggered yet. fn idle_migration_dto() -> MigrationStateDto { MigrationStateDto { diff --git a/src/interfaces/api/mod.rs b/src/interfaces/api/mod.rs index e9c510d7..25f8918b 100644 --- a/src/interfaces/api/mod.rs +++ b/src/interfaces/api/mod.rs @@ -225,7 +225,8 @@ use crate::interfaces::api::handlers::file_handler::MoveFilePayload; handlers::admin_handler::start_migration, handlers::admin_handler::pause_migration, handlers::admin_handler::resume_migration, - handlers::admin_handler::verify_migration, + // handlers::admin_handler::verify_migration retired in + // slice 7 — superseded by `blobs_consistency?storage=`. handlers::admin_handler::generate_encryption_key, // JobRegistry admin surface — production, always-on, // audit-logged. Retired the `/internal/trigger-*` handlers in From f949939508f4d3682e07bc9a6abf2fa34d8f3a80 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 14:32:33 +0200 Subject: [PATCH 08/17] feat(storage): add cmd option --select-storage --- .../services/blobs_consistency_service.rs | 4 +- src/main.rs | 182 ++++++++++++++++-- 2 files changed, 170 insertions(+), 16 deletions(-) diff --git a/src/infrastructure/services/blobs_consistency_service.rs b/src/infrastructure/services/blobs_consistency_service.rs index 9112352b..ea142bbb 100644 --- a/src/infrastructure/services/blobs_consistency_service.rs +++ b/src/infrastructure/services/blobs_consistency_service.rs @@ -198,9 +198,7 @@ impl RecoverableJobHandler for BlobsConsistencyCheck { && let Err(e) = store.set_string_param(PROBED_STORAGE_PARAM, n).await { return RunOutcome::Failed { - message: format!( - "persist {PROBED_STORAGE_PARAM} to params: {e}" - ), + message: format!("persist {PROBED_STORAGE_PARAM} to params: {e}"), }; } name diff --git a/src/main.rs b/src/main.rs index 8e33fc4a..275edf3c 100644 --- a/src/main.rs +++ b/src/main.rs @@ -127,14 +127,22 @@ fn make_socket(addr: &SocketAddr, reuse_port: bool) -> std::io::Result { fn main() -> Result<(), Box> { // Minimal CLI: - // --version Print version + branch + commit hash and exit. - // --config Load env from this file. When given, the default - // `./.env` probe is INTENTIONALLY skipped — tests - // use this to isolate from a developer's repo-root - // `.env`, and operators get a reproducible "this - // file and nothing else" boot. + // --version Print version + branch + commit hash and exit. + // --config Load env from this file. When given, the default + // `./.env` probe is INTENTIONALLY skipped — tests + // use this to isolate from a developer's repo-root + // `.env`, and operators get a reproducible "this + // file and nothing else" boot. + // --select-storage One-shot repair: verify the named entry exists + // in the current .env, UPDATE + // admin_settings.storage.active_backend_name in + // the DB, and exit. Does NOT boot the server. + // Use to recover from the "boot fails on missing + // entry" case — see + // `docs/plan/storage-multi-entry.md` §Fallback. let mut args = std::env::args().skip(1); let mut config_path: Option = None; + let mut select_storage: Option = None; while let Some(arg) = args.next() { match arg.as_str() { "--version" | "-V" => { @@ -153,11 +161,15 @@ fn main() -> Result<(), Box> { }; config_path = Some(p); } + "--select-storage" => { + let Some(name) = args.next() else { + eprintln!("--select-storage requires an entry name"); + std::process::exit(2); + }; + select_storage = Some(name); + } "--help" | "-h" => { - println!( - "OxiCloud v{}\n\nUSAGE:\n oxicloud [--config ]\n oxicloud --version\n oxicloud --help\n", - env!("CARGO_PKG_VERSION"), - ); + print_help(); return Ok(()); } other => { @@ -195,13 +207,157 @@ fn main() -> Result<(), Box> { } } - // Build the Tokio runtime explicitly (not via `#[tokio::main]`) so the - // worker + blocking pools are sized from the cgroup CPU quota and bounded - // — with the `.env` loaded above already in scope. See `build_runtime`. + // Build the Tokio runtime — needed either way (the repair flag also + // needs async DB access). Sized from cgroup CPU quota — see + // `build_runtime`. let runtime = build_runtime()?; + + // Repair-flag short-circuit. `--select-storage` runs the small + // "verify entry + UPDATE pointer + exit" path and NEVER falls + // through to booting the server — the operator restarts normally + // after this exits. + if let Some(name) = select_storage { + return runtime.block_on(run_select_storage(&name)); + } + runtime.block_on(run()) } +/// Print the `--help` output. Kept as a fn (not an inline string) so +/// the layout is easy to eyeball + doesn't clutter the argv match arm. +/// +/// One `println!` per output line — source layout matches what the +/// user sees. Longer than one big raw string but grep-friendly (a +/// specific flag description shows up as its own hit) and diffs stay +/// line-local. +/// +/// Sections: USAGE (invocation shapes) → OPTIONS (per-flag +/// explanations) → ENVIRONMENT (the small set of env vars an operator +/// checks at first boot). Everything else lives in `example.env` — +/// listing all 80+ env knobs here would rot every release. +fn print_help() { + println!( + "OxiCloud v{} (branch={} commit={})", + env!("CARGO_PKG_VERSION"), + env!("GIT_BRANCH"), + env!("GIT_HASH"), + ); + println!(); + println!("USAGE:"); + println!(" oxicloud [--config ] Boot the server. This is the normal"); + println!(" invocation for a docker/systemd unit."); + println!(); + println!(" oxicloud --select-storage One-shot repair — set the active"); + println!(" storage entry in the DB and exit."); + println!(); + println!(" oxicloud --version Print version + commit and exit."); + println!(); + println!(" oxicloud --help Print this help and exit."); + println!(); + println!(); + println!("OPTIONS:"); + println!(" --config "); + println!(" Load environment variables from instead of the default `./.env`."); + println!(" When set, the file WINS over any pre-existing shell exports (the"); + println!(" overriding variant of dotenvy). Use for reproducible CI / systemd"); + println!(" unit boots where a leaked shell export must not silently corrupt"); + println!(" config. Without this flag, the default `./.env` probe is"); + println!(" non-overriding — shell exports win — matching dev convenience."); + println!(); + println!(" --select-storage "); + println!(" Verify is declared in `OXICLOUD_STORAGE_ENTRIES`, then set"); + println!(" `admin_settings.storage.active_backend_name = ` in the DB and"); + println!(" exit. Does NOT boot the server. Use to unblock boot after renaming"); + println!(" or removing a storage entry in `.env` while the DB still points at"); + println!(" the old name (the server aborts boot with a pointer to this flag"); + println!(" when that happens). See `docs/plan/storage-multi-entry.md`"); + println!(" §Fallback for the full recovery flow."); + println!(); + println!(" --version, -V"); + println!(" Print the version, git branch, and commit hash. Exits 0."); + println!(); + println!(" --help, -h"); + println!(" Print this help. Exits 0."); + println!(); + println!(); + println!("ENVIRONMENT:"); + println!(" DATABASE_URL PostgreSQL connection string (required for boot and"); + println!(" for --select-storage)."); + println!(); + println!(" OXICLOUD_SERVER_HOST Bind host (default: 127.0.0.1)."); + println!(" OXICLOUD_SERVER_PORT Bind port (default: 8086)."); + println!(); + println!(" OXICLOUD_STORAGE_ENTRIES=,,..."); + println!(" Comma-separated list of named storage entries. Each"); + println!(" entry N declared here reads its config from"); + println!(" `OXICLOUD_STORAGE__BACKEND`,"); + println!(" `OXICLOUD_STORAGE__S3_BUCKET`, etc. See"); + println!(" `docs/plan/storage-multi-entry.md` for the full"); + println!(" contract. When unset (with legacy flat storage"); + println!(" vars present) one entry named `default` is"); + println!(" synthesised."); + println!(); + println!("The full env-var surface is documented in `example.env` at the repo root."); +} + +/// Repair-flag body. Loads env config, parses entries, verifies the +/// requested name is declared, connects to PG, upserts +/// `admin_settings.storage.active_backend_name`. Never touches the +/// server — the operator restarts after this exits. +/// +/// Exit codes: +/// - `0` on success. +/// - Non-zero via `std::process::exit` on every failure path (name +/// not declared, DB unreachable, upsert failed). Printed to stderr. +async fn run_select_storage(name: &str) -> Result<(), Box> { + use common::config::AppConfig; + use infrastructure::services::entry_backend::persist_active_backend_name; + + // Parse entries + validate `name` is declared. Loading AppConfig + // here re-runs the same env-parse the server does at boot, so a + // successful --select-storage guarantees a subsequent normal + // boot will find the entry (no drift between the two code paths). + let config = AppConfig::from_env(); + if config.storage_entries.is_empty() { + eprintln!( + "OXICLOUD_STORAGE_ENTRIES is not set (or synthesised — legacy path). \ + `--select-storage` needs at least one named entry to switch to." + ); + std::process::exit(2); + } + if !config.storage_entries.iter().any(|e| e.name == name) { + let available = config + .storage_entries + .iter() + .map(|e| e.name.as_str()) + .collect::>() + .join(", "); + eprintln!( + "entry `{name}` is not declared in OXICLOUD_STORAGE_ENTRIES. Available: [{available}]" + ); + std::process::exit(2); + } + + // Connect to PG using the same DATABASE_URL the server uses. + let db_url = std::env::var("DATABASE_URL").map_err( + |_| "DATABASE_URL not set — `--select-storage` needs the same DB the server would boot on", + )?; + let pool = sqlx::PgPool::connect(&db_url) + .await + .map_err(|e| format!("failed to connect to DATABASE_URL: {e}"))?; + + persist_active_backend_name(&pool, name) + .await + .map_err(|e| { + format!("failed to write admin_settings.storage.active_backend_name = `{name}`: {e}") + })?; + + println!( + "active_backend_name = `{name}` written to admin_settings. Restart the server to switch." + ); + Ok(()) +} + /// Construct the multi-threaded Tokio runtime with explicit, CFS-quota-aware /// pool sizes. /// From cc439aaff9783405ee2aef4da0ada7b18cdbeb13 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 16:20:49 +0200 Subject: [PATCH 09/17] fix(dedup): fix informations --- src/infrastructure/services/dedup_service.rs | 33 ++++++++++++++------ 1 file changed, 24 insertions(+), 9 deletions(-) diff --git a/src/infrastructure/services/dedup_service.rs b/src/infrastructure/services/dedup_service.rs index 6d3c7f7c..158a1550 100644 --- a/src/infrastructure/services/dedup_service.rs +++ b/src/infrastructure/services/dedup_service.rs @@ -2142,16 +2142,31 @@ impl DedupService { /// Get deduplication statistics (CDC + legacy). pub async fn get_stats(&self) -> DedupStatsDto { - // Physical storage (all blobs = chunks + legacy) - let (total_blobs, total_bytes_stored): (i64, i64) = - sqlx::query_as("SELECT COUNT(*), COALESCE(SUM(size), 0) FROM storage.blobs") - .fetch_one(self.pool.as_ref()) - .await - .unwrap_or((0, 0)); + // Physical storage (all blobs = chunks + legacy). + // + // The `::bigint` casts on the SUM columns are load-bearing: + // Postgres's `SUM(bigint)` returns `numeric` (not bigint), and + // sqlx has no default decode from `numeric` into Rust's `i64`. + // Without the cast, this `query_as` FAILS on a non-empty + // `storage.blobs` table — decode error → the outer + // `unwrap_or((0, 0))` silently swallows it and every operator + // sees `total_blobs = 0, total_bytes_stored = 0` in the admin + // UI while their disk holds gigabytes. On an EMPTY table SUM + // is NULL, COALESCE inlines the literal integer `0`, the row + // decodes fine, and the bug never surfaces during dev — hence + // it lasted so long. + let (total_blobs, total_bytes_stored): (i64, i64) = sqlx::query_as( + "SELECT COUNT(*), COALESCE(SUM(size), 0)::bigint FROM storage.blobs", + ) + .fetch_one(self.pool.as_ref()) + .await + .unwrap_or((0, 0)); - // Referenced bytes from CDC manifests + // Referenced bytes from CDC manifests. Same `numeric`-vs-`bigint` + // gotcha: `SUM(numeric)` → `numeric`; wrap the whole sum in + // `::bigint` so `query_scalar::<_, i64>` decodes cleanly. let manifest_referenced: i64 = sqlx::query_scalar( - "SELECT COALESCE(SUM(total_size::BIGINT * ref_count), 0) FROM storage.chunk_manifests", + "SELECT COALESCE(SUM(total_size::BIGINT * ref_count), 0)::bigint FROM storage.chunk_manifests", ) .fetch_one(self.pool.as_ref()) .await @@ -2161,7 +2176,7 @@ impl DedupService { // A legacy blob has its hash directly in storage.files.blob_hash. // We approximate by subtracting manifest-attributed storage. let all_blob_referenced: i64 = sqlx::query_scalar( - "SELECT COALESCE(SUM(size::BIGINT * ref_count), 0) FROM storage.blobs", + "SELECT COALESCE(SUM(size::BIGINT * ref_count), 0)::bigint FROM storage.blobs", ) .fetch_one(self.pool.as_ref()) .await From 8329b4aa56b58e583f2972ca045ac9f21741bf8e Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 16:37:23 +0200 Subject: [PATCH 10/17] feat(storage): improve admin panel --- docs/config/admin-settings.md | 44 + docs/config/env.md | 58 +- example.env | 75 +- frontend/src/lib/api/endpoints/admin.ts | 9 +- frontend/src/lib/api/endpoints/adminJobs.ts | 7 +- .../src/routes/admin/[[tab]]/+page.svelte | 946 ++++++++---------- .../src/routes/admin/[[tab]]/page.test.ts | 38 +- src/application/dtos/settings_dto.rs | 66 +- .../services/storage_settings_service.rs | 172 ++-- src/common/config.rs | 110 ++ src/infrastructure/services/dedup_service.rs | 11 +- .../services/s3_blob_backend.rs | 98 +- tests/api/run.sh | 1 + tests/api/storage_multi_entry.hurl | 108 ++ tests/common/server.env | 16 + 15 files changed, 1106 insertions(+), 653 deletions(-) create mode 100644 tests/api/storage_multi_entry.hurl diff --git a/docs/config/admin-settings.md b/docs/config/admin-settings.md index a04f81f6..e874f81e 100644 --- a/docs/config/admin-settings.md +++ b/docs/config/admin-settings.md @@ -64,6 +64,50 @@ If a value is overridden by environment variables, the admin API can expose that Successful responses include discovered endpoints such as the authorization endpoint, token endpoint, and userinfo endpoint. +## Storage & Migration + +The admin storage tab operates on the **named storage entries** declared in `.env` (see [Storage Entries](/config/env#storage-entries-multi-entry-recommended)). The set of entries is immutable per-deploy — adding or removing one requires a server restart. Runtime behaviour is driven by a single DB row that names which entry is currently active. + +### Endpoints + +| Method | Path | Description | +| --- | --- | --- | +| `GET` | `/api/admin/settings/storage` | List entries + active pointer + read-only flag + basic stats | +| `POST` | `/api/admin/settings/storage/test` | Reachability + round-trip test against the currently-effective backend | +| `POST` | `/api/admin/storage/migration/start` | Trigger a cross-entry migration. Body: `{"target_name": ""}` | +| `POST` | `/api/admin/storage/migration/pause` | Cooperative cancel — handler yields at the next batch boundary | +| `POST` | `/api/admin/storage/migration/resume` | Resume a paused run (target read from `params.target_name`, no body needed) | +| `GET` | `/api/admin/storage/migration` | Poll the current run's progress | + +Runs are recoverable — status, cursor, and per-blob failure findings all live in `jobs.recoverable_runs` / `jobs.run_findings`. The same run history is browsable via `GET /api/admin/jobs/storage_migration/runs`. + +### Cutover flow (moving the active pointer) + +1. Declare the target entry in `.env` and restart so `OXICLOUD_STORAGE_ENTRIES` picks it up. +2. Admin storage tab → pick the target from the dropdown → **Start migration**. The server engages global read-only mode (writes refused across the whole app; reads keep working), then copies blobs from source → target. +3. On `Completed`, the server writes `admin_settings.storage.active_backend_name = `. Read-only stays ON — writes on the OLD backend would strand data now that the pointer says the new one is active. +4. **Operator restarts the server.** Boot picks the new active entry, and the boot-clear rule drops the read-only flag (`no in-flight run + booted-entry matches DB pointer`). Server writable again, on the new backend. + +### Repair flag — pointer / entry drift + +If an entry is renamed or removed from `.env` while the DB pointer still names the old one, boot aborts with a clear error pointing at: + +``` +oxicloud --select-storage +``` + +This one-shot repair command re-runs the same env-parse the server does at boot, verifies `` is declared in `OXICLOUD_STORAGE_ENTRIES`, updates `admin_settings.storage.active_backend_name` in the DB, and exits. Operator then restarts normally. See [Environment Variables — Storage Entries](/config/env#storage-entries-multi-entry-recommended) for the model, and [`oxicloud --help`](https://github.com/oxicloud/oxicloud/blob/main/src/main.rs) for the full flag list. + +### Auditing entries other than the active one + +`blobs_consistency` and `backend_consistency` (recoverable jobs on the Jobs tab) accept `?storage=` to probe any declared entry — not just the live one. Use this to verify a migration target before cutover, or to audit an old backend after cutover but before decommissioning: + +``` +POST /api/admin/jobs/blobs_consistency/trigger?storage= +``` + +Unknown names 400 at the HTTP layer. + ## Data Storage Runtime settings are stored in `auth.admin_settings`. diff --git a/docs/config/env.md b/docs/config/env.md index 0b5d4ea1..c39651e7 100644 --- a/docs/config/env.md +++ b/docs/config/env.md @@ -78,7 +78,59 @@ Most runtime variables use the `OXICLOUD_` prefix. A few build-time or allocator | `OXICLOUD_GRANT_CLEANUP_INTERVAL_HOURS` | `24` | How often the grant-cleanup daemon fires. Clamped to a minimum of 1 hour. Adjusting this doesn't change what gets deleted — only how promptly. Daily is fine for any realistic grant volume. | | `OXICLOUD_WEBDAV_DRIVE_LISTING_PREFIX` | `@drive` | Native WebDAV URL segment that renders the caller's drive list. Sanitized by trimming leading/trailing `/`. Three shapes: (1) default `@drive` — `/webdav/…` addresses the caller's default personal drive (back-compat), `/webdav/@drive/` returns the drive listing, `/webdav/@drive//…` targets a specific drive. (2) empty string `""` — `/webdav/` IS the drive listing, `/webdav//…` targets a specific drive, no default-drive shortcut. (3) any other string (e.g. `drives`) — same shape as `@drive` with that segment substituted. Only drives the caller has Read on via `role_grants` resolve. | -## Storage Backend +## Storage Entries (multi-entry, recommended) + +Declare one or more **named** storage backends. The one the app runs on is picked from the DB (`admin_settings.storage.active_backend_name`); the admin panel's storage tab flips the pointer, and cross-backend migration is a recoverable job that copies blobs between two entries with a read-only safety window. See [Admin Settings — Storage & Migration](/config/admin-settings) for the operator flow and the [multi-entry design doc](https://github.com/oxicloud/oxicloud/blob/main/docs/plan/storage-multi-entry.md) for the full model. + +| Variable | Default | Description | +|---|---|---| +| `OXICLOUD_STORAGE_ENTRIES` | — | Comma-separated allowlist of entry names. Names must match `[a-z0-9_-]{1,32}` and be unique. Order is preserved (the first entry is the fallback when the DB pointer is unset — fresh install). | + +Each declared name `` then reads its own set of per-entry variables: + +| Variable | Default | Description | +|---|---|---| +| `OXICLOUD_STORAGE__BACKEND` | — | Backend type for entry ``: `local` \| `s3` \| `azure` (required per entry) | +| `OXICLOUD_STORAGE__ROOT_DIR` | `OXICLOUD_STORAGE_PATH` | Local-only: root directory for this entry's `.blobs/`. Falls back to the ambient `OXICLOUD_STORAGE_PATH` when unset. | +| `OXICLOUD_STORAGE__S3_BUCKET` | — | S3-only: bucket name (required when backend=s3) | +| `OXICLOUD_STORAGE__S3_REGION` | `us-east-1` | S3-only: AWS region | +| `OXICLOUD_STORAGE__S3_ENDPOINT_URL` | — | S3-only: custom endpoint for non-AWS providers | +| `OXICLOUD_STORAGE__S3_ACCESS_KEY` | — | S3-only: access key ID | +| `OXICLOUD_STORAGE__S3_SECRET_KEY` | — | S3-only: secret access key | +| `OXICLOUD_STORAGE__S3_FORCE_PATH_STYLE` | `false` | S3-only: path-style URLs (required for MinIO, R2) | +| `OXICLOUD_STORAGE__AZURE_ACCOUNT_NAME` | — | Azure-only: storage account name | +| `OXICLOUD_STORAGE__AZURE_ACCOUNT_KEY` | — | Azure-only: storage account key | +| `OXICLOUD_STORAGE__AZURE_CONTAINER` | — | Azure-only: blob container name (required when backend=azure) | +| `OXICLOUD_STORAGE__AZURE_SAS_TOKEN` | — | Azure-only: SAS token (alternative to account key) | +| `OXICLOUD_STORAGE__AZURE_ENDPOINT_URL` | — | Azure-only: custom endpoint (Azurite, private deployments) | +| `OXICLOUD_STORAGE__ENCRYPTION_KEY` | — | Base64-encoded 32-byte AES-256 key. **Presence implies encryption is enabled** on this entry — no separate enable flag. Bad base64 / wrong length aborts boot. | +| `OXICLOUD_STORAGE__ENCRYPTION_CIPHER` | `aes-256-gcm` when `_ENCRYPTION_KEY` is set | Cipher choice for this entry. Only `aes-256-gcm` is accepted today (future-proofing knob — the enum is ready for a second cipher, the implementation still hardcodes AES-256-GCM). Setting the cipher without a key aborts boot. | + +**Fail-fast rules** (boot aborts with actionable message): + +- A declared name whose required per-entry fields are missing (`_BACKEND` never set, S3 with no `_S3_BUCKET`, Azure with no `_AZURE_CONTAINER`). +- Setting `OXICLOUD_STORAGE_ENTRIES` alongside any of the legacy flat vars below (`OXICLOUD_STORAGE_BACKEND`, `OXICLOUD_S3_*`, `OXICLOUD_AZURE_*`, `OXICLOUD_STORAGE_ENCRYPTION_*`). Pick one mode; the error lists every conflicting var to remove. +- A DB pointer (`admin_settings.storage.active_backend_name`) that names an entry not in the current `_ENTRIES`. The error points at the repair flag `oxicloud --select-storage ` — verify + UPDATE DB + exit. + +**Example** — two entries, local disk plus an S3 target for planned migration: + +``` +OXICLOUD_STORAGE_ENTRIES=local_main,s3_prod + +OXICLOUD_STORAGE_local_main_BACKEND=local +OXICLOUD_STORAGE_local_main_ROOT_DIR=/srv/oxicloud + +OXICLOUD_STORAGE_s3_prod_BACKEND=s3 +OXICLOUD_STORAGE_s3_prod_S3_BUCKET=my-oxicloud-bucket +OXICLOUD_STORAGE_s3_prod_S3_REGION=us-east-1 +OXICLOUD_STORAGE_s3_prod_S3_ACCESS_KEY=… +OXICLOUD_STORAGE_s3_prod_S3_SECRET_KEY=… +OXICLOUD_STORAGE_s3_prod_ENCRYPTION_KEY=… # openssl rand -base64 32 +``` + +## Storage Backend (DEPRECATED — legacy single-backend) + +> ⚠️ **Deprecated.** Use [Storage Entries](#storage-entries-multi-entry-recommended) above for new deployments. These flat variables still work when `OXICLOUD_STORAGE_ENTRIES` is **unset** — the parser then synthesises one entry named `default` from them, keeping pre-multi-entry `.env` files booting unchanged. Booting via this path emits a `storage.legacy_flat_vars_deprecated` warning so operators see it in logs. Removal target: not yet fixed; migrate at your convenience by moving each variable below into `OXICLOUD_STORAGE__*` form under an entry declared in `OXICLOUD_STORAGE_ENTRIES`. **Setting any variable from this section alongside `OXICLOUD_STORAGE_ENTRIES` is a fail-fast boot error** — pick one mode. | Variable | Default | Description | |---|---|---| @@ -118,10 +170,12 @@ A least-recently-used disk cache that can speed up repeated reads from S3 or Azu | `OXICLOUD_STORAGE_CACHE_MAX_SIZE` | `53687091200` | Max cache size in bytes (50 GB) | | `OXICLOUD_STORAGE_CACHE_PATH` | `{STORAGE_PATH}/.blob-cache` | Cache directory | -### Client-Side Encryption +### Client-Side Encryption (DEPRECATED — per-entry key is the new home) AES-256-GCM encryption applied to blobs before they are written to any backend. +> ⚠️ **Deprecated.** Prefer per-entry `OXICLOUD_STORAGE__ENCRYPTION_KEY` under an entry declared in `OXICLOUD_STORAGE_ENTRIES` — presence of the key implies encryption is enabled on that entry (no separate flag), and multi-entry enables cross-key rotation via migration to a new entry. The flat vars below still work in zero-entries mode and get folded into the synthesised `default` entry, alongside the same deprecation warning at boot. + | Variable | Default | Description | |---|---|---| | `OXICLOUD_STORAGE_ENCRYPTION_ENABLED` | `false` | Enable at-rest blob encryption | diff --git a/example.env b/example.env index 9c673fe0..f84b531a 100644 --- a/example.env +++ b/example.env @@ -353,8 +353,73 @@ DATABASE_URL=postgres://postgres:postgres@localhost:5432/oxicloud #OXICLOUD_PLUGIN_LOG_QUEUE_CAPACITY=1024 # ----------------------------------------------------------------------------- -# STORAGE BACKEND +# STORAGE ENTRIES (multi-entry, recommended) # ----------------------------------------------------------------------------- +# +# Declare one or more NAMED storage backends. The one the app runs on is +# picked from the DB (`admin_settings.storage.active_backend_name`) — the +# admin panel's storage tab flips it, and cross-backend migration is a +# recoverable-run job that copies blobs between two entries. See +# `docs/plan/storage-multi-entry.md` for the full model. +# +# Rules: +# * `OXICLOUD_STORAGE_ENTRIES` is a comma-separated allowlist of names. +# Names must match `[a-z0-9_-]{1,32}` and be unique. Order is +# preserved (the first entry is the fallback when no active pointer +# is set in the DB yet — e.g. fresh install). +# * For each name `N`, the parser reads +# `OXICLOUD_STORAGE__BACKEND` (local | s3 | azure) plus the +# backend-specific fields below. A missing required field aborts +# boot with the exact var name in the error message. +# * Presence of `OXICLOUD_STORAGE__ENCRYPTION_KEY` implies AES-256 +# encryption is enabled on that entry (no separate enable flag). +# Bad base64 / wrong length aborts boot with the entry name. +# * SETTING `_ENTRIES` alongside the legacy flat vars below (e.g. +# `OXICLOUD_STORAGE_BACKEND` + `OXICLOUD_S3_BUCKET`) is a FAIL-FAST +# boot error — pick one mode. Migrate any leftover flat vars into +# per-entry `_STORAGE__*` form. +# +# Example: local disk today, S3 target for a planned migration. +# +#OXICLOUD_STORAGE_ENTRIES=local_main,s3_prod +# +#OXICLOUD_STORAGE_local_main_BACKEND=local +#OXICLOUD_STORAGE_local_main_ROOT_DIR=/srv/oxicloud +# +#OXICLOUD_STORAGE_s3_prod_BACKEND=s3 +#OXICLOUD_STORAGE_s3_prod_S3_BUCKET=my-oxicloud-bucket +#OXICLOUD_STORAGE_s3_prod_S3_REGION=us-east-1 +#OXICLOUD_STORAGE_s3_prod_S3_ENDPOINT_URL=https://s3.example.com +#OXICLOUD_STORAGE_s3_prod_S3_ACCESS_KEY= +#OXICLOUD_STORAGE_s3_prod_S3_SECRET_KEY= +#OXICLOUD_STORAGE_s3_prod_S3_FORCE_PATH_STYLE=false +#OXICLOUD_STORAGE_s3_prod_ENCRYPTION_KEY= # generate: openssl rand -base64 32 +# Cipher declaration — future-proofing. Today only `aes-256-gcm` is +# accepted (and it's the default when `_ENCRYPTION_KEY` is set), so +# this line can be omitted. Explicit here as documentation. +#OXICLOUD_STORAGE_s3_prod_ENCRYPTION_CIPHER=aes-256-gcm +# +# Repair flag: if you rename an entry in .env while the DB still points +# at the old name, boot aborts with an actionable error pointing at: +# +# oxicloud --select-storage +# +# which verifies the entry exists in `_ENTRIES` and updates the DB +# pointer without booting the server. See §Fallback in the plan doc. + +# ----------------------------------------------------------------------------- +# STORAGE BACKEND — DEPRECATED (single-backend flat vars) +# ----------------------------------------------------------------------------- +# +# ⚠️ DEPRECATED. Use the STORAGE ENTRIES section above for new deployments. +# These flat variables still work when `OXICLOUD_STORAGE_ENTRIES` is UNSET — +# the parser then synthesises a single entry named `default` from them AND +# emits a boot-time deprecation warning +# (`storage.legacy_flat_vars_deprecated`) so operators see it in logs. +# Removal target: not yet fixed. Migrate at your convenience by moving +# each `OXICLOUD_STORAGE_BACKEND` / `OXICLOUD_S3_*` / `OXICLOUD_AZURE_*` / +# `OXICLOUD_STORAGE_ENCRYPTION_*` into `OXICLOUD_STORAGE__*` under +# an entry declared in `OXICLOUD_STORAGE_ENTRIES`. # Blob storage backend: local (default), s3, or azure #OXICLOUD_STORAGE_BACKEND=local @@ -400,9 +465,15 @@ DATABASE_URL=postgres://postgres:postgres@localhost:5432/oxicloud # Cache directory (default: {STORAGE_PATH}/.blob-cache) #OXICLOUD_STORAGE_CACHE_PATH= -# --- Client-Side Encryption --- +# --- Client-Side Encryption --- DEPRECATED (per-entry key is the new home) # AES-256-GCM encryption applied to blobs before writing to any backend. # WARNING: losing the key means losing all data. Back it up securely. +# +# ⚠️ DEPRECATED. Prefer per-entry `OXICLOUD_STORAGE__ENCRYPTION_KEY` +# under an entry declared in `OXICLOUD_STORAGE_ENTRIES` (see the top +# multi-entry section). The flat vars below still work in +# zero-entries mode and get folded into the synthesised `default` +# entry, alongside a deprecation warning at boot. # Enable at-rest blob encryption (default: false) #OXICLOUD_STORAGE_ENCRYPTION_ENABLED=false diff --git a/frontend/src/lib/api/endpoints/admin.ts b/frontend/src/lib/api/endpoints/admin.ts index b15c5bae..a4001438 100644 --- a/frontend/src/lib/api/endpoints/admin.ts +++ b/frontend/src/lib/api/endpoints/admin.ts @@ -476,14 +476,7 @@ export interface StorageEntrySummary { } export interface StorageSettings { - backend: string; - s3_endpoint_url?: string | null; - s3_bucket?: string | null; - s3_region?: string | null; - s3_access_key_set?: boolean; - s3_secret_key_set?: boolean; - s3_force_path_style?: boolean; - env_overrides?: string[]; + // Live stats — what the running process reports. current_backend?: string; total_blobs?: number; total_bytes_stored?: number; diff --git a/frontend/src/lib/api/endpoints/adminJobs.ts b/frontend/src/lib/api/endpoints/adminJobs.ts index fb7f4764..97db6419 100644 --- a/frontend/src/lib/api/endpoints/adminJobs.ts +++ b/frontend/src/lib/api/endpoints/adminJobs.ts @@ -53,11 +53,16 @@ export function listJobs(): Promise { */ export async function triggerJob( name: string, - opts: { force?: boolean; deep?: boolean } = {} + opts: { force?: boolean; deep?: boolean; storage?: string } = {} ): Promise { const params = new URLSearchParams(); if (opts.force) params.set('force', 'true'); if (opts.deep) params.set('deep', 'true'); + // `storage` scopes tenants that respect JobRunArgs.storage — + // currently blobs_consistency / backend_consistency (probes the + // named entry instead of the live backend). See + // `docs/plan/storage-multi-entry.md` slice 7. + if (opts.storage) params.set('storage', opts.storage); const q = params.toString(); const url = `/api/admin/jobs/${encodeURIComponent(name)}/trigger${q ? `?${q}` : ''}`; const res = await apiFetch(url, { diff --git a/frontend/src/routes/admin/[[tab]]/+page.svelte b/frontend/src/routes/admin/[[tab]]/+page.svelte index 8b28a792..f6e2d890 100644 --- a/frontend/src/routes/admin/[[tab]]/+page.svelte +++ b/frontend/src/routes/admin/[[tab]]/+page.svelte @@ -26,7 +26,6 @@ resetUserPassword, saveOidc, savePluginRetention, - saveStorage, sendSmtpTest, setPluginEnabled, setRegistrationEnabled, @@ -75,6 +74,7 @@ DrivePoliciesPartial, User } from '$lib/api/types'; + import { triggerJob } from '$lib/api/endpoints/adminJobs'; import AdminJobsPanel from '$lib/components/AdminJobsPanel.svelte'; import Icon from '$lib/icons/Icon.svelte'; import Modal from '$lib/components/Modal.svelte'; @@ -370,129 +370,72 @@ } } - // Storage - const STORAGE_PRESETS: Record = - { - custom: { endpoint: '', region: '', pathStyle: false }, - aws: { endpoint: '', region: 'us-east-1', pathStyle: false }, - backblaze: { - endpoint: 'https://s3.{region}.backblazeb2.com', - region: 'us-west-004', - pathStyle: false - }, - 'cloudflare-r2': { - endpoint: 'https://{accountId}.r2.cloudflarestorage.com', - region: 'auto', - pathStyle: true - }, - minio: { endpoint: 'http://localhost:9000', region: 'us-east-1', pathStyle: true }, - digitalocean: { - endpoint: 'https://{region}.digitaloceanspaces.com', - region: 'nyc3', - pathStyle: false - }, - wasabi: { - endpoint: 'https://s3.{region}.wasabisys.com', - region: 'us-east-1', - pathStyle: false - } - }; + // Storage — multi-entry read-only view. + // + // Post `docs/plan/storage-multi-entry.md`, the .env is the SOLE + // place to declare backends. The admin storage tab is now: + // - a read-only list of the entries the server booted with, + // - a per-entry test button (round-trip against that entry), + // - a per-entry audit action (triggers blobs_consistency?storage=), + // - a per-non-active migrate+activate button, + // - the migration status line + cutover hint. + // No form. No save. The retired save endpoint / DTO are still on + // the backend during the deprecation window but the UI never + // hits them. let storage = $state(null); - let sForm = $state({ - backend: 'local', - preset: 'custom', - endpoint: '', - bucket: '', - region: '', - accessKey: '', - secretKey: '', - pathStyle: false - }); let storageMsg = $state<{ text: string; ok: boolean } | null>(null); - let storageBusy = $state(false); + // Per-entry test state — keyed by entry name so the buttons don't + // step on each other and the last result stays visible per row. + let entryTest = $state< + Record + >({}); async function loadStorage() { try { storage = await getStorageSettings(); - sForm = { - backend: storage.backend ?? 'local', - preset: 'custom', - endpoint: storage.s3_endpoint_url ?? '', - bucket: storage.s3_bucket ?? '', - region: storage.s3_region ?? '', - accessKey: '', - secretKey: '', - pathStyle: storage.s3_force_path_style ?? false + } catch (e) { + storageMsg = { text: errorMessage(e), ok: false }; + } + } + + async function doTestEntry(name: string) { + entryTest = { ...entryTest, [name]: { busy: true } }; + try { + const r: StorageTestResult = await testStorage({ entry_name: name }); + entryTest = { ...entryTest, [name]: { busy: false, result: r } }; + } catch (e) { + entryTest = { ...entryTest, [name]: { busy: false, error: errorMessage(e) } }; + } + } + + async function doAuditEntry(name: string) { + try { + await triggerJob('blobs_consistency', { storage: name }); + storageMsg = { + text: t( + 'admin.storage_audit_triggered', + { name }, + 'blobs_consistency triggered for `{{name}}` — watch it on the Jobs tab.' + ), + ok: true }; } catch (e) { storageMsg = { text: errorMessage(e), ok: false }; } } - function applyPreset() { - const p = STORAGE_PRESETS[sForm.preset]; - if (!p) return; - if (p.endpoint) sForm.endpoint = p.endpoint; - if (p.region) sForm.region = p.region; - sForm.pathStyle = p.pathStyle; - } - function storageBody() { - return { - backend: sForm.backend, - s3_endpoint_url: sForm.endpoint.trim() || null, - s3_bucket: sForm.bucket.trim() || null, - s3_region: sForm.region.trim() || null, - s3_access_key: sForm.accessKey || null, - s3_secret_key: sForm.secretKey || null, - s3_force_path_style: sForm.pathStyle - }; - } - async function doSaveStorage() { - storageBusy = true; - storageMsg = null; - try { - await saveStorage(storageBody()); - storageMsg = { text: t('admin.storage_saved', 'Storage settings saved.'), ok: true }; - await loadStorage(); - } catch (e) { - storageMsg = { text: errorMessage(e), ok: false }; - } finally { - storageBusy = false; - } - } - async function doTestStorage() { - storageBusy = true; - storageMsg = null; - try { - const r: StorageTestResult = await testStorage(storageBody()); - // Backend now performs BOTH reachability (health-check) - // and a full read/write round-trip. `connected` gets - // flipped to false by the service if the round-trip - // itself fails, so a single boolean covers the whole - // pass/fail signal. `roundtrip_passed` distinguishes the - // two flavours of failure for the operator. - const ok = r.connected ?? r.success ?? false; - if (ok) { - let text = t('admin.storage_test_success', 'Connection + read/write OK'); - if (r.backend_type) text += ` (${r.backend_type})`; - if (r.roundtrip_elapsed_ms != null) text += ` — round-trip ${r.roundtrip_elapsed_ms} ms`; - if (r.available_bytes != null) - text += ` · ${formatBytes(r.available_bytes)} ${t('admin.available', 'available')}`; - if (r.cleanup_ok === false) - text += ` · ⚠ cleanup DELETE failed — orphan test blob left on backend`; - storageMsg = { text, ok: true }; - } else { - const phase = r.phase_reached ? ` [phase: ${r.phase_reached}]` : ''; - const label = - r.roundtrip_passed === false - ? t('admin.storage_test_failure', 'Read/write test failed') - : t('admin.storage_test_failure', 'Connection failed'); - storageMsg = { text: `${label}${phase}: ${r.message ?? ''}`, ok: false }; - } - } catch (e) { - storageMsg = { text: errorMessage(e), ok: false }; - } finally { - storageBusy = false; - } + + async function doMigrateActivate(name: string) { + if ( + !confirm( + t( + 'admin.storage_migrate_confirm', + { name }, + 'Migrate all blobs to `{{name}}` and set it as the active entry? The server enters read-only mode during the copy; restart is required to finish cutover.' + ) + ) + ) + return; + await doMigration('start', name); } // Migration @@ -526,96 +469,18 @@ } } - // ── Multi-entry migration target picker (slice 6) ──────────────── - // - // Multi-entry mode requires the admin to name the target entry - // before starting a migration. Backend rejects an unnamed start - // with 400. Dropdown shows every non-active entry; picking one - // enables the Start button. - let migrationTarget = $state(''); - const availableTargets = $derived( - (storage?.entries ?? []).filter((e) => !e.is_active).map((e) => e.name) - ); - // Sync target when the entries list first appears — pick the first - // non-active entry by default so the operator can just click Start - // on a simple two-entry setup. - $effect(() => { - if (!migrationTarget && availableTargets.length > 0) { - migrationTarget = availableTargets[0]; - } - // Also unset when the previously-chosen target became active - // (cutover completed under our feet). - if (migrationTarget && !availableTargets.includes(migrationTarget)) { - migrationTarget = availableTargets[0] ?? ''; - } - }); + // The old target-name picker state was retired — the entries + // table now has per-row "Migrate & activate" buttons on + // non-active entries. Simpler mental model; no picker to sync. - // ── Post-migration .env cutover hint ───────────────────────────── - // - // Migration copies blobs to the target backend, but boot-time - // backend selection reads env vars only — never the DB config the - // admin filled in. So the app keeps running on the SOURCE backend - // even after the copy completes. To actually cut over, the - // operator has to add the equivalent env vars to `.env` and - // restart. This hint block spells out those lines with a - // copy-to-clipboard button. - // - // Shown only when: - // - a migration has completed successfully, AND - // - the live backend still differs from the configured target - // (so we're actually pending cutover), AND - // - the backend env var isn't ALREADY overriding (which would - // mean the admin already updated .env or the platform sets it). - const cutoverPending = $derived( - migration?.status === 'completed' && - !!storage && - storage.current_backend != null && - storage.current_backend !== storage.backend && - !(storage.env_overrides ?? []).includes('backend') - ); - - // Env-var lines the admin needs to paste. Credentials are NEVER - // echoed — the storage-settings DTO only returns `_set` booleans - // for access/secret keys (not the values), so we render a - // placeholder line the admin fills in from their own records. - // Local backend still gets a line for completeness, but a Local - // deployment typically has no reason to explicitly set the var - // (default is Local). - const cutoverEnvLines = $derived.by((): string[] => { - if (!storage) return []; - const lines: string[] = []; - switch (storage.backend) { - case 's3': - lines.push('OXICLOUD_STORAGE_BACKEND=s3'); - if (storage.s3_endpoint_url) - lines.push(`OXICLOUD_S3_ENDPOINT_URL=${storage.s3_endpoint_url}`); - if (storage.s3_bucket) lines.push(`OXICLOUD_S3_BUCKET=${storage.s3_bucket}`); - if (storage.s3_region) lines.push(`OXICLOUD_S3_REGION=${storage.s3_region}`); - if (storage.s3_access_key_set) - lines.push('OXICLOUD_S3_ACCESS_KEY='); - if (storage.s3_secret_key_set) - lines.push('OXICLOUD_S3_SECRET_KEY='); - if (storage.s3_force_path_style) lines.push('OXICLOUD_S3_FORCE_PATH_STYLE=true'); - break; - case 'local': - lines.push('OXICLOUD_STORAGE_BACKEND=local'); - break; - // Azure not yet exposed in the admin form; add here when it is. - } - return lines; - }); - - let cutoverCopied = $state(false); - async function copyCutoverEnv() { - try { - await navigator.clipboard.writeText(cutoverEnvLines.join('\n')); - cutoverCopied = true; - setTimeout(() => (cutoverCopied = false), 2000); - } catch { - // Clipboard permission denied — silent; the block is - // selectable so the operator can copy manually. - } - } + // Retired: the .env cutover-hint state (cutoverPending + + // cutoverEnvLines + cutoverCopied + copyCutoverEnv). It served + // the pre-multi-entry flow that made admins paste env vars + // into .env after migration. Post-multi-entry, the server + // writes `active_backend_name` to the DB automatically on + // migration completion; the operator just restarts. The new + // short "restart to switch" hint is rendered inline in the + // entries card template, no derived state needed. // Migration integrity verification retired in slice 7 — the // sample-based /storage/migration/verify endpoint is replaced by @@ -2011,136 +1876,36 @@ {/if} {:else if tab === 'storage'} +
-

{t('admin.storage_tab', 'Storage')}

+

{t('admin.storage_tab', 'Storage entries')}

{#if !storage}

{t('common.loading', 'Loading…')}

- {:else} -
(e.preventDefault(), doSaveStorage())} - > - - {#if sForm.backend === 's3'} - - - - - - - - {/if} - {#if storageMsg}

- {storageMsg.text} -

{/if} -
- - - -
-
+ {:else if !storage.entries || storage.entries.length === 0} + +

+ + {t( + 'admin.storage_no_entries', + { backend: storage.current_backend ?? '?' }, + 'No OXICLOUD_STORAGE_ENTRIES declared. Running on the legacy single-backend fallback ({{backend}}). Migrate to the multi-entry model — see docs/config/env.md.' + )} +

{t('admin.storage_current', 'Current backend')}
{storage.current_backend ?? '—'}
@@ -2153,15 +1918,7 @@
{t('admin.storage_dedup', 'Dedup ratio')}
{storage.dedup_ratio != null ? `${storage.dedup_ratio.toFixed(2)}x` : '—'}
- {/if} -
- -
-

{t('admin.migration', 'Storage migration')}

- - {#if storage?.entries && storage.entries.length > 0} + {:else} {#if storage.migration_readonly}
{/if} - - - - - - - - - - - - {#each storage.entries as entry (entry.name)} - - - - - - - - {/each} - -
{t('admin.entry_name', 'Entry')}{t('admin.entry_backend', 'Backend')}{t('admin.entry_location', 'Location')}{t('admin.entry_encryption', 'Encryption')}{t('admin.entry_status', 'Status')}
{entry.name}{entry.backend}{entry.location_hint ?? '—'} - {#if entry.encryption_enabled} - AES-256 - {:else} - — - {/if} - + + + {@const migrationInFlight = + migration != null && (migration.status === 'running' || migration.status === 'paused')} +
+ {#each storage.entries as entry (entry.name)} + {@const test = entryTest[entry.name]} +
+
+
+ {entry.name} {#if entry.is_active} - {t('admin.entry_active', 'active')} + + + {t('admin.entry_active', 'active')} + {:else} - {t('admin.entry_inactive', 'available')} + + {t('admin.entry_inactive', 'available')} + {/if} -
- {/if} - {#if !migration} -

{t('common.loading', 'Loading…')}

- {:else} -

{t('admin.status', 'Status')}: {migration.status}

- {#if migration.total_blobs > 0} -
-
-
-

- {migration.migrated_blobs} / {migration.total_blobs} ({migrationPct}%) · - {formatBytes(migration.migrated_bytes)} - {#if migration.throughput_bytes_per_sec && migration.status === 'running'} - · {formatBytes(Math.round(migration.throughput_bytes_per_sec))}/s - {/if} - {#if migrationEtaMin != null} - · {t('admin.mig_eta', { min: migrationEtaMin }, `~${migrationEtaMin} min remaining`)} - {/if} -

- {/if} - {#if migration.failed_blobs && migration.failed_blobs.length > 0} -
- - {t( - 'admin.mig_failed', - { n: migration.failed_blobs.length }, - `${migration.failed_blobs.length} failed blobs` - )} - -
{migration.failed_blobs.join('\n')}
-
- {/if} -
- - {#if migration.status !== 'running' && migration.status !== 'paused' && migration.status !== 'completed'} - {#if storage?.entries && storage.entries.length > 0} - - - {:else} - - {/if} - {/if} - {#if migration.status === 'running'} - - {/if} - {#if migration.status === 'paused'} - - {/if} - +
+ +
+ + + {#if !entry.is_active && !migrationInFlight} + + {:else} + + {/if} +
+ +
+
{t('admin.entry_backend', 'Backend')}
+
{entry.backend}
+
{t('admin.entry_location', 'Location')}
+
{entry.location_hint ?? '—'}
+ {#if entry.is_active} +
{t('admin.storage_blobs', 'Blobs')}
+
{storage.total_blobs ?? '—'}
+
{t('admin.storage_size', 'Stored')}
+
+ {storage.total_bytes_stored != null + ? formatBytes(storage.total_bytes_stored) + : '—'} +
+
{t('admin.storage_dedup', 'Dedup ratio')}
+
+ {storage.dedup_ratio != null ? `${storage.dedup_ratio.toFixed(2)}x` : '—'} +
+ {/if} +
+ {#if test?.result != null || test?.error != null} +
+ {#if test.error} + {test.error} + {:else if test.result} + {@const ok = test.result.connected ?? false} + {@const rt = test.result.roundtrip_elapsed_ms} + {@const cleanup = test.result.cleanup_ok} + + + {ok + ? t('admin.storage_test_success', 'Read/write OK') + : t('admin.storage_test_failure', 'Test failed')} + {#if rt != null} + · {t('admin.storage_test_elapsed', { ms: rt }, '{{ms}} ms')} + {/if} + {#if cleanup === false} + · ⚠ {t('admin.storage_test_cleanup_warn', 'cleanup DELETE failed')} + {/if} + {#if !ok} + — {test.result.message} + {/if} + + {/if} +
+ {/if} + + {/each}
- {#if cutoverPending} - + +
+

+ {t('admin.mig_status', 'Migration status')}: + {migration?.status ?? '—'} + {#if migration?.status === 'running'} + + {/if} + {#if migration?.status === 'paused'} + + {/if} +

+ {#if migration && migration.total_blobs > 0} +
+
+
+

+ {migration.migrated_blobs} / {migration.total_blobs} ({migrationPct}%) + {#if migrationEtaMin != null} + · {t( + 'admin.mig_eta', + { min: migrationEtaMin }, + `~${migrationEtaMin} min remaining` + )} + {/if} +

+ {/if} + {#if migration?.failed_blobs && migration.failed_blobs.length > 0} +
+ + {t( + 'admin.mig_failed', + { n: migration.failed_blobs.length }, + `${migration.failed_blobs.length} failed blobs` + )} + +
{migration.failed_blobs.join('\n')}
+
+ {/if} +
+ + {#if migration?.status === 'completed' && storage.migration_readonly} +

- - {t('admin.mig_cutover_title', 'Cutover pending — update .env and restart')} + + {t('admin.mig_cutover_done_title', 'Migration complete — restart to switch')}

{t( - 'admin.mig_cutover_body', - { target: storage?.backend ?? '?', live: storage?.current_backend ?? '?' }, - 'Blobs are now on {{target}} but the server is still running on {{live}}. To switch, add these lines to your .env and restart the server.' + 'admin.mig_cutover_done_body', + { active: storage.active_entry_name ?? '?' }, + 'The DB pointer now names `{{active}}` as the active backend, but the running process is still bound to the previous entry. Restart the server to complete the cutover; boot picks up the new active entry and clears read-only mode automatically.' )}

-
{cutoverEnvLines.join('\n')}
-
- -

- {t( - 'admin.mig_cutover_secret_note', - 'The access key and secret key are placeholders — paste the values you entered when saving these settings. Credentials are never displayed here.' - )} -

-
{/if} {/if} + {#if storageMsg} +

{storageMsg.text}

+ {/if}
@@ -2361,7 +2168,7 @@

{t( 'admin.encryption_hint', - 'Generate an AES-256 key for at-rest blob encryption, then set it as OXICLOUD_STORAGE_ENCRYPTION_KEY in your server environment.' + 'Generate an AES-256 key for at-rest blob encryption. Set it as OXICLOUD_STORAGE__ENCRYPTION_KEY under an entry declared in OXICLOUD_STORAGE_ENTRIES — presence of the key implies encryption is enabled on that entry (no separate flag).' )}

@@ -4050,6 +4099,17 @@ background: var(--color-danger-bg, var(--color-bg-muted)); } + /* On the readonly banner the danger-tinted background swallows + `.muted` (which is a light grey). Use the strong text color + instead so the body message stays legible in both themes. + `color-danger-text` if the design system publishes one, else + fall back to the regular text color which still meets WCAG + contrast against the muted-red/pink bg tokens. */ + .cutover-hint__readonly-body { + margin: 0; + color: var(--color-danger-text, var(--color-text)); + } + .entries-list { display: flex; flex-direction: column; diff --git a/src/infrastructure/services/mod.rs b/src/infrastructure/services/mod.rs index 924c3115..b7ab9895 100644 --- a/src/infrastructure/services/mod.rs +++ b/src/infrastructure/services/mod.rs @@ -11,7 +11,6 @@ pub mod dedup_service; pub mod drives_consistency_service; pub mod encrypted_blob_backend; pub mod entry_backend; -pub mod swappable_blob_backend; pub mod exif_service; pub mod face_geometry; pub mod face_indexing_service; @@ -47,6 +46,7 @@ pub mod search_index; pub mod share_unlock_cookie; pub mod smtp_email_sender; pub mod storage_migration_service; +pub mod swappable_blob_backend; pub mod thumbnail_service; #[cfg(test)] mod thumbnail_service_test; diff --git a/src/infrastructure/services/s3_blob_backend.rs b/src/infrastructure/services/s3_blob_backend.rs index 36b60e8b..7e507546 100644 --- a/src/infrastructure/services/s3_blob_backend.rs +++ b/src/infrastructure/services/s3_blob_backend.rs @@ -421,10 +421,7 @@ impl BlobStorageBackend for S3BlobBackend { Ok(StorageHealthStatus { connected: false, backend_type: "s3".to_string(), - message: format!( - "S3 bucket '{}' is not accessible: {detail}", - self.bucket - ), + message: format!("S3 bucket '{}' is not accessible: {detail}", self.bucket), available_bytes: None, }) } @@ -595,7 +592,10 @@ where } SdkError::TimeoutError(_) => "timeout".to_string(), SdkError::ResponseError(r) => { - format!("malformed response (HTTP {}): {r:?}", r.raw().status().as_u16()) + format!( + "malformed response (HTTP {}): {r:?}", + r.raw().status().as_u16() + ) } SdkError::ConstructionFailure(c) => format!("request construction failed: {c:?}"), _ => format!("unknown SDK error: {err:?}"), diff --git a/src/infrastructure/services/storage_migration_service.rs b/src/infrastructure/services/storage_migration_service.rs index a8cfac4d..5a5f45aa 100644 --- a/src/infrastructure/services/storage_migration_service.rs +++ b/src/infrastructure/services/storage_migration_service.rs @@ -119,9 +119,8 @@ pub struct StorageMigrationService { /// restart. Shared with `CoreServices.blob_backend_hot_swap` — /// same instance the coerced `blob_backend: Arc` /// delegates through. - blob_backend_hot_swap: Arc< - crate::infrastructure::services::swappable_blob_backend::SwappableBlobBackend, - >, + blob_backend_hot_swap: + Arc, } impl StorageMigrationService { @@ -700,8 +699,9 @@ impl StorageMigrationService { // 4. Drop read-only. In this order (after swap) so no write // slips through against the OLD backend between "readonly // off" and "backend swapped". - let readonly_persisted = - persist_migration_readonly(self.pool.as_ref(), false).await.is_ok(); + let readonly_persisted = persist_migration_readonly(self.pool.as_ref(), false) + .await + .is_ok(); self.migration_readonly.store(false, Ordering::Relaxed); if !readonly_persisted { diff --git a/src/infrastructure/services/swappable_blob_backend.rs b/src/infrastructure/services/swappable_blob_backend.rs index 40c30d2c..e28abadb 100644 --- a/src/infrastructure/services/swappable_blob_backend.rs +++ b/src/infrastructure/services/swappable_blob_backend.rs @@ -126,11 +126,7 @@ impl BlobStorageBackend for SwappableBlobBackend { Box::pin(async move { inner.put_blob(&hash, &source_path).await }) } - fn put_blob_from_bytes( - &self, - hash: &str, - data: Bytes, - ) -> BoxFut<'_, Result> { + fn put_blob_from_bytes(&self, hash: &str, data: Bytes) -> BoxFut<'_, Result> { let inner = self.current(); let hash = hash.to_owned(); Box::pin(async move { inner.put_blob_from_bytes(&hash, data).await }) @@ -222,4 +218,3 @@ impl BlobStorageBackend for SwappableBlobBackend { Box::pin(async move { inner.list_blob_hashes(cursor, limit).await }) } } - From bbfb106a3244aa0638ea8fff9fddfd1c97a495d4 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 18:01:10 +0200 Subject: [PATCH 13/17] feat(maintenance): add a maintenance notification during backend migration --- frontend/src/lib/api/client.ts | 25 +++- frontend/src/lib/components/AppShell.svelte | 17 +++ .../src/lib/components/ReadOnlyBanner.svelte | 79 +++++++++--- .../src/lib/stores/serverStatus.svelte.ts | 67 ++++++++++ frontend/static/locales/ar.json | 6 + frontend/static/locales/de.json | 6 + frontend/static/locales/en.json | 6 + frontend/static/locales/es.json | 6 + frontend/static/locales/fa.json | 6 + frontend/static/locales/fr.json | 6 + frontend/static/locales/hi.json | 6 + frontend/static/locales/it.json | 6 + frontend/static/locales/ja.json | 6 + frontend/static/locales/ko.json | 6 + frontend/static/locales/nl.json | 6 + frontend/static/locales/pl.json | 6 + frontend/static/locales/pt.json | 6 + frontend/static/locales/ru.json | 6 + frontend/static/locales/zh-TW.json | 6 + frontend/static/locales/zh.json | 6 + src/common/di.rs | 10 ++ src/common/migration_progress.rs | 66 ++++++++++ src/common/mod.rs | 1 + .../services/storage_migration_service.rs | 54 ++++++++ src/interfaces/api/routes.rs | 25 ++-- src/interfaces/middleware/mod.rs | 1 + src/interfaces/middleware/server_status.rs | 119 ++++++++++++++++++ 27 files changed, 537 insertions(+), 23 deletions(-) create mode 100644 frontend/src/lib/stores/serverStatus.svelte.ts create mode 100644 src/common/migration_progress.rs create mode 100644 src/interfaces/middleware/server_status.rs diff --git a/frontend/src/lib/api/client.ts b/frontend/src/lib/api/client.ts index f388550c..d9024c67 100644 --- a/frontend/src/lib/api/client.ts +++ b/frontend/src/lib/api/client.ts @@ -18,6 +18,15 @@ */ import { getCsrfHeaders } from './csrf'; +import { updateFromHeader } from '$lib/stores/serverStatus.svelte'; + +/** + * Name of the response header the server stamps while a + * maintenance event is live. Case-insensitive on the wire — the + * Fetch API's `Headers.get` matches irrespective of case, so this + * constant matches whatever axum emits. + */ +const SERVER_STATUS_HEADER = 'x-server-status'; const REFRESH_ENDPOINT = '/api/auth/refresh'; @@ -93,6 +102,18 @@ export function createApiFetch(deps: ApiClientDeps): FetchFn { const apiFetch: FetchFn = async (input, init) => { const origin = deps.origin ?? globalThis.location?.origin ?? 'http://localhost'; const response = await rawFetch(input, init); + // Server-status header piggyback — the server stamps + // `x-server-status` on every response while a maintenance + // event is in progress (see middleware::server_status). Read + // it and update the reactive store; the AppShell banner + // subscribes and shows/hides itself. Absent header = nothing + // happening; the update fn resets the store to default in + // that case so a lingering banner disappears. + // + // Runs on EVERY response including a 401 (below) so a session + // refresh doesn't accidentally clear a live banner. + updateFromHeader(response.headers.get(SERVER_STATUS_HEADER)); + if (response.status !== 401) return response; const urlStr = urlString(input as RequestInfo | URL); @@ -104,7 +125,9 @@ export function createApiFetch(deps: ApiClientDeps): FetchFn { onSessionExpired(); throw new Error('Session expired'); } - return rawFetch(input, init); + const retryResponse = await rawFetch(input, init); + updateFromHeader(retryResponse.headers.get(SERVER_STATUS_HEADER)); + return retryResponse; }; return apiFetch; diff --git a/frontend/src/lib/components/AppShell.svelte b/frontend/src/lib/components/AppShell.svelte index 934610e8..d669b4dd 100644 --- a/frontend/src/lib/components/AppShell.svelte +++ b/frontend/src/lib/components/AppShell.svelte @@ -11,10 +11,12 @@ import type { FileItem, FolderItem, ItemType } from '$lib/api/types'; import { lazyComponent } from '$lib/composables/lazyComponent.svelte'; import DrivePicker from '$lib/components/DrivePicker.svelte'; + import ReadOnlyBanner from '$lib/components/ReadOnlyBanner.svelte'; import Icon from '$lib/icons/Icon.svelte'; import { dateTimeFormatFor, iconNameFromClass } from '$lib/utils/display'; import { userInitials, avatarColorIndex } from '$lib/utils/avatar'; import { i18n, LANGUAGES, setLocale, t, type Locale } from '$lib/i18n/index.svelte'; + import { serverStatus } from '$lib/stores/serverStatus.svelte'; import { apiFetch } from '$lib/api/client'; import { dialogs } from '$lib/stores/dialogs.svelte'; import { files as filesStore } from '$lib/stores/files.svelte'; @@ -1025,6 +1027,21 @@
+ + {#if serverStatus().readonly} + + {/if} {@render children()}
diff --git a/frontend/src/lib/components/ReadOnlyBanner.svelte b/frontend/src/lib/components/ReadOnlyBanner.svelte index cfee3dae..d7fd5491 100644 --- a/frontend/src/lib/components/ReadOnlyBanner.svelte +++ b/frontend/src/lib/components/ReadOnlyBanner.svelte @@ -1,6 +1,8 @@
- {#if driveName} + {#if variant === 'maintenance'} + {t('server_status.readonly_title', 'Server maintenance in progress')} + {:else if driveName} {t( 'drive.read_only_banner.title_named', { name: driveName }, @@ -56,10 +85,30 @@ {/if} - {t( - 'drive.read_only_banner.body', - 'Uploads, edits, deletes, renames, sharing and membership changes are refused. Reads and downloads keep working. Contact an administrator to un-freeze the drive.' - )} + {#if variant === 'maintenance'} + {#if progress} + {t( + 'server_status.readonly_progress', + { + target: progress.target, + migrated: progress.migrated, + total: progress.total, + percent: progress.percent + }, + 'Migrating storage to `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blobs). Uploads, renames, deletes, and shares are refused; reads and downloads work as normal.' + )} + {:else} + {t( + 'server_status.readonly_body', + 'Uploads, renames, deletes, and shares are refused temporarily. Reads and downloads work as normal.' + )} + {/if} + {:else} + {t( + 'drive.read_only_banner.body', + 'Uploads, edits, deletes, renames, sharing and membership changes are refused. Reads and downloads keep working. Contact an administrator to un-freeze the drive.' + )} + {/if}
diff --git a/frontend/src/lib/stores/serverStatus.svelte.ts b/frontend/src/lib/stores/serverStatus.svelte.ts new file mode 100644 index 00000000..a16d449b --- /dev/null +++ b/frontend/src/lib/stores/serverStatus.svelte.ts @@ -0,0 +1,67 @@ +/** + * Reactive server-status store. + * + * Populated by the `apiFetch` wrapper, which reads the + * `x-server-status` header off every API response and calls + * `updateFromHeader(...)`. When no migration is running the header + * is absent and the store stays at its default (readonly=false, no + * migration info). See `middleware::server_status` on the server + * for the header spec. + * + * The AppShell subscribes to this store to show/hide the + * maintenance banner without polling — the state travels back to + * the client on the piggyback of whatever API request the user was + * making anyway. Zero extra network cost. + */ + +/** + * JSON shape emitted in the `x-server-status` header. Optional + * `migration` field is present only while a migration is running. + */ +export interface ServerStatus { + readonly: boolean; + migration?: { + target: string; + migrated: number; + total: number; + percent: number; + }; +} + +const DEFAULT: ServerStatus = { readonly: false }; + +// Rune-based reactive state — `$state` in a `.svelte.ts` module. +let current = $state(DEFAULT); + +/** Current server status. Reactively updates when apiFetch sees a new header. */ +export function serverStatus(): ServerStatus { + return current; +} + +/** + * Parse the raw header value and update the store. Silently + * tolerates a missing header (resets to default: nothing to + * broadcast means nothing wrong) and a malformed one (keeps the + * previous value rather than surface a parse error to users). + * + * Called by `apiFetch` after every response — see `client.ts`. + */ +export function updateFromHeader(rawHeader: string | null): void { + if (rawHeader == null) { + // No header on this response = server not in maintenance + // mode = reset the store to the default so any lingering + // banner disappears. Cheap idempotent write. + if (current.readonly || current.migration) current = DEFAULT; + return; + } + try { + const parsed = JSON.parse(rawHeader) as ServerStatus; + // Basic shape validation — server should never send a + // missing `readonly`, but be defensive. + if (typeof parsed.readonly === 'boolean') { + current = parsed; + } + } catch { + // Malformed header — keep previous state rather than churn. + } +} diff --git a/frontend/static/locales/ar.json b/frontend/static/locales/ar.json index ca126d05..ed4f3d50 100644 --- a/frontend/static/locales/ar.json +++ b/frontend/static/locales/ar.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "صيانة الخادم جارية", + "readonly_title": "صيانة الخادم جارية", + "readonly_progress": "جارٍ نقل التخزين إلى `{{target}}` — {{percent}}٪ ({{migrated}} / {{total}} كتلة). الرفع وإعادة التسمية والحذف والمشاركة مرفوضة؛ القراءة والتنزيل تعملان بشكل طبيعي.", + "readonly_body": "الرفع وإعادة التسمية والحذف والمشاركة مرفوضة مؤقتًا. القراءة والتنزيل يعملان بشكل طبيعي." + }, "app": { "title": "OxiCloud", "description": "نظام تخزين سحابي بسيط" diff --git a/frontend/static/locales/de.json b/frontend/static/locales/de.json index 8ee18eb0..a635472d 100644 --- a/frontend/static/locales/de.json +++ b/frontend/static/locales/de.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Serverwartung läuft", + "readonly_title": "Serverwartung läuft", + "readonly_progress": "Speicher wird auf `{{target}}` migriert — {{percent}} % ({{migrated}} / {{total}} Blöcke). Uploads, Umbenennungen, Löschungen und Freigaben werden abgelehnt; Lesen und Herunterladen funktionieren normal.", + "readonly_body": "Uploads, Umbenennungen, Löschungen und Freigaben werden vorübergehend abgelehnt. Lesen und Herunterladen funktionieren normal." + }, "app": { "title": "OxiCloud", "description": "Minimalistisches Cloud-Speichersystem" diff --git a/frontend/static/locales/en.json b/frontend/static/locales/en.json index 78189299..ed307588 100644 --- a/frontend/static/locales/en.json +++ b/frontend/static/locales/en.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Server maintenance in progress", + "readonly_title": "Server maintenance in progress", + "readonly_progress": "Migrating storage to `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blobs). Uploads, renames, deletes, and shares are refused; reads and downloads work as normal.", + "readonly_body": "Uploads, renames, deletes, and shares are refused temporarily. Reads and downloads work as normal." + }, "app": { "title": "OxiCloud", "description": "Minimalist cloud storage system" diff --git a/frontend/static/locales/es.json b/frontend/static/locales/es.json index f72df3f0..84d0de60 100644 --- a/frontend/static/locales/es.json +++ b/frontend/static/locales/es.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Mantenimiento del servidor en curso", + "readonly_title": "Mantenimiento del servidor en curso", + "readonly_progress": "Migrando el almacenamiento a `{{target}}` — {{percent}} % ({{migrated}} / {{total}} bloques). Las subidas, renombres, eliminaciones y comparticiones se rechazan; las lecturas y descargas funcionan con normalidad.", + "readonly_body": "Las subidas, renombres, eliminaciones y comparticiones se rechazan temporalmente. Las lecturas y descargas funcionan con normalidad." + }, "app": { "title": "OxiCloud", "description": "Sistema de almacenamiento en la nube minimalista" diff --git a/frontend/static/locales/fa.json b/frontend/static/locales/fa.json index fdabf73f..6b6b467b 100644 --- a/frontend/static/locales/fa.json +++ b/frontend/static/locales/fa.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "نگهداری سرور در حال انجام است", + "readonly_title": "نگهداری سرور در حال انجام است", + "readonly_progress": "در حال انتقال حافظه به `{{target}}` — {{percent}}٪ ({{migrated}} / {{total}} بلاک). آپلود، تغییر نام، حذف و اشتراک‌گذاری رد می‌شوند؛ خواندن و دانلود عادی کار می‌کنند.", + "readonly_body": "آپلود، تغییر نام، حذف و اشتراک‌گذاری موقتاً رد می‌شوند. خواندن و دانلود عادی کار می‌کنند." + }, "app": { "title": "OxiCloud", "description": "سیستم ذخیره‌سازی ابری ساده‌گرا" diff --git a/frontend/static/locales/fr.json b/frontend/static/locales/fr.json index b406e717..100ce288 100644 --- a/frontend/static/locales/fr.json +++ b/frontend/static/locales/fr.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Maintenance du serveur en cours", + "readonly_title": "Maintenance du serveur en cours", + "readonly_progress": "Migration du stockage vers `{{target}}` — {{percent}} % ({{migrated}} / {{total}} blocs). Les téléversements, renommages, suppressions et partages sont refusés ; la lecture et le téléchargement continuent normalement.", + "readonly_body": "Les téléversements, renommages, suppressions et partages sont temporairement refusés. La lecture et le téléchargement fonctionnent normalement." + }, "app": { "title": "OxiCloud", "description": "Système de stockage cloud minimaliste" diff --git a/frontend/static/locales/hi.json b/frontend/static/locales/hi.json index 84b3af75..f079ddaf 100644 --- a/frontend/static/locales/hi.json +++ b/frontend/static/locales/hi.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "सर्वर रखरखाव प्रगति पर है", + "readonly_title": "सर्वर रखरखाव प्रगति पर है", + "readonly_progress": "स्टोरेज को `{{target}}` पर माइग्रेट किया जा रहा है — {{percent}}% ({{migrated}} / {{total}} ब्लॉब्स)। अपलोड, नाम बदलना, हटाना और साझा करना अस्वीकृत हैं; पढ़ना और डाउनलोड सामान्य रूप से काम करते हैं।", + "readonly_body": "अपलोड, नाम बदलना, हटाना और साझा करना अस्थायी रूप से अस्वीकृत हैं। पढ़ना और डाउनलोड सामान्य रूप से काम करते हैं।" + }, "app": { "title": "OxiCloud", "description": "न्यूनतम क्लाउड स्टोरेज सिस्टम" diff --git a/frontend/static/locales/it.json b/frontend/static/locales/it.json index 25ae6019..92ed7dca 100644 --- a/frontend/static/locales/it.json +++ b/frontend/static/locales/it.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Manutenzione del server in corso", + "readonly_title": "Manutenzione del server in corso", + "readonly_progress": "Migrazione dello storage verso `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blob). Caricamenti, rinomine, eliminazioni e condivisioni sono rifiutati; le letture e i download funzionano normalmente.", + "readonly_body": "Caricamenti, rinomine, eliminazioni e condivisioni sono temporaneamente rifiutati. Le letture e i download funzionano normalmente." + }, "app": { "title": "OxiCloud", "description": "Sistema di archiviazione cloud minimalista" diff --git a/frontend/static/locales/ja.json b/frontend/static/locales/ja.json index 000379b7..6a4e8206 100644 --- a/frontend/static/locales/ja.json +++ b/frontend/static/locales/ja.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "サーバーメンテナンス中", + "readonly_title": "サーバーメンテナンス中", + "readonly_progress": "ストレージを `{{target}}` に移行中 — {{percent}}%({{migrated}} / {{total}} ブロブ)。アップロード、名前変更、削除、共有は拒否されます。読み取りとダウンロードは通常どおり動作します。", + "readonly_body": "アップロード、名前変更、削除、共有は一時的に拒否されます。読み取りとダウンロードは通常どおり動作します。" + }, "app": { "title": "OxiCloud", "description": "ミニマリストクラウドストレージシステム" diff --git a/frontend/static/locales/ko.json b/frontend/static/locales/ko.json index 916b843b..56b1a5d3 100644 --- a/frontend/static/locales/ko.json +++ b/frontend/static/locales/ko.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "서버 유지 관리 진행 중", + "readonly_title": "서버 유지 관리 진행 중", + "readonly_progress": "저장소를 `{{target}}`(으)로 마이그레이션 중 — {{percent}}% ({{migrated}} / {{total}} 블롭). 업로드, 이름 변경, 삭제 및 공유가 거부됩니다. 읽기 및 다운로드는 정상적으로 작동합니다.", + "readonly_body": "업로드, 이름 변경, 삭제 및 공유가 일시적으로 거부됩니다. 읽기 및 다운로드는 정상적으로 작동합니다." + }, "app": { "title": "OxiCloud", "description": "미니멀리스트 클라우드 스토리지 시스템" diff --git a/frontend/static/locales/nl.json b/frontend/static/locales/nl.json index 4cfc5da7..c62bc1d5 100644 --- a/frontend/static/locales/nl.json +++ b/frontend/static/locales/nl.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Serveronderhoud bezig", + "readonly_title": "Serveronderhoud bezig", + "readonly_progress": "Opslag wordt gemigreerd naar `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blobs). Uploads, hernoemingen, verwijderingen en delen worden geweigerd; lezen en downloaden werken normaal.", + "readonly_body": "Uploads, hernoemingen, verwijderingen en delen worden tijdelijk geweigerd. Lezen en downloaden werken normaal." + }, "app": { "title": "OxiCloud", "description": "Minimalistisch cloudopslagsysteem" diff --git a/frontend/static/locales/pl.json b/frontend/static/locales/pl.json index 7dc8fec3..7b3ab098 100644 --- a/frontend/static/locales/pl.json +++ b/frontend/static/locales/pl.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Trwa konserwacja serwera", + "readonly_title": "Trwa konserwacja serwera", + "readonly_progress": "Migracja pamięci do `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blobów). Przesyłanie, zmiana nazwy, usuwanie i udostępnianie są odrzucane; odczyt i pobieranie działają normalnie.", + "readonly_body": "Przesyłanie, zmiana nazwy, usuwanie i udostępnianie są tymczasowo odrzucane. Odczyt i pobieranie działają normalnie." + }, "app": { "title": "OxiCloud", "description": "Minimalistyczny cloud storage" diff --git a/frontend/static/locales/pt.json b/frontend/static/locales/pt.json index 0b9c4ade..c7670445 100644 --- a/frontend/static/locales/pt.json +++ b/frontend/static/locales/pt.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Manutenção do servidor em curso", + "readonly_title": "Manutenção do servidor em curso", + "readonly_progress": "A migrar o armazenamento para `{{target}}` — {{percent}}% ({{migrated}} / {{total}} blobs). Envios, renomeações, eliminações e partilhas são recusados; leituras e transferências funcionam normalmente.", + "readonly_body": "Envios, renomeações, eliminações e partilhas são temporariamente recusados. Leituras e transferências funcionam normalmente." + }, "app": { "title": "OxiCloud", "description": "Sistema de armazenamento em nuvem minimalista" diff --git a/frontend/static/locales/ru.json b/frontend/static/locales/ru.json index adf05ad2..87f760fc 100644 --- a/frontend/static/locales/ru.json +++ b/frontend/static/locales/ru.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "Идёт обслуживание сервера", + "readonly_title": "Идёт обслуживание сервера", + "readonly_progress": "Миграция хранилища на `{{target}}` — {{percent}}% ({{migrated}} / {{total}} блобов). Загрузки, переименования, удаления и общий доступ отклоняются; чтение и скачивание работают как обычно.", + "readonly_body": "Загрузки, переименования, удаления и общий доступ временно отклоняются. Чтение и скачивание работают как обычно." + }, "app": { "title": "OxiCloud", "description": "Минималистичная система облачного хранения" diff --git a/frontend/static/locales/zh-TW.json b/frontend/static/locales/zh-TW.json index a11e7ba5..b57dc99f 100644 --- a/frontend/static/locales/zh-TW.json +++ b/frontend/static/locales/zh-TW.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "伺服器維護進行中", + "readonly_title": "伺服器維護進行中", + "readonly_progress": "正在將儲存遷移至 `{{target}}` — {{percent}}%({{migrated}} / {{total}} 個 blob)。上傳、重新命名、刪除和分享會被拒絕;讀取和下載正常運作。", + "readonly_body": "上傳、重新命名、刪除和分享暫時被拒絕。讀取和下載正常運作。" + }, "app": { "title": "OxiCloud", "description": "極簡雲端儲存系統" diff --git a/frontend/static/locales/zh.json b/frontend/static/locales/zh.json index 90c2effb..5cc129f8 100644 --- a/frontend/static/locales/zh.json +++ b/frontend/static/locales/zh.json @@ -38,6 +38,12 @@ } } }, + "server_status": { + "readonly_banner_aria": "服务器维护进行中", + "readonly_title": "服务器维护进行中", + "readonly_progress": "正在将存储迁移到 `{{target}}` — {{percent}}%({{migrated}} / {{total}} 个 blob)。上传、重命名、删除和共享被拒绝;读取和下载正常工作。", + "readonly_body": "上传、重命名、删除和共享暂时被拒绝。读取和下载正常工作。" + }, "app": { "title": "OxiCloud", "description": "极简云存储系统" diff --git a/src/common/di.rs b/src/common/di.rs index acc79189..ca555373 100644 --- a/src/common/di.rs +++ b/src/common/di.rs @@ -2028,6 +2028,7 @@ impl AppServiceFactory { crate::infrastructure::services::webdav_dead_property_store::create_dead_property_store(pool.clone()), authorization: authorization.clone(), migration_readonly: migration_readonly.clone(), + migration_progress: Arc::new(std::sync::RwLock::new(None)), drive_repo: drive_repo.clone(), drive_management_service: Arc::new( crate::application::services::drive_management_service::DriveManagementService::new( @@ -2254,6 +2255,7 @@ impl AppServiceFactory { self.storage_path.clone(), app_state.migration_readonly.clone(), app_state.core.blob_backend_hot_swap.clone(), + app_state.migration_progress.clone(), ), ) .register_recoverable_job(&app_state.core.job_registry, &job_store_provider_dyn) @@ -2785,6 +2787,14 @@ pub struct AppState { /// memory in sync. See `docs/plan/storage-multi-entry.md` /// §"Read-only mode". pub migration_readonly: Arc, + /// Live progress snapshot for the storage-migration handler. + /// `Some(_)` while a migration is running; `None` otherwise. + /// Updated by the handler on every batch checkpoint (cheap + /// in-memory write, no DB read on the request path). The + /// server-status header middleware reads it to inform every + /// user's session banner about maintenance progress without + /// polling. See `MigrationProgress` for the field shape. + pub migration_progress: Arc>>, /// Drive entity repository — `GET /api/drives`, the personal-drive /// lifecycle hook, and (post-D2) shared-drive creation flow all read /// through this. Backing table is `storage.drives`; membership is diff --git a/src/common/migration_progress.rs b/src/common/migration_progress.rs new file mode 100644 index 00000000..9f78b7ff --- /dev/null +++ b/src/common/migration_progress.rs @@ -0,0 +1,66 @@ +//! Shared in-memory snapshot of the running storage-migration. +//! +//! Consumed by the server-status middleware to build the +//! `X-Server-Status` response header on every authenticated +//! request. That header lets every logged-in user's session banner +//! show current maintenance progress without any polling — +//! the state travels back on the piggyback of whatever API call +//! the user was going to make anyway. +//! +//! Written by the migration handler on each batch checkpoint (a +//! cheap `RwLock::write` + a small struct copy — no DB access on +//! the request path). Cleared on `RunOutcome::Completed` / +//! `Paused` / `Failed`. `None` means "no migration is running"; +//! middleware omits the header entirely in that case. + +use serde::Serialize; + +/// One snapshot of a running migration. Every field is a scalar so +/// the whole struct copies cheaply under the `RwLock::write` guard. +#[derive(Debug, Clone, Serialize)] +pub struct MigrationProgress { + /// The entry name blobs are being copied INTO. Used by the + /// user-facing banner text so admins/users know what the + /// server is switching to. + pub target_name: String, + /// Blobs migrated so far this run. Starts at 0 on a Fresh + /// open; on Resume the checkpointed value is loaded from the + /// run row's `stats.scanned_count`. + pub migrated_blobs: u64, + /// Total blobs in the current DB snapshot. Captured once at + /// run start via `SELECT COUNT(*) FROM storage.blobs`. Doesn't + /// change during the run — new uploads are refused while + /// read-only is engaged, so the denominator stays honest. + pub total_blobs: u64, + /// Convenience: `migrated_blobs * 100 / total_blobs`, clamped + /// to 0..=100. Middleware could compute it but it's tiny and + /// makes the JSON payload obvious. + pub percent: u8, +} + +impl MigrationProgress { + pub fn new(target_name: String, total_blobs: u64) -> Self { + Self { + target_name, + migrated_blobs: 0, + total_blobs, + percent: 0, + } + } + + /// Update the counter + recompute `percent`. Called by the + /// migration handler after each batch checkpoint. + pub fn bump(&mut self, migrated_delta: u64) { + self.migrated_blobs = self.migrated_blobs.saturating_add(migrated_delta); + self.recompute_percent(); + } + + fn recompute_percent(&mut self) { + self.percent = if self.total_blobs == 0 { + 0 + } else { + ((self.migrated_blobs.min(self.total_blobs) as u128 * 100) / self.total_blobs as u128) + as u8 + }; + } +} diff --git a/src/common/mod.rs b/src/common/mod.rs index b581e78a..ed8cc6a5 100644 --- a/src/common/mod.rs +++ b/src/common/mod.rs @@ -3,6 +3,7 @@ pub mod di; pub mod errors; pub mod fmt; pub mod locale; +pub mod migration_progress; pub mod mime_detect; pub mod runtime; pub mod stubs; diff --git a/src/infrastructure/services/storage_migration_service.rs b/src/infrastructure/services/storage_migration_service.rs index 5a5f45aa..6436ed47 100644 --- a/src/infrastructure/services/storage_migration_service.rs +++ b/src/infrastructure/services/storage_migration_service.rs @@ -121,6 +121,12 @@ pub struct StorageMigrationService { /// delegates through. blob_backend_hot_swap: Arc, + /// Shared in-memory progress snapshot. `Some(_)` during a + /// running/paused migration, `None` otherwise. Read by the + /// server-status header middleware to broadcast maintenance + /// state to every user's session without polling. + migration_progress: + Arc>>, } impl StorageMigrationService { @@ -135,6 +141,9 @@ impl StorageMigrationService { blob_backend_hot_swap: Arc< crate::infrastructure::services::swappable_blob_backend::SwappableBlobBackend, >, + migration_progress: Arc< + std::sync::RwLock>, + >, ) -> Self { Self { pool, @@ -144,6 +153,7 @@ impl StorageMigrationService { storage_path_fallback, migration_readonly, blob_backend_hot_swap, + migration_progress, } } @@ -384,6 +394,27 @@ impl RecoverableJobHandler for StorageMigrationService { cutover hot-swap completes" ); + // Seed the shared progress snapshot for the header + // middleware. Total blob count is a one-shot SELECT COUNT(*) + // — best-effort; if it fails we still push a snapshot with + // total=0 so the banner at least shows *something* is + // happening. + let total_blobs: u64 = sqlx::query_scalar::<_, i64>("SELECT COUNT(*) FROM storage.blobs") + .fetch_one(self.pool.as_ref()) + .await + .map(|n| n.max(0) as u64) + .unwrap_or(0); + { + let mut guard = self + .migration_progress + .write() + .unwrap_or_else(std::sync::PoisonError::into_inner); + *guard = Some(crate::common::migration_progress::MigrationProgress::new( + target_name.clone(), + total_blobs, + )); + } + let source_kind = self.source.backend_type(); let target_kind = target.backend_type(); tracing::info!( @@ -613,6 +644,19 @@ impl RecoverableJobHandler for StorageMigrationService { message: format!("checkpoint: {e}"), }; } + // Bump the shared progress snapshot so the server-status + // header middleware surfaces fresh numbers on every + // user's next API call. Guard is held only for a struct + // update — microseconds. + { + let mut guard = self + .migration_progress + .write() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if let Some(progress) = guard.as_mut() { + progress.bump(batch_len); + } + } if (rows.len() as i64) < BATCH_SIZE { return self @@ -703,6 +747,16 @@ impl StorageMigrationService { .await .is_ok(); self.migration_readonly.store(false, Ordering::Relaxed); + // Clear the shared progress snapshot so the server-status + // header stops emitting on subsequent requests. Guard held + // only for the assignment. + { + let mut guard = self + .migration_progress + .write() + .unwrap_or_else(std::sync::PoisonError::into_inner); + *guard = None; + } if !readonly_persisted { tracing::warn!( diff --git a/src/interfaces/api/routes.rs b/src/interfaces/api/routes.rs index 8a72d582..10ad9281 100644 --- a/src/interfaces/api/routes.rs +++ b/src/interfaces/api/routes.rs @@ -687,13 +687,24 @@ pub fn create_api_routes(app_state: &Arc) -> Router> { // them on every overlapping request. router = router.route("/{*rest}", any(api_not_found)); - // No per-router layers: the global `TraceLayer` + request-id stack in - // `main.rs` wraps the whole app (this `/api` router is nested into it), - // so a second `TraceLayer` here just double-wrapped every `/api` - // request in a redundant span + response-future poll (benches/ROUND13.md - // §H1). Compression is likewise the global layer's job — re-applying it - // here (no predicate) would compress media downloads, burning CPU for - // ~0 gain and stripping `Content-Length`. + // Server-status header. Stamps `X-Server-Status` on every + // response so the frontend's fetch wrapper can update a + // reactive store — banner shows/hides without polling. + // Sub-nanosecond on the hot path (single atomic load), a few + // µs on the cold path (only during a running migration). See + // `middleware::server_status`. + let router = router.layer(axum::middleware::from_fn_with_state( + app_state.clone(), + crate::interfaces::middleware::server_status::server_status_middleware, + )); + + // No per-router layers beyond that: the global `TraceLayer` + request-id + // stack in `main.rs` wraps the whole app (this `/api` router is nested + // into it), so a second `TraceLayer` here just double-wrapped every + // `/api` request in a redundant span + response-future poll + // (benches/ROUND13.md §H1). Compression is likewise the global layer's + // job — re-applying it here (no predicate) would compress media + // downloads, burning CPU for ~0 gain and stripping `Content-Length`. router } diff --git a/src/interfaces/middleware/mod.rs b/src/interfaces/middleware/mod.rs index 2ef51d39..8dbe66e3 100644 --- a/src/interfaces/middleware/mod.rs +++ b/src/interfaces/middleware/mod.rs @@ -3,6 +3,7 @@ pub mod auth; pub mod csrf; pub mod locale; pub mod rate_limit; +pub mod server_status; pub mod trace_span; pub mod trusted_proxy; pub mod user; diff --git a/src/interfaces/middleware/server_status.rs b/src/interfaces/middleware/server_status.rs new file mode 100644 index 00000000..2df6c57a --- /dev/null +++ b/src/interfaces/middleware/server_status.rs @@ -0,0 +1,119 @@ +//! Middleware that stamps `X-Server-Status` on every response. +//! +//! Consumed by the frontend `apiFetch` wrapper — every API round-trip +//! carries the current server maintenance state back to the client +//! (no polling, no dedicated endpoint). The banner in the app shell +//! subscribes to a store the wrapper updates and shows/hides itself +//! reactively. See `docs/plan/storage-multi-entry.md` §"Read-only mode" +//! for the broader design. +//! +//! ## Cost model +//! +//! On the *hot path* (no migration running — the ~100% case in normal +//! operation) this middleware does: +//! 1. one `AtomicBool::load(Relaxed)` — sub-nanosecond; +//! 2. an early return when `false`. +//! +//! No allocation, no lock, no formatting. Adds no measurable latency +//! at any user count. +//! +//! On the *cold path* (migration in progress) this middleware does: +//! 1. the atomic load above; +//! 2. one `RwLock::read` (uncontended — writers are the migration +//! handler, one per batch every ~100 blobs); +//! 3. one small `serde_json::to_string` call on a 4-field struct +//! (a few dozen bytes); +//! 4. one header insertion. +//! +//! Total per-request work in this branch: microseconds. + +use axum::extract::Request; +use axum::extract::State; +use axum::http::HeaderValue; +use axum::middleware::Next; +use axum::response::Response; +use std::sync::Arc; +use std::sync::atomic::Ordering; + +use crate::common::di::AppState; + +/// Name of the response header the frontend reads. Kept short — an +/// admin browser session may keep this header around in every open +/// tab's dev-tools network view during a migration; the value is +/// small JSON but the name should not add bloat. +pub const SERVER_STATUS_HEADER: &str = "x-server-status"; + +/// Compact JSON shape written into the header. Fields are documented +/// in `common::migration_progress::MigrationProgress`. +/// +/// Kept internal so the wire format can evolve. Frontend treats the +/// header as opaque JSON and pattern-matches on the fields it +/// currently understands. +#[derive(serde::Serialize)] +struct HeaderPayload { + readonly: bool, + #[serde(skip_serializing_if = "Option::is_none")] + migration: Option, +} + +#[derive(serde::Serialize)] +struct MigrationHeader { + // `target` is owned here — the RwLock guard is released before + // serialisation, so a borrowed slice wouldn't survive. Names + // are small (`[a-z0-9_-]{1,32}`) so the copy is trivial. + target: String, + migrated: u64, + total: u64, + percent: u8, +} + +pub async fn server_status_middleware( + State(state): State>, + request: Request, + next: Next, +) -> Response { + // Hot-path fast return. When no migration is running the flag is + // false and there's nothing to emit — a bare atomic load and out. + let readonly = state.migration_readonly.load(Ordering::Relaxed); + let mut response = next.run(request).await; + if !readonly { + return response; + } + + // Cold path — build the payload from the shared progress + // snapshot. If the snapshot is absent (readonly is true but the + // handler hasn't seeded progress yet, or a restart-during- + // migration scenario) we still emit `readonly: true` so the + // banner shows — the frontend renders a "maintenance in progress" + // message even when specific numbers aren't available. + let payload = { + let guard = state + .migration_progress + .read() + .unwrap_or_else(std::sync::PoisonError::into_inner); + HeaderPayload { + readonly: true, + migration: guard.as_ref().map(|p| MigrationHeader { + target: p.target_name.clone(), + migrated: p.migrated_blobs, + total: p.total_blobs, + percent: p.percent, + }), + } + }; + + // `serde_json::to_string` on this 4-field struct is a few + // dozen-byte allocation — negligible against the response body. + // A serialize failure here would be a programming bug (all + // fields are trivially serializable), so we degrade to a + // minimal `readonly: true` string rather than skipping the + // header entirely. + let value = + serde_json::to_string(&payload).unwrap_or_else(|_| r#"{"readonly":true}"#.to_string()); + if let Ok(header_value) = HeaderValue::from_str(&value) { + response + .headers_mut() + .insert(SERVER_STATUS_HEADER, header_value); + } + response +} From c31b8b814d2c18e481a868f0b005e37605aea8c6 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 19:58:38 +0200 Subject: [PATCH 14/17] fix(oidc): change the test --- tests/common/server-with-oidc.env | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/common/server-with-oidc.env b/tests/common/server-with-oidc.env index a0931733..8b33152e 100644 --- a/tests/common/server-with-oidc.env +++ b/tests/common/server-with-oidc.env @@ -80,4 +80,4 @@ OXICLOUD_OIDC_PROVIDER_NAME=MockSSO OXICLOUD_OIDC_ADMIN_GROUPS=admin-users OXICLOUD_AUTH_METHODS=password,magic_link -OXICLOUD_REQUIRE_VERIFIED_EMAIL=false +OXICLOUD_REQUIRE_VERIFIED_EMAIL=true From 88921c975a0fe1bf288adf5042d7c23538bcf121 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 20:08:17 +0200 Subject: [PATCH 15/17] fix(hurl test): add new job --- tests/api/admin_jobs.hurl | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/api/admin_jobs.hurl b/tests/api/admin_jobs.hurl index 03fe890f..35991bc3 100644 --- a/tests/api/admin_jobs.hurl +++ b/tests/api/admin_jobs.hurl @@ -95,13 +95,16 @@ jsonpath "$..interval_ms" count == 3 # files_consistency, blobs_consistency, backend_consistency — # wrapped by RecoverableAdapter so they appear here alongside the # periodics) + 1 coordinator (consistency_batch — a plain -# JobHandler that dispatches every registered `*_consistency`). +# JobHandler that dispatches every registered `*_consistency`) + +# 1 on-demand admin op (storage_migration — recoverable, no +# periodic tick, triggered by the admin panel's backend cutover). # Bump when a new tenant registers. -jsonpath "$..running" count == 10 +jsonpath "$..running" count == 11 jsonpath "$[*].name" contains "drives_consistency" jsonpath "$[*].name" contains "folders_consistency" jsonpath "$[*].name" contains "files_consistency" jsonpath "$[*].name" contains "consistency_batch" +jsonpath "$[*].name" contains "storage_migration" # ───────────────────────────────────────────────────────────── From 836c7a57c1acfa875dee5c5529d7b780ee080280 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 20:30:33 +0200 Subject: [PATCH 16/17] audit(RUSTSEC-2026-0222): inhibit alert, wasmtime plugin are not used per today --- .cargo/audit.toml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/.cargo/audit.toml b/.cargo/audit.toml index 57ac4780..7a63df25 100644 --- a/.cargo/audit.toml +++ b/.cargo/audit.toml @@ -40,6 +40,24 @@ ignore = [ "RUSTSEC-2026-0195", "RUSTSEC-2026-0194", + # wasmtime 43.0.2 — "Stores can mix up type indices between engines" + # (GHSA-hgjw-h833-99q9). Transitive via extism 1.30.0 (latest published; + # extism `main` still pins wasmtime 43, no upgrade path). The advisory + # has no patched 43.x — fix requires wasmtime >=46.0.2 or >=47.0.3, and + # forcing that via [patch.crates-io] would break extism (three major + # wasmtime API bumps between 43 and 46). Real fix waits on extism + # upstream to migrate. + # + # Runtime exposure is zero in default deployments: + # - `plugins` is an OPT-IN build feature; default builds and the CI + # release binary don't link wasmtime at all. + # - Runtime activation additionally requires OXICLOUD_ENABLE_PLUGINS=true. + # - Plugin binaries are ADMIN-SUPPLIED, not attacker input. + # - The advisory's attack pattern is multi-Engine Store-sharing; + # OxiCloud's plugin runtime creates one fresh Plugin per invocation + # with its own Store (see infrastructure/services/plugins/runtime.rs). + "RUSTSEC-2026-0222", + # astral-tokio-tar 0.5.6 — tar extraction advisories, transitive via # testcontainers → testcontainers-modules, a DEV-dependency used only by # the `--cfg integration_tests` harness to spin up throwaway Postgres From f1c72f8837b8baef580d84b8cc50fea32e771000 Mon Sep 17 00:00:00 2001 From: Edouard Vanbelle Date: Sat, 1 Aug 2026 20:35:04 +0200 Subject: [PATCH 17/17] test(storage): adapt playwright admin tests --- tests/e2e/spa/admin.spec.ts | 47 +++++++++++++------------------------ 1 file changed, 16 insertions(+), 31 deletions(-) diff --git a/tests/e2e/spa/admin.spec.ts b/tests/e2e/spa/admin.spec.ts index 26285b89..3a749771 100644 --- a/tests/e2e/spa/admin.spec.ts +++ b/tests/e2e/spa/admin.spec.ts @@ -30,7 +30,12 @@ test('walk every admin tab', async ({ page }) => { await expect(page.getByTestId('admin-oidc-form')).toBeVisible(); await page.goto('/admin/storage'); - await expect(page.getByTestId('admin-storage-form')).toBeVisible(); + // Storage panel is now a card-per-entry list (multi-entry config + // landed with `feat(storage): add multi entry in config`). The old + // single `admin-storage-form` was retired along with the in-UI save + // flow — backend is env-configured per entry, and the panel exposes + // Test / Blob-consistency / Migrate actions per card. + await expect(page.getByTestId('admin-storage-entries-list')).toBeVisible(); await page.goto('/admin/smtp'); await expect(page.getByTestId('admin-smtp-send-btn')).toBeVisible(); @@ -48,20 +53,17 @@ test('open the create-user form', async ({ page }) => { await expect(page.getByTestId('admin-create-user-form')).toBeVisible({ timeout: 15_000 }); }); -test('storage tab: change backend select', async ({ page }) => { +test('storage tab: test button probes the active entry', async ({ page }) => { await page.goto('/admin/storage'); - await expect(page.getByTestId('admin-storage-form')).toBeVisible({ timeout: 15_000 }); - await page.getByTestId('admin-storage-backend-select').selectOption({ index: 1 }).catch(() => {}); -}); - -test('storage tab: save the local backend settings', async ({ page }) => { - await page.goto('/admin/storage'); - await expect(page.getByTestId('admin-storage-form')).toBeVisible({ timeout: 15_000 }); - // Keep the (safe) local backend and save — exercises the save handler without - // reconfiguring storage to a remote backend. - await page.getByTestId('admin-storage-backend-select').selectOption('local').catch(() => {}); - await page.getByTestId('admin-storage-save-btn').click().catch(() => {}); - await expect(page.getByTestId('admin-storage-form')).toBeVisible(); + await expect(page.getByTestId('admin-storage-entries-list')).toBeVisible({ timeout: 15_000 }); + // Test button is safe — the handler does a read/write/delete probe on the + // backend without mutating any user data. There's always at least one entry + // in the E2E env (the default `local` storage entry seeded from example.env). + const testBtn = page.locator('[data-testid^="admin-storage-test-"]').first(); + await testBtn.click(); + // Result footer renders (either status--ok or status--error) once the probe + // completes; either outcome exercises the endpoint + render path. + await expect(page.locator('.entry-card__test-result').first()).toBeVisible({ timeout: 10_000 }); }); test('oidc tab: toggle enabled and fill issuer', async ({ page }) => { @@ -247,23 +249,6 @@ test('send a test email from the smtp tab', async ({ page }) => { await expect(page.getByTestId('admin-smtp-send-btn')).toBeVisible(); }); -test('storage tab: fill the S3 backend fields', async ({ page }) => { - await page.goto('/admin/storage'); - await expect(page.getByTestId('admin-storage-form')).toBeVisible({ timeout: 15_000 }); - - // Switch to S3 to reveal + fill the conditional fields (no save — that would - // reconfigure storage to an unreachable backend). - await page.getByTestId('admin-storage-backend-select').selectOption('s3').catch(() => {}); - await page.getByTestId('admin-storage-endpoint-input').fill('https://s3.example.test', { timeout: 2_000 }).catch(() => {}); - await page.getByTestId('admin-storage-bucket-input').fill('e2e-bucket', { timeout: 2_000 }).catch(() => {}); - await page.getByTestId('admin-storage-region-input').fill('us-east-1', { timeout: 2_000 }).catch(() => {}); - await page.getByTestId('admin-storage-access-key-input').fill('AKIA', { timeout: 2_000 }).catch(() => {}); - await page.getByTestId('admin-storage-secret-key-input').fill('secret', { timeout: 2_000 }).catch(() => {}); - await page.getByTestId('admin-storage-path-style-checkbox').click({ timeout: 2_000 }).catch(() => {}); - // Switch back to the safe local backend. - await page.getByTestId('admin-storage-backend-select').selectOption('local').catch(() => {}); -}); - test('oidc tab: run discovery against a bogus issuer', async ({ page }) => { await page.goto('/admin/oidc'); await expect(page.getByTestId('admin-oidc-form')).toBeVisible({ timeout: 15_000 });