fix(storage migration): overwrite target blob if mismatch

if target blob already exists, migration will check the blob
    if header is already with the targeted key (or no key if not ciphered)  no need write
    otherwise write the blob (that will convert any blob with no header into the correct version)
This commit is contained in:
Edouard Vanbelle
2026-08-02 14:13:55 +02:00
parent 9902a6f8fe
commit a10d3254bf
5 changed files with 169 additions and 80 deletions
+13 -4
View File
@@ -152,10 +152,19 @@ pub trait BlobStorageBackend: Send + Sync + 'static {
/// overwrites atomically). /// overwrites atomically).
/// ///
/// Default: delegates to `put_blob_from_bytes`. That default is /// Default: delegates to `put_blob_from_bytes`. That default is
/// CORRECT for backends whose `put_blob_from_bytes` already /// WRONG for every current backend — Local uses `O_EXCL` and
/// overwrites (S3, Azure — object stores overwrite on PUT by /// S3/Azure both HEAD-probe before writing, so `put_blob_from_bytes`
/// default). Backends whose `put_blob_from_bytes` is /// is silently a no-op when the target already exists. Every
/// idempotent-skip (like `LocalBlobBackend`) MUST override. /// production backend MUST override this to guarantee overwrite:
///
/// - `LocalBlobBackend` — tempfile + rename(2) + fsync
/// - `S3BlobBackend` / `AzureBlobBackend` — unconditional PUT
/// (same body as `put_blob_from_bytes_unsynced`, which already
/// skips the HEAD probe; object-store PUTs are durable on
/// return so no separate sync is needed)
///
/// The default remains only for the `#[cfg(test)]` mocks that
/// never exercise rotate/migrate.
fn put_blob_from_bytes_replace( fn put_blob_from_bytes_replace(
&self, &self,
hash: &str, hash: &str,
@@ -173,6 +173,23 @@ impl BlobStorageBackend for AzureBlobBackend {
}) })
} }
/// Atomic overwrite path used by `storage_rotate` and
/// `storage_migration` when re-writing an already-present blob
/// under a new head key/format. Trait default delegates to
/// `put_blob_from_bytes` which `get_properties`-probes and
/// silently skips — exactly wrong for the rotate/migrate use
/// case (the whole point is to replace the existing bytes).
/// Override delegates to the same unconditional PUT as
/// `put_blob_from_bytes_unsynced` — Azure's PUT is durable on
/// return, no separate sync barrier needed.
fn put_blob_from_bytes_replace(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
self.put_blob_from_bytes_unsynced(hash, data)
}
fn get_blob_stream( fn get_blob_stream(
&self, &self,
hash: &str, hash: &str,
@@ -237,6 +237,48 @@ impl EncryptedBlobBackend {
} }
} }
/// Smart-skip probe used by `storage_migration` (and potentially
/// `storage_rotate` if it ever gains a fast-path). Reads the
/// first [`HEADER_SIZE`] bytes of the on-disk blob and returns
/// `true` iff:
///
/// * the blob exists, AND
/// * its first bytes carry the v1 header (`OXCPT | 0001`), AND
/// * the header's `key_fp` matches this wrapper's head key
/// (encrypted-v1 → `head_key_fp`; plaintext-v1 → all-zero).
///
/// Any error (blob missing, short read, IO failure) returns
/// `Ok(false)` — the caller re-writes, which is always safe.
/// Only propagates errors that the caller genuinely cannot
/// distinguish from "blob absent" and needs to see.
///
/// Backend-agnostic: reads through the trait's
/// `get_blob_range_stream` on the *inner* backend (bypasses this
/// wrapper's decrypt so we see the raw on-disk header bytes).
/// Local pays one `pread` syscall; S3 pays one HEAD/GET with
/// `Range: bytes=0-14`; Azure the same.
pub async fn is_at_head_format(&self, hash: &str) -> Result<bool, DomainError> {
// Skip the probe entirely for backends that don't stream —
// `get_blob_range_stream` would fail on a legit-missing blob
// with `NotFound` which we want to translate to `Ok(false)`.
let stream = match self
.inner
.get_blob_range_stream(hash, 0, Some(HEADER_SIZE as u64))
.await
{
Ok(s) => s,
Err(_) => return Ok(false),
};
let raw = match collect_stream(stream).await {
Ok(b) => b,
Err(_) => return Ok(false),
};
if raw.len() < HEADER_SIZE {
return Ok(false);
}
Ok(BlobFormat::classify(&raw) == self.head_format())
}
/// Fetch, classify, and decrypt a blob in one round-trip. Used by /// Fetch, classify, and decrypt a blob in one round-trip. Used by
/// K3's `storage_rotate` per-blob step: it needs both the /// K3's `storage_rotate` per-blob step: it needs both the
/// plaintext (to re-encrypt under the head pair) AND the current /// plaintext (to re-encrypt under the head pair) AND the current
@@ -206,6 +206,11 @@ impl BlobStorageBackend for S3BlobBackend {
/// dedup layer already filtered out chunks the database knows about, /// dedup layer already filtered out chunks the database knows about,
/// so the probe was a pure extra round-trip on every NEW chunk of /// so the probe was a pure extra round-trip on every NEW chunk of
/// every upload (2 RTTs -> 1, benches/S3-PUT.md). /// every upload (2 RTTs -> 1, benches/S3-PUT.md).
///
/// Shares the body with `put_blob_from_bytes_replace` below —
/// S3 PUT is durable on return, so "unsynced" and "replace"
/// collapse to the same semantics here (unlike Local, where
/// `_replace` needs tempfile-rename + fsync).
fn put_blob_from_bytes_unsynced( fn put_blob_from_bytes_unsynced(
&self, &self,
hash: &str, hash: &str,
@@ -232,6 +237,23 @@ impl BlobStorageBackend for S3BlobBackend {
}) })
} }
/// Atomic overwrite path used by `storage_rotate` and
/// `storage_migration` when re-writing an already-present blob
/// under a new head key/format. Trait default delegates to
/// `put_blob_from_bytes` which HEAD-probes and silently skips —
/// exactly wrong for the rotate/migrate use case (the whole
/// point is to replace the existing bytes). Override delegates
/// to the same unconditional PUT as `put_blob_from_bytes_unsynced`
/// — S3's PUT is durable on return, no separate sync barrier
/// needed.
fn put_blob_from_bytes_replace(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
self.put_blob_from_bytes_unsynced(hash, data)
}
fn get_blob_stream( fn get_blob_stream(
&self, &self,
hash: &str, hash: &str,
@@ -46,11 +46,12 @@
//! cancel + cursor discipline; the batch loop is I/O-bound anyway. //! cancel + cursor discipline; the batch loop is I/O-bound anyway.
//! Add concurrency later if a real throughput need appears. //! Add concurrency later if a real throughput need appears.
use std::path::{Path, PathBuf}; use std::path::PathBuf;
use std::sync::Arc; use std::sync::Arc;
use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::atomic::{AtomicBool, Ordering};
use async_trait::async_trait; use async_trait::async_trait;
use bytes::Bytes;
use futures::StreamExt; use futures::StreamExt;
use sqlx::PgPool; use sqlx::PgPool;
@@ -61,8 +62,9 @@ use crate::infrastructure::scheduler::{
JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome, JobRegistry, JobRunArgs, JobStore, JobStoreProvider, RecoverableJobHandler, RunOutcome,
RunStatus, record_or_log, RunStatus, record_or_log,
}; };
use crate::infrastructure::services::encrypted_blob_backend::EncryptedBlobBackend;
use crate::infrastructure::services::entry_backend::{ use crate::infrastructure::services::entry_backend::{
build_entry_backend, persist_active_backend_name, persist_migration_readonly, build_entry_backend_typed, persist_active_backend_name, persist_migration_readonly,
}; };
pub const STORAGE_MIGRATION_JOB_NAME: &str = "storage_migration"; pub const STORAGE_MIGRATION_JOB_NAME: &str = "storage_migration";
@@ -413,7 +415,13 @@ impl RecoverableJobHandler for StorageMigrationService {
// Build target backend via the shared factory — same code // Build target backend via the shared factory — same code
// path boot uses, so the encryption decorator wrapping is // path boot uses, so the encryption decorator wrapping is
// uniform. // uniform.
let target = build_entry_backend(target_entry, &self.storage_path_fallback); // Typed variant: we need the wrapper's `is_at_head_format`
// + `put_blob_from_bytes_replace` for the smart-skip probe
// and overwrite path (the trait-object `put_blob` silently
// no-ops on existing target blobs — that's the bug this
// commit fixes end-to-end). The `Arc<dyn>` coercion is free
// for the swap-hot-swap call in `finish_completed`.
let target = build_entry_backend_typed(target_entry, &self.storage_path_fallback);
if let Err(e) = target.initialize().await { if let Err(e) = target.initialize().await {
return RunOutcome::Failed { return RunOutcome::Failed {
message: format!("target backend init: {e}"), message: format!("target backend init: {e}"),
@@ -503,14 +511,13 @@ impl RecoverableJobHandler for StorageMigrationService {
}; };
let mut copied_count = 0u64; let mut copied_count = 0u64;
// K1.2: with the target-skip short-circuit gone (see the // Populated by the smart-skip probe below: target blob
// detailed comment further down), no blob is ever "skipped" // already exists at the current head format+key, so a
// during a migration walk today. The counter stays wired // rewrite would be identical bytes. Cheap (15-byte range
// through the log lines + `finish_completed` so K3's // read via `is_at_head_format`), massive latency win on
// format-aware smart-skip can re-populate it without // resume + on backends where the source was rotated to the
// touching the observability surface. Not mutated in this // same key as the target already had.
// slice — hence no `mut`. let mut skipped_count: u64 = 0;
let skipped_count: u64 = 0;
let mut failed_count = 0u64; let mut failed_count = 0u64;
let mut source_missing_count = 0u64; let mut source_missing_count = 0u64;
@@ -638,35 +645,41 @@ impl RecoverableJobHandler for StorageMigrationService {
} }
} }
// Always copy — do NOT short-circuit on // Smart skip via v1 header inspection: read the
// `target.blob_exists(hash)`. Ed hit this on 2026-08-01 // target's first 15 bytes and only rewrite if the
// during S3 → local migration testing with encryption // stored format+key_fp differs from the target's
// enabled on the target: the target had pre-existing // current head. This is the "if file exists and
// plaintext blobs from an earlier local-active session, // destination key mismatch, overwrite it" rule.
// so `blob_exists` returned true and the migration
// silently skipped them. Result: the "encrypted"
// target ended up with mixed plaintext + ciphertext
// blobs — undetectable until a subsequent read failed.
// //
// The old skip was justified by two use cases: // Failure of the probe → treat as "not at head" →
// (a) resume idempotency — the last cursor-checkpoint // fall through to overwrite. Safe because the
// window (~100 blobs) gets re-processed on resume; // overwrite is atomic (Local: tempfile + rename;
// (b) target-side dedup — same content already present. // S3/Azure: unconditional PUT via the newly-fixed
// `put_blob_from_bytes_replace` overrides).
// //
// Both are now handled by unconditional overwrite: the // Historical note: earlier we had `blob_exists`
// re-copy is bounded by the checkpoint window (small), // skip (K1.0), which silently skipped mismatched
// and dedup-hit content is rare in practice // keys and produced mixed-encryption backends. K1.2
// (content-addressability means duplicate blobs ARE // removed that skip and made every blob rewrite
// the same blob unless two backends were seeded // unconditionally. This slice replaces the
// separately from the same source). // unconditional rewrite with a smart skip that's
// // both correct (checks the header, not just
// K3's `storage_rotate` job will restore a smart skip // existence) AND fast (skips the ~99% of resume
// via the v1 header's `<key_fp>` field — "already at // blobs already at head format).
// head format+key" then becomes cheaply detectable if target.is_at_head_format(hash).await.unwrap_or(false) {
// without reading target bytes. Until then, correct > skipped_count += 1;
// fast. tracing::debug!(
target: "oxicloud::migration",
event = "storage_migration.blob_skipped_head_match",
run_id = %store.run_id(),
hash = %hash,
head_format = %target.head_format(),
"target blob already at head format — skipping rewrite"
);
continue;
}
match copy_blob(self.source.as_ref(), target.as_ref(), hash).await { match copy_blob(self.source.as_ref(), target.clone(), hash).await {
Ok(()) => { Ok(()) => {
copied_count += 1; copied_count += 1;
} }
@@ -979,53 +992,39 @@ fn entry_identity(entry: &NamedStorageEntry) -> String {
/// cleans up on reboot). /// cleans up on reboot).
async fn copy_blob( async fn copy_blob(
source: &dyn BlobStorageBackend, source: &dyn BlobStorageBackend,
target: &dyn BlobStorageBackend, target: Arc<EncryptedBlobBackend>,
hash: &str, hash: &str,
) -> Result<(), DomainError> { ) -> Result<(), DomainError> {
let tmp_dir = std::env::temp_dir().join("oxicloud-migration"); // Read the full plaintext from source (source is wrapped in
tokio::fs::create_dir_all(&tmp_dir).await.map_err(|e| { // `EncryptedBlobBackend`, so its `get_blob_stream` decrypts
DomainError::internal_error( // transparently — even legacy plaintext blobs come out clean
"StorageMigration", // via the BLAKE3 rescue). Collected in-memory because the
format!("create temp dir {}: {e}", tmp_dir.display()), // target's `put_blob_from_bytes_replace` needs a `Bytes`.
) //
})?; // For CDC chunks (every blob written since chunking landed)
let tmp_path = tmp_dir.join(format!("{hash}.tmp")); // this is ≤ 1 MiB. Legacy whole-file blobs pay a full-blob
// buffer here; acceptable given rotate/migration are admin-
if let Err(e) = write_source_to_tmp(source, hash, &tmp_path).await { // triggered operations. Streaming through a temp file (the
let _ = tokio::fs::remove_file(&tmp_path).await; // old shape) doesn't help — target still needs the bytes.
return Err(e); let stream = source.get_blob_stream(hash).await?;
} let plaintext = collect_stream_bytes(stream).await?;
target
let put_result = target.put_blob(hash, &tmp_path).await; .put_blob_from_bytes_replace(hash, plaintext)
let _ = tokio::fs::remove_file(&tmp_path).await; .await
put_result.map(|_bytes_written| ()) .map(|_bytes_written| ())
} }
async fn write_source_to_tmp( async fn collect_stream_bytes(
source: &dyn BlobStorageBackend, stream: crate::application::ports::blob_storage_ports::BlobStream,
hash: &str, ) -> Result<Bytes, DomainError> {
tmp_path: &Path, use bytes::BytesMut;
) -> Result<(), DomainError> { let mut buf = BytesMut::new();
use tokio::io::AsyncWriteExt;
let stream = source.get_blob_stream(hash).await?;
let mut file = tokio::fs::File::create(tmp_path).await.map_err(|e| {
DomainError::internal_error(
"StorageMigration",
format!("create temp file {}: {e}", tmp_path.display()),
)
})?;
let mut stream = std::pin::pin!(stream); let mut stream = std::pin::pin!(stream);
while let Some(chunk) = stream.next().await { while let Some(chunk) = stream.next().await {
let bytes = chunk.map_err(|e| { let bytes = chunk.map_err(|e| {
DomainError::internal_error("StorageMigration", format!("source stream read: {e}")) DomainError::internal_error("StorageMigration", format!("source stream read: {e}"))
})?; })?;
file.write_all(&bytes).await.map_err(|e| { buf.extend_from_slice(&bytes);
DomainError::internal_error("StorageMigration", format!("temp file write: {e}"))
})?;
} }
file.flush() Ok(buf.freeze())
.await
.map_err(|e| DomainError::internal_error("StorageMigration", format!("temp flush: {e}")))?;
Ok(())
} }