feat(storage key rot): add rotate services
This commit is contained in:
@@ -85,6 +85,14 @@ pub fn admin_routes() -> Router<Arc<AppState>> {
|
||||
.route("/storage/migration/start", post(start_migration))
|
||||
.route("/storage/migration/pause", post(pause_migration))
|
||||
.route("/storage/migration/resume", post(resume_migration))
|
||||
// K3 (storage-key-rotation): per-entry rotate trigger.
|
||||
// Normalises every blob on the named entry to its head-pair
|
||||
// format (legacy → v1, plaintext ↔ encrypted, old-key →
|
||||
// new-key). No readonly mode; safe under normal traffic.
|
||||
.route(
|
||||
"/storage/entries/{name}/rotate",
|
||||
post(trigger_storage_rotate),
|
||||
)
|
||||
// NOTE: /storage/migration/verify retired in slice 7 (see the
|
||||
// comment near where `verify_migration` used to live). Use
|
||||
// `POST /api/admin/jobs/blobs_consistency/trigger?storage=<name>`.
|
||||
@@ -621,6 +629,118 @@ async fn trigger_storage_migration(
|
||||
.into_response())
|
||||
}
|
||||
|
||||
/// POST /api/admin/storage/entries/{name}/rotate — trigger the
|
||||
/// `storage_rotate` recoverable job on a specific entry.
|
||||
///
|
||||
/// Normalises every blob on `<name>` to the entry's head-pair
|
||||
/// format: legacy → v1, plaintext ↔ encrypted, old-key → new-key.
|
||||
/// See `docs/plan/storage-key-rotation.md` §"The rotation job".
|
||||
///
|
||||
/// Unlike migration, rotation does NOT engage read-only mode —
|
||||
/// rewrites happen in place under normal traffic. Concurrent user
|
||||
/// writes coexist safely.
|
||||
///
|
||||
/// The handler validates the entry name synchronously (400 on
|
||||
/// unknown entry); the actual walk detaches into a
|
||||
/// `tokio::spawn` so the HTTP call returns immediately.
|
||||
#[utoipa::path(
|
||||
post,
|
||||
path = "/api/admin/storage/entries/{name}/rotate",
|
||||
responses(
|
||||
(status = 202, description = "Rotation dispatched"),
|
||||
(status = 400, description = "Unknown entry"),
|
||||
(status = 401, description = "Unauthorized"),
|
||||
(status = 403, description = "Admin required")
|
||||
),
|
||||
params(
|
||||
("name" = String, Path, description = "Storage entry name to rotate")
|
||||
),
|
||||
security(("bearerAuth" = [])),
|
||||
tag = "admin"
|
||||
)]
|
||||
pub async fn trigger_storage_rotate(
|
||||
State(state): State<Arc<AppState>>,
|
||||
axum::extract::Path(name): axum::extract::Path<String>,
|
||||
) -> Result<impl IntoResponse, AppError> {
|
||||
use crate::infrastructure::scheduler::JobRunArgs;
|
||||
use crate::infrastructure::services::storage_migration_service::STORAGE_MIGRATION_JOB_NAME;
|
||||
use crate::infrastructure::services::storage_rotate_service::STORAGE_ROTATE_JOB_NAME;
|
||||
|
||||
// Synchronous entry-existence check — a bad name would fail the
|
||||
// run anyway, but returning 400 here spares the operator an
|
||||
// audit-log round-trip.
|
||||
let entries = &state.core.config.storage_entries;
|
||||
if entries.iter().all(|e| e.name != name) {
|
||||
let available = if entries.is_empty() {
|
||||
"(none)".to_string()
|
||||
} else {
|
||||
entries
|
||||
.iter()
|
||||
.map(|e| e.name.as_str())
|
||||
.collect::<Vec<_>>()
|
||||
.join(", ")
|
||||
};
|
||||
return Err(AppError::bad_request(format!(
|
||||
"unknown storage entry `{name}`. Available: [{available}]"
|
||||
)));
|
||||
}
|
||||
|
||||
// Concurrency guard per plan: at most one encryption-touching
|
||||
// recoverable run at a time across the whole app. Rotation
|
||||
// rewrites blobs in place; migration copies + swaps; running
|
||||
// both simultaneously could interleave writes on the same
|
||||
// hash. Cheap check — `list_runs` limit 1 with the status
|
||||
// filter is an index scan.
|
||||
let provider = state.core.job_store_provider.clone();
|
||||
for job_name in [STORAGE_ROTATE_JOB_NAME, STORAGE_MIGRATION_JOB_NAME] {
|
||||
let in_flight = provider
|
||||
.list_runs(job_name, 5)
|
||||
.await
|
||||
.map_err(AppError::from)?
|
||||
.into_iter()
|
||||
.any(|r| {
|
||||
matches!(
|
||||
r.status,
|
||||
crate::infrastructure::scheduler::RunStatus::Running
|
||||
| crate::infrastructure::scheduler::RunStatus::Paused
|
||||
| crate::infrastructure::scheduler::RunStatus::CancelRequested
|
||||
)
|
||||
});
|
||||
if in_flight {
|
||||
return Err(AppError::bad_request(format!(
|
||||
"cannot start storage_rotate on `{name}` — `{job_name}` is already Running / \
|
||||
Paused / CancelRequested. Wait for it to finish (or cancel via \
|
||||
`POST /api/admin/jobs/{job_name}/cancel`)."
|
||||
)));
|
||||
}
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
target: "audit",
|
||||
event = "storage_rotate.trigger_requested",
|
||||
target_name = %name,
|
||||
"👮🏻♂️ Admin triggered storage_rotate on `{name}`"
|
||||
);
|
||||
|
||||
let registry = state.core.job_registry.clone();
|
||||
let args = JobRunArgs {
|
||||
storage: Some(name.clone()),
|
||||
..JobRunArgs::default()
|
||||
};
|
||||
tokio::spawn(async move {
|
||||
registry.trigger(STORAGE_ROTATE_JOB_NAME, &args).await;
|
||||
});
|
||||
|
||||
Ok((
|
||||
StatusCode::ACCEPTED,
|
||||
Json(serde_json::json!({
|
||||
"message": format!("Rotation dispatched on `{name}` — poll GET /api/admin/jobs/{STORAGE_ROTATE_JOB_NAME} for progress"),
|
||||
"detached": true,
|
||||
})),
|
||||
)
|
||||
.into_response())
|
||||
}
|
||||
|
||||
/// Idle-state DTO — no run has been triggered yet.
|
||||
fn idle_migration_dto() -> MigrationStateDto {
|
||||
MigrationStateDto {
|
||||
|
||||
@@ -9,23 +9,22 @@
|
||||
//!
|
||||
//! ## Cost model
|
||||
//!
|
||||
//! On the *hot path* (no migration running — the ~100% case in normal
|
||||
//! operation) this middleware does:
|
||||
//! On the *hot path* (no migration AND no rotation running — the
|
||||
//! ~100% case in normal operation) this middleware does:
|
||||
//! 1. one `AtomicBool::load(Relaxed)` — sub-nanosecond;
|
||||
//! 2. an early return when `false`.
|
||||
//! 2. one `RwLock::read` on `rotation_progress` — uncontended;
|
||||
//! 3. an early return when both are inactive.
|
||||
//!
|
||||
//! No allocation, no lock, no formatting. Adds no measurable latency
|
||||
//! at any user count.
|
||||
//! No allocation, no formatting on the hot path. The rotation-check
|
||||
//! `RwLock::read` is cheap because writers only fire on batch
|
||||
//! checkpoints (~every 100 blobs); worst-case contention is
|
||||
//! sub-microsecond.
|
||||
//!
|
||||
//! On the *cold path* (migration in progress) this middleware does:
|
||||
//! 1. the atomic load above;
|
||||
//! 2. one `RwLock::read` (uncontended — writers are the migration
|
||||
//! handler, one per batch every ~100 blobs);
|
||||
//! 3. one small `serde_json::to_string` call on a 4-field struct
|
||||
//! (a few dozen bytes);
|
||||
//! 4. one header insertion.
|
||||
//! On the *cold path* (migration OR rotation in progress) the
|
||||
//! payload builder pulls the progress snapshot(s), formats a small
|
||||
//! JSON struct (~a few dozen bytes) and inserts the header.
|
||||
//!
|
||||
//! Total per-request work in this branch: microseconds.
|
||||
//! Total per-request work on cold path: microseconds.
|
||||
|
||||
use axum::extract::Request;
|
||||
use axum::extract::State;
|
||||
@@ -53,11 +52,20 @@ pub const SERVER_STATUS_HEADER: &str = "x-server-status";
|
||||
struct HeaderPayload {
|
||||
readonly: bool,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
migration: Option<MigrationHeader>,
|
||||
migration: Option<ProgressHeader>,
|
||||
/// K3: independent of `readonly` — rotation does NOT engage the
|
||||
/// app-wide read-only flag, so the frontend needs a distinct
|
||||
/// signal to know "rotation is running, show the rotation
|
||||
/// banner instead of migration banner".
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
rotation: Option<ProgressHeader>,
|
||||
}
|
||||
|
||||
/// Shared progress shape used by both `migration` and `rotation`
|
||||
/// header fields — same struct name, same JSON field names. Frontend
|
||||
/// treats them identically at the render layer.
|
||||
#[derive(serde::Serialize)]
|
||||
struct MigrationHeader {
|
||||
struct ProgressHeader {
|
||||
// `target` is owned here — the RwLock guard is released before
|
||||
// serialisation, so a borrowed slice wouldn't survive. Names
|
||||
// are small (`[a-z0-9_-]{1,32}`) so the copy is trivial.
|
||||
@@ -67,49 +75,76 @@ struct MigrationHeader {
|
||||
percent: u8,
|
||||
}
|
||||
|
||||
impl ProgressHeader {
|
||||
fn from_snapshot(p: &crate::common::migration_progress::MigrationProgress) -> Self {
|
||||
Self {
|
||||
target: p.target_name.clone(),
|
||||
migrated: p.migrated_blobs,
|
||||
total: p.total_blobs,
|
||||
percent: p.percent,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn server_status_middleware(
|
||||
State(state): State<Arc<AppState>>,
|
||||
request: Request,
|
||||
next: Next,
|
||||
) -> Response {
|
||||
// Hot-path fast return. When no migration is running the flag is
|
||||
// false and there's nothing to emit — a bare atomic load and out.
|
||||
let readonly = state.migration_readonly.load(Ordering::Relaxed);
|
||||
|
||||
// Rotation snapshot check — cheap uncontended `read`; if `None`
|
||||
// and readonly is also false, hot-path returns without a header.
|
||||
let rotation_active = state
|
||||
.rotation_progress
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.is_some();
|
||||
|
||||
let mut response = next.run(request).await;
|
||||
if !readonly {
|
||||
if !readonly && !rotation_active {
|
||||
return response;
|
||||
}
|
||||
|
||||
// Cold path — build the payload from the shared progress
|
||||
// snapshot. If the snapshot is absent (readonly is true but the
|
||||
// handler hasn't seeded progress yet, or a restart-during-
|
||||
// migration scenario) we still emit `readonly: true` so the
|
||||
// banner shows — the frontend renders a "maintenance in progress"
|
||||
// message even when specific numbers aren't available.
|
||||
// Cold path — build the payload from whichever snapshots are
|
||||
// active. `readonly:true` fires the migration banner even if
|
||||
// the migration handler hasn't seeded its progress yet
|
||||
// (restart-mid-migration scenario). `rotation` is populated
|
||||
// independently.
|
||||
let payload = {
|
||||
let guard = state
|
||||
.migration_progress
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let migration = if readonly {
|
||||
state
|
||||
.migration_progress
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.as_ref()
|
||||
.map(ProgressHeader::from_snapshot)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let rotation = if rotation_active {
|
||||
state
|
||||
.rotation_progress
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.as_ref()
|
||||
.map(ProgressHeader::from_snapshot)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
HeaderPayload {
|
||||
readonly: true,
|
||||
migration: guard.as_ref().map(|p| MigrationHeader {
|
||||
target: p.target_name.clone(),
|
||||
migrated: p.migrated_blobs,
|
||||
total: p.total_blobs,
|
||||
percent: p.percent,
|
||||
}),
|
||||
readonly,
|
||||
migration,
|
||||
rotation,
|
||||
}
|
||||
};
|
||||
|
||||
// `serde_json::to_string` on this 4-field struct is a few
|
||||
// dozen-byte allocation — negligible against the response body.
|
||||
// A serialize failure here would be a programming bug (all
|
||||
// fields are trivially serializable), so we degrade to a
|
||||
// minimal `readonly: true` string rather than skipping the
|
||||
// header entirely.
|
||||
// `serde_json::to_string` on this struct is a few dozen-byte
|
||||
// allocation — negligible against the response body. A
|
||||
// serialize failure here would be a programming bug, so we
|
||||
// degrade to a minimal string rather than skipping the header.
|
||||
let value =
|
||||
serde_json::to_string(&payload).unwrap_or_else(|_| r#"{"readonly":true}"#.to_string());
|
||||
serde_json::to_string(&payload).unwrap_or_else(|_| r#"{"readonly":false}"#.to_string());
|
||||
if let Ok(header_value) = HeaderValue::from_str(&value) {
|
||||
response
|
||||
.headers_mut()
|
||||
|
||||
Reference in New Issue
Block a user