Merge branch 'main' into idp-auto-redirect

This commit is contained in:
Markus Schmidt
2026-07-18 20:16:54 +02:00
committed by GitHub
180 changed files with 23338 additions and 2554 deletions
+195 -105
View File
@@ -56,14 +56,32 @@ fn parse_caldav_datetime(value: &str) -> Option<DateTime<Utc>> {
/// `None` if either tag is missing (malformed body) so callers
/// can fall back safely.
pub(crate) fn extract_vevent_chunk(ical_data: &str) -> Option<&str> {
let upper = ical_data.to_ascii_uppercase();
let begin = upper.find("BEGIN:VEVENT")?;
// End marker: the line-start of END:VEVENT after `begin`, plus
// the length of "END:VEVENT" itself, then find the next CRLF/LF
// to include the terminator line.
let after_begin = &upper[begin..];
let rel_end = after_begin.find("END:VEVENT")?;
let end_tag_end = begin + rel_end + "END:VEVENT".len();
// Byte index of the first ASCII-case-insensitive occurrence of
// `needle` in `hay` at or after `from`. Every stored body OxiCloud
// itself writes carries uppercase tags, so try the memchr-backed
// exact `find` first; only genuinely mixed-case foreign bodies pay
// the manual scan. Either way this replaces the old
// `to_ascii_uppercase()` of the ENTIRE body — one full-copy String
// allocation per event per REPORT/GET, done purely to locate two
// tags.
fn find_ci(hay: &str, needle: &str, from: usize) -> Option<usize> {
if let Some(i) = hay[from..].find(needle) {
return Some(from + i);
}
let h = hay.as_bytes();
let n = needle.as_bytes();
if h.len() < n.len() {
return None;
}
(from..=h.len() - n.len()).find(|&i| h[i..i + n.len()].eq_ignore_ascii_case(n))
}
let begin = find_ci(ical_data, "BEGIN:VEVENT", 0)?;
// End marker: the first END:VEVENT after `begin`, plus the length
// of "END:VEVENT" itself, then any immediate CRLF/LF to include
// the terminator line.
let rel_end = find_ci(ical_data, "END:VEVENT", begin)?;
let end_tag_end = rel_end + "END:VEVENT".len();
// Include any immediate line terminator so the chunk stays a
// well-formed line even when the caller concatenates.
let mut end = end_tag_end;
@@ -88,21 +106,27 @@ pub(crate) fn extract_vevent_chunk(ical_data: &str) -> Option<&str> {
pub(crate) fn group_events_by_uid<'a>(
events: &'a [CalendarEventDto],
) -> Vec<Vec<&'a CalendarEventDto>> {
let mut order: Vec<String> = Vec::new();
let mut buckets: std::collections::HashMap<String, Vec<&'a CalendarEventDto>> =
// Keys borrow from the DTO slice (which outlives every local) — the
// old String-keyed map cloned every event's UID (twice for first
// appearances) on every REPORT / collection PROPFIND / GET.
let mut order: Vec<&'a str> = Vec::new();
let mut buckets: std::collections::HashMap<&'a str, Vec<&'a CalendarEventDto>> =
std::collections::HashMap::new();
for event in events {
let key = event.ical_uid.clone();
if !buckets.contains_key(&key) {
order.push(key.clone());
let key = event.ical_uid.as_str();
match buckets.entry(key) {
std::collections::hash_map::Entry::Vacant(slot) => {
order.push(key);
slot.insert(vec![event]);
}
std::collections::hash_map::Entry::Occupied(mut slot) => slot.get_mut().push(event),
}
buckets.entry(key).or_default().push(event);
}
let mut out = Vec::with_capacity(order.len());
for uid in order {
let mut bucket = buckets.remove(&uid).unwrap_or_default();
let mut bucket = buckets.remove(uid).unwrap_or_default();
// Master first (recurrence_id None), exceptions in insertion order.
bucket.sort_by_key(|e| e.recurrence_id.is_some());
out.push(bucket);
@@ -1040,71 +1064,142 @@ impl CalDavAdapter {
// Write the calendar collection itself
Self::write_calendar_response(&mut xml_writer, calendar, request, base_href, caller_id)?;
// If depth > 0, include event resources — folded per UID
// so a recurring event's master + per-instance exception
// overrides share ONE D:response (RFC 4791 §4.1 + RFC
// 5545 §3.6.1). Pre-fix this loop emitted one D:response
// per DB row, and since master + exception share the
// same href (base + uid.ics) clients saw a duplicate
// href and deduped — the exception appeared to have
// vanished.
// If depth > 0, include event resources — see
// `write_collection_event_page`, which the streaming emitter
// reuses page by page.
if depth != "0" {
for bundle in group_events_by_uid(events) {
// The master (sorted first by group_events_by_uid)
// supplies the ETag anchor + getlastmodified. If
// the bundle is all exceptions (no master row),
// fall back to the first exception.
let anchor = match bundle.first() {
Some(e) => *e,
None => continue,
};
let event_href = format!("{}{}.ics", base_href, anchor.ical_uid);
xml_writer.write_event(Event::Start(BytesStart::new("D:response")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:href")))?;
xml_writer.write_event(Event::Text(BytesText::new(&event_href)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:href")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:propstat")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:prop")))?;
// resourcetype (empty for non-collection)
xml_writer.write_event(Event::Empty(BytesStart::new("D:resourcetype")))?;
// getetag — anchor row's id
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer
.write_event(Event::Text(BytesText::new(&format!("\"{}\"", anchor.id))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
// getcontenttype
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontenttype")))?;
xml_writer.write_event(Event::Text(BytesText::new(
"text/calendar; component=vevent",
)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontenttype")))?;
// getlastmodified — anchor row's updated_at
xml_writer.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
xml_writer
.write_event(Event::Text(BytesText::new(&anchor.updated_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:prop")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:status")))?;
xml_writer.write_event(Event::Text(BytesText::new("HTTP/1.1 200 OK")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:status")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:propstat")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:response")))?;
}
Self::write_collection_event_page(&mut xml_writer, events, base_href)?;
}
Self::write_caldav_multistatus_end(&mut xml_writer)?;
Ok(())
}
/// Multistatus opening + the calendar collection's own
/// `D:response` — the head of a depth-1 collection PROPFIND. The
/// streaming emitter calls this once, then
/// [`Self::write_collection_event_page`] per hydrated UID page,
/// then [`Self::write_caldav_multistatus_end`].
pub fn write_collection_head<W: Write>(
xml_writer: &mut Writer<W>,
calendar: &CalendarDto,
request: &PropFindRequest,
base_href: &str,
caller_id: &str,
) -> Result<()> {
Self::write_caldav_multistatus_start(xml_writer)?;
Self::write_calendar_response(xml_writer, calendar, request, base_href, caller_id)
}
/// One depth-1 collection page: event resources folded per UID so a
/// recurring master + per-instance exception overrides share ONE
/// `D:response` (RFC 4791 §4.1 + RFC 5545 §3.6.1) — emitting one
/// response per DB row made clients dedupe the shared href and the
/// exception appeared to vanish. Callers guarantee same-UID rows
/// arrive within a single page.
pub fn write_collection_event_page<W: Write>(
xml_writer: &mut Writer<W>,
events: &[CalendarEventDto],
base_href: &str,
) -> Result<()> {
for bundle in group_events_by_uid(events) {
// The master (sorted first by group_events_by_uid)
// supplies the ETag anchor + getlastmodified. If
// the bundle is all exceptions (no master row),
// fall back to the first exception.
let anchor = match bundle.first() {
Some(e) => *e,
None => continue,
};
let event_href = format!("{}{}.ics", base_href, anchor.ical_uid);
xml_writer.write_event(Event::Start(BytesStart::new("D:response")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:href")))?;
xml_writer.write_event(Event::Text(BytesText::new(&event_href)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:href")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:propstat")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:prop")))?;
// resourcetype (empty for non-collection)
xml_writer.write_event(Event::Empty(BytesStart::new("D:resourcetype")))?;
// getetag — anchor row's id
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!("\"{}\"", anchor.id))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
// getcontenttype
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontenttype")))?;
xml_writer.write_event(Event::Text(BytesText::new(
"text/calendar; component=vevent",
)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontenttype")))?;
// getlastmodified — anchor row's updated_at
xml_writer.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
xml_writer.write_event(Event::Text(BytesText::new(&anchor.updated_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:prop")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:status")))?;
xml_writer.write_event(Event::Text(BytesText::new("HTTP/1.1 200 OK")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:status")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:propstat")))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:response")))?;
}
Ok(())
}
/// Write the CalDAV `<D:multistatus>` opening tag (DAV + CalDAV +
/// CalendarServer namespaces). Streaming emitters call this once,
/// then [`Self::write_report_page`] per hydrated UID page, then
/// [`Self::write_caldav_multistatus_end`].
pub fn write_caldav_multistatus_start<W: Write>(xml_writer: &mut Writer<W>) -> Result<()> {
xml_writer.write_event(Event::Start(
BytesStart::new("D:multistatus").with_attributes([
("xmlns:D", "DAV:"),
("xmlns:C", "urn:ietf:params:xml:ns:caldav"),
("xmlns:CS", "http://calendarserver.org/ns/"),
]),
))?;
Ok(())
}
/// Close the multistatus opened by
/// [`Self::write_caldav_multistatus_start`].
pub fn write_caldav_multistatus_end<W: Write>(xml_writer: &mut Writer<W>) -> Result<()> {
xml_writer.write_event(Event::End(BytesEnd::new("D:multistatus")))?;
Ok(())
}
/// One REPORT page: group `events` per UID and emit one
/// `D:response` per bundle. Callers guarantee same-UID rows arrive
/// within a single page (the uid-keyset pager does).
pub fn write_report_page<W: Write>(
xml_writer: &mut Writer<W>,
events: &[CalendarEventDto],
request: &CalDavReportType,
base_href: &str,
) -> Result<()> {
let props = match request {
CalDavReportType::CalendarQuery { props, .. } => props,
CalDavReportType::CalendarMultiget { props, .. } => props,
CalDavReportType::SyncCollection { props, .. } => props,
};
for bundle in group_events_by_uid(events) {
let anchor = match bundle.first() {
Some(e) => *e,
None => continue,
};
let href = format!("{}{}.ics", base_href, anchor.ical_uid);
Self::write_event_response(xml_writer, &bundle, props, &href)?;
}
Ok(())
}
/// Generate a response for calendar events
pub fn generate_calendar_events_response<W: Write>(
writer: W,
@@ -1114,40 +1209,15 @@ impl CalDavAdapter {
) -> Result<()> {
let mut xml_writer = Writer::new(writer);
// Start multistatus response
xml_writer.write_event(Event::Start(
BytesStart::new("D:multistatus").with_attributes([
("xmlns:D", "DAV:"),
("xmlns:C", "urn:ietf:params:xml:ns:caldav"),
("xmlns:CS", "http://calendarserver.org/ns/"),
]),
))?;
Self::write_caldav_multistatus_start(&mut xml_writer)?;
// Determine which properties to include based on request type
let props = match request {
CalDavReportType::CalendarQuery { props, .. } => props.clone(),
CalDavReportType::CalendarMultiget { props, .. } => props.clone(),
CalDavReportType::SyncCollection { props, .. } => props.clone(),
};
// Responses folded per UID so a recurring master + exception
// overrides share ONE D:response (RFC 4791 §4.1) — see
// `write_report_page`, which the streaming emitters reuse
// page by page.
Self::write_report_page(&mut xml_writer, events, request, base_href)?;
// Add responses for events — folded per UID so a
// recurring master + per-instance exception overrides
// share ONE D:response with all VEVENTs concatenated
// into the calendar-data payload (RFC 4791 §4.1). Pre-
// fix this loop emitted one D:response per DB row, so
// master + exception carried duplicate hrefs and clients
// deduped, hiding the exception from the resulting sync.
for bundle in group_events_by_uid(events) {
let anchor = match bundle.first() {
Some(e) => *e,
None => continue,
};
let href = format!("{}{}.ics", base_href, anchor.ical_uid);
Self::write_event_response(&mut xml_writer, &bundle, &props, &href)?;
}
// End multistatus
xml_writer.write_event(Event::End(BytesEnd::new("D:multistatus")))?;
Self::write_caldav_multistatus_end(&mut xml_writer)?;
Ok(())
}
@@ -1418,6 +1488,26 @@ impl CalDavAdapter {
}
}
// ─────────────────────────────────────────────────────────────
// Bench support
// ─────────────────────────────────────────────────────────────
/// Thin public wrappers over the `pub(crate)` read-side helpers so
/// `examples/bench_caldav_parse.rs` can measure them. Gated behind the
/// `bench` feature — adds nothing to prod builds.
#[cfg(feature = "bench")]
pub mod bench {
use super::*;
pub fn extract_vevent_chunk(ical_data: &str) -> Option<&str> {
super::extract_vevent_chunk(ical_data)
}
pub fn group_events_by_uid(events: &[CalendarEventDto]) -> Vec<Vec<&CalendarEventDto>> {
super::group_events_by_uid(events)
}
}
// ─────────────────────────────────────────────────────────────
// Tests
// ─────────────────────────────────────────────────────────────
+126 -74
View File
@@ -278,6 +278,27 @@ impl CardDavAdapter {
) -> Result<()> {
let mut xml_writer = Writer::new(writer);
Self::write_collection_head(&mut xml_writer, address_book, request, base_href)?;
// Write contacts if depth > 0
if depth != "0" {
Self::write_collection_contact_page(&mut xml_writer, contacts, base_href)?;
}
Self::write_carddav_multistatus_end(&mut xml_writer)
}
/// Multistatus opening (DAV + CardDAV + CalendarServer namespaces)
/// plus the address book's own `D:response` — the head of a depth-1
/// collection PROPFIND. Streaming emitters call this once, then
/// [`Self::write_collection_contact_page`] per cursor page, then
/// [`Self::write_carddav_multistatus_end`].
pub fn write_collection_head<W: Write>(
xml_writer: &mut Writer<W>,
address_book: &AddressBookDto,
request: &PropFindRequest,
base_href: &str,
) -> Result<()> {
xml_writer.write_event(Event::Start(
BytesStart::new("D:multistatus").with_attributes([
("xmlns:D", "DAV:"),
@@ -285,19 +306,25 @@ impl CardDavAdapter {
("xmlns:CS", "http://calendarserver.org/ns/"),
]),
))?;
Self::write_addressbook_response(xml_writer, address_book, request, base_href)
}
// Write the address book itself
Self::write_addressbook_response(&mut xml_writer, address_book, request, base_href)?;
// Write contacts if depth > 0
if depth != "0" {
for contact in contacts {
let contact_href = format!("{}{}.vcf", base_href, contact.uid);
Self::write_contact_response(&mut xml_writer, contact, &[], &contact_href)?;
}
/// One depth-1 collection page of contact entries (standard props;
/// href buffer reused across the page).
pub fn write_collection_contact_page<W: Write>(
xml_writer: &mut Writer<W>,
contacts: &[ContactDto],
base_href: &str,
) -> Result<()> {
let mut href = String::with_capacity(base_href.len() + 48);
for contact in contacts {
href.clear();
let _ = std::fmt::Write::write_fmt(
&mut href,
format_args!("{}{}.vcf", base_href, contact.uid),
);
Self::write_contact_response(xml_writer, contact, &[], &href)?;
}
xml_writer.write_event(Event::End(BytesEnd::new("D:multistatus")))?;
Ok(())
}
@@ -648,47 +675,65 @@ impl CardDavAdapter {
}
/// Generate response for contacts (for REPORT)
pub fn generate_contacts_response<W: Write>(
writer: W,
contacts: &[ContactDto],
vcards: &[(String, String)], // (uid, vcard_data)
report: &CardDavReportType,
base_href: &str,
) -> Result<()> {
let mut xml_writer = Writer::new(writer);
/// REPORT `<D:multistatus>` opening tag (DAV + CardDAV namespaces).
/// Streaming emitters call this once, then
/// [`Self::write_contacts_report_page`] per cursor page, then
/// [`Self::write_carddav_multistatus_end`].
pub fn write_report_multistatus_start<W: Write>(xml_writer: &mut Writer<W>) -> Result<()> {
xml_writer.write_event(Event::Start(
BytesStart::new("D:multistatus").with_attributes([
("xmlns:D", "DAV:"),
("xmlns:CR", "urn:ietf:params:xml:ns:carddav"),
]),
))?;
Ok(())
}
let props = match report {
CardDavReportType::AddressbookQuery { props } => props.clone(),
CardDavReportType::AddressbookMultiget { props, .. } => props.clone(),
CardDavReportType::SyncCollection { props, .. } => props.clone(),
};
for contact in contacts {
let href = format!("{}{}.vcf", base_href, contact.uid);
let vcard = vcards
.iter()
.find(|(uid, _)| *uid == contact.uid)
.map(|(_, data)| data.as_str())
.unwrap_or("");
Self::write_contact_response(&mut xml_writer, contact, &props, &href)?;
// If address-data is requested, include vcard
if props.iter().any(|p| p.name == "address-data") || props.is_empty() {
// Already handled in write_contact_response
}
let _ = vcard; // suppress warning - used via contact_to_vcard fallback
}
/// Close a multistatus opened by either start writer.
pub fn write_carddav_multistatus_end<W: Write>(xml_writer: &mut Writer<W>) -> Result<()> {
xml_writer.write_event(Event::End(BytesEnd::new("D:multistatus")))?;
Ok(())
}
/// One REPORT page of contact responses. Props are borrowed from
/// the request; one href buffer is reused across the page.
pub fn write_contacts_report_page<W: Write>(
xml_writer: &mut Writer<W>,
contacts: &[ContactDto],
report: &CardDavReportType,
base_href: &str,
) -> Result<()> {
let props = match report {
CardDavReportType::AddressbookQuery { props } => props,
CardDavReportType::AddressbookMultiget { props, .. } => props,
CardDavReportType::SyncCollection { props, .. } => props,
};
let mut href = String::with_capacity(base_href.len() + 48);
for contact in contacts {
href.clear();
let _ = std::fmt::Write::write_fmt(
&mut href,
format_args!("{}{}.vcf", base_href, contact.uid),
);
// `write_contact_response` generates the vCard on demand when (and
// only when) address-data is actually requested.
Self::write_contact_response(xml_writer, contact, props, &href)?;
}
Ok(())
}
pub fn generate_contacts_response<W: Write>(
writer: W,
contacts: &[ContactDto],
report: &CardDavReportType,
base_href: &str,
) -> Result<()> {
let mut xml_writer = Writer::new(writer);
Self::write_report_multistatus_start(&mut xml_writer)?;
Self::write_contacts_report_page(&mut xml_writer, contacts, report, base_href)?;
Self::write_carddav_multistatus_end(&mut xml_writer)
}
/// Write a single contact response element
fn write_contact_response<W: Write>(
xml_writer: &mut Writer<W>,
@@ -710,10 +755,11 @@ impl CardDavAdapter {
xml_writer.write_event(Event::Empty(BytesStart::new("D:resourcetype")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!(
"\"{}\"",
contact.etag
))))?;
let mut quoted = String::with_capacity(contact.etag.len() + 2);
quoted.push('"');
quoted.push_str(&contact.etag);
quoted.push('"');
xml_writer.write_event(Event::Text(BytesText::new(&quoted)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontenttype")))?;
@@ -733,10 +779,11 @@ impl CardDavAdapter {
}
("DAV:", "getetag") => {
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!(
"\"{}\"",
contact.etag
))))?;
let mut quoted = String::with_capacity(contact.etag.len() + 2);
quoted.push('"');
quoted.push_str(&contact.etag);
quoted.push('"');
xml_writer.write_event(Event::Text(BytesText::new(&quoted)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
}
("DAV:", "getcontenttype") => {
@@ -868,20 +915,25 @@ impl CardDavAdapter {
/// Convert a ContactDto to vCard 3.0 format
pub fn contact_to_vcard(contact: &ContactDto) -> String {
// `write!` into a String is infallible; `let _ =` discards the Ok(()).
// Formatting straight into the buffer avoids one temporary String per
// vCard line compared to `push_str(&format!(…))`.
use std::fmt::Write as _;
let mut vcard = String::from("BEGIN:VCARD\r\nVERSION:3.0\r\n");
vcard.push_str(&format!("UID:{}\r\n", contact.uid));
let _ = write!(vcard, "UID:{}\r\n", contact.uid);
if let (Some(last), Some(first)) = (&contact.last_name, &contact.first_name) {
vcard.push_str(&format!("N:{};{};;;\r\n", last, first));
let _ = write!(vcard, "N:{};{};;;\r\n", last, first);
} else if let Some(last) = &contact.last_name {
vcard.push_str(&format!("N:{};;;;\r\n", last));
let _ = write!(vcard, "N:{};;;;\r\n", last);
} else if let Some(first) = &contact.first_name {
vcard.push_str(&format!("N:;{};;;\r\n", first));
let _ = write!(vcard, "N:;{};;;\r\n", first);
}
if let Some(fn_name) = &contact.full_name {
vcard.push_str(&format!("FN:{}\r\n", fn_name));
let _ = write!(vcard, "FN:{}\r\n", fn_name);
} else {
// FN is mandatory in vCard 3.0
let fn_name = format!(
@@ -892,68 +944,68 @@ pub fn contact_to_vcard(contact: &ContactDto) -> String {
.trim()
.to_string();
if !fn_name.is_empty() {
vcard.push_str(&format!("FN:{}\r\n", fn_name));
let _ = write!(vcard, "FN:{}\r\n", fn_name);
} else {
vcard.push_str("FN:Unknown\r\n");
}
}
if let Some(nickname) = &contact.nickname {
vcard.push_str(&format!("NICKNAME:{}\r\n", nickname));
let _ = write!(vcard, "NICKNAME:{}\r\n", nickname);
}
for email in &contact.email {
vcard.push_str(&format!(
let _ = write!(
vcard,
"EMAIL;TYPE={}:{}\r\n",
email.r#type.to_uppercase(),
email.email
));
);
}
for phone in &contact.phone {
vcard.push_str(&format!(
let _ = write!(
vcard,
"TEL;TYPE={}:{}\r\n",
phone.r#type.to_uppercase(),
phone.number
));
);
}
for addr in &contact.address {
let adr = format!(
";;{};{};{};{};{}",
let _ = write!(
vcard,
"ADR;TYPE={}:;;{};{};{};{};{}\r\n",
addr.r#type.to_uppercase(),
addr.street.as_deref().unwrap_or(""),
addr.city.as_deref().unwrap_or(""),
addr.state.as_deref().unwrap_or(""),
addr.postal_code.as_deref().unwrap_or(""),
addr.country.as_deref().unwrap_or(""),
);
vcard.push_str(&format!(
"ADR;TYPE={}:{}\r\n",
addr.r#type.to_uppercase(),
adr
));
}
if let Some(org) = &contact.organization {
vcard.push_str(&format!("ORG:{}\r\n", org));
let _ = write!(vcard, "ORG:{}\r\n", org);
}
if let Some(title) = &contact.title {
vcard.push_str(&format!("TITLE:{}\r\n", title));
let _ = write!(vcard, "TITLE:{}\r\n", title);
}
if let Some(notes) = &contact.notes {
vcard.push_str(&format!("NOTE:{}\r\n", notes.replace('\n', "\\n")));
let _ = write!(vcard, "NOTE:{}\r\n", notes.replace('\n', "\\n"));
}
if let Some(bday) = &contact.birthday {
vcard.push_str(&format!("BDAY:{}\r\n", bday.format("%Y-%m-%d")));
let _ = write!(vcard, "BDAY:{}\r\n", bday.format("%Y-%m-%d"));
}
if let Some(photo) = &contact.photo_url {
vcard.push_str(&format!("PHOTO;VALUE=URI:{}\r\n", photo));
let _ = write!(vcard, "PHOTO;VALUE=URI:{}\r\n", photo);
}
vcard.push_str(&format!(
let _ = write!(
vcard,
"REV:{}\r\n",
contact.updated_at.format("%Y%m%dT%H%M%SZ")
));
);
vcard.push_str("END:VCARD\r\n");
vcard
@@ -484,10 +484,6 @@ mod tests {
#[test]
fn test_generate_contacts_response() {
let contacts = vec![sample_contact()];
let vcards = vec![(
"contact-001".to_string(),
contact_to_vcard(&sample_contact()),
)];
let report = CardDavReportType::AddressbookQuery {
props: vec![
QualifiedName {
@@ -505,7 +501,6 @@ mod tests {
let result = CardDavAdapter::generate_contacts_response(
&mut output,
&contacts,
&vcards,
&report,
"/carddav/ab-001",
);
@@ -528,14 +523,12 @@ mod tests {
#[test]
fn test_generate_empty_contacts_response() {
let contacts: Vec<ContactDto> = vec![];
let vcards: Vec<(String, String)> = vec![];
let report = CardDavReportType::AddressbookQuery { props: vec![] };
let mut output = Vec::new();
let result = CardDavAdapter::generate_contacts_response(
&mut output,
&contacts,
&vcards,
&report,
"/carddav/ab-001",
);
+147 -120
View File
@@ -627,17 +627,19 @@ impl WebDavAdapter {
// RFC 4918 §9.2: known props → 200 propstat; unknown → 404 propstat.
// Props found in the dead store are returned in the dead 200 propstat,
// so exclude them from the 404 propstat to avoid duplicate reporting.
let (known, unknown): (Vec<_>, Vec<_>) = props
// Single pass: the requested-props writer skips unknown
// names itself (its match arms mirror
// `folder_prop_is_known` exactly), so only the usually
// empty 404 list needs materialising — the old
// `partition` built two throwaway Vecs per row.
let truly_unknown: Vec<_> = props
.iter()
.partition(|p| Self::folder_prop_is_known(p, quota));
let truly_unknown: Vec<_> = unknown
.into_iter()
.filter(|p| !dead_name_set.contains(*p))
.filter(|p| !Self::folder_prop_is_known(p, quota) && !dead_name_set.contains(p))
.collect();
xml_writer.write_event(Event::Start(BytesStart::new("D:propstat")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:prop")))?;
Self::write_folder_requested_props(xml_writer, folder, &known, quota)?;
Self::write_folder_requested_props(xml_writer, folder, props, quota)?;
xml_writer.write_event(Event::End(BytesEnd::new("D:prop")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:status")))?;
xml_writer.write_event(Event::Text(BytesText::new("HTTP/1.1 200 OK")))?;
@@ -714,16 +716,19 @@ impl WebDavAdapter {
// RFC 4918 §9.2: known props → 200 propstat; unknown → 404 propstat.
// Props found in the dead store are returned in the dead 200 propstat,
// so exclude them from the 404 propstat to avoid duplicate reporting.
let (known, unknown): (Vec<_>, Vec<_>) =
props.iter().partition(|p| Self::file_prop_is_known(p));
let truly_unknown: Vec<_> = unknown
.into_iter()
.filter(|p| !dead_name_set.contains(*p))
// Single pass: the requested-props writer skips unknown
// names itself (its match arms mirror `file_prop_is_known`
// exactly), so only the usually empty 404 list needs
// materialising — the old `partition` built two throwaway
// Vecs per row.
let truly_unknown: Vec<_> = props
.iter()
.filter(|p| !Self::file_prop_is_known(p) && !dead_name_set.contains(p))
.collect();
xml_writer.write_event(Event::Start(BytesStart::new("D:propstat")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:prop")))?;
Self::write_file_requested_props(xml_writer, file, &known)?;
Self::write_file_requested_props(xml_writer, file, props)?;
xml_writer.write_event(Event::End(BytesEnd::new("D:prop")))?;
xml_writer.write_event(Event::Start(BytesStart::new("D:status")))?;
xml_writer.write_event(Event::Text(BytesText::new("HTTP/1.1 200 OK")))?;
@@ -759,6 +764,71 @@ impl WebDavAdapter {
Ok(())
}
// ── Per-row formatted-value writers (stack-rendered) ─────────────
//
// PROPFIND emits two formatted dates, a size and a quoted etag for
// EVERY row of every listing. `to_rfc3339()`/`to_rfc2822()` ran
// chrono's format-spec interpreter and allocated a String each;
// `to_string()`/`format!` added two more. These render the same
// bytes from stack buffers (`common::fmt`); out-of-range timestamps
// keep the old chrono path as a byte-identical fallback.
fn write_creationdate<W: Write>(xml_writer: &mut Writer<W>, secs: u64) -> Result<()> {
xml_writer.write_event(Event::Start(BytesStart::new("D:creationdate")))?;
let secs = secs as i64;
let mut buf = [0u8; 25];
match crate::common::fmt::rfc3339_utc(&mut buf, secs) {
Some(s) => xml_writer.write_event(Event::Text(BytesText::new(s)))?,
None => {
let s = chrono::DateTime::<Utc>::from_timestamp(secs, 0)
.unwrap_or_else(Utc::now)
.to_rfc3339();
xml_writer.write_event(Event::Text(BytesText::new(&s)))?;
}
}
xml_writer.write_event(Event::End(BytesEnd::new("D:creationdate")))?;
Ok(())
}
fn write_lastmodified<W: Write>(xml_writer: &mut Writer<W>, secs: u64) -> Result<()> {
xml_writer.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
let secs = secs as i64;
let mut buf = [0u8; 31];
match crate::common::fmt::rfc2822_utc(&mut buf, secs) {
Some(s) => xml_writer.write_event(Event::Text(BytesText::new(s)))?,
None => {
let s = chrono::DateTime::<Utc>::from_timestamp(secs, 0)
.unwrap_or_else(Utc::now)
.to_rfc2822();
xml_writer.write_event(Event::Text(BytesText::new(&s)))?;
}
}
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
Ok(())
}
fn write_etag_quoted<W: Write>(xml_writer: &mut Writer<W>, etag: &str) -> Result<()> {
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
// One exactly-sized allocation instead of format!'s grow-from-empty.
let mut quoted = String::with_capacity(etag.len() + 2);
quoted.push('"');
quoted.push_str(etag);
quoted.push('"');
xml_writer.write_event(Event::Text(BytesText::new(&quoted)))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
Ok(())
}
fn write_contentlength<W: Write>(xml_writer: &mut Writer<W>, size: u64) -> Result<()> {
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontentlength")))?;
let mut buf = [0u8; 20];
xml_writer.write_event(Event::Text(BytesText::new(crate::common::fmt::u64_str(
&mut buf, size,
))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontentlength")))?;
Ok(())
}
/// Write standard folder properties
fn write_folder_standard_props<W: Write>(
xml_writer: &mut Writer<W>,
@@ -776,31 +846,15 @@ impl WebDavAdapter {
xml_writer.write_event(Event::End(BytesEnd::new("D:displayname")))?;
// Creation date
xml_writer.write_event(Event::Start(BytesStart::new("D:creationdate")))?;
// Convert u64 timestamp to DateTime
let created_at = chrono::DateTime::<Utc>::from_timestamp(folder.created_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer.write_event(Event::Text(BytesText::new(&created_at.to_rfc3339())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:creationdate")))?;
Self::write_creationdate(xml_writer, folder.created_at)?;
// Last modified
xml_writer.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
// Convert u64 timestamp to DateTime
let modified_at = chrono::DateTime::<Utc>::from_timestamp(folder.modified_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer.write_event(Event::Text(BytesText::new(&modified_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
Self::write_lastmodified(xml_writer, folder.modified_at)?;
// ETag — routes through `FolderDto::etag` (= `Folder::etag()`)
// so every WebDAV emitter and HEAD response agree on a single
// value for the same folder.
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!("\"{}\"", folder.etag))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
Self::write_etag_quoted(xml_writer, &folder.etag)?;
// Content length (0 for directories)
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontentlength")))?;
@@ -829,13 +883,19 @@ impl WebDavAdapter {
used_bytes: i64,
available_bytes: Option<i64>,
) -> Result<()> {
let mut buf = [0u8; 21];
xml_writer.write_event(Event::Start(BytesStart::new("D:quota-used-bytes")))?;
xml_writer.write_event(Event::Text(BytesText::new(&used_bytes.to_string())))?;
xml_writer.write_event(Event::Text(BytesText::new(crate::common::fmt::i64_str(
&mut buf, used_bytes,
))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:quota-used-bytes")))?;
if let Some(available_bytes) = available_bytes {
xml_writer.write_event(Event::Start(BytesStart::new("D:quota-available-bytes")))?;
xml_writer.write_event(Event::Text(BytesText::new(&available_bytes.to_string())))?;
xml_writer.write_event(Event::Text(BytesText::new(crate::common::fmt::i64_str(
&mut buf,
available_bytes,
))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:quota-available-bytes")))?;
}
@@ -861,36 +921,18 @@ impl WebDavAdapter {
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontenttype")))?;
// Content length
xml_writer.write_event(Event::Start(BytesStart::new("D:getcontentlength")))?;
xml_writer.write_event(Event::Text(BytesText::new(&file.size.to_string())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontentlength")))?;
Self::write_contentlength(xml_writer, file.size)?;
// Creation date
xml_writer.write_event(Event::Start(BytesStart::new("D:creationdate")))?;
// Convert u64 timestamp to DateTime
let created_at = chrono::DateTime::<Utc>::from_timestamp(file.created_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer.write_event(Event::Text(BytesText::new(&created_at.to_rfc3339())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:creationdate")))?;
Self::write_creationdate(xml_writer, file.created_at)?;
// Last modified
xml_writer.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
// Convert u64 timestamp to DateTime
let modified_at = chrono::DateTime::<Utc>::from_timestamp(file.modified_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer.write_event(Event::Text(BytesText::new(&modified_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
Self::write_lastmodified(xml_writer, file.modified_at)?;
// ETag — routes through `FileDto::etag` (= `File::etag()`) so
// PROPFIND, GET, HEAD, PUT-response, and MOVE all emit
// byte-identical values for the same file.
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!("\"{}\"", file.etag))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
Self::write_etag_quoted(xml_writer, &file.etag)?;
Ok(())
}
@@ -936,7 +978,7 @@ impl WebDavAdapter {
fn write_folder_requested_props<W: Write>(
xml_writer: &mut Writer<W>,
folder: &FolderDto,
props: &[&QualifiedName],
props: &[QualifiedName],
quota: Option<(i64, Option<i64>)>,
) -> Result<()> {
for prop in props {
@@ -953,37 +995,13 @@ impl WebDavAdapter {
xml_writer.write_event(Event::End(BytesEnd::new("D:displayname")))?;
}
"creationdate" => {
xml_writer.write_event(Event::Start(BytesStart::new("D:creationdate")))?;
// Convert u64 timestamp to DateTime
let created_at =
chrono::DateTime::<Utc>::from_timestamp(folder.created_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer
.write_event(Event::Text(BytesText::new(&created_at.to_rfc3339())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:creationdate")))?;
Self::write_creationdate(xml_writer, folder.created_at)?;
}
"getlastmodified" => {
xml_writer
.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
// Convert u64 timestamp to DateTime
let modified_at =
chrono::DateTime::<Utc>::from_timestamp(folder.modified_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer
.write_event(Event::Text(BytesText::new(&modified_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
Self::write_lastmodified(xml_writer, folder.modified_at)?;
}
"getetag" => {
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!(
"\"{}\"",
folder.etag
))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
Self::write_etag_quoted(xml_writer, &folder.etag)?;
}
"getcontentlength" => {
xml_writer
@@ -1000,21 +1018,25 @@ impl WebDavAdapter {
}
"quota-used-bytes" => {
if let Some((used, _)) = quota {
let mut buf = [0u8; 21];
xml_writer
.write_event(Event::Start(BytesStart::new("D:quota-used-bytes")))?;
xml_writer
.write_event(Event::Text(BytesText::new(&used.to_string())))?;
xml_writer.write_event(Event::Text(BytesText::new(
crate::common::fmt::i64_str(&mut buf, used),
)))?;
xml_writer
.write_event(Event::End(BytesEnd::new("D:quota-used-bytes")))?;
}
}
"quota-available-bytes" => {
if let Some((_, Some(available))) = quota {
let mut buf = [0u8; 21];
xml_writer.write_event(Event::Start(BytesStart::new(
"D:quota-available-bytes",
)))?;
xml_writer
.write_event(Event::Text(BytesText::new(&available.to_string())))?;
xml_writer.write_event(Event::Text(BytesText::new(
crate::common::fmt::i64_str(&mut buf, available),
)))?;
xml_writer.write_event(Event::End(BytesEnd::new(
"D:quota-available-bytes",
)))?;
@@ -1035,7 +1057,7 @@ impl WebDavAdapter {
fn write_file_requested_props<W: Write>(
xml_writer: &mut Writer<W>,
file: &FileDto,
props: &[&QualifiedName],
props: &[QualifiedName],
) -> Result<()> {
for prop in props {
if prop.namespace == "DAV:" {
@@ -1055,44 +1077,16 @@ impl WebDavAdapter {
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontenttype")))?;
}
"getcontentlength" => {
xml_writer
.write_event(Event::Start(BytesStart::new("D:getcontentlength")))?;
xml_writer
.write_event(Event::Text(BytesText::new(&file.size.to_string())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getcontentlength")))?;
Self::write_contentlength(xml_writer, file.size)?;
}
"creationdate" => {
xml_writer.write_event(Event::Start(BytesStart::new("D:creationdate")))?;
// Convert u64 timestamp to DateTime
let created_at =
chrono::DateTime::<Utc>::from_timestamp(file.created_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer
.write_event(Event::Text(BytesText::new(&created_at.to_rfc3339())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:creationdate")))?;
Self::write_creationdate(xml_writer, file.created_at)?;
}
"getlastmodified" => {
xml_writer
.write_event(Event::Start(BytesStart::new("D:getlastmodified")))?;
// Convert u64 timestamp to DateTime
let modified_at =
chrono::DateTime::<Utc>::from_timestamp(file.modified_at as i64, 0)
.unwrap_or_else(Utc::now);
xml_writer
.write_event(Event::Text(BytesText::new(&modified_at.to_rfc2822())))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getlastmodified")))?;
Self::write_lastmodified(xml_writer, file.modified_at)?;
}
"getetag" => {
xml_writer.write_event(Event::Start(BytesStart::new("D:getetag")))?;
xml_writer.write_event(Event::Text(BytesText::new(&format!(
"\"{}\"",
file.etag
))))?;
xml_writer.write_event(Event::End(BytesEnd::new("D:getetag")))?;
Self::write_etag_quoted(xml_writer, &file.etag)?;
}
_ => {
// Unknown prop — skipped here; caller writes 404 propstat.
@@ -1586,3 +1580,36 @@ impl WebDavAdapter {
Self::write_file_response_with_dead_props(writer, file, request, href, dead_props)
}
}
/// Thin public wrappers over the private per-row PROPFIND writers so
/// `examples/bench_propfind_xml.rs` can measure them. Gated behind the
/// `bench` feature — adds nothing to prod builds.
#[cfg(feature = "bench")]
pub mod bench {
use super::*;
pub fn write_file_propfind_row<W: Write>(
xml_writer: &mut Writer<W>,
file: &FileDto,
request: &PropFindRequest,
href: &str,
dead_props: &[(QualifiedName, Option<String>)],
) -> Result<()> {
WebDavAdapter::write_file_response_with_dead_props(
xml_writer, file, request, href, dead_props,
)
}
pub fn write_folder_propfind_row<W: Write>(
xml_writer: &mut Writer<W>,
folder: &FolderDto,
request: &PropFindRequest,
href: &str,
dead_props: &[(QualifiedName, Option<String>)],
quota: Option<(i64, Option<i64>)>,
) -> Result<()> {
WebDavAdapter::write_folder_response_with_dead_props(
xml_writer, folder, request, href, dead_props, quota,
)
}
}
+228 -5
View File
@@ -8,6 +8,175 @@
//! then fall back to the file extension when the MIME is generic
//! (`application/octet-stream` or empty).
use std::collections::HashMap;
use std::fmt::Write as _;
use std::sync::{Arc, LazyLock};
// ─── Arc<str> interning for closed-set display values ────────────────
//
// `FileDto` / `FolderDto` store their display fields as `Arc<str>` so DTO
// clones are O(1). But `Arc::<str>::from(&str)` always allocates + copies,
// so building the DTO paid 3-4 heap allocations per row even though the
// value space is a small closed set. Interning turns each conversion into
// a HashMap lookup + refcount bump.
/// Every `&'static str` that [`icon_class_for`], [`icon_special_class_for`]
/// and [`category_for`] can return, plus the folder-DTO constants.
///
/// Keep this table in sync when adding a value to those functions — a
/// missing entry is not a bug (callers fall back to `Arc::from`, same
/// bytes, one extra allocation), just a lost optimization.
static DISPLAY_INTERN: LazyLock<HashMap<&'static str, Arc<str>>> = LazyLock::new(|| {
const CLOSED_SET: &[&str] = &[
// icon_class_for
"fas fa-file-pdf",
"fas fa-file-word",
"fas fa-file-excel",
"fas fa-file-powerpoint",
"fas fa-file-archive",
"fas fa-file-code",
"fas fa-hdd",
"fas fa-file-image",
"fas fa-file-video",
"fas fa-file-audio",
"fas fa-file-alt",
"fas fa-terminal",
"fas fa-file",
// icon_special_class_for
"pdf-icon",
"doc-icon",
"spreadsheet-icon",
"presentation-icon",
"archive-icon",
"code-icon json-icon",
"code-icon js-icon",
"code-icon ts-icon",
"code-icon html-icon",
"code-icon sql-icon",
"code-icon config-icon",
"code-icon php-icon",
"script-icon",
"installer-icon",
"image-icon",
"video-icon",
"audio-icon",
"code-icon py-icon",
"code-icon rust-icon",
"code-icon",
"code-icon go-icon",
"code-icon ruby-icon",
"code-icon md-icon",
"code-icon css-icon",
"code-icon java-icon",
"code-icon c-icon",
"code-icon cs-icon",
"code-icon swift-icon",
"",
// category_for
"PDF",
"Document",
"Spreadsheet",
"Presentation",
"Archive",
"Code",
"Installer",
"Image",
"Video",
"Audio",
"Markdown",
"Text",
// FolderDto constants
"fas fa-folder",
"folder-icon",
"Folder",
];
CLOSED_SET.iter().map(|s| (*s, Arc::from(*s))).collect()
});
/// Returns a shared `Arc<str>` for a display value from the closed sets
/// above (icon class, icon special class, category). Lookup + refcount
/// bump instead of alloc + copy; unknown values (future additions not
/// yet in the table) fall back to `Arc::from` with identical bytes.
pub fn intern_display(s: &'static str) -> Arc<str> {
DISPLAY_INTERN
.get(s)
.cloned()
.unwrap_or_else(|| Arc::from(s))
}
/// The MIME types that dominate real storage rows. Exotic types fall back
/// to a per-row `Arc::from` — correctness is unaffected, only the alloc is.
static MIME_INTERN: LazyLock<HashMap<&'static str, Arc<str>>> = LazyLock::new(|| {
const COMMON_MIMES: &[&str] = &[
"",
"directory",
"application/octet-stream",
// Images
"image/jpeg",
"image/png",
"image/gif",
"image/webp",
"image/svg+xml",
"image/heic",
"image/heif",
"image/avif",
"image/bmp",
"image/tiff",
"image/x-icon",
// Video
"video/mp4",
"video/quicktime",
"video/webm",
"video/x-matroska",
"video/x-msvideo",
// Audio
"audio/mpeg",
"audio/mp4",
"audio/ogg",
"audio/flac",
"audio/wav",
"audio/x-wav",
"audio/aac",
// Documents
"application/pdf",
"application/msword",
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/vnd.ms-excel",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-powerpoint",
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
"application/vnd.oasis.opendocument.text",
"application/vnd.oasis.opendocument.spreadsheet",
// Text / code
"text/plain",
"text/csv",
"text/html",
"text/css",
"text/markdown",
"text/xml",
"application/json",
"application/javascript",
"application/xml",
"application/x-yaml",
// Archives
"application/zip",
"application/gzip",
"application/x-tar",
"application/x-7z-compressed",
"application/x-rar-compressed",
];
COMMON_MIMES.iter().map(|s| (*s, Arc::from(*s))).collect()
});
/// Returns a shared `Arc<str>` for the given MIME type. Common types hit
/// the intern table (refcount bump); exotic ones allocate as before.
pub fn intern_mime(mime: &str) -> Arc<str> {
MIME_INTERN
.get(mime)
.cloned()
.unwrap_or_else(|| Arc::from(mime))
}
// ─── Private: extract lowercase extension from a filename ────────────
fn ext_of(name: &str) -> Option<&str> {
let name = name.rsplit('/').next().unwrap_or(name); // strip path
@@ -388,11 +557,21 @@ pub fn format_file_size(bytes: u64) -> String {
let value = bytes as f64 / K.powi(i as i32);
// Two decimal places, then strip trailing zeros (matches JS parseFloat behaviour)
let formatted = format!("{:.2}", value);
let formatted = formatted.trim_end_matches('0').trim_end_matches('.');
format!("{} {}", formatted, SIZES[i])
// Single buffer: write the 2-decimal value, strip trailing zeros in
// place (matches JS parseFloat behaviour), then append the unit.
// 16 chars covers the worst case ("16777216 TB" for u64::MAX,
// "1023.99 Bytes" for the longest unit), so no realloc occurs.
let mut out = String::with_capacity(16);
let _ = write!(out, "{:.2}", value);
while out.ends_with('0') {
out.pop();
}
if out.ends_with('.') {
out.pop();
}
out.push(' ');
out.push_str(SIZES[i]);
out
}
#[cfg(test)]
@@ -506,6 +685,50 @@ mod tests {
);
}
/// Every value the closed-set display functions can return must hit
/// the intern table (same bytes, shared allocation) — a miss is only
/// a lost optimization, but this test keeps the table in sync.
#[test]
fn test_intern_display_covers_closed_sets_and_shares_storage() {
for s in [
"fas fa-file-pdf",
"fas fa-file",
"fas fa-terminal",
"fas fa-folder",
"code-icon rust-icon",
"folder-icon",
"",
"PDF",
"Folder",
"Document",
"Markdown",
] {
let a = intern_display(s);
let b = intern_display(s);
assert_eq!(&*a, s, "interned bytes must be identical");
assert!(
Arc::ptr_eq(&a, &b),
"closed-set value {s:?} must come from the intern table"
);
}
}
#[test]
fn test_intern_mime_common_hits_table_exotic_falls_back() {
let a = intern_mime("image/jpeg");
let b = intern_mime("image/jpeg");
assert_eq!(&*a, "image/jpeg");
assert!(Arc::ptr_eq(&a, &b), "common MIME must be interned");
let exotic = intern_mime("chemical/x-pdb");
assert_eq!(&*exotic, "chemical/x-pdb");
let exotic2 = intern_mime("chemical/x-pdb");
assert!(
!Arc::ptr_eq(&exotic, &exotic2),
"exotic MIME falls back to a fresh Arc"
);
}
#[test]
fn test_ext_of() {
assert_eq!(ext_of("file.txt"), Some("txt"));
+14 -9
View File
@@ -6,7 +6,8 @@ use utoipa::ToSchema;
use uuid::Uuid;
use super::display_helpers::{
category_for, format_file_size, icon_class_for, icon_special_class_for,
category_for, format_file_size, icon_class_for, icon_special_class_for, intern_display,
intern_mime,
};
/// DTO for file responses
@@ -101,11 +102,15 @@ impl From<File> for FileDto {
// for id, name, path, folder_id (previously 4× .to_string()).
let parts = file.into_parts();
let icon_class = Arc::from(icon_class_for(&parts.name, &parts.mime_type));
let icon_special_class = Arc::from(icon_special_class_for(&parts.name, &parts.mime_type));
let category = Arc::from(category_for(&parts.name, &parts.mime_type));
// Display fields come from closed static tables and MIME values
// repeat massively across rows — intern instead of allocating a
// fresh Arc<str> per row (`Arc::from(&str)` always allocs+copies).
let icon_class = intern_display(icon_class_for(&parts.name, &parts.mime_type));
let icon_special_class =
intern_display(icon_special_class_for(&parts.name, &parts.mime_type));
let category = intern_display(category_for(&parts.name, &parts.mime_type));
let size_formatted = format_file_size(parts.size);
let mime_type = Arc::from(parts.mime_type.as_str());
let mime_type = intern_mime(&parts.mime_type);
Self {
id: parts.id,
@@ -169,13 +174,13 @@ impl FileDto {
name: "stub-file".to_string(),
path: "/stub/path".to_string(),
size: 0,
mime_type: Arc::from("application/octet-stream"),
mime_type: intern_mime("application/octet-stream"),
folder_id: None,
created_at: 0,
modified_at: 0,
icon_class: Arc::from("fas fa-file"),
icon_special_class: Arc::from(""),
category: Arc::from("Document"),
icon_class: intern_display("fas fa-file"),
icon_special_class: intern_display(""),
category: intern_display("Document"),
size_formatted: "0 Bytes".to_string(),
content_hash: String::new(),
etag: String::new(),
+27 -17
View File
@@ -1,6 +1,7 @@
use std::sync::Arc;
use crate::application::dtos::cursor::{CursorListResponse, CursorQuery, PageCursor};
use crate::application::dtos::display_helpers::intern_display;
use crate::application::dtos::grant_dto::{ResourceContentDto, ResourceTypeDto};
use crate::domain::entities::folder::Folder;
use crate::domain::services::authorization::ResourceKind;
@@ -99,24 +100,33 @@ pub struct FolderDto {
impl From<Folder> for FolderDto {
fn from(folder: Folder) -> Self {
let is_root = folder.parent_id().is_none();
let etag = folder.etag().to_string();
// Consume the entity by moving all fields — zero heap allocations
// for id, name, path, parent_id (previously 3-4× .to_string()).
let parts = folder.into_parts();
let is_root = parts.parent_id.is_none();
// Single-allocation ETag straight from the owned parts. The old
// shape (`folder.etag().to_string()`) built the String and then
// cloned it — a pure double-alloc.
let etag = Folder::compute_etag(&parts.id, parts.tree_modified_at);
Self {
id: folder.id().to_string(),
name: folder.name().to_string(),
path: folder.path_string().to_string(),
parent_id: folder.parent_id().map(String::from),
drive_id: folder.drive_id(),
created_at: folder.created_at(),
modified_at: folder.modified_at(),
id: parts.id,
name: parts.name,
path: parts.path_string,
parent_id: parts.parent_id,
drive_id: parts.drive_id,
created_at: parts.created_at,
modified_at: parts.modified_at,
is_root,
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
// Constant display fields: refcount bump on interned statics
// instead of 3 fresh Arc allocations per row.
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
etag,
created_by: folder.created_by(),
updated_by: folder.updated_by(),
created_by: parts.created_by,
updated_by: parts.updated_by,
}
}
}
@@ -163,9 +173,9 @@ impl FolderDto {
created_at: 0,
modified_at: 0,
is_root: true,
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
etag: String::new(),
created_by: None,
updated_by: None,
+15 -7
View File
@@ -1,4 +1,5 @@
use serde::{Deserialize, Serialize};
use std::sync::Arc;
use utoipa::ToSchema;
/**
@@ -109,8 +110,10 @@ pub struct SearchFileResultDto {
pub path: String,
/// Size in bytes
pub size: u64,
/// MIME type
pub mime_type: String,
/// MIME type — `Arc<str>` so enrichment reuses `FileDto`'s interned
/// value (an atomic increment) instead of allocating per result row.
#[schema(value_type = String)]
pub mime_type: Arc<str>,
/// Parent folder ID
pub folder_id: Option<String>,
/// Creation timestamp
@@ -122,11 +125,14 @@ pub struct SearchFileResultDto {
/// Human-readable file size (e.g., "2.5 MB")
pub size_formatted: String,
/// CSS icon class for the file type (e.g., "fas fa-file-pdf")
pub icon_class: String,
#[schema(value_type = String)]
pub icon_class: Arc<str>,
/// Extra CSS class for icon styling (e.g., "pdf-icon", "code-icon js-icon")
pub icon_special_class: String,
#[schema(value_type = String)]
pub icon_special_class: Arc<str>,
/// Content category: "document", "image", "video", "audio", "archive", "code", "other"
pub category: String,
#[schema(value_type = String)]
pub category: Arc<str>,
/// Raw BLAKE3 content hash. Feeds `FileDto::content_hash` and
/// `File::compute_etag` when search results are converted to
/// `FileDto` (NC REPORT/SEARCH response). Defaults to `String::new()`
@@ -267,9 +273,11 @@ pub struct SearchSuggestionItem {
/// Path for context
pub path: String,
/// CSS icon class
pub icon_class: String,
#[schema(value_type = String)]
pub icon_class: Arc<str>,
/// Extra CSS class for icon styling
pub icon_special_class: String,
#[schema(value_type = String)]
pub icon_special_class: Arc<str>,
/// Relevance score
pub relevance_score: u32,
}
+105 -35
View File
@@ -16,6 +16,28 @@ use crate::domain::services::authorization::{
ResourceKind, Role, Subject,
};
/// Discriminates the two denial shapes surfaced by
/// [`AuthorizationEngine::require_visible`] in the `authz.denied` audit line.
/// Log-aggregation consumers key off the string form via `as_str`; keep the
/// values stable — a new denial shape means a new variant, never a renamed
/// existing one.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
pub enum AuthzDenialVisibility {
/// Caller has `Read` on the resource — 403 Forbidden.
Visible,
/// Caller has no `Read` — 404 anti-enum.
Hidden,
}
impl AuthzDenialVisibility {
pub fn as_str(self) -> &'static str {
match self {
Self::Visible => "visible",
Self::Hidden => "hidden",
}
}
}
pub trait AuthorizationEngine: Send + Sync + 'static {
/// Returns true if `subject` has `permission` on `resource`, considering
/// owner short-circuit AND cascading from folder ancestors.
@@ -53,9 +75,28 @@ pub trait AuthorizationEngine: Send + Sync + 'static {
Ok(allowed)
}
/// Convenience wrapper around `check`: returns `Ok(())` when allowed and
/// `DomainError::not_found` when denied (anti-enumeration — same error as
/// "resource doesn't exist" so attackers can't probe IDs by error shape).
/// Graduated-denial wrapper around `check`. Semantics:
///
/// - `permission` granted → `Ok(())`
/// - `permission` denied, `Read` also denied → `DomainError::not_found`
/// (404, anti-enumeration — same shape as "doesn't exist" so a probing
/// caller can't distinguish "wrong id" from "no access")
/// - `permission` denied, `Read` granted → `DomainError::access_denied`
/// (403 — the caller can already see the resource, so hiding existence
/// leaks nothing new; a clear 403 beats a confusing 404 for UX and for
/// API-first clients like rclone)
///
/// Special case: when `permission == Read`, the visibility gate collapses
/// onto itself — a `Read` denial IS a "hidden" outcome by definition, so
/// the method short-circuits to the strict anti-enum 404 without a second
/// DB round-trip. That's why there's only one method: strict Read-denial
/// and graduated write-denial fall out of the same signature.
///
/// Do NOT use this in search / enumeration paths where existence itself is
/// the attack vector — those must filter at the SQL/index layer, never
/// touch this method with per-row ids. Cross-tenant probes on ids the
/// caller has no prior read handle for degrade to the 404 shape naturally
/// (Read denied → `Hidden`).
async fn require(
&self,
subject: Subject,
@@ -80,39 +121,68 @@ pub trait AuthorizationEngine: Send + Sync + 'static {
permission,
resource
);
Ok(())
return Ok(());
}
// Visibility probe. Short-circuit: when the target permission IS
// `Read` and the check above returned false, we already know Read is
// denied — visibility is `Hidden` by definition, no second DB hop.
// Otherwise probe Read; a DB-hop failure here degrades to `Hidden` so
// the caller sees the strict anti-enum shape (safe default).
let visibility = if permission == Permission::Read {
AuthzDenialVisibility::Hidden
} else if self
.check(subject, Permission::Read, resource)
.await
.unwrap_or(false)
{
AuthzDenialVisibility::Visible
} else {
let (kind, id) = match resource {
Resource::Folder(id) => ("Folder", id),
Resource::File(id) => ("File", id),
Resource::Drive(id) => ("Drive", id),
Resource::Calendar(id) => ("Calendar", id),
Resource::AddressBook(id) => ("AddressBook", id),
Resource::Playlist(id) => ("Playlist", id),
};
// Audit-worthy: denials are the interesting signal. Routed
// through the `audit` tracing target so log aggregators can
// surface them separately from operational debug traffic.
// Span context (request_id, client_ip, user_id) is attached
// automatically by the request-scope span set in
// `interfaces/middleware/trace_span.rs`, so this log line
// doesn't need to duplicate those fields — they appear in
// the structured output of every log written inside the
// request span.
tracing::info!(
target: "audit",
event = "authz.denied",
subject_type = subject.type_str(),
subject_id = %subject.id(),
permission = permission.as_str(),
resource_type = resource.type_str(),
resource_id = %resource.id(),
"👮🏻‍♂️ perms: ⛔ Subject '{}' hasn't permission to '{}' on resource '{}'",
subject,
permission,
resource
);
Err(DomainError::not_found(kind, id.to_string()))
AuthzDenialVisibility::Hidden
};
let (kind, id) = match resource {
Resource::Folder(id) => ("Folder", id),
Resource::File(id) => ("File", id),
Resource::Drive(id) => ("Drive", id),
Resource::Calendar(id) => ("Calendar", id),
Resource::AddressBook(id) => ("AddressBook", id),
Resource::Playlist(id) => ("Playlist", id),
};
// Audit-worthy: denials are the interesting signal. Routed through
// the `audit` tracing target so log aggregators can surface them
// separately from operational debug traffic. Span context
// (request_id, client_ip, user_id) comes from the request-scope
// span set in `interfaces/middleware/trace_span.rs`, so this line
// doesn't need to duplicate those fields.
//
// The `visibility` field discriminates the two denial shapes for
// operators grepping exists-but-denied vs fully-hidden. `visible`
// denials are the ones surfaced to the caller as 403 (and safe to
// detail in the UI); `hidden` denials are the 404 anti-enum path.
tracing::info!(
target: "audit",
event = "authz.denied",
visibility = visibility.as_str(),
subject_type = subject.type_str(),
subject_id = %subject.id(),
permission = permission.as_str(),
resource_type = resource.type_str(),
resource_id = %resource.id(),
"👮🏻‍♂️ perms: ⛔ Subject '{}' hasn't permission to '{}' on resource '{}' (visibility={})",
subject,
permission,
resource,
visibility.as_str()
);
match visibility {
AuthzDenialVisibility::Visible => Err(DomainError::access_denied(
kind,
format!("Missing '{}' permission on {} {}", permission, kind, id),
)),
AuthzDenialVisibility::Hidden => Err(DomainError::not_found(kind, id.to_string())),
}
}
+22
View File
@@ -34,6 +34,12 @@ pub trait CalendarStoragePort: Send + Sync + 'static {
) -> Result<CalendarDto, DomainError>;
async fn delete_calendar(&self, calendar_id: &str) -> Result<(), DomainError>;
async fn get_calendar(&self, calendar_id: &str) -> Result<CalendarDto, DomainError>;
/// Batch sibling of [`Self::get_calendar`]: hydrate a page of
/// grant-derived calendar ids in ONE storage round-trip. Missing
/// rows (deleted/trashed race) drop out silently; ordering is not
/// guaranteed.
async fn get_calendars_by_ids(&self, ids: &[Uuid]) -> Result<Vec<CalendarDto>, DomainError>;
async fn list_calendars_by_owner(
&self,
owner_id: Uuid,
@@ -110,6 +116,12 @@ pub trait CalendarStoragePort: Send + Sync + 'static {
&self,
calendar_id: &str,
) -> Result<Vec<CalendarEventDto>, DomainError>;
/// Cursor stream over the calendar's events in bundle order (see
/// the repository doc) — feeds the streaming CalDAV emitters.
fn stream_events_uid_order(
&self,
calendar_id: &str,
) -> futures::stream::BoxStream<'static, Result<CalendarEventDto, DomainError>>;
async fn list_events_by_calendar_paginated(
&self,
calendar_id: &str,
@@ -212,6 +224,16 @@ pub trait CalendarUseCase: Send + Sync + 'static {
offset: Option<i64>,
user_id: Uuid,
) -> Result<Vec<CalendarEventDto>, DomainError>;
/// Streaming support: cursor over the calendar's events in bundle
/// order, behind the same Read authz gate as [`Self::list_events`].
async fn stream_events_uid_order(
&self,
calendar_id: &str,
user_id: Uuid,
) -> Result<
futures::stream::BoxStream<'static, Result<CalendarEventDto, DomainError>>,
DomainError,
>;
async fn get_events_in_range(
&self,
calendar_id: &str,
+21
View File
@@ -37,6 +37,12 @@ pub trait ContactStoragePort: Send + Sync + 'static {
) -> Result<AddressBook, DomainError>;
async fn delete_address_book(&self, id: &Uuid) -> Result<(), DomainError>;
async fn get_address_book_by_id(&self, id: &Uuid) -> Result<Option<AddressBook>, DomainError>;
/// Batch sibling of [`Self::get_address_book_by_id`]: hydrate a page
/// of grant-derived ids in ONE storage round-trip. Missing rows drop
/// out silently; ordering is not guaranteed.
async fn get_address_books_by_ids(&self, ids: &[Uuid])
-> Result<Vec<AddressBook>, DomainError>;
async fn get_public_address_books(&self) -> Result<Vec<AddressBook>, DomainError>;
// ── Contacts ─────────────────────────────────────────────────
@@ -60,6 +66,12 @@ pub trait ContactStoragePort: Send + Sync + 'static {
&self,
address_book_id: &Uuid,
) -> Result<Vec<Contact>, DomainError>;
/// Cursor stream over the book's contacts in listing order — feeds
/// the streaming CardDAV emitters.
fn stream_contacts_by_book(
&self,
address_book_id: Uuid,
) -> futures::stream::BoxStream<'static, Result<Contact, DomainError>>;
async fn get_contacts_by_address_book_paginated(
&self,
address_book_id: &Uuid,
@@ -168,6 +180,15 @@ pub trait ContactUseCase: Send + Sync + 'static {
/// List contacts in an address book. `limit`/`offset` bound the
/// result for paginated callers (REST API); `None` returns the full
/// book, which the CardDAV listing/sync paths rely on.
/// Streaming support: cursor over the book's contacts (same Read
/// gate as [`Self::list_contacts`], checked once before the cursor
/// opens).
async fn stream_contacts_by_book(
&self,
address_book_id: &str,
user_id: Uuid,
) -> Result<futures::stream::BoxStream<'static, Result<ContactDto, DomainError>>, DomainError>;
async fn list_contacts(
&self,
address_book_id: &str,
+19
View File
@@ -60,6 +60,25 @@ pub trait FileUploadUseCase: Send + Sync + 'static {
caller_id: Uuid,
) -> Result<FileDto, DomainError>;
/// `_with_perms` variant of `upload_file_streaming` — enforces
/// `Create` on the target folder before registering the row.
///
/// AuthZ audit #17 (2026-07-12): the chunked-upload `complete`
/// path called plain `upload_file_streaming` at finalize; a grant
/// revoked between session open and finalize stayed effective
/// until the caller landed the final chunk (up to 24h JWT TTL,
/// forever with app-passwords). Handlers now call this variant
/// so the engine re-checks at finalize regardless of how long
/// the session was open.
async fn upload_file_streaming_with_perms(
&self,
name: String,
folder_id: Option<String>,
content_type: String,
blob: StoredBlob,
caller_id: Uuid,
) -> Result<FileDto, DomainError>;
/// Replace the content of the file at `path` with an already-ingested
/// blob, or create the file when it doesn't exist (WebDAV/WOPI PUT).
///
+25
View File
@@ -77,6 +77,31 @@ pub trait FolderUseCase: Send + Sync + 'static {
pagination: &crate::application::dtos::pagination::PaginationRequestDto,
) -> Result<crate::application::dtos::pagination::PaginatedResponseDto<FolderDto>, DomainError>;
/// Keyset-paged sub-folder listing in name order, scoped to a caller —
/// `name > after_name LIMIT limit`, `has_next = len() == limit`.
///
/// Used by streaming WebDAV/NC PROPFIND: O(page) per page off the
/// `idx_folders_unique_name` index instead of the quadratic
/// `COUNT(*) OVER() … LIMIT/OFFSET` walk (benches/FOLDER-KEYSET.md).
///
/// The default implementation falls back to `list_folders_with_perms`
/// + in-memory slice so stubs and mocks compile without changes.
async fn list_folders_batch_with_perms(
&self,
parent_id: Option<&str>,
caller_id: Uuid,
after_name: Option<&str>,
limit: usize,
) -> Result<Vec<FolderDto>, DomainError> {
let mut all = self.list_folders_with_perms(parent_id, caller_id).await?;
all.sort_by(|a, b| a.name.cmp(&b.name));
Ok(all
.into_iter()
.filter(|f| after_name.is_none_or(|a| f.name.as_str() > a))
.take(limit)
.collect())
}
/// Renames a folder (ownership verified against caller_id)
async fn rename_folder_with_perms(
&self,
+4
View File
@@ -27,11 +27,15 @@ pub trait SearchUseCase: Send + Sync + 'static {
) -> Result<Arc<SearchResultsDto>, DomainError>;
/// Returns quick suggestions for autocomplete (lightweight, fast).
/// `caller_id` scopes results to drives the caller can Read — without
/// it the endpoint leaks names + paths across every tenant on the
/// instance (AuthZ audit finding #1, 2026-07-12).
async fn suggest(
&self,
query: &str,
folder_id: Option<&str>,
limit: usize,
caller_id: Uuid,
) -> Result<SearchSuggestionsDto, DomainError>;
/// Clears the search results cache.
+5
View File
@@ -104,6 +104,11 @@ pub trait MusicStoragePort: Send + Sync {
async fn get_playlist(&self, playlist_id: &str) -> Result<Option<PlaylistDto>, DomainError>;
/// Batch sibling of [`Self::get_playlist`]: hydrate a page of
/// grant-derived ids in ONE storage round-trip. Missing rows drop
/// out silently; ordering is not guaranteed.
async fn get_playlists_by_ids(&self, ids: &[Uuid]) -> Result<Vec<PlaylistDto>, DomainError>;
async fn list_playlists_by_owner(
&self,
owner_id: Uuid,
+10 -2
View File
@@ -205,13 +205,21 @@ pub trait FileReadPort: Send + Sync + 'static {
/// Results are ordered by relevance (exact > starts-with > contains) so the
/// caller can use them directly for autocomplete suggestions.
///
/// The default implementation falls back to `list_files` + in-memory filter
/// so that stubs and mocks compile without changes.
/// `caller_id` scopes results to files whose owning drive the caller can
/// Read (direct or group-mediated `role_grants`). Without it the endpoint
/// leaks names + paths across every tenant on the instance — closed as
/// AuthZ audit finding #1 (2026-07-12).
///
/// The default implementation falls back to `list_files` + in-memory
/// filter so that stubs and mocks compile without changes. Stub-mode
/// callers already operate against a single tenant's data, so ignoring
/// `caller_id` here is safe; the PG impl enforces the real scope.
async fn suggest_files_by_name(
&self,
folder_id: Option<&str>,
query: &str,
limit: usize,
_caller_id: Uuid,
) -> Result<Vec<File>, DomainError> {
let all = self.list_files(folder_id).await?;
let q = query.to_lowercase();
@@ -304,12 +304,45 @@ impl AppPasswordService {
let cache_key: [u8; 32] =
blake3::hash(format!("{}:{}", username, password).as_bytes()).into();
// ── 2. Cache hit → return immediately ────────────────────────
if let Some(cached) = self.auth_cache.get(&cache_key).await {
return Ok((cached.user_id, cached.username, cached.email, cached.role));
}
// ── 2. Single-flight cache lookup ─────────────────────────────
// Concurrent misses on the same credential coalesce into ONE
// full verification: DAV sync clients hold 4-8 parallel
// connections, so an expiring cache entry used to fan out into
// K simultaneous Argon2id runs (~100-300 ms CPU + 64 MiB RAM
// apiece) every TTL — a recurring p99 spike on every DAV
// surface (8 -> 1 verifications, benches/AUTH-HERD.md).
// `try_get_with` caches only `Ok` results, so failed
// verifications are still never cached, preserving the full
// Argon2id cost as a brute-force deterrent.
let result = self
.auth_cache
.try_get_with(
cache_key,
self.verify_basic_auth_uncached(username, password),
)
.await
.map_err(
|e: std::sync::Arc<DomainError>| match std::sync::Arc::try_unwrap(e) {
Ok(err) => err,
// Another coalesced waiter still holds the Arc — rebuild
// an equivalent error (the source chain isn't clonable).
Err(shared) => {
DomainError::new(shared.kind, shared.entity_type, shared.message.clone())
}
},
)?;
Ok((result.user_id, result.username, result.email, result.role))
}
// ── 3. Cache miss → full verification ────────────────────────
/// The uncached Basic Auth slow path: user lookup, prefix-scoped
/// candidate fetch, Argon2id verification. Runs at most once per
/// credential per TTL — `verify_basic_auth` coalesces concurrent
/// callers onto a single in-flight instance of this future.
async fn verify_basic_auth_uncached(
&self,
username: &str,
password: &str,
) -> Result<CachedBasicAuthResult, DomainError> {
let user = self
.user_repo
.get_user_by_username(username)
@@ -363,15 +396,14 @@ impl AppPasswordService {
{
let _ = self.repo.touch_last_used(ap.id).await;
let result = CachedBasicAuthResult {
// Caching happens in `verify_basic_auth`: `try_get_with`
// stores this value under the blake3 key on return.
return Ok(CachedBasicAuthResult {
user_id: user.id(),
username: user.username().unwrap_or("").to_string(),
email: user.email().to_string(),
role: user.role().to_string(),
};
self.auth_cache.insert(cache_key, result.clone()).await;
return Ok((result.user_id, result.username, result.email, result.role));
});
}
}
@@ -147,8 +147,12 @@ pub struct AuthApplicationService {
/// request. The short TTL keeps the "role changes apply without token
/// rotation" property within seconds while removing one DB round-trip
/// per request; the known mutation paths (`change_user_role`,
/// `set_user_active`) also invalidate eagerly.
user_flags_cache: Cache<Uuid, UserFlags>,
/// `set_user_active`) also invalidate eagerly. `moka::future` so
/// concurrent misses for one user coalesce into a single DB lookup
/// (`try_get_with` single-flight) — every authenticated request
/// calls this, so each 30 s TTL expiry used to fan out one SELECT
/// per in-flight request of that user.
user_flags_cache: moka::future::Cache<Uuid, UserFlags>,
/// Self-service auth-method allowlist (mirrors
/// `AuthConfig::allowed_auth_methods`). Empty = both methods
/// allowed. Consulted by login / register / magic-link handlers via
@@ -198,7 +202,7 @@ impl AuthApplicationService {
.time_to_live(Duration::from_secs(120))
.build(),
magic_link_repo: None,
user_flags_cache: Cache::builder()
user_flags_cache: moka::future::Cache::builder()
.max_capacity(10_000)
.time_to_live(USER_FLAGS_CACHE_TTL)
.build(),
@@ -1363,7 +1367,7 @@ impl AuthApplicationService {
// Invalidate the flags cache so subsequent per-request guards
// observe the new `is_external=false` without waiting for the
// 30-second TTL. Same pattern as `change_user_role`.
self.user_flags_cache.invalidate(&caller_id);
self.user_flags_cache.invalidate(&caller_id).await;
// Dispatch — home-drive provisioning happens here. Log-and-
// continue: a provisioning failure leaves the row updated and
@@ -1508,12 +1512,20 @@ impl AuthApplicationService {
/// Staleness is bounded by [`USER_FLAGS_CACHE_TTL`]; role and active
/// changes made through this service invalidate the entry eagerly.
pub async fn get_user_flags(&self, user_id: Uuid) -> Result<UserFlags, DomainError> {
if let Some(flags) = self.user_flags_cache.get(&user_id) {
return Ok(flags);
}
let flags = self.user_storage.get_user_flags(user_id).await?;
self.user_flags_cache.insert(user_id, flags);
Ok(flags)
// Single-flight: concurrent misses for the same user coalesce
// into ONE storage lookup; errors are never cached (same herd
// shape ROUND3 fixed for basic-auth, minus the Argon2 cost).
self.user_flags_cache
.try_get_with(user_id, async {
Ok::<_, DomainError>(self.user_storage.get_user_flags(user_id).await?)
})
.await
// try_get_with hands back `Arc<DomainError>` shared by all
// waiters; DomainError isn't Clone, so rebuild a fresh one
// preserving the kind / entity / message.
.map_err(|shared: std::sync::Arc<DomainError>| {
DomainError::new(shared.kind, shared.entity_type, shared.message.clone())
})
}
/// Apply a profile update on behalf of the calling user (PR 24).
@@ -1912,6 +1924,59 @@ impl AuthApplicationService {
))
}
/// Username-keyed sibling of [`Self::get_user_profile`], routing every
/// lookup through the same visibility check as the user-profile REST
/// endpoint. Preserves the anti-enum shape end-to-end: whether the
/// username doesn't exist OR the caller has no visibility path, the
/// response is `NotFound`.
///
/// AuthZ audit #11 (2026-07-12): NextCloud OCS user-provisioning
/// (`nextcloud/ocs_handler.rs::user_provisioning_response`) used to
/// resolve `userid` via bare `get_user_by_username`, gated only by a
/// bespoke `caller.role == "admin"` shortcut. Admins bypassed the
/// `expose_system_users` gate; non-admins got a `403 Insufficient
/// privileges` for any cross-user probe (leaking existence via the
/// differential vs a genuine 404); zero audit lines. This wrapper
/// closes all three.
///
/// The username→id resolution happens here so the target isn't
/// leaked through the audit line as a plaintext username on failure:
/// the `target_username_not_found` event carries the string
/// (unavoidable — we resolved it, we log it), but every other
/// downstream event keys off `target_id` after resolution, matching
/// the id-based endpoint.
pub async fn get_user_profile_by_username_with_perms(
&self,
caller_id: Uuid,
username: &str,
expose_system_users: bool,
pool: &sqlx::PgPool,
) -> Result<UserDto, DomainError> {
let target = match self.user_storage.get_user_by_username(username).await {
Ok(u) => u,
Err(e) if e.kind == ErrorKind::NotFound => {
tracing::info!(
target: "audit",
event = "user_profile.rejected",
reason = "target_username_not_found",
caller_id = %caller_id,
target_username = %username,
"👮🏻‍♂️ user-profile rejected: username '{}' does not exist (caller {})",
username,
caller_id,
);
return Err(DomainError::new(
ErrorKind::NotFound,
"User",
"User not found",
));
}
Err(e) => return Err(e),
};
self.get_user_profile(caller_id, target.id(), expose_system_users, pool)
.await
}
// New method to get user by username - needed for admin user handling
pub async fn get_user_by_username(&self, username: &str) -> Result<UserDto, DomainError> {
let user = self.user_storage.get_user_by_username(username).await?;
@@ -2226,7 +2291,7 @@ impl AuthApplicationService {
self.user_storage
.set_user_active_status(user_id, active)
.await?;
self.user_flags_cache.invalidate(&user_id);
self.user_flags_cache.invalidate(&user_id).await;
Ok(())
}
@@ -2240,7 +2305,7 @@ impl AuthApplicationService {
));
}
self.user_storage.change_role(user_id, role).await?;
self.user_flags_cache.invalidate(&user_id);
self.user_flags_cache.invalidate(&user_id).await;
Ok(())
}
+29 -11
View File
@@ -189,17 +189,13 @@ impl CalendarUseCase for CalendarService {
})
.collect();
// Hydrate DTOs. `get_calendar` misses on trashed / deleted
// calendars — those are dropped from the listing rather than
// erroring, so a lifecycle-race doesn't turn a PROPFIND into
// a 5xx.
let mut out = Vec::with_capacity(calendar_ids.len());
for id in calendar_ids {
if let Ok(dto) = self.calendar_storage.get_calendar(&id.to_string()).await {
out.push(dto);
}
}
Ok(out)
// Hydrate DTOs in ONE `= ANY` round-trip (was one point SELECT
// per accessible calendar — K serial round-trips on every
// CalDAV discovery poll). Missing rows (deleted/trashed race)
// drop out of the result set instead of erroring, so a
// lifecycle-race still doesn't turn a PROPFIND into a 5xx.
let ids: Vec<Uuid> = calendar_ids.into_iter().collect();
self.calendar_storage.get_calendars_by_ids(&ids).await
}
async fn list_public_calendars(
@@ -360,6 +356,28 @@ impl CalendarUseCase for CalendarService {
}
}
async fn stream_events_uid_order(
&self,
calendar_id: &str,
user_id: Uuid,
) -> Result<
futures::stream::BoxStream<'static, Result<CalendarEventDto, DomainError>>,
DomainError,
> {
// Same Read gate as `list_events`, checked ONCE before the
// cursor opens — the stream itself carries no further authz
// (single request, same caller, same resource).
let calendar = self.calendar_storage.get_calendar(calendar_id).await?;
let allowed = calendar.is_public
|| self
.has_calendar_perm(calendar_id, user_id, Permission::Read)
.await?;
if !allowed {
return Err(DomainError::not_found("Calendar", calendar_id));
}
Ok(self.calendar_storage.stream_events_uid_order(calendar_id))
}
async fn get_events_in_range(
&self,
calendar_id: &str,
+55 -16
View File
@@ -494,12 +494,14 @@ impl AddressBookUseCase for ContactService {
let mut address_book_map = std::collections::HashMap::new();
for id in book_ids {
// Missing rows (deleted / trashed race) drop out silently
// — matches the calendar-listing carve-out.
if let Ok(Some(book)) = self.contact_storage.get_address_book_by_id(&id).await {
address_book_map.insert(*book.id(), book);
}
// Hydrate in ONE `= ANY` round-trip (was one point SELECT per
// accessible book — K serial round-trips on every CardDAV
// discovery poll). Missing rows (deleted / trashed race) drop
// out of the result set — matches the calendar-listing
// carve-out.
let ids: Vec<Uuid> = book_ids.into_iter().collect();
for book in self.contact_storage.get_address_books_by_ids(&ids).await? {
address_book_map.insert(*book.id(), book);
}
// Public address books surface for every authenticated caller
@@ -533,10 +535,16 @@ impl ContactUseCase for ContactService {
let address_book_id = Uuid::parse_str(&dto.address_book_id)
.map_err(|_| DomainError::validation_error("Invalid address book ID format"))?;
// Check if user has write access to the address book
// AuthZ audit #19 (2026-07-12): previously required
// `Permission::Update`, which is NOT in the Contributor bundle
// (Read + Create) — Contributor grantees on a shared address
// book couldn't add contacts via REST or CardDAV PUT despite
// holding the intended Create permission. `Delete` uses Delete
// (audit #13, above); creation must use Create. Same fix
// applied to `create_contact_from_vcard` + `create_group`.
let caller_id = Uuid::parse_str(&dto.user_id)
.map_err(|_| DomainError::validation_error("Invalid user ID format"))?;
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Update)
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Create)
.await?;
// Convert DTOs to domain entities
@@ -612,10 +620,13 @@ impl ContactUseCase for ContactService {
let address_book_id = Uuid::parse_str(&dto.address_book_id)
.map_err(|_| DomainError::validation_error("Invalid address book ID format"))?;
// Check if user has write access to the address book
// AuthZ audit #19 — see the sibling `create_contact` above.
// This is the CardDAV `PUT contact.vcf` entry point; the fix
// unblocks Contributor grantees creating contacts through the
// CardDAV protocol as well as the REST surface.
let caller_id = Uuid::parse_str(&dto.user_id)
.map_err(|_| DomainError::validation_error("Invalid user ID format"))?;
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Update)
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Create)
.await?;
// Parse vCard data
@@ -754,8 +765,14 @@ impl ContactUseCase for ContactService {
.await?
.ok_or_else(|| DomainError::not_found("Contact", "not found"))?;
// Check if user has write access to the address book
self.require_address_book_perm(contact.address_book_id(), &user_id, Permission::Update)
// AuthZ audit #13 (2026-07-12): previously required
// `Permission::Update`, which the Editor role bundle satisfies
// (Read + Comment + Create + Update). Every Editor grantee on a
// shared address book could delete individual contacts — a
// silent privilege escalation because the intent for CardDAV
// deletion is Delete, not Update. Sibling
// `CalendarService::delete_event` was the ground-truth pattern.
self.require_address_book_perm(contact.address_book_id(), &user_id, Permission::Delete)
.await?;
// Delete the contact
@@ -823,6 +840,25 @@ impl ContactUseCase for ContactService {
Ok(contacts.into_iter().map(ContactDto::from).collect())
}
async fn stream_contacts_by_book(
&self,
address_book_id: &str,
user_id: Uuid,
) -> Result<futures::stream::BoxStream<'static, Result<ContactDto, DomainError>>, DomainError>
{
use futures::StreamExt;
let id = Uuid::parse_str(address_book_id)
.map_err(|_| DomainError::validation_error("Invalid address book ID format"))?;
// Same Read gate as `list_contacts`, once, before the cursor.
self.require_address_book_read_or_public(&id, &user_id)
.await?;
Ok(Box::pin(
self.contact_storage
.stream_contacts_by_book(id)
.map(|r| r.map(ContactDto::from)),
))
}
async fn list_contacts(
&self,
address_book_id: &str,
@@ -881,10 +917,10 @@ impl ContactUseCase for ContactService {
let address_book_id = Uuid::parse_str(&dto.address_book_id)
.map_err(|_| DomainError::validation_error("Invalid address book ID format"))?;
// Check if user has write access to the address book
// AuthZ audit #19 — see the sibling `create_contact` above.
let caller_id = Uuid::parse_str(&dto.user_id)
.map_err(|_| DomainError::validation_error("Invalid user ID format"))?;
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Update)
self.require_address_book_perm(&address_book_id, &caller_id, Permission::Create)
.await?;
let group = ContactGroup::new(address_book_id, dto.name);
@@ -938,8 +974,11 @@ impl ContactUseCase for ContactService {
.await?
.ok_or_else(|| DomainError::not_found("Contact group", "not found"))?;
// Check if user has write access to the address book
self.require_address_book_perm(group.address_book_id(), &user_id, Permission::Update)
// AuthZ audit #13 (2026-07-12): see the sibling `delete_contact`
// above — required `Update` (in the Editor bundle) instead of
// `Delete`, letting any Editor on a shared address book delete
// groups they shouldn't.
self.require_address_book_perm(group.address_book_id(), &user_id, Permission::Delete)
.await?;
// Delete the group
@@ -263,6 +263,12 @@ impl DriveManagementService {
self.authz
.invalidate_drive_role_cache_for_drive(drive_id)
.await;
// Same freshness contract for the repo's readable-drives cache:
// the subject's drive list changed with this grant.
match subject {
Subject::User(uid) => self.drive_repo.invalidate_readable_for_user(uid).await,
_ => self.drive_repo.invalidate_readable_all(),
}
// D6 §11: canonical `drive.member_added` audit event covers
// every successful membership write (add + role-refresh, since
@@ -335,6 +341,12 @@ impl DriveManagementService {
self.authz
.invalidate_drive_role_cache_for_drive(drive_id)
.await;
// And the repo's readable-drives cache: the drive must vanish
// from the removed subject's list immediately.
match subject {
Subject::User(uid) => self.drive_repo.invalidate_readable_for_user(uid).await,
_ => self.drive_repo.invalidate_readable_all(),
}
// D6 §11: canonical `drive.member_removed` audit event covers
// every successful removal (owner-driven or admin bypass).
@@ -468,6 +480,11 @@ impl DriveManagementService {
/// supplied is overwritten. Returns the post-merge typed view.
/// Audit emits `drive.policy_changed` with the post-merge bag for
/// steady-state observability.
///
/// Ed's call, 2026-07-17: intentional deviation from the AGENTS.md
/// "AuthZ in service layer" rule for this specific endpoint —
/// the handler-layer admin check stays, this method stays trusting.
/// See memory `feedback_drive_policies_admin_at_handler`.
pub async fn update_policies(
&self,
caller_id: Uuid,
@@ -145,6 +145,12 @@ impl FavoritesUseCase for FavoritesService {
// valid (partial success would leak the same oracle we
// closed on the single-item path). See
// `docs/plan/authz_audit/rest_storage.md`.
//
// Deliberately serial: a `try_join_all` fan-out measured WORSE
// on both the cold (drive_of point-SELECTs) and warm (all-moka)
// paths — future orchestration + pool-acquire contention cost
// more than the local round trips they overlap. Rejected by
// `bench_favorites_authz`; numbers in benches/ROUND6.md.
for (item_id, item_type) in items {
let resource = Resource::parse(item_type, item_id)?;
self.authorization
@@ -287,22 +287,6 @@ impl FileRetrievalService {
Ok(files.into_iter().map(FileDto::from).collect())
}
/// Range read that first consults the RAM content cache (see
/// [`Self::get_file_range_preloaded`]).
pub async fn get_file_range_preloaded_with_perms(
&self,
dto: &FileDto,
caller_id: Uuid,
start: u64,
end: Option<u64>,
) -> Result<RangeContent, DomainError> {
self.require_file(&dto.id, Permission::Read, caller_id)
.await?;
// Same throttled Recent recording as the streaming variant.
self.notify_file_accessed(caller_id, &dto.id);
self.get_file_range_preloaded(dto, start, end).await
}
/// Range read for HTTP Range Requests, cache-aware.
///
/// Media players and PDF viewers fetch these files *exclusively* through
@@ -457,6 +457,44 @@ impl FileUploadUseCase for FileUploadService {
Ok(dto)
}
/// AuthZ audit #17 — `Create` on target folder is re-verified here
/// so mid-session grant revocations take effect at finalize. When
/// `folder_id` is `None` the write lands at drive-root; the drive
/// resolution for that case isn't plumbed through the chunked-
/// upload session (`UploadSession.folder_id` alone), so we fall
/// back to the pre-audit behaviour there. That drive-root path is
/// tracked separately as part of the D0 folder-id-walking work;
/// closing it here would require session-scoped drive_id.
async fn upload_file_streaming_with_perms(
&self,
name: String,
folder_id: Option<String>,
content_type: String,
blob: StoredBlob,
caller_id: Uuid,
) -> Result<FileDto, DomainError> {
if let Some(fid) = folder_id.as_deref() {
let Some(authz) = &self.authorization else {
return Err(DomainError::internal_error(
"FileUpload",
"upload_file_streaming_with_perms called without authorization engine wired",
));
};
let folder_uuid = Uuid::parse_str(fid)
.map_err(|_| DomainError::not_found("Folder", fid.to_string()))?;
authz
.require(
Subject::User(caller_id),
Permission::Create,
Resource::Folder(folder_uuid),
)
.await?;
}
self.upload_file_streaming(name, folder_id, content_type, blob, caller_id)
.await
}
/// Swap the content of the file at `path` to an already-ingested blob,
/// creating the file when it doesn't exist (WebDAV/NextCloud/WOPI PUT).
///
+77 -2
View File
@@ -429,6 +429,62 @@ impl FolderUseCase for FolderService {
Ok(response)
}
/// Keyset-paged sub-folder listing (name order), caller-scoped.
///
/// AuthZ mirrors `list_folders_paginated_with_perms`: one
/// `authz.require(Read)` on the parent per batch; root scope goes
/// through the caller's drive-membership listing.
async fn list_folders_batch_with_perms(
&self,
parent_id: Option<&str>,
caller_id: Uuid,
after_name: Option<&str>,
limit: usize,
) -> Result<Vec<FolderDto>, DomainError> {
match parent_id {
Some(pid) => {
self.authz
.require(
Subject::User(caller_id),
Permission::Read,
Self::folder_resource(pid)?,
)
.await?;
let folders = self
.folder_storage
.list_folders_batch(parent_id, after_name, limit)
.await
.map_err(|e| {
DomainError::internal_error(
"FolderStorage",
format!("Failed to batch-list folders in parent {pid}: {e}"),
)
})?;
Ok(folders.into_iter().map(FolderDto::from).collect())
}
None => {
// Root scope: one row per readable drive — a handful.
let mut all = self
.folder_storage
.list_root_folders_for_caller(caller_id)
.await
.map_err(|e| {
DomainError::internal_error(
"FolderStorage",
format!("Failed to batch-list root folders for '{caller_id}': {e}"),
)
})?;
all.sort_by(|a, b| a.name().cmp(b.name()));
Ok(all
.into_iter()
.filter(|f| after_name.is_none_or(|a| f.name() > a))
.take(limit)
.map(FolderDto::from)
.collect())
}
}
}
/// Lists folders with pagination, scoped to a specific owner.
async fn list_folders_paginated_with_perms(
&self,
@@ -525,7 +581,7 @@ impl FolderUseCase for FolderService {
)
.await?;
let folder = self
let renamed = self
.folder_storage
.rename_folder(id, dto.name, caller_id)
.await
@@ -536,7 +592,26 @@ impl FolderUseCase for FolderService {
)
})?;
Ok(FolderDto::from(folder))
// Root folders double as the drive's display name (see the
// `required_perm` branch above and `drive_pg_repository.rs`
// `readable_cache` + `default_drive_cache` docs).
// `drives.name` is sourced from `folders.name` of the root
// folder, so a rename affects BOTH caches — every user's
// readable-drive list AND the per-user default-drive lookup.
// Both are 30 s TTL; without the invalidation, `GET /api/drives`
// returns the stale name for up to that window after a root
// rename. Surfaced by `tests/api/drives_membership.hurl`
// Step 23. Regression from commit `12dc648c` ("perf: round 4 —
// drive-selector cache") which added the caches without
// wiring the root-rename invalidation.
if folder.parent_id().is_none()
&& let Some(drive_repo) = &self.drive_repo
{
drive_repo.invalidate_readable_all();
drive_repo.invalidate_default_drive_all();
}
Ok(FolderDto::from(renamed))
}
/// Moves a folder to a new parent. Requires `Update` on the source and
+11 -8
View File
@@ -196,15 +196,18 @@ impl MusicUseCase for MusicService {
// only. Owner is a grant like any other in `role_grants`, so we
// filter the aggregated set against the owner_id stamped on
// each row after hydration — cheaper than a second SQL round-trip.
let mut playlists: Vec<PlaylistDto> = Vec::with_capacity(playlist_ids.len());
// Hydrate in ONE `= ANY` round-trip (was one point SELECT per
// accessible playlist). Missing rows (deleted race) drop out of
// the result set silently, as before.
let user_str = user_id.to_string();
for id in playlist_ids.drain() {
if let Ok(Some(p)) = self.storage.get_playlist(&id.to_string()).await
&& (include_shared || p.owner_id == user_str)
{
playlists.push(p);
}
}
let ids: Vec<Uuid> = playlist_ids.drain().collect();
let mut playlists: Vec<PlaylistDto> = self
.storage
.get_playlists_by_ids(&ids)
.await?
.into_iter()
.filter(|p| include_shared || p.owner_id == user_str)
.collect();
if include_public {
let public = self.storage.list_public_playlists(limit, offset).await?;
@@ -41,55 +41,50 @@ impl NextcloudFileIdService {
/// Resolve — creating when absent — stable numeric file IDs for many
/// UUIDs at once. Cache hits cost nothing; the misses are resolved with a
/// single backing query. The returned map is keyed by the caller's
/// original id strings; unresolvable inputs are simply absent (mirroring
/// the `.ok()` behaviour the callers relied on).
pub async fn get_or_create_file_ids(
&self,
file_ids: &[String],
) -> Result<HashMap<String, i64>> {
/// single backing query. The returned map is keyed by parsed UUID;
/// unparseable/unresolvable inputs are simply absent (mirroring the
/// `.ok()` behaviour the callers relied on).
pub async fn get_or_create_file_ids(&self, file_ids: &[&str]) -> Result<HashMap<Uuid, i64>> {
self.get_or_create_many("file", file_ids).await
}
/// Folder counterpart of [`Self::get_or_create_file_ids`].
pub async fn get_or_create_folder_ids(
&self,
folder_ids: &[String],
) -> Result<HashMap<String, i64>> {
folder_ids: &[&str],
) -> Result<HashMap<Uuid, i64>> {
self.get_or_create_many("folder", folder_ids).await
}
async fn get_or_create_many(
&self,
object_type: &str,
raw_ids: &[String],
) -> Result<HashMap<String, i64>> {
raw_ids: &[&str],
) -> Result<HashMap<Uuid, i64>> {
let mut result = HashMap::with_capacity(raw_ids.len());
// Parsed-UUID → caller's original string; also dedupes the miss list.
let mut pending: HashMap<Uuid, String> = HashMap::new();
let mut misses: Vec<Uuid> = Vec::new();
for raw in raw_ids {
let Ok(uuid) = Uuid::parse_str(raw) else {
continue; // Unparseable ids never had a mapping — skip silently.
};
if let Some(id) = self.cache.get(&uuid).await {
result.insert(raw.clone(), id);
result.insert(uuid, id);
} else {
pending.entry(uuid).or_insert_with(|| raw.clone());
misses.push(uuid);
}
}
if !pending.is_empty() {
let misses: Vec<Uuid> = pending.keys().copied().collect();
if !misses.is_empty() {
misses.sort_unstable();
misses.dedup();
let resolved = self
.repo()?
.get_or_create_many(object_type, &misses)
.await?;
for (uuid, id) in resolved {
self.cache.insert(uuid, id).await;
if let Some(original) = pending.get(&uuid) {
result.insert(original.clone(), id);
}
result.insert(uuid, id);
}
}
@@ -184,10 +179,7 @@ mod tests {
#[tokio::test]
async fn test_get_or_create_file_ids_skips_unparseable() {
let svc = NextcloudFileIdService::new_stub();
let map = svc
.get_or_create_file_ids(&["not-a-uuid".to_string()])
.await
.unwrap();
let map = svc.get_or_create_file_ids(&["not-a-uuid"]).await.unwrap();
assert!(map.is_empty());
}
}
+263 -87
View File
@@ -2,9 +2,7 @@ use std::cmp::Reverse;
use std::sync::Arc;
use std::time::{Duration, Instant};
use crate::application::dtos::display_helpers::{
category_for, icon_class_for, icon_special_class_for,
};
use crate::application::dtos::display_helpers::intern_display;
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::FolderDto;
use crate::application::dtos::search_dto::{
@@ -67,9 +65,80 @@ pub struct SearchService {
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
///
/// **Byte-bounded**, not entry-bounded: entries are weighed by
/// [`search_results_entry_weight`] and `max_capacity` is a byte budget.
/// Keys span user × query × offset × limit, and each page holds up to 500
/// enriched rows (~500–900 B of owned Strings each) — an entry-count bound
/// let hundreds of MB of result pages accumulate invisibly.
search_cache: moka::future::Cache<u64, Arc<SearchResultsDto>>,
}
// ─── Search-results cache (byte-bounded) ─────────────────────────────────
/// Approximate heap bytes retained by one cached search page.
///
/// With a `weigher` installed, moka's `max_capacity` is the sum of entry
/// *weights*, so this converts the cache bound from "number of entries" to
/// real bytes: the length of every owned `String` in each file/folder row,
/// plus a fixed per-row and per-entry overhead for struct fields, the 24-B
/// `String` headers, `Vec` slots and allocator slop. Same pattern as the
/// file-content cache and the dedup manifest cache.
///
/// `pub` so `examples/bench_search_cache_mem.rs` can recompute retained
/// bytes with the exact production formula.
pub fn search_results_entry_weight(_key: &u64, value: &Arc<SearchResultsDto>) -> u32 {
/// Fixed per-row overhead: struct scalars + one 24-B header per `String`
/// field (12 on a file row, 4 on a folder row) + `Vec` slot + allocator
/// slop. Deliberately a round upper-ish estimate — under-weighing is the
/// failure mode that re-opens the memory hole.
const ROW_OVERHEAD: usize = 200;
/// Fixed per-entry overhead: `Arc` + `SearchResultsDto` scalars + `Vec`
/// headers + moka's own bookkeeping per entry.
const ENTRY_OVERHEAD: usize = 256;
fn opt_len(s: &Option<String>) -> usize {
s.as_deref().map_or(0, str::len)
}
let mut bytes = ENTRY_OVERHEAD + value.sort_by.len();
for f in &value.files {
bytes += ROW_OVERHEAD
+ f.id.len()
+ f.name.len()
+ f.path.len()
+ f.mime_type.len()
+ opt_len(&f.folder_id)
+ f.size_formatted.len()
+ f.icon_class.len()
+ f.icon_special_class.len()
+ f.category.len()
+ f.blob_hash.len()
+ opt_len(&f.snippet)
+ opt_len(&f.match_source);
}
for d in &value.folders {
bytes += ROW_OVERHEAD + d.id.len() + d.name.len() + d.path.len() + opt_len(&d.parent_id);
}
bytes.min(u32::MAX as usize) as u32
}
/// Build the search-results cache exactly as production wires it: a byte
/// budget enforced through [`search_results_entry_weight`], plus TTL.
///
/// Shared with `examples/bench_search_cache_mem.rs` so the benchmark
/// measures the identical cache configuration that serves requests.
pub fn build_search_results_cache(
cache_ttl_secs: u64,
max_bytes: u64,
) -> moka::future::Cache<u64, Arc<SearchResultsDto>> {
moka::future::Cache::builder()
.max_capacity(max_bytes)
.weigher(search_results_entry_weight)
.time_to_live(Duration::from_secs(cache_ttl_secs))
.build()
}
// ─── Utility functions (pure, no self — computed on the server) ─────────
/// Compute relevance score (0–100) for a name against a query.
@@ -138,28 +207,15 @@ fn format_bytes(bytes: u64) -> String {
}
}
/// Get Font Awesome icon class for a file based on extension and MIME type.
/// Delegates to the centralised `display_helpers` so every API surface is
/// consistent.
fn get_icon_class(name: &str, mime: &str) -> String {
icon_class_for(name, mime).to_string()
}
/// Get CSS special class for icon styling.
fn get_icon_special_class(name: &str, mime: &str) -> String {
icon_special_class_for(name, mime).to_string()
}
/// Get category label from centralised helpers.
fn get_category(name: &str, mime: &str) -> String {
category_for(name, mime).to_string()
}
// ─── SearchService implementation ───────────────────────────────────────
impl SearchService {
/**
* Creates a new instance of the search service.
*
* `max_cache_bytes` is the byte budget for the results cache (weigher-
* bounded, see [`search_results_entry_weight`]) — it replaced the old
* entry-count capacity, which was blind to how big each cached page is.
*/
pub fn new(
file_repository: Arc<FileBlobReadRepository>,
@@ -168,12 +224,9 @@ impl SearchService {
authorization: Option<Arc<crate::infrastructure::services::pg_acl_engine::PgAclEngine>>,
drive_repo: Option<Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>>,
cache_ttl: u64,
max_cache_size: usize,
max_cache_bytes: u64,
) -> Self {
let search_cache = moka::future::Cache::builder()
.max_capacity(max_cache_size as u64)
.time_to_live(Duration::from_secs(cache_ttl))
.build();
let search_cache = build_search_results_cache(cache_ttl, max_cache_bytes);
Self {
file_repository,
@@ -195,8 +248,14 @@ impl SearchService {
/// Enrich a FileDto → SearchFileResultDto with server-computed metadata.
///
/// Consumes the DTO: every `String` moves and the interned display
/// fields (`mime_type`/`icon_class`/`icon_special_class`/`category`,
/// already computed once in `FileDto::from`) transfer as refcount
/// bumps — the old borrow-based version cloned all of them AND re-ran
/// the three display classifiers per result row.
///
/// `query_lower` must already be lowercased (empty string when no query).
fn enrich_file(file: &FileDto, query_lower: &str) -> SearchFileResultDto {
fn enrich_file(file: FileDto, query_lower: &str) -> SearchFileResultDto {
let relevance = if query_lower.is_empty() {
50
} else {
@@ -204,23 +263,23 @@ impl SearchService {
};
SearchFileResultDto {
id: file.id.clone(),
name: file.name.clone(),
path: file.path.clone(),
id: file.id,
name: file.name,
path: file.path,
size: file.size,
mime_type: file.mime_type.to_string(),
folder_id: file.folder_id.clone(),
mime_type: file.mime_type,
folder_id: file.folder_id,
created_at: file.created_at,
modified_at: file.modified_at,
relevance_score: relevance,
size_formatted: format_bytes(file.size),
icon_class: get_icon_class(&file.name, &file.mime_type),
icon_special_class: get_icon_special_class(&file.name, &file.mime_type),
category: get_category(&file.name, &file.mime_type),
icon_class: file.icon_class,
icon_special_class: file.icon_special_class,
category: file.category,
// Carry the content hash through so REPORT/SEARCH
// responses on the NC surface can emit the same ETag
// (`File::compute_etag`) as PROPFIND/GET would.
blob_hash: file.content_hash.clone(),
blob_hash: file.content_hash,
snippet: None,
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
}
@@ -228,8 +287,10 @@ impl SearchService {
/// Enrich a FolderDto → SearchFolderResultDto with server-computed metadata.
///
/// Consumes the DTO so the owned strings move instead of cloning.
///
/// `query_lower` must already be lowercased (empty string when no query).
fn enrich_folder(folder: &FolderDto, query_lower: &str) -> SearchFolderResultDto {
fn enrich_folder(folder: FolderDto, query_lower: &str) -> SearchFolderResultDto {
let relevance = if query_lower.is_empty() {
50
} else {
@@ -237,10 +298,10 @@ impl SearchService {
};
SearchFolderResultDto {
id: folder.id.clone(),
name: folder.name.clone(),
path: folder.path.clone(),
parent_id: folder.parent_id.clone(),
id: folder.id,
name: folder.name,
path: folder.path,
parent_id: folder.parent_id,
drive_id: folder.drive_id,
created_at: folder.created_at,
modified_at: folder.modified_at,
@@ -287,7 +348,7 @@ impl SearchService {
// grants are honoured inline by `storage.caller_group_ids` on
// the SQL side, so no Rust-side subject expansion here.
let accessible_drives: Vec<Uuid> = match drive_repo.list_readable_by(user_id).await {
Ok(drives) => drives.into_iter().map(|d| d.drive.id).collect(),
Ok(drives) => drives.iter().map(|d| d.drive.id).collect(),
Err(e) => {
tracing::warn!("Content-index: drive lookup failed — degrading to empty: {e}");
return Vec::new();
@@ -409,9 +470,10 @@ impl SearchService {
let Some(hit) = by_id.get(dto.id.as_str()) else {
continue;
};
let mut enriched = Self::enrich_file(&dto, "");
enriched.relevance_score = content_relevance(hit.score, max_score);
enriched.snippet = hit.snippet.clone();
let (score, snippet) = (hit.score, hit.snippet.clone());
let mut enriched = Self::enrich_file(dto, "");
enriched.relevance_score = content_relevance(score, max_score);
enriched.snippet = snippet;
enriched.match_source = Some("content".to_string());
enriched_files.push(enriched);
added += 1;
@@ -425,20 +487,28 @@ impl SearchService {
/// Quick suggestions search — returns up to `limit` name suggestions
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
/// so only a handful of rows cross the DB→app boundary.
pub async fn suggest(
///
/// `caller_id` scopes the underlying repo queries to drives the caller
/// can Read. Without it (the pre-fix shape) any authenticated user —
/// including external magic-link recipients — could autocomplete both
/// names and full paths across every tenant on the instance (AuthZ
/// audit finding #1, 2026-07-12). Named `_with_perms` per the
/// AGENTS.md AuthZ convention.
pub async fn suggest_with_perms(
&self,
query: &str,
folder_id: Option<&str>,
limit: usize,
caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
let start = Instant::now();
// Ask SQL for at most `limit` best-matching files and folders
let (files, folders) = tokio::join!(
self.file_repository
.suggest_files_by_name(folder_id, query, limit),
.suggest_files_by_name(folder_id, query, limit, caller_id),
self.folder_repository
.suggest_folders_by_name(folder_id, query, limit),
.suggest_folders_by_name(folder_id, query, limit, caller_id),
);
let files = files?;
let folders = folders?;
@@ -449,30 +519,36 @@ impl SearchService {
// Pre-compute once — avoids N heap allocations inside the loops.
let query_lower = query.to_lowercase();
for file in &files {
let file_dto = FileDto::from(file.clone());
// Consume the entities: the old loop deep-cloned every File into
// the DTO conversion and then cloned name/id/path AGAIN into the
// suggestion — 3 field clones + a full entity clone per row on
// an every-keystroke path.
for file in files {
let file_dto = FileDto::from(file);
let score = compute_relevance(&file_dto.name, &query_lower);
suggestions.push(SearchSuggestionItem {
name: file_dto.name.clone(),
name: file_dto.name,
item_type: "file".to_string(),
id: file_dto.id.clone(),
path: file_dto.path.clone(),
icon_class: get_icon_class(&file_dto.name, &file_dto.mime_type),
icon_special_class: get_icon_special_class(&file_dto.name, &file_dto.mime_type),
id: file_dto.id,
path: file_dto.path,
// Interned in `FileDto::from` — reuse instead of re-running
// the display classifiers per keystroke suggestion.
icon_class: file_dto.icon_class,
icon_special_class: file_dto.icon_special_class,
relevance_score: score,
});
}
for folder in &folders {
let folder_dto = FolderDto::from(folder.clone());
for folder in folders {
let folder_dto = FolderDto::from(folder);
let score = compute_relevance(&folder_dto.name, &query_lower);
suggestions.push(SearchSuggestionItem {
name: folder_dto.name.clone(),
name: folder_dto.name,
item_type: "folder".to_string(),
id: folder_dto.id.clone(),
path: folder_dto.path.clone(),
icon_class: "fas fa-folder".to_string(),
icon_special_class: "folder-icon".to_string(),
id: folder_dto.id,
path: folder_dto.path,
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
relevance_score: score,
});
}
@@ -489,6 +565,22 @@ impl SearchService {
}
}
// ─── Bench-only public wrappers (feature = "bench") ──────────────────────
#[cfg(feature = "bench")]
impl SearchService {
/// Public wrapper over the private `enrich_file` so
/// `examples/bench_search_enrich.rs` can measure it.
pub fn enrich_file_for_bench(file: FileDto, query_lower: &str) -> SearchFileResultDto {
Self::enrich_file(file, query_lower)
}
/// Public wrapper over the private `enrich_folder` for the same bench.
pub fn enrich_folder_for_bench(folder: FolderDto, query_lower: &str) -> SearchFolderResultDto {
Self::enrich_folder(folder, query_lower)
}
}
// ─── SearchUseCase trait implementation ──────────────────────────────────
impl SearchUseCase for SearchService {
@@ -541,11 +633,11 @@ impl SearchUseCase for SearchService {
.search_files_paginated(criteria.folder_id.as_deref(), &criteria, user_id)
.await?;
// Convert to DTOs and enrich with metadata
let file_dtos: Vec<FileDto> = files.into_iter().map(FileDto::from).collect();
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
// Convert to DTOs and enrich with metadata — one fused
// pass, no intermediate Vec<FileDto> materialization.
let mut enriched_files: Vec<SearchFileResultDto> = files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.collect();
// Get folders for this folder (non-recursive, filtered in SQL)
@@ -559,13 +651,10 @@ impl SearchUseCase for SearchService {
)
.await?;
let filtered_folders: Vec<FolderDto> =
folders.into_iter().map(FolderDto::from).collect();
// For folders, apply sorting and pagination in memory (usually fewer folders)
let mut enriched_folders: Vec<SearchFolderResultDto> = filtered_folders
.iter()
.map(|f| Self::enrich_folder(f, &query_lower))
let mut enriched_folders: Vec<SearchFolderResultDto> = folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.collect();
// Sort folders (cached_key avoids O(N log N) temporary String allocations)
@@ -646,17 +735,15 @@ impl SearchUseCase for SearchService {
.await?;
// ── Convert to DTOs and enrich with server-computed metadata ──
let file_dtos: Vec<FileDto> = found_files.into_iter().map(FileDto::from).collect();
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
// Fused single pass: no intermediate DTO Vec materialization.
let mut enriched_files: Vec<SearchFileResultDto> = found_files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.collect();
let folder_dtos: Vec<FolderDto> =
found_folders.into_iter().map(FolderDto::from).collect();
let mut enriched_folders: Vec<SearchFolderResultDto> = folder_dtos
.iter()
.map(|f| Self::enrich_folder(f, &query_lower))
let mut enriched_folders: Vec<SearchFolderResultDto> = found_folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.collect();
// ── Sort folders (cached_key avoids O(N log N) temporary String allocations) ──
@@ -724,14 +811,20 @@ impl SearchUseCase for SearchService {
})
}
/// Returns quick suggestions for autocomplete.
/// Returns quick suggestions for autocomplete. Delegates to the
/// inherent `suggest_with_perms` — the trait method is preserved as
/// the polymorphic entry point (e.g. for `StubSearchUseCase` in
/// tests); production callers can equivalently call the inherent
/// method directly.
async fn suggest(
&self,
query: &str,
folder_id: Option<&str>,
limit: usize,
caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
self.suggest(query, folder_id, limit).await
self.suggest_with_perms(query, folder_id, limit, caller_id)
.await
}
/// Clears the search results cache.
@@ -763,6 +856,7 @@ impl SearchService {
_query: &str,
_folder_id: Option<&str>,
_limit: usize,
_caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
Ok(SearchSuggestionsDto {
suggestions: Vec::new(),
@@ -800,21 +894,103 @@ mod tests {
name: name.to_string(),
path: format!("/{name}"),
size,
mime_type: "text/plain".to_string(),
mime_type: "text/plain".into(),
folder_id: None,
created_at: 0,
modified_at,
relevance_score: relevance,
size_formatted: String::new(),
icon_class: String::new(),
icon_special_class: String::new(),
category: String::new(),
icon_class: "".into(),
icon_special_class: "".into(),
category: "".into(),
blob_hash: String::new(),
snippet: None,
match_source: None,
}
}
#[test]
fn entry_weight_counts_every_owned_string_plus_overheads() {
// Empty page: entry overhead + sort_by ("relevance" = 9 bytes).
let empty = Arc::new(SearchResultsDto::empty());
let base = search_results_entry_weight(&0, &empty) as usize;
assert_eq!(base, 256 + 9);
// One file row: base + row overhead + its owned string bytes
// (id 7 + name 7 + path 8 + mime 10; the rest are empty/None).
let one_file = Arc::new(SearchResultsDto::new(
vec![dto("abc.txt", 50, 10, 1)],
Vec::new(),
100,
0,
Some(1),
0,
"relevance".to_string(),
));
let w = search_results_entry_weight(&0, &one_file) as usize;
assert_eq!(w, base + 200 + 7 + 7 + 8 + 10);
// Folder rows weigh too (id 2 + name 4 + path 5 + parent 6 = 17).
let one_folder = Arc::new(SearchResultsDto::new(
Vec::new(),
vec![SearchFolderResultDto {
id: "f1".to_string(),
name: "docs".to_string(),
path: "/docs".to_string(),
parent_id: Some("parent".to_string()),
drive_id: Uuid::nil(),
created_at: 0,
modified_at: 0,
is_root: false,
relevance_score: 50,
}],
100,
0,
Some(1),
0,
"relevance".to_string(),
));
let w = search_results_entry_weight(&0, &one_folder) as usize;
assert_eq!(w, base + 200 + 2 + 4 + 5 + 6);
}
#[tokio::test]
async fn cache_evicts_down_to_the_byte_budget() {
// Budget fits ~2 of these entries; inserting 20 must never let the
// weighted size settle above the budget.
let entry = |i: usize| {
Arc::new(SearchResultsDto::new(
(0..50)
.map(|r| dto(&format!("file_{i}_{r}_{}", "x".repeat(100)), 50, 1, 1))
.collect(),
Vec::new(),
50,
0,
Some(50),
0,
"relevance".to_string(),
))
};
let per_entry = search_results_entry_weight(&0, &entry(0)) as u64;
let budget = per_entry * 2 + per_entry / 2;
let cache = build_search_results_cache(300, budget);
for i in 0..20u64 {
cache.insert(i, entry(i as usize)).await;
}
cache.run_pending_tasks().await;
let retained: u64 = cache
.iter()
.map(|(k, v)| search_results_entry_weight(&k, &v) as u64)
.sum();
assert!(
retained <= budget,
"retained {retained} B exceeds budget {budget} B"
);
assert!(cache.entry_count() <= 2);
}
#[test]
fn merged_files_resort_by_relevance_and_by_column() {
let mut files = vec![
@@ -19,6 +19,13 @@ use uuid::Uuid;
pub struct StorageUsageService {
pool: Arc<PgPool>,
user_repository: Arc<UserPgRepository>,
/// Optional so DI can wire it lazily and older test constructors
/// keep compiling. When `Some`, every write path that mutates
/// `drives.used_bytes` or `users.storage_used_bytes` invalidates
/// the drive lookup caches so `GET /api/drives` reflects the new
/// usage on the next call (see the invalidation calls in the
/// delta / sweep methods below).
drive_repo: Option<Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>>,
}
impl StorageUsageService {
@@ -27,6 +34,44 @@ impl StorageUsageService {
Self {
pool,
user_repository,
drive_repo: None,
}
}
/// Wires the drive repository used for cache-invalidation-on-write.
/// Production DI calls this in `common::di`; tests without a real
/// drive repo leave it `None` and the invalidation calls no-op.
pub fn with_drive_repo(
mut self,
drive_repo: Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>,
) -> Self {
self.drive_repo = Some(drive_repo);
self
}
/// Drop the per-caller readable-drive listing cache and the
/// per-user default-drive cache so `GET /api/drives` and the
/// WebDAV / NextCloud / WOPI drive-lookup paths re-read fresh
/// values.
///
/// **Called only from the reconciliation sweep**, not from the
/// hot-path `add_drive_storage_usage_delta*` methods. The design
/// (Ed's call, 2026-07-17): keep the cache useful under active
/// upload load — per-mutation invalidation would nuke the cache
/// on every file upload, defeating the point. `used_bytes` on
/// `GET /api/drives` therefore lags by up to the cache TTL (30 s),
/// which matches the sibling caches' accepted UX phantom for
/// drive-name staleness. Tests / operators that need immediate
/// freshness call `POST /api/admin/internal/trigger-sweep`, which
/// runs `update_all_drives_storage_usage` → this method.
///
/// Security posture unaffected: `check_drive_quota` reads
/// directly from SQL, bypassing the cache entirely, so quota
/// enforcement is honest regardless of listing staleness.
fn invalidate_drive_lookup_caches(&self) {
if let Some(repo) = &self.drive_repo {
repo.invalidate_readable_all();
repo.invalidate_default_drive_all();
}
}
@@ -209,6 +254,9 @@ impl StorageUsageService {
.execute(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("StorageUsage", format!("drive delta: {e}")))?;
// Deliberate no-invalidate here — see the class doc on
// `invalidate_drive_lookup_caches`. Delta writes lag the
// cache by up to the TTL; the sweep is the escape hatch.
Ok(())
}
@@ -285,6 +333,7 @@ impl StorageUsageService {
.map_err(|e| {
DomainError::internal_error("StorageUsage", format!("drive delta by folder: {e}"))
})?;
// See `add_drive_storage_usage_delta` — deliberate no-invalidate.
Ok(())
}
@@ -595,6 +644,20 @@ impl StorageUsagePort for StorageUsageService {
"Drive storage-usage reconciliation corrected {} drive(s)",
result.rows_affected()
);
// Unconditional invalidation — do NOT gate on
// `rows_affected() > 0`. When a fire-and-forget delta has
// already made SQL correct BEFORE the sweep runs, the sweep
// touches zero rows but the cache may still hold the
// pre-delta value from an earlier `GET /api/drives`. Gating
// means the cache stays stale in exactly the case
// `trigger-sweep` is called to fix. The invalidation cost is
// small (moka `invalidate_all` on both caches); the
// correctness guarantee matters. Regression avoidance:
// drive_quota.hurl Step 6 exercises this race — 2nd upload's
// delta lands during the 200 ms delay, sweep sees SQL is
// already right → zero rows → without unconditional
// invalidation, cache stays at the previous step's value.
self.invalidate_drive_lookup_caches();
Ok(())
}
@@ -613,6 +676,7 @@ impl Clone for StorageUsageService {
Self {
pool: Arc::clone(&self.pool),
user_repository: Arc::clone(&self.user_repository),
drive_repo: self.drive_repo.clone(),
}
}
}
@@ -44,6 +44,11 @@ pub struct SubjectGroupService {
/// 30 s TTL. Without this, fresh group-mediated drive grants
/// don't appear in `/api/drives` for up to 30 s after `add_member`.
engine: Arc<crate::infrastructure::services::pg_acl_engine::PgAclEngine>,
/// Same freshness contract for the drive repository's per-user
/// readable-drives cache: a membership change on a group that holds
/// drive grants changes every affected user's visible drive list,
/// so the cached lists drop alongside `user_groups_cache`.
drive_repo: Arc<crate::infrastructure::repositories::pg::DrivePgRepository>,
}
impl SubjectGroupService {
@@ -52,12 +57,14 @@ impl SubjectGroupService {
pool: Arc<PgPool>,
user_storage: Arc<UserPgRepository>,
engine: Arc<crate::infrastructure::services::pg_acl_engine::PgAclEngine>,
drive_repo: Arc<crate::infrastructure::repositories::pg::DrivePgRepository>,
) -> Self {
Self {
repo,
pool,
user_storage,
engine,
drive_repo,
}
}
@@ -426,6 +433,7 @@ impl SubjectGroupService {
// call for up to 30 s.
for uid in self.invalidation_targets(member).await? {
self.engine.invalidate_user_groups_cache(uid).await;
self.drive_repo.invalidate_readable_for_user(uid).await;
}
tracing::info!(
@@ -525,6 +533,7 @@ impl SubjectGroupService {
// for up to 30 s, surfacing grants they no longer have.
for uid in self.invalidation_targets(member).await? {
self.engine.invalidate_user_groups_cache(uid).await;
self.drive_repo.invalidate_readable_for_user(uid).await;
}
tracing::info!(
@@ -634,7 +643,9 @@ mod integration_tests {
// future test starts exercising real authz lookups.
let engine =
Arc::new(crate::infrastructure::services::pg_acl_engine::PgAclEngine::new_stub());
SubjectGroupService::new(repo, pool, user_storage, engine)
let drive_repo =
Arc::new(crate::infrastructure::repositories::pg::DrivePgRepository::new(pool.clone()));
SubjectGroupService::new(repo, pool, user_storage, engine, drive_repo)
}
async fn first_admin(pool: &sqlx::PgPool) -> Uuid {
+20 -17
View File
@@ -662,22 +662,25 @@ impl TrashUseCase for TrashService {
async fn empty_trash_for_drive(&self, user_id: Uuid, drive_id: Uuid) -> Result<()> {
// Per-drive trash empty — the Drive group-by on `/trash` exposes
// this as a per-row affordance so multi-drive owners can clear
// one drive without touching the others. Refuses with
// `NotFound` (anti-enum) when the caller lacks Delete on the
// named drive — same shape as the user-facing drive listing
// would emit for an unknown id.
let allowed = self.drives_with_delete_for(user_id).await?;
if !allowed.contains(&drive_id) {
tracing::info!(
target: "audit",
event = "trash.empty_drive_rejected",
reason = "no_delete_on_drive",
user_id = %user_id,
drive_id = %drive_id,
"👮🏻‍♂️ refused per-drive empty — caller lacks Delete on this drive",
);
return Err(DomainError::not_found("Drive", drive_id.to_string()));
}
// one drive without touching the others.
//
// Route through `authz.require(Delete, Drive)` so the denial
// shape stays consistent with every other write verb: 403 when
// the caller has Read on the drive (viewer/editor holding no
// Delete), 404 when they don't (anti-enum). Before 2026-07-16
// this method rolled its own `drives_with_delete_for` check +
// hardcoded `NotFound` — that predated the graduated-denial
// engine change and returned 404 unconditionally even for a
// Viewer who could see the drive in `/api/drives`. The engine
// now emits `authz.denied` with `visibility="visible"|"hidden"`
// and the standard mapping renders it as 403 or 404.
self.authz
.require(
Subject::User(user_id),
Permission::Delete,
Resource::Drive(drive_id),
)
.await?;
info!("Emptying trash for drive {} (user {})", drive_id, user_id);
self.clear_trash_in(&[drive_id], user_id).await
}
@@ -801,7 +804,7 @@ impl TrashService {
// role_grants on resource_type='drive', including group-mediated
// grants). Empty set → empty page without a SQL round-trip.
let drive_ids: Vec<Uuid> = match self.drive_repo.list_readable_by(user_id).await {
Ok(drives) => drives.into_iter().map(|d| d.drive.id).collect(),
Ok(drives) => drives.iter().map(|d| d.drive.id).collect(),
Err(e) => {
return Err(DomainError::internal_error(
"Trash",
+42
View File
@@ -310,6 +310,10 @@ pub struct AzureStorageConfig {
pub container: String,
/// Optional SAS token (alternative to account key).
pub sas_token: Option<String>,
/// Optional custom endpoint (Azurite emulator, private deployments,
/// benches). `None` = the public cloud URL derived from the account
/// name. Mirrors S3's `endpoint_url`.
pub endpoint_url: Option<String>,
}
/// LRU local disk cache configuration for remote blob backends.
@@ -1250,6 +1254,33 @@ impl Default for ContentSearchConfig {
}
}
/// Search-results cache configuration — the per-user results-page cache
/// inside `SearchService`, not the Tantivy content index above.
///
/// The cache is **byte-bounded**: each entry is weighed by the approximate
/// heap size of its result page (see `search_results_entry_weight`) and moka
/// evicts once the summed weight exceeds `max_bytes` — the same byte-budget
/// pattern the file-content cache and the dedup manifest cache use. This
/// replaced an entry-count capacity: with cache keys spanning
/// user × query × offset × limit and up to 500 enriched rows per page, an
/// entry count said nothing about resident memory (1000 entries could pin
/// ~300 MB for the TTL). No entry-count knob is kept — bytes are the only
/// dimension that matters here.
#[derive(Debug, Clone)]
pub struct SearchCacheConfig {
/// Byte budget for cached search-result pages. Default: 32 MiB.
/// Env: `OXICLOUD_SEARCH_CACHE_MAX_BYTES`.
pub max_bytes: u64,
}
impl Default for SearchCacheConfig {
fn default() -> Self {
Self {
max_bytes: 32 * 1024 * 1024,
}
}
}
/// WASM plugin runtime configuration (M0 walking skeleton).
///
/// The runtime is doubly gated: it is only compiled when the `plugins` cargo
@@ -1375,6 +1406,8 @@ pub struct AppConfig {
pub i18n: I18nConfig,
/// Content-search configuration (embedded full-text index)
pub content_search: ContentSearchConfig,
/// Search-results cache configuration (byte-bounded moka cache)
pub search_cache: SearchCacheConfig,
/// WASM plugin runtime configuration
pub plugins: PluginConfig,
/// Face-recognition (People) model configuration
@@ -1431,6 +1464,7 @@ impl Default for AppConfig {
magic_link: MagicLinkConfig::default(),
i18n: I18nConfig::default(),
content_search: ContentSearchConfig::default(),
search_cache: SearchCacheConfig::default(),
plugins: PluginConfig::default(),
faces: FacesConfig::default(),
}
@@ -1934,6 +1968,13 @@ impl AppConfig {
config.content_search.max_text_bytes = val;
}
// Search-results cache (byte-bounded)
if let Ok(v) = env::var("OXICLOUD_SEARCH_CACHE_MAX_BYTES").map(|v| v.parse::<u64>())
&& let Ok(val) = v
{
config.search_cache.max_bytes = val;
}
// WASM plugin runtime
if let Ok(v) = env::var("OXICLOUD_ENABLE_PLUGINS").map(|v| v.parse::<bool>())
&& let Ok(val) = v
@@ -2103,6 +2144,7 @@ impl AppConfig {
account_key: env::var("OXICLOUD_AZURE_ACCOUNT_KEY").unwrap_or_default(),
container,
sas_token: env::var("OXICLOUD_AZURE_SAS_TOKEN").ok(),
endpoint_url: env::var("OXICLOUD_AZURE_ENDPOINT_URL").ok(),
});
}
+21 -3
View File
@@ -656,8 +656,12 @@ impl AppServiceFactory {
content_index_port,
Some(authz.clone()),
Some(drive_repo.clone()),
300, // Cache TTL in seconds (5 minutes)
1000, // Maximum cache entries
300, // Cache TTL in seconds (5 minutes)
// Byte budget for cached result pages (weigher-bounded, 32 MiB
// default; env OXICLOUD_SEARCH_CACHE_MAX_BYTES). Replaces the old
// entry-count capacity, which let 500-row pages keyed by
// user×query×offset×limit pin hundreds of MB for the TTL.
self.config.search_cache.max_bytes,
)));
tracing::info!("Application services initialized");
@@ -1052,14 +1056,26 @@ impl AppServiceFactory {
_repos: &RepositoryServices,
db_pool: &Arc<PgPool>,
maintenance_pool: &Arc<PgPool>,
drive_repo: Arc<crate::infrastructure::repositories::pg::DrivePgRepository>,
) -> Arc<StorageUsageService> {
let user_repository = Arc::new(
crate::infrastructure::repositories::pg::UserPgRepository::new(db_pool.clone()),
);
// The `drive_repo` passed in is the SAME instance held on
// `AppState`, so its `readable_cache` / `default_drive_cache`
// are the caches the request path reads from. A separately
// constructed `DrivePgRepository` would have its OWN caches
// and invalidation would be a no-op observed by nobody —
// this is the trap that regressed the used_bytes freshness
// after perf commit `12dc648c`.
let service = Arc::new(
crate::application::services::storage_usage_service::StorageUsageService::new(
maintenance_pool.clone(),
user_repository,
)
.with_drive_repo(
drive_repo
as Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>,
),
);
// Keep cached storage usage fresh off the request path: GET /api/auth/me
@@ -1246,7 +1262,8 @@ impl AppServiceFactory {
// 3c. Storage usage / quota service (needed by the instant-upload
// path inside the application services, and re-exposed on AppState
// for the handler-side quota checks of the byte-upload paths).
let storage_usage = self.create_storage_usage_service(&repos, &pool, &maintenance_pool);
let storage_usage =
self.create_storage_usage_service(&repos, &pool, &maintenance_pool, drive_repo.clone());
// 3d. Content index (embedded Tantivy) — opened before application
// services so SearchService can hold the query port; the feeding
@@ -1678,6 +1695,7 @@ impl AppServiceFactory {
),
),
authorization.clone(),
drive_repo.clone(),
),
)),
email_sender: None, // populated below
+288
View File
@@ -0,0 +1,288 @@
//! Heap-free fixed-layout formatters for the hot XML/HTTP emit paths.
//!
//! PROPFIND writes two formatted dates, a size and a quoted etag for
//! EVERY row of every listing; `to_rfc3339()` / `to_rfc2822()` run
//! chrono's format-spec interpreter and allocate a `String` each, and
//! `u64::to_string()` allocates another. These helpers render the same
//! bytes into a caller-provided stack buffer: zero heap traffic, no
//! interpreter.
//!
//! Byte-identity with chrono (for whole-second in-range UTC datetimes)
//! is asserted by the unit tests below and by the equivalence gate in
//! `examples/bench_propfind_xml.rs`. Out-of-range seconds (negative or
//! year > 9999, where the fixed-width layout no longer applies) return
//! `None` — callers keep the old chrono path as fallback, so exotic
//! values change nothing observable.
/// Seconds range rendering to a fixed-width 4-digit year: 1970-01-01
/// through 9999-12-31 23:59:59 UTC.
const MAX_4DIGIT_YEAR_SECS: i64 = 253_402_300_799;
const MONTHS: [&[u8; 3]; 12] = [
b"Jan", b"Feb", b"Mar", b"Apr", b"May", b"Jun", b"Jul", b"Aug", b"Sep", b"Oct", b"Nov", b"Dec",
];
const WEEKDAYS: [&[u8; 3]; 7] = [b"Thu", b"Fri", b"Sat", b"Sun", b"Mon", b"Tue", b"Wed"];
/// Civil date from days since 1970-01-01 (Howard Hinnant's algorithm).
fn civil_from_days(z: i64) -> (i64, u32, u32) {
let z = z + 719_468;
let era = z.div_euclid(146_097);
let doe = z.rem_euclid(146_097); // day-of-era [0, 146096]
let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; // [0, 399]
let y = yoe + era * 400;
let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); // [0, 365]
let mp = (5 * doy + 2) / 153; // [0, 11]
let d = (doy - (153 * mp + 2) / 5 + 1) as u32; // [1, 31]
let m = if mp < 10 { mp + 3 } else { mp - 9 } as u32; // [1, 12]
(if m <= 2 { y + 1 } else { y }, m, d)
}
#[inline]
fn push2(out: &mut [u8], pos: usize, v: u32) {
out[pos] = b'0' + (v / 10) as u8;
out[pos + 1] = b'0' + (v % 10) as u8;
}
#[inline]
fn push4(out: &mut [u8], pos: usize, v: i64) {
out[pos] = b'0' + (v / 1000 % 10) as u8;
out[pos + 1] = b'0' + (v / 100 % 10) as u8;
out[pos + 2] = b'0' + (v / 10 % 10) as u8;
out[pos + 3] = b'0' + (v % 10) as u8;
}
/// Split epoch seconds into (days, y, m, d, hh, mm, ss).
#[inline]
fn split(secs: i64) -> (i64, i64, u32, u32, u32, u32, u32) {
let days = secs.div_euclid(86_400);
let sod = secs.rem_euclid(86_400);
let (y, m, d) = civil_from_days(days);
(
days,
y,
m,
d,
(sod / 3600) as u32,
(sod / 60 % 60) as u32,
(sod % 60) as u32,
)
}
/// `chrono::DateTime<Utc>::to_rfc3339()` for a whole-second timestamp:
/// `2026-07-17T11:47:14+00:00` (25 bytes) written into `buf`.
///
/// Returns `None` when `secs` is outside the fixed-width range —
/// callers fall back to chrono.
pub fn rfc3339_utc(buf: &mut [u8; 25], secs: i64) -> Option<&str> {
if !(0..=MAX_4DIGIT_YEAR_SECS).contains(&secs) {
return None;
}
let (_days, y, m, d, hh, mm, ss) = split(secs);
push4(buf, 0, y);
buf[4] = b'-';
push2(buf, 5, m);
buf[7] = b'-';
push2(buf, 8, d);
buf[10] = b'T';
push2(buf, 11, hh);
buf[13] = b':';
push2(buf, 14, mm);
buf[16] = b':';
push2(buf, 17, ss);
buf[19..25].copy_from_slice(b"+00:00");
// SAFETY-free: every byte written above is ASCII.
Some(std::str::from_utf8(&buf[..]).expect("ascii"))
}
/// `chrono::DateTime<Utc>::to_rfc2822()` for a whole-second timestamp:
/// `Fri, 17 Jul 2026 11:47:14 +0000` written into `buf`.
///
/// chrono does NOT zero-pad the day (`Thu, 1 Jan 1970 …`), so the
/// rendered length is 30 or 31 bytes — the round-4 PROPFIND equivalence
/// gate caught an early padded version of this function; the sweep test
/// below pins parity byte-for-byte across 60 years.
pub fn rfc2822_utc(buf: &mut [u8; 31], secs: i64) -> Option<&str> {
if !(0..=MAX_4DIGIT_YEAR_SECS).contains(&secs) {
return None;
}
let (days, y, m, d, hh, mm, ss) = split(secs);
let weekday = WEEKDAYS[days.rem_euclid(7) as usize];
buf[0..3].copy_from_slice(weekday);
buf[3] = b',';
buf[4] = b' ';
let mut p = 5;
if d >= 10 {
buf[p] = b'0' + (d / 10) as u8;
p += 1;
}
buf[p] = b'0' + (d % 10) as u8;
p += 1;
buf[p] = b' ';
p += 1;
buf[p..p + 3].copy_from_slice(MONTHS[(m - 1) as usize]);
p += 3;
buf[p] = b' ';
p += 1;
push4(buf, p, y);
p += 4;
buf[p] = b' ';
p += 1;
push2(buf, p, hh);
p += 2;
buf[p] = b':';
p += 1;
push2(buf, p, mm);
p += 2;
buf[p] = b':';
p += 1;
push2(buf, p, ss);
p += 2;
buf[p..p + 6].copy_from_slice(b" +0000");
p += 6;
Some(std::str::from_utf8(&buf[..p]).expect("ascii"))
}
/// `u64::to_string()` without the heap `String`: renders into `buf`,
/// returns the populated tail slice.
pub fn u64_str(buf: &mut [u8; 20], mut v: u64) -> &str {
let mut pos = buf.len();
loop {
pos -= 1;
buf[pos] = b'0' + (v % 10) as u8;
v /= 10;
if v == 0 {
break;
}
}
std::str::from_utf8(&buf[pos..]).expect("ascii")
}
/// `i64::to_string()` without the heap `String` (quota bytes are `i64`).
pub fn i64_str(buf: &mut [u8; 21], v: i64) -> &str {
let mut u = [0u8; 20];
let digits = u64_str(&mut u, v.unsigned_abs());
let neg = v < 0;
let start = 21 - digits.len() - usize::from(neg);
if neg {
buf[start] = b'-';
}
buf[start + usize::from(neg)..].copy_from_slice(digits.as_bytes());
std::str::from_utf8(&buf[start..]).expect("ascii")
}
/// Lower-case hex of `bytes` into one preallocated `String`.
///
/// Replaces the `.map(|b| format!("{b:02x}")).collect()` shape, which heap-
/// allocates a 2-byte `String` per digest byte (16 for MD5, 32 for SHA-256)
/// before collect concatenates them.
pub fn hex_lower(bytes: &[u8]) -> String {
const HEX: &[u8; 16] = b"0123456789abcdef";
let mut out = String::with_capacity(bytes.len() * 2);
for &b in bytes {
out.push(HEX[(b >> 4) as usize] as char);
out.push(HEX[(b & 0x0f) as usize] as char);
}
out
}
#[cfg(test)]
mod tests {
use super::*;
use chrono::{TimeZone, Utc};
/// `hex_lower` must match the `format!("{b:02x}")`-per-byte shape it
/// replaced, byte for byte.
#[test]
fn hex_lower_matches_format() {
let cases: [&[u8]; 5] = [
&[],
&[0x00],
&[0xff, 0x00, 0xab],
&(0u8..=255).collect::<Vec<u8>>(),
b"The quick brown fox",
];
for bytes in cases {
let reference: String = bytes.iter().map(|b| format!("{b:02x}")).collect();
assert_eq!(hex_lower(bytes), reference);
}
}
/// Edge-heavy corpus: epoch, single-digit day (padding!), leap day,
/// end-of-year, DST-irrelevant midsummer, far future, max in-range.
const CASES: [i64; 12] = [
0,
1,
86_399,
86_400,
951_782_400, // 2000-02-29 (leap)
1_120_176_000, // 2005-07-01 (day < 10 → chrono pads)
1_752_753_434,
2_147_483_647,
4_102_444_799, // 2099-12-31 23:59:59
7_258_118_400,
250_000_000_000,
MAX_4DIGIT_YEAR_SECS,
];
#[test]
fn rfc3339_matches_chrono() {
for &secs in &CASES {
let dt = Utc.timestamp_opt(secs, 0).unwrap();
let mut buf = [0u8; 25];
assert_eq!(
rfc3339_utc(&mut buf, secs).expect("in range"),
dt.to_rfc3339(),
"secs={secs}"
);
}
}
#[test]
fn rfc2822_matches_chrono() {
for &secs in &CASES {
let dt = Utc.timestamp_opt(secs, 0).unwrap();
let mut buf = [0u8; 31];
assert_eq!(
rfc2822_utc(&mut buf, secs).expect("in range"),
dt.to_rfc2822(),
"secs={secs}"
);
}
}
#[test]
fn out_of_range_falls_back() {
let mut b3 = [0u8; 25];
let mut b2 = [0u8; 31];
assert!(rfc3339_utc(&mut b3, -1).is_none());
assert!(rfc2822_utc(&mut b2, -1).is_none());
assert!(rfc3339_utc(&mut b3, MAX_4DIGIT_YEAR_SECS + 1).is_none());
}
#[test]
fn ints_match_std() {
let mut b = [0u8; 20];
for v in [0u64, 1, 9, 10, 42, 1024, u64::MAX] {
assert_eq!(u64_str(&mut b, v), v.to_string());
}
let mut b = [0u8; 21];
for v in [0i64, -1, 42, -1024, i64::MIN, i64::MAX] {
assert_eq!(i64_str(&mut b, v), v.to_string());
}
}
/// Exhaustive-ish sweep: every 6h13m across 60 years — catches any
/// weekday / month-boundary drift against chrono.
#[test]
fn sweep_matches_chrono() {
let mut secs: i64 = 0;
while secs < 60 * 366 * 86_400 {
let dt = Utc.timestamp_opt(secs, 0).unwrap();
let mut b3 = [0u8; 25];
let mut b2 = [0u8; 31];
assert_eq!(rfc3339_utc(&mut b3, secs).unwrap(), dt.to_rfc3339());
assert_eq!(rfc2822_utc(&mut b2, secs).unwrap(), dt.to_rfc2822());
secs += 22_380; // 6h13m — walks through all times of day + weekdays
}
}
}
+1
View File
@@ -1,6 +1,7 @@
pub mod config;
pub mod di;
pub mod errors;
pub mod fmt;
pub mod locale;
pub mod mime_detect;
pub mod runtime;
+12
View File
@@ -511,6 +511,17 @@ impl FileUploadUseCase for StubFileUploadUseCase {
) -> Result<FileDto, DomainError> {
Ok(FileDto::default())
}
async fn upload_file_streaming_with_perms(
&self,
_name: String,
_folder_id: Option<String>,
_content_type: String,
_blob: StoredBlob,
_caller_id: Uuid,
) -> Result<FileDto, DomainError> {
Ok(FileDto::default())
}
}
// ---------------------------------------------------------------------------
@@ -725,6 +736,7 @@ impl SearchUseCase for StubSearchUseCase {
_query: &str,
_folder_id: Option<&str>,
_limit: usize,
_caller_id: Uuid,
) -> Result<SearchSuggestionsDto, DomainError> {
Ok(SearchSuggestionsDto {
suggestions: Vec::new(),
+123 -58
View File
@@ -250,25 +250,36 @@ impl CalendarEvent {
* @return Result containing the new CalendarEvent or a domain error
*/
pub fn from_ical(calendar_id: Uuid, ical_data: String) -> Result<Self> {
// This implementation would require a proper iCalendar parser
// For brevity, we're using a simplified version here
// Parse the body ONCE and read every property from the parsed
// component. The previous shape funnelled each of the 8 property
// lookups below through `extract_ical_property[_with_params]`,
// which re-ran the full `IcalParser` (line unfolding + component
// tree build) per property — 8 complete parses per VEVENT on
// every CalDAV PUT / import. A missing-or-unparseable body maps
// to the same "Missing SUMMARY" error the old first lookup
// produced, preserving error parity.
let event = Self::parse_first_vevent(&ical_data);
// Extract required fields from iCalendar data
let summary = Self::extract_ical_property(&ical_data, "SUMMARY").ok_or_else(|| {
DomainError::new(
ErrorKind::InvalidInput,
"CalendarEvent",
"Missing SUMMARY in iCalendar data",
)
})?;
// Extract required fields from the parsed component
let summary = event
.as_ref()
.and_then(|e| Self::prop_value(e, "SUMMARY"))
.ok_or_else(|| {
DomainError::new(
ErrorKind::InvalidInput,
"CalendarEvent",
"Missing SUMMARY in iCalendar data",
)
})?;
let event = event.expect("prop_value returned Some, so the parse succeeded");
// DTSTART / DTEND: use the params-aware extractor so we can
// detect `VALUE=DATE` (all-day) from the property parameters
// rather than scanning the raw property line. The pre-parser-
// rewrite substring scan couldn't see param-carrying lines at
// all — see #528.
let (dtstart_value, dtstart_params) =
Self::extract_ical_property_with_params(&ical_data, "DTSTART").ok_or_else(|| {
let (dtstart_value, dtstart_params) = Self::prop_with_params(&event, "DTSTART")
.ok_or_else(|| {
DomainError::new(
ErrorKind::InvalidInput,
"CalendarEvent",
@@ -277,7 +288,7 @@ impl CalendarEvent {
})?;
let (dtend_value, _dtend_params) =
Self::extract_ical_property_with_params(&ical_data, "DTEND").ok_or_else(|| {
Self::prop_with_params(&event, "DTEND").ok_or_else(|| {
DomainError::new(
ErrorKind::InvalidInput,
"CalendarEvent",
@@ -313,13 +324,13 @@ impl CalendarEvent {
})?;
// Extract optional fields
let description = Self::extract_ical_property(&ical_data, "DESCRIPTION");
let location = Self::extract_ical_property(&ical_data, "LOCATION");
let rrule = Self::extract_ical_property(&ical_data, "RRULE");
let description = Self::prop_value(&event, "DESCRIPTION");
let location = Self::prop_value(&event, "LOCATION");
let rrule = Self::prop_value(&event, "RRULE");
// Extract UID or generate a new one
let ical_uid = Self::extract_ical_property(&ical_data, "UID")
.unwrap_or_else(|| Uuid::new_v4().to_string());
let ical_uid =
Self::prop_value(&event, "UID").unwrap_or_else(|| Uuid::new_v4().to_string());
// RECURRENCE-ID (RFC 5545 §3.8.4.4). When present, this VEVENT
// is an override for a specific occurrence of a recurring
@@ -329,17 +340,16 @@ impl CalendarEvent {
// gets stored, just as a plain event (worst case a client sync
// treats it as a new master, which the DB uniqueness will
// refuse; better a persistence error than a silent split).
let recurrence_id =
match Self::extract_ical_property_with_params(&ical_data, "RECURRENCE-ID") {
Some((value, params)) => {
let is_date = params
.get("VALUE")
.map(|vs| vs.iter().any(|v| v.eq_ignore_ascii_case("DATE")))
.unwrap_or(false);
Self::parse_ical_datetime(&value, is_date).ok()
}
None => None,
};
let recurrence_id = match Self::prop_with_params(&event, "RECURRENCE-ID") {
Some((value, params)) => {
let is_date = params
.get("VALUE")
.map(|vs| vs.iter().any(|v| v.eq_ignore_ascii_case("DATE")))
.unwrap_or(false);
Self::parse_ical_datetime(&value, is_date).ok()
}
None => None,
};
let now = Utc::now();
@@ -627,18 +637,28 @@ impl CalendarEvent {
));
}
// Extract and update properties from iCalendar data
if let Some(summary) = Self::extract_ical_property(&ical_data, "SUMMARY") {
// Parse the body ONCE and update every property from the parsed
// component (same 8-parses→1 collapse as `from_ical`). An
// unparseable body behaves exactly like the old per-property
// lookups all returning `None`: optional fields clear, required
// fields keep their previous values.
let event = Self::parse_first_vevent(&ical_data);
if let Some(summary) = event.as_ref().and_then(|e| Self::prop_value(e, "SUMMARY")) {
self.summary = summary;
}
self.description = Self::extract_ical_property(&ical_data, "DESCRIPTION");
self.location = Self::extract_ical_property(&ical_data, "LOCATION");
self.description = event
.as_ref()
.and_then(|e| Self::prop_value(e, "DESCRIPTION"));
self.location = event.as_ref().and_then(|e| Self::prop_value(e, "LOCATION"));
// Extract DTSTART with parameters — needed for the all-day
// detection below AND for the DTSTART/DTEND datetime parsers
// (they need to know whether the value is a date or a datetime).
let dtstart_pair = Self::extract_ical_property_with_params(&ical_data, "DTSTART");
let dtstart_pair = event
.as_ref()
.and_then(|e| Self::prop_with_params(e, "DTSTART"));
let all_day = dtstart_pair
.as_ref()
.and_then(|(_v, params)| params.get("VALUE"))
@@ -652,15 +672,17 @@ impl CalendarEvent {
self.start_time = start_time;
}
if let Some((value, _params)) = Self::extract_ical_property_with_params(&ical_data, "DTEND")
if let Some((value, _params)) = event
.as_ref()
.and_then(|e| Self::prop_with_params(e, "DTEND"))
&& let Ok(end_time) = Self::parse_ical_datetime(&value, all_day)
{
self.end_time = end_time;
}
self.rrule = Self::extract_ical_property(&ical_data, "RRULE");
self.rrule = event.as_ref().and_then(|e| Self::prop_value(e, "RRULE"));
if let Some(uid) = Self::extract_ical_property(&ical_data, "UID") {
if let Some(uid) = event.as_ref().and_then(|e| Self::prop_value(e, "UID")) {
self.ical_uid = uid;
}
@@ -756,43 +778,74 @@ impl CalendarEvent {
* @param property_name The name of the property to extract
* @return Option containing the property value if found
*/
#[cfg(test)]
fn extract_ical_property(ical_data: &str, property_name: &str) -> Option<String> {
Self::extract_ical_property_with_params(ical_data, property_name).map(|(v, _p)| v)
Self::prop_value(&Self::parse_first_vevent(ical_data)?, property_name)
}
/// Extract a property's value AND parameter map. Same lookup rules
/// as `extract_ical_property`; the second element is a map keyed by
/// parameter name (`"VALUE"`, `"TZID"`, `"CN"`, …) whose value is
/// the list of parameter values (parameters can be multi-valued —
/// `MEMBER="mailto:a@x","mailto:b@x"` — hence the `Vec<String>`
/// per key).
///
/// Callers that only need the value should use `extract_ical_property`;
/// this variant is for DTSTART / DTEND / RECURRENCE-ID which need
/// `VALUE=DATE` detection to distinguish all-day from timed events.
/// Test-only sibling of [`Self::prop_with_params`] that parses the
/// raw body first. Production callers (`from_ical`,
/// `update_ical_data`) parse ONCE and use the by-reference helpers.
#[cfg(test)]
fn extract_ical_property_with_params(
ical_data: &str,
property_name: &str,
) -> Option<(String, std::collections::HashMap<String, Vec<String>>)> {
let event = Self::parse_first_vevent(ical_data)?;
Self::prop_with_params(&Self::parse_first_vevent(ical_data)?, property_name)
}
/// Read a property's trimmed value from an already-parsed VEVENT.
///
/// Value-only lookups skip the parameter-map build entirely; use
/// [`Self::prop_with_params`] for DTSTART / DTEND / RECURRENCE-ID
/// which need `VALUE=DATE` detection.
///
/// Returns `None` when the property is missing or its value is
/// empty after trimming — the same rules the old per-property
/// full-parse extractors applied.
fn prop_value(
event: &ical::parser::ical::component::IcalEvent,
property_name: &str,
) -> Option<String> {
let prop = event
.properties
.into_iter()
.iter()
.find(|p| p.name.eq_ignore_ascii_case(property_name))?;
let value = prop.value?;
if value.trim().is_empty() {
let trimmed = prop.value.as_deref()?.trim();
if trimmed.is_empty() {
return None;
}
Some(trimmed.to_string())
}
/// Read a property's trimmed value AND parameter map from an
/// already-parsed VEVENT. The map is keyed by parameter name
/// (`"VALUE"`, `"TZID"`, `"CN"`, …) whose value is the list of
/// parameter values (parameters can be multi-valued —
/// `MEMBER="mailto:a@x","mailto:b@x"` — hence the `Vec<String>`
/// per key).
fn prop_with_params(
event: &ical::parser::ical::component::IcalEvent,
property_name: &str,
) -> Option<(String, std::collections::HashMap<String, Vec<String>>)> {
let prop = event
.properties
.iter()
.find(|p| p.name.eq_ignore_ascii_case(property_name))?;
let trimmed = prop.value.as_deref()?.trim();
if trimmed.is_empty() {
return None;
}
let mut params: std::collections::HashMap<String, Vec<String>> =
std::collections::HashMap::new();
if let Some(param_list) = prop.params {
if let Some(param_list) = &prop.params {
for (name, values) in param_list {
// RFC 5545 property parameter names are ASCII case-insensitive.
// Normalise to UPPER so callers key on a canonical form.
params.insert(name.to_ascii_uppercase(), values);
params.insert(name.to_ascii_uppercase(), values.clone());
}
}
Some((value.trim().to_string(), params))
Some((trimmed.to_string(), params))
}
/// Parse a VCALENDAR body containing one or more VEVENT components
@@ -847,6 +900,18 @@ impl CalendarEvent {
let mut in_event = false;
let mut current = String::new();
// Allocation-free case-insensitive prefix test. `to_ascii_uppercase`
// maps ASCII bytes in place and leaves multi-byte chars untouched,
// so "first N bytes uppercased equal TAG" ⇔ "first N bytes
// ASCII-case-insensitively equal TAG"; `get(..N)` returning `None`
// (char straddling the boundary) implies the prefix can't be the
// all-ASCII tag. The old per-line `to_ascii_uppercase()` allocated
// a String for every line of every uploaded body.
fn starts_with_ci(line: &str, tag: &str) -> bool {
line.get(..tag.len())
.is_some_and(|p| p.eq_ignore_ascii_case(tag))
}
for raw_line in ical_data.split('\n') {
let line = raw_line.trim_end_matches('\r');
// Match the tag ignoring case, allowing surrounding
@@ -854,9 +919,9 @@ impl CalendarEvent {
// continuations — the raw-line scan sees those but they
// won't start with BEGIN/END so they slot through as
// in-event content, which is correct).
let upper = line.trim_start().to_ascii_uppercase();
let tag_area = line.trim_start();
if upper.starts_with("BEGIN:VEVENT") {
if starts_with_ci(tag_area, "BEGIN:VEVENT") {
in_event = true;
current.clear();
}
@@ -866,7 +931,7 @@ impl CalendarEvent {
current.push_str("\r\n");
}
if in_event && upper.starts_with("END:VEVENT") {
if in_event && starts_with_ci(tag_area, "END:VEVENT") {
blocks.push(std::mem::take(&mut current));
in_event = false;
}
+81 -13
View File
@@ -1,7 +1,7 @@
use uuid::Uuid;
use crate::domain::services::path_service::{
StoragePath, normalize_storage_name, validate_storage_name,
StoragePath, normalize_storage_name_owned, validate_storage_name,
};
// Re-export entity errors from the centralized module
@@ -122,7 +122,7 @@ impl File {
mime_type: String,
folder_id: Option<String>,
) -> FileResult<Self> {
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FileError::InvalidFileName(format!("{name}: {reason}")));
}
@@ -133,7 +133,7 @@ impl File {
.as_secs();
// Store the path string for serialization compatibility
let path_string = storage_path.to_string();
let path_string = storage_path.to_path_string();
Ok(Self {
id,
@@ -160,13 +160,13 @@ impl File {
created_at: u64,
modified_at: u64,
) -> FileResult<Self> {
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FileError::InvalidFileName(format!("{name}: {reason}")));
}
// Store the path string for serialization compatibility
let path_string = storage_path.to_string();
let path_string = storage_path.to_path_string();
Ok(Self {
id,
@@ -252,13 +252,64 @@ impl File {
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> FileResult<Self> {
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FileError::InvalidFileName(format!("{name}: {reason}")));
}
// Store the path string for serialization compatibility
let path_string = storage_path.to_string();
let path_string = storage_path.to_path_string();
Ok(Self {
id,
name,
storage_path,
path_string,
size,
mime_type,
folder_id,
created_at,
modified_at,
blob_hash,
created_by,
updated_by,
})
}
/// PG-row constructor: the per-listing-row hot path.
///
/// Builds `storage_path` **and** `path_string` in one pass from the
/// materialized folder path via
/// [`StoragePath::from_folder_and_name`], instead of the old chain
/// (`format!` temp → `from_string` split → `Display` re-join) that
/// allocated the full path three times per row. The owned `name` is
/// NFC-normalized without the always-copy of the borrowing variant
/// (DB rows are NFC by invariant, so this is a zero-alloc check).
///
/// The path is built from the raw incoming name and the name field is
/// normalized afterwards — the exact observable sequence of the old
/// `make_file_path` + constructor pair, byte-identical for every
/// input (for DB rows the two names coincide: stored names are NFC).
#[allow(clippy::too_many_arguments)]
pub fn from_materialized_row(
id: String,
name: String,
folder_path: Option<&str>,
size: u64,
mime_type: String,
folder_id: Option<String>,
created_at: u64,
modified_at: u64,
blob_hash: String,
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> FileResult<Self> {
let (storage_path, path_string) = StoragePath::from_folder_and_name(folder_path, &name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FileError::InvalidFileName(format!("{name}: {reason}")));
}
Ok(Self {
id,
@@ -352,8 +403,25 @@ impl File {
/// formula here changes it everywhere — that is the property
/// we want.
pub fn compute_etag(blob_hash: &str, modified_at: u64) -> String {
let prefix: String = blob_hash.chars().take(16).collect();
format!("{}-{}", prefix, modified_at)
use std::fmt::Write as _;
// Byte index just past the 16th char (whole string when shorter).
// `blob_hash` is lowercase hex ASCII in practice, so this is
// effectively `min(len, 16)`, but `char_indices` keeps the slice
// char-boundary-safe for exotic fixture values — byte-identical
// to the old `chars().take(16).collect::<String>()` without the
// intermediate allocation.
let end = match blob_hash.char_indices().nth(16) {
Some((i, _)) => i,
None => blob_hash.len(),
};
// Single allocation: prefix + '-' + up to 20 digits (u64::MAX).
let mut etag = String::with_capacity(end + 1 + 20);
etag.push_str(&blob_hash[..end]);
etag.push('-');
let _ = write!(etag, "{modified_at}");
etag
}
// Getters
@@ -425,7 +493,7 @@ impl File {
// Create directly without validation to avoid errors in DTO
// conversions. Still NFC-normalize so even DTO-reconstructed
// entities maintain the storage invariant.
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
Self {
id,
@@ -449,7 +517,7 @@ impl File {
/// Creates a new version of the file with updated name
pub fn with_name(mut self, new_name: String) -> FileResult<Self> {
let new_name = normalize_storage_name(&new_name);
let new_name = normalize_storage_name_owned(new_name);
if let Err(reason) = validate_storage_name(&new_name) {
return Err(FileError::InvalidFileName(format!("{new_name}: {reason}")));
}
@@ -468,7 +536,7 @@ impl File {
// Consume `self` and mutate in place — only the path, name and mtime
// change; id / mime_type / folder_id / blob_hash are carried over
// without the per-field clone the old `&self` builder paid.
self.path_string = new_storage_path.to_string();
self.path_string = new_storage_path.to_path_string();
self.storage_path = new_storage_path;
self.name = new_name;
self.modified_at = now;
@@ -493,7 +561,7 @@ impl File {
.as_secs();
// Consume `self`: only the path, folder_id and mtime change.
self.path_string = new_storage_path.to_string();
self.path_string = new_storage_path.to_path_string();
self.storage_path = new_storage_path;
self.folder_id = folder_id;
self.modified_at = now;
+116 -11
View File
@@ -1,12 +1,36 @@
use uuid::Uuid;
use crate::domain::services::path_service::{
StoragePath, normalize_storage_name, validate_storage_name,
StoragePath, normalize_storage_name_owned, validate_storage_name,
};
// Re-export entity errors from the centralized module
pub use super::entity_errors::{FolderError, FolderResult};
/// Owned parts of a [`Folder`] entity, produced by [`Folder::into_parts()`].
///
/// Consuming a `Folder` into `FolderParts` **moves** every field without
/// cloning, eliminating the 3-4 heap allocations that previously occurred
/// when converting `Folder → FolderDto` via `.to_string()` on each getter.
/// Mirrors [`super::file::FileParts`].
pub struct FolderParts {
pub id: String,
pub name: String,
pub storage_path: StoragePath,
pub path_string: String,
pub parent_id: Option<String>,
/// Drive that owns this folder. See [`Folder::drive_id`].
pub drive_id: Uuid,
pub created_at: u64,
pub modified_at: u64,
/// Descendant-rollup timestamp. See [`Folder::tree_modified_at`].
pub tree_modified_at: u64,
/// §14 provenance: original creator. See [`Folder::created_by`].
pub created_by: Option<Uuid>,
/// §14 provenance: most recent mutator. See [`Folder::updated_by`].
pub updated_by: Option<Uuid>,
}
/// Represents a folder entity in the domain
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Folder {
@@ -96,7 +120,7 @@ impl Folder {
storage_path: StoragePath,
parent_id: Option<String>,
) -> FolderResult<Self> {
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FolderError::InvalidFolderName(format!("{name}: {reason}")));
}
@@ -106,7 +130,7 @@ impl Folder {
.unwrap_or_default()
.as_secs();
let path_string = storage_path.to_string();
let path_string = storage_path.to_path_string();
Ok(Self {
id,
@@ -197,12 +221,12 @@ impl Folder {
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> FolderResult<Self> {
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FolderError::InvalidFolderName(format!("{name}: {reason}")));
}
let path_string = storage_path.to_string();
let path_string = storage_path.to_path_string();
Ok(Self {
id,
@@ -219,6 +243,70 @@ impl Folder {
})
}
/// PG-row constructor: the per-listing-row hot path.
///
/// Takes the materialized `storage.folders.path` column by value and
/// splits it once via [`StoragePath::from_joined`] — when the stored
/// path is already canonical (every row the repository writes), the
/// input `String` is reused as `path_string` with zero copies,
/// replacing the old `from_string` split + `Display` re-join pair.
/// The owned `name` is NFC-normalized without the always-copy of the
/// borrowing variant (DB rows are NFC by invariant).
#[allow(clippy::too_many_arguments)]
pub fn from_materialized_row(
id: String,
name: String,
path: String,
parent_id: Option<String>,
drive_id: Uuid,
created_at: u64,
modified_at: u64,
tree_modified_at: u64,
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> FolderResult<Self> {
let name = normalize_storage_name_owned(name);
if let Err(reason) = validate_storage_name(&name) {
return Err(FolderError::InvalidFolderName(format!("{name}: {reason}")));
}
let (storage_path, path_string) = StoragePath::from_joined(path);
Ok(Self {
id,
name,
storage_path,
path_string,
parent_id,
drive_id,
created_at,
modified_at,
tree_modified_at,
created_by,
updated_by,
})
}
/// Consume the entity and return all fields by ownership.
///
/// Use this when converting `Folder` into a DTO to avoid cloning
/// every `String` field (saves 3-4 heap allocations per folder).
pub fn into_parts(self) -> FolderParts {
FolderParts {
id: self.id,
name: self.name,
storage_path: self.storage_path,
path_string: self.path_string,
parent_id: self.parent_id,
drive_id: self.drive_id,
created_at: self.created_at,
modified_at: self.modified_at,
tree_modified_at: self.tree_modified_at,
created_by: self.created_by,
updated_by: self.updated_by,
}
}
// Getters
pub fn id(&self) -> &str {
&self.id
@@ -326,8 +414,25 @@ impl Folder {
/// changed; the folder's own value stays untouched
/// (self-exclusion).
pub fn compute_etag(id: &str, tree_modified_at: u64) -> String {
let prefix: String = id.chars().take(16).collect();
format!("{}-{}", prefix, tree_modified_at)
use std::fmt::Write as _;
// Byte index just past the 16th char (whole string when shorter).
// `id` is a UUID string (ASCII) in practice, so this is
// effectively `min(len, 16)`, but `char_indices` keeps the slice
// char-boundary-safe for exotic fixture values — byte-identical
// to the old `chars().take(16).collect::<String>()` without the
// intermediate allocation.
let end = match id.char_indices().nth(16) {
Some((i, _)) => i,
None => id.len(),
};
// Single allocation: prefix + '-' + up to 20 digits (u64::MAX).
let mut etag = String::with_capacity(end + 1 + 20);
etag.push_str(&id[..end]);
etag.push('-');
let _ = write!(etag, "{tree_modified_at}");
etag
}
/// Creates a new Folder instance from a DTO
@@ -350,7 +455,7 @@ impl Folder {
// round-trips lose the real rollup signal, so callers that
// need a freshly-rolled-up etag must reload from the
// repository.
let name = normalize_storage_name(&name);
let name = normalize_storage_name_owned(name);
Self {
id,
name,
@@ -376,7 +481,7 @@ impl Folder {
/// Creates a new version of the folder with updated name
pub fn with_name(&self, new_name: String) -> FolderResult<Self> {
let new_name = normalize_storage_name(&new_name);
let new_name = normalize_storage_name_owned(new_name);
if let Err(reason) = validate_storage_name(&new_name) {
return Err(FolderError::InvalidFolderName(format!(
"{new_name}: {reason}"
@@ -391,7 +496,7 @@ impl Folder {
};
// Update string representation
let new_path_string = new_storage_path.to_string();
let new_path_string = new_storage_path.to_path_string();
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
@@ -431,7 +536,7 @@ impl Folder {
};
// Update string representation
let new_path_string = new_storage_path.to_string();
let new_path_string = new_storage_path.to_path_string();
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
@@ -24,6 +24,14 @@ pub trait AddressBookRepository: Send + Sync + 'static {
address_book: AddressBook,
) -> AddressBookRepositoryResult<AddressBook>;
async fn delete_address_book(&self, id: &Uuid) -> AddressBookRepositoryResult<()>;
/// Batch sibling of `get_address_book_by_id`: one `= ANY($1)`
/// round-trip for a page of grant-derived ids. Missing ids drop
/// out; ordering is not guaranteed.
async fn get_address_books_by_ids(
&self,
ids: &[Uuid],
) -> AddressBookRepositoryResult<Vec<AddressBook>>;
async fn get_address_book_by_id(
&self,
id: &Uuid,
@@ -25,6 +25,18 @@ pub trait CalendarEventRepository: Send + Sync + 'static {
/// Finds a calendar event by its ID
async fn find_event_by_id(&self, id: &Uuid) -> CalendarEventRepositoryResult<CalendarEvent>;
/// Cursor stream over every event of `calendar_id` in bundle order:
/// rows sorted by `(first occurrence per UID, uid, master-first,
/// start_time)` so a recurring master + its exception overrides
/// arrive adjacent and bundles appear in the first-appearance order
/// the buffered `start_time` listing produced. ONE scan+sort on the
/// server; the streaming CalDAV emitters cut pages at UID
/// boundaries so only a page of rows is ever resident.
fn stream_events_uid_order(
&self,
calendar_id: Uuid,
) -> futures::stream::BoxStream<'static, CalendarEventRepositoryResult<CalendarEvent>>;
/// Lists all events in a specific calendar
async fn list_events_by_calendar(
&self,
@@ -25,6 +25,12 @@ pub trait CalendarRepository: Send + Sync + 'static {
/// Finds a calendar by its ID
async fn find_calendar_by_id(&self, id: &Uuid) -> CalendarRepositoryResult<Calendar>;
/// Batch sibling of [`Self::find_calendar_by_id`]: one `= ANY($1)`
/// round-trip for a page of grant-derived ids. Missing ids drop out
/// (no per-id NotFound), matching the listing carve-out for
/// deleted/trashed races. Ordering is not guaranteed.
async fn find_calendars_by_ids(&self, ids: &[Uuid]) -> CalendarRepositoryResult<Vec<Calendar>>;
/// Lists all calendars owned by a specific user. Post-Round-3 the
/// service layer prefers `authz.list_incoming_grants` (surfaces
/// owned + shared in one union), but this direct lookup remains
@@ -25,6 +25,14 @@ pub trait ContactRepository: Send + Sync + 'static {
address_book_id: &Uuid,
uids: &[String],
) -> ContactRepositoryResult<Vec<Contact>>;
/// Cursor stream over every contact of the book in the listing
/// order (`full_name, first_name, last_name`) — ONE scan+sort on
/// the server; the streaming CardDAV emitters page over it.
fn stream_contacts_by_book(
&self,
address_book_id: Uuid,
) -> futures::stream::BoxStream<'static, ContactRepositoryResult<Contact>>;
async fn get_contacts_by_address_book(
&self,
address_book_id: &Uuid,
+28 -1
View File
@@ -172,10 +172,14 @@ pub trait DriveRepository: Send + Sync + 'static {
/// Returns rows in a stable order: default drive first (if any),
/// then by display name. The `/api/drives` handler relies on that
/// order for the picker UI without a follow-up sort.
/// Returned as `Arc<Vec<…>>`: warm hits are a refcount bump straight
/// off the per-user cache instead of a deep clone of every row's
/// Strings — this runs per DAV request with an explicit drive
/// selector.
async fn list_readable_by(
&self,
caller_id: Uuid,
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError>;
) -> Result<std::sync::Arc<Vec<DriveWithRootName>>, DriveRepositoryError>;
/// `true` when the drive holds no live (non-trashed) folders other
/// than its own root and no live files at all. Used by
@@ -184,6 +188,29 @@ pub trait DriveRepository: Send + Sync + 'static {
/// content first so a single click can't wipe a populated drive.
async fn is_empty(&self, drive_id: Uuid) -> Result<bool, DriveRepositoryError>;
/// Drop the cached readable-drive list for one user. Called by
/// service-layer code paths that mutate state affecting a specific
/// caller's drive listing (grant writes, membership changes) but
/// don't reach through the drive-repo itself. Default no-op — the
/// no-cache stubs need no plumbing.
async fn invalidate_readable_for_user(&self, _user_id: Uuid) {}
/// Drop every cached readable-drive list. Called when the affected
/// user set is unknown at this layer — group-subject grants, drive
/// deletion, policy edits, root-folder renames (drive.name is
/// sourced from the root folder, so a rename affects the listing
/// for every user with a grant on the drive). Default no-op.
fn invalidate_readable_all(&self) {}
/// Drop every entry in the "default drive per user" cache. Called
/// from paths that mutate a drive's display name or its root
/// folder id at the concrete cache level (root-folder rename is
/// the only one today). Same class of bug as
/// `invalidate_readable_all` — the cache holds a `DriveWithRootName`
/// with `root_folder_name` baked in, so a rename would otherwise
/// stay stale for the cache TTL. Default no-op.
fn invalidate_default_drive_all(&self) {}
/// Hard-delete a drive: its `role_grants` rows, its root folder,
/// and the drive row itself, in one transaction. Caller is
/// responsible for ensuring `is_empty` first; this method does
+35 -1
View File
@@ -98,6 +98,32 @@ pub trait FolderRepository: Send + Sync + 'static {
include_total: bool,
) -> Result<(Vec<Folder>, Option<usize>), DomainError>;
/// Keyset-paged listing of `parent_id`'s direct sub-folders in name
/// order — `name > $after_name ORDER BY name LIMIT $limit`, one bounded
/// index-range read per page off the partial unique index
/// `idx_folders_unique_name`. Streaming PROPFIND drains sub-folders
/// with this instead of `COUNT(*) OVER() … LIMIT/OFFSET`, which
/// window-aggregated and rescanned all N sub-folders on every page
/// (4.5x on a 5k-dir parent, benches/FOLDER-KEYSET.md). `has_next`
/// falls out of `rows.len() == limit` — no total needed.
///
/// The default implementation falls back to `list_folders` + in-memory
/// slice so stubs and mocks compile without changes.
async fn list_folders_batch(
&self,
parent_id: Option<&str>,
after_name: Option<&str>,
limit: usize,
) -> Result<Vec<Folder>, DomainError> {
let mut all = self.list_folders(parent_id).await?;
all.sort_by(|a, b| a.name().cmp(b.name()));
Ok(all
.into_iter()
.filter(|f| after_name.is_none_or(|a| f.name() > a))
.take(limit)
.collect())
}
/// Renames a folder. `caller_id` is stamped into `updated_by`
/// alongside the `updated_at = NOW()` bump (§14 provenance).
async fn rename_folder(
@@ -243,13 +269,21 @@ pub trait FolderRepository: Send + Sync + 'static {
/// Results are ordered by relevance (exact > starts-with > contains) for
/// autocomplete suggestions.
///
/// `caller_id` scopes results to folders whose owning drive the caller
/// can Read (direct or group-mediated `role_grants`). Without it the
/// endpoint leaked names + paths across every tenant on the instance —
/// closed as AuthZ audit finding #1 (2026-07-12).
///
/// The default implementation falls back to `list_folders` + in-memory
/// filter so that stubs and mocks compile without changes.
/// filter so that stubs and mocks compile without changes. Stub-mode
/// callers already operate against a single tenant's data, so ignoring
/// `caller_id` here is safe; the PG impl enforces the real scope.
async fn suggest_folders_by_name(
&self,
parent_id: Option<&str>,
query: &str,
limit: usize,
_caller_id: uuid::Uuid,
) -> Result<Vec<Folder>, DomainError> {
let all = self.list_folders(parent_id).await?;
let q = query.to_lowercase();
@@ -13,6 +13,11 @@ pub trait PlaylistRepository: Send + Sync + 'static {
async fn find_playlist_by_id(&self, id: &Uuid) -> PlaylistRepositoryResult<Playlist>;
/// Batch sibling of [`Self::find_playlist_by_id`]: one `= ANY($1)`
/// round-trip for a page of grant-derived ids. Missing ids drop
/// out; ordering is not guaranteed.
async fn find_playlists_by_ids(&self, ids: &[Uuid]) -> PlaylistRepositoryResult<Vec<Playlist>>;
async fn list_playlists_by_owner(
&self,
owner_id: Uuid,
+127 -3
View File
@@ -40,6 +40,22 @@ pub fn normalize_storage_name(name: &str) -> String {
name.nfc().collect()
}
/// Owned-input sibling of [`normalize_storage_name`].
///
/// The borrowing variant must always allocate a fresh `String` even when
/// the input is already NFC — which is every name loaded back from
/// PostgreSQL (DB invariant) and every ASCII name. Callers that own the
/// `String` (entity constructors receive `name: String` by value) were
/// paying that copy only to drop the original immediately. This variant
/// returns the input unchanged on the fast path: zero allocations per
/// row on every listing (PROPFIND, photos timeline, search).
pub fn normalize_storage_name_owned(name: String) -> String {
if is_nfc_quick(name.chars()) == IsNormalized::Yes {
return name;
}
name.nfc().collect()
}
/// Validates a single file or folder name component.
///
/// Returns `Err` with a human-readable reason if the name is rejected.
@@ -102,6 +118,88 @@ impl StoragePath {
Self { segments }
}
/// One-pass builder for PG listing rows: materialized folder path +
/// file name → `(StoragePath, path_string)`.
///
/// Replaces the old per-row chain
/// `StoragePath::from_string(&format!("{fp}/{name}"))` +
/// `storage_path.to_string()`, which allocated a joined temporary,
/// split it back into per-segment `String`s, and then re-joined those
/// segments (via `join` + `write!`) into the `path_string` the DTOs
/// actually serve. Here both representations are built in a single
/// pass with exactly one `String` for the joined form and no
/// intermediate temporaries.
///
/// Byte-equivalence with the old chain holds because concatenating
/// with a `/` separator distributes over `split('/')`:
/// `(fp + "/" + name).split('/') == fp.split('/') ⧺ name.split('/')`,
/// and the joined form is exactly `Display`'s `/`-prefixed rendering
/// of the surviving segments (root renders as `"/"`).
pub fn from_folder_and_name(folder_path: Option<&str>, file_name: &str) -> (Self, String) {
let fp = folder_path.unwrap_or("");
// Upper bounds: every byte of both inputs survives at most once,
// plus one leading '/' per segment (≤ segment count) — sizing to
// input length + 2 covers the worst case without a second scan.
let mut joined = String::with_capacity(fp.len() + file_name.len() + 2);
let mut segments: Vec<String> =
Vec::with_capacity(fp.bytes().filter(|&b| b == b'/').count() + 2);
for seg in fp
.split('/')
.chain(file_name.split('/'))
.filter(|s| Self::is_safe_segment(s))
{
joined.push('/');
joined.push_str(seg);
segments.push(seg.to_string());
}
if segments.is_empty() {
joined.push('/');
}
(Self { segments }, joined)
}
/// One-pass splitter for a pre-joined materialized path (the
/// `storage.folders.path` column) → `(StoragePath, path_string)`.
///
/// When the input is already in canonical joined form (leading `/`,
/// no empty/`.`/`..` segments, no trailing `/`) — which is every row
/// the repository writes — the input `String` is reused as the
/// `path_string` with zero copies. Non-canonical inputs fall back to
/// the filtering rebuild and produce exactly what
/// `from_string(&path).to_string()` used to.
pub fn from_joined(path: String) -> (Self, String) {
if Self::is_canonical_joined(&path) {
let segments: Vec<String> = if path.len() == 1 {
Vec::new()
} else {
path[1..].split('/').map(str::to_string).collect()
};
return (Self { segments }, path);
}
// Fallback: identical to the old from_string + to_string pair.
let segments: Vec<String> = path
.split('/')
.filter(|s| Self::is_safe_segment(s))
.map(str::to_string)
.collect();
let sp = Self { segments };
let joined = sp.to_path_string();
(sp, joined)
}
/// `true` when `path` is exactly `Display`'s canonical rendering of
/// its own segments: `"/"` alone, or `/seg(/seg)*` where every
/// segment is safe. One scan, no allocations.
fn is_canonical_joined(path: &str) -> bool {
if path == "/" {
return true;
}
if !path.starts_with('/') || path.ends_with('/') {
return false;
}
path[1..].split('/').all(Self::is_safe_segment)
}
/// Creates a path from a PathBuf
pub fn from(path_buf: PathBuf) -> Self {
let segments = path_buf
@@ -152,14 +250,40 @@ impl StoragePath {
impl std::fmt::Display for StoragePath {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
if self.segments.is_empty() {
write!(f, "/")
} else {
write!(f, "/{}", self.segments.join("/"))
return f.write_str("/");
}
// Write segments directly — the old `self.segments.join("/")`
// allocated a full joined temporary inside every `format!`/
// `to_string` of a path.
for seg in &self.segments {
f.write_str("/")?;
f.write_str(seg)?;
}
Ok(())
}
}
impl StoragePath {
/// The canonical joined form (`Display`'s output) in exactly one
/// pre-sized allocation.
///
/// `to_string()` routes through `Display` into an unsized `String`
/// that grows geometrically (multiple reallocs + copies for typical
/// path lengths). Entity constructors call this once per row on
/// every listing, so the sized single-alloc variant is the default
/// there.
pub fn to_path_string(&self) -> String {
if self.segments.is_empty() {
return "/".to_string();
}
let mut s = String::with_capacity(self.segments.iter().map(|seg| seg.len() + 1).sum());
for seg in &self.segments {
s.push('/');
s.push_str(seg);
}
s
}
/// Returns the path representation as a string
pub fn as_str(&self) -> &str {
// Note: The implementation should really store the string,
@@ -115,6 +115,11 @@ impl CalendarStoragePort for CalendarStorageAdapter {
Ok(CalendarDto::from(calendar))
}
async fn get_calendars_by_ids(&self, ids: &[Uuid]) -> Result<Vec<CalendarDto>, DomainError> {
let calendars = self.calendar_repository.find_calendars_by_ids(ids).await?;
Ok(calendars.into_iter().map(CalendarDto::from).collect())
}
async fn list_calendars_by_owner(
&self,
owner_id: Uuid,
@@ -442,6 +447,30 @@ impl CalendarStoragePort for CalendarStorageAdapter {
Ok(events.into_iter().map(CalendarEventDto::from).collect())
}
fn stream_events_uid_order(
&self,
calendar_id: &str,
) -> futures::stream::BoxStream<'static, Result<CalendarEventDto, DomainError>> {
use futures::StreamExt;
let uuid = match Uuid::parse_str(calendar_id) {
Ok(u) => u,
Err(_) => {
return Box::pin(futures::stream::once(async {
Err(DomainError::new(
ErrorKind::InvalidInput,
"Calendar",
"Invalid calendar ID format",
))
}));
}
};
Box::pin(
self.event_repository
.stream_events_uid_order(uuid)
.map(|r| r.map(CalendarEventDto::from)),
)
}
async fn list_events_by_calendar_paginated(
&self,
calendar_id: &str,
@@ -84,6 +84,15 @@ impl ContactStoragePort for ContactStorageAdapter {
.await
}
async fn get_address_books_by_ids(
&self,
ids: &[Uuid],
) -> Result<Vec<AddressBook>, DomainError> {
self.address_book_repository
.get_address_books_by_ids(ids)
.await
}
async fn get_public_address_books(&self) -> Result<Vec<AddressBook>, DomainError> {
self.address_book_repository
.get_public_address_books()
@@ -137,6 +146,14 @@ impl ContactStoragePort for ContactStorageAdapter {
.await
}
fn stream_contacts_by_book(
&self,
address_book_id: Uuid,
) -> futures::stream::BoxStream<'static, Result<Contact, DomainError>> {
self.contact_repository
.stream_contacts_by_book(address_book_id)
}
async fn get_contacts_by_address_book_paginated(
&self,
address_book_id: &Uuid,
@@ -95,6 +95,11 @@ impl MusicStoragePort for MusicStorageAdapter {
}
}
async fn get_playlists_by_ids(&self, ids: &[Uuid]) -> Result<Vec<PlaylistDto>, DomainError> {
let playlists = self.playlist_repository.find_playlists_by_ids(ids).await?;
Ok(playlists.into_iter().map(PlaylistDto::from).collect())
}
async fn list_playlists_by_owner(
&self,
owner_id: Uuid,
@@ -110,6 +110,45 @@ impl AddressBookRepository for AddressBookPgRepository {
Ok(())
}
async fn get_address_books_by_ids(
&self,
ids: &[Uuid],
) -> AddressBookRepositoryResult<Vec<AddressBook>> {
if ids.is_empty() {
return Ok(Vec::new());
}
let rows = sqlx::query(
r#"
SELECT id, name, owner_id, description, color, is_public, created_at, updated_at
FROM carddav.address_books
WHERE id = ANY($1)
"#,
)
.bind(ids)
.fetch_all(&*self.pool)
.await
.map_err(|e| {
DomainError::database_error(format!("Failed to get address books by ids: {}", e))
})?;
Ok(rows
.iter()
.map(|row| {
let owner_id: Uuid = row.get("owner_id");
AddressBook::from_raw(
row.get("id"),
row.get("name"),
owner_id.to_string(),
row.get("description"),
row.get("color"),
row.get("is_public"),
row.get("created_at"),
row.get("updated_at"),
)
})
.collect())
}
async fn get_address_book_by_id(
&self,
id: &Uuid,
@@ -16,6 +16,31 @@ impl CalendarEventPgRepository {
pub fn new(pool: Arc<PgPool>) -> Self {
Self { pool }
}
/// Shared row → entity mapping (the inline shape every listing
/// method uses, factored for the cursor stream).
fn row_to_event(row: &sqlx::postgres::PgRow) -> CalendarEventRepositoryResult<CalendarEvent> {
let mut event = CalendarEvent::with_id(
row.get("id"),
row.get("calendar_id"),
row.get("summary"),
row.get::<Option<String>, _>("description"),
row.get::<Option<String>, _>("location"),
row.get("start_time"),
row.get("end_time"),
row.get("all_day"),
row.get::<Option<String>, _>("rrule"),
row.get("ical_uid"),
row.get("ical_data"),
row.get("created_at"),
row.get("updated_at"),
)
.map_err(|e| {
DomainError::database_error(format!("Error creating calendar event: {}", e))
})?;
event.set_recurrence_id(row.get::<Option<DateTime<Utc>>, _>("recurrence_id"));
Ok(event)
}
}
impl CalendarEventRepository for CalendarEventPgRepository {
@@ -547,6 +572,57 @@ impl CalendarEventRepository for CalendarEventPgRepository {
Ok(result.rows_affected() as i64)
}
fn stream_events_uid_order(
&self,
calendar_id: Uuid,
) -> futures::stream::BoxStream<'static, CalendarEventRepositoryResult<CalendarEvent>> {
// ONE ordered scan for the whole calendar, served through a PG
// cursor (`fetch`) so only a window of rows is in flight. The
// window function puts every UID's rows adjacent, bundles
// ordered by first occurrence — exactly the first-appearance
// order the buffered `ORDER BY start_time` listing produced
// after grouping — with the master row first inside each UID.
//
// The first streaming shape hydrated pages via
// `ical_uid = ANY(page)`: ~20 µs per index descent made the
// total wall 3-4x the buffered single scan (measured in
// benches/ROUND5.md). This keeps the buffered path's one
// scan+sort while bounding memory to a page.
let pool = self.pool.clone();
let stream: futures::stream::BoxStream<
'static,
CalendarEventRepositoryResult<CalendarEvent>,
> = Box::pin(async_stream::try_stream! {
let mut conn = pool.acquire().await.map_err(|e| {
DomainError::database_error(format!("Failed to acquire connection: {}", e))
})?;
let mut rows = sqlx::query(
r#"
SELECT
id, calendar_id, summary, description, location,
start_time, end_time, all_day, rrule,
created_at, updated_at, ical_uid, ical_data, recurrence_id
FROM caldav.calendar_events
WHERE calendar_id = $1
ORDER BY MIN(start_time) OVER (PARTITION BY ical_uid),
ical_uid,
(recurrence_id IS NOT NULL),
start_time
"#,
)
.bind(calendar_id)
.fetch(&mut *conn);
use futures::TryStreamExt;
while let Some(row) = rows.try_next().await.map_err(|e| {
DomainError::database_error(format!("Failed to stream events: {}", e))
})? {
yield Self::row_to_event(&row)?;
}
});
stream
}
async fn list_events_by_calendar_paginated(
&self,
calendar_id: &Uuid,
@@ -138,6 +138,42 @@ impl CalendarRepository for CalendarPgRepository {
Ok(calendar)
}
async fn find_calendars_by_ids(&self, ids: &[Uuid]) -> CalendarRepositoryResult<Vec<Calendar>> {
if ids.is_empty() {
return Ok(Vec::new());
}
let rows = sqlx::query(
r#"
SELECT id, name, owner_id, description, color, is_public, created_at, updated_at
FROM caldav.calendars
WHERE id = ANY($1)
"#,
)
.bind(ids)
.fetch_all(&*self.pool)
.await
.map_err(|e| {
DomainError::database_error(format!("Failed to get calendars by ids: {}", e))
})?;
rows.iter()
.map(|row| {
Calendar::with_id(
row.get("id"),
row.get("name"),
row.get("owner_id"),
row.get("description"),
row.get("color"),
row.get("created_at"),
row.get("updated_at"),
)
.map_err(|e| {
DomainError::database_error(format!("Failed to create calendar object: {}", e))
})
})
.collect()
}
async fn list_calendars_by_owner(
&self,
owner_id: Uuid,
@@ -278,6 +278,45 @@ impl ContactRepository for ContactPgRepository {
Ok(contacts)
}
fn stream_contacts_by_book(
&self,
address_book_id: Uuid,
) -> futures::stream::BoxStream<'static, ContactRepositoryResult<Contact>> {
// ONE ordered scan served through a PG cursor — the CardDAV
// multistatus emitters page over this stream so only a page of
// contacts is resident (same design as the CalDAV round-5
// cursor; contacts have no master/exception bundling, so pages
// can cut anywhere).
let pool = self.pool.clone();
let stream: futures::stream::BoxStream<'static, ContactRepositoryResult<Contact>> =
Box::pin(async_stream::try_stream! {
let mut conn = pool.acquire().await.map_err(|e| {
DomainError::database_error(format!("Failed to acquire connection: {}", e))
})?;
let mut rows = sqlx::query(
r#"
SELECT
id, address_book_id, uid, full_name, first_name, last_name, nickname,
email, phone, address, organization, title, notes, photo_url,
birthday, anniversary, vcard, etag, created_at, updated_at
FROM carddav.contacts
WHERE address_book_id = $1
ORDER BY full_name, first_name, last_name
"#,
)
.bind(address_book_id)
.fetch(&mut *conn);
use futures::TryStreamExt;
while let Some(row) = rows.try_next().await.map_err(|e| {
DomainError::database_error(format!("Failed to stream contacts: {}", e))
})? {
yield Self::row_to_contact(&row)?;
}
});
stream
}
async fn get_contacts_by_address_book(
&self,
address_book_id: &Uuid,
@@ -25,10 +25,11 @@ use crate::domain::repositories::drive_repository::{
/// policy edits — all of which invalidate explicitly below), yet it is
/// re-resolved on EVERY NextCloud request (basic-auth chroot), every
/// native `/webdav` request (Mode-B scope resolution) and every WOPI
/// call. 30 s mirrors `drive_role_cache` in `pg_acl_engine.rs` and bounds
/// the one non-invalidated staleness source: a root-folder *rename*,
/// which doesn't pass through this repository. Measured in
/// `benches/CHROOT-CACHE.md`.
/// call. 30 s mirrors `drive_role_cache` in `pg_acl_engine.rs`. Root-
/// folder renames — which don't pass through this repository directly
/// — invalidate via the `DriveRepository::invalidate_default_drive_all`
/// trait hook called from `folder_service::rename_folder_with_perms`
/// when `parent_id IS NULL`. Measured in `benches/CHROOT-CACHE.md`.
const DEFAULT_DRIVE_CACHE_TTL: Duration = Duration::from_secs(30);
/// One entry per active user; entries are small (a `Drive` + a name).
@@ -41,6 +42,35 @@ pub struct DrivePgRepository {
/// provisioning idempotency check (`NotFound` → create) always sees
/// the live table.
default_drive_cache: Cache<Uuid, DriveWithRootName>,
/// caller_id → every drive the caller can read (the full
/// role_grants ⋈ drives ⋈ folders join of [`list_readable_by`],
/// including the transitive-group expansion).
///
/// Re-resolved before this cache existed on EVERY native `/webdav`
/// request that names an explicit drive selector (all verbs; MOVE
/// and COPY twice), plus per-request in search, trash listing and
/// the `GET /api/drives` picker — the heaviest per-request query
/// left on the DAV path after CHROOT-CACHE. Concurrent misses are
/// coalesced (`try_get_with`), errors are never cached.
///
/// Freshness: every membership/lifecycle mutation that flows
/// through this repository or `DriveManagementService` invalidates
/// explicitly (per-user when the subject is a User, whole cache for
/// Group subjects, whose transitive membership is not resolvable
/// here). Root-folder renames — which update `drive.name` because it
/// reads through `folders.name` of the root row — also invalidate,
/// via the trait's `invalidate_readable_all` hook called from
/// `folder_service::rename_folder_with_perms` when
/// `parent_id IS NULL`. That path was missed by the perf commit
/// that introduced this cache (`12dc648c`) and surfaced by
/// `drives_membership.hurl` Step 23; the trait hook closes it
/// without folder_service knowing about the concrete moka cache.
///
/// Residual staleness — a grant written by a path that can't reach
/// this cache — is bounded by the same 30 s TTL the sibling caches
/// accept; actual permission enforcement is unaffected (the ACL
/// engine re-checks per operation with its own invalidation).
readable_cache: Cache<Uuid, Arc<Vec<DriveWithRootName>>>,
}
impl DrivePgRepository {
@@ -51,9 +81,36 @@ impl DrivePgRepository {
.max_capacity(DEFAULT_DRIVE_CACHE_CAPACITY)
.time_to_live(DEFAULT_DRIVE_CACHE_TTL)
.build(),
readable_cache: Cache::builder()
.max_capacity(DEFAULT_DRIVE_CACHE_CAPACITY)
.time_to_live(DEFAULT_DRIVE_CACHE_TTL)
.build(),
}
}
/// Drop the cached readable-drive list for one user (their grant set
/// changed: membership write, personal-drive provisioning, …).
pub async fn invalidate_readable_for_user(&self, user_id: Uuid) {
self.readable_cache.invalidate(&user_id).await;
}
/// Drop every cached readable-drive list. Used when the affected
/// user set is unknown at this layer: group-subject grants, drive
/// deletion, policy edits. All are admin-rare; repopulation costs
/// one join per active caller.
pub fn invalidate_readable_all(&self) {
self.readable_cache.invalidate_all();
}
/// Drop every cached `default_drive_cache` entry. Exposed as a
/// `pub` sibling of the whole-cache invalidators above so trait
/// callers holding a `dyn DriveRepository` can trigger the same
/// cleanup path (e.g. `folder_service` on root-folder rename —
/// see `impl DriveRepository` below).
pub fn invalidate_default_drive_all(&self) {
self.default_drive_cache.invalidate_all();
}
fn map_sqlx_err(context: &'static str, e: sqlx::Error) -> DriveRepositoryError {
if let sqlx::Error::Database(ref dberr) = e
&& let Some(code) = dberr.code()
@@ -112,10 +169,83 @@ impl DrivePgRepository {
dwr.caller_role = role_str.as_deref().and_then(Role::parse);
Ok(dwr)
}
/// The uncached grants join behind [`DriveRepository::list_readable_by`].
///
/// Joining role_grants → drives → folders returns every drive the
/// caller can read, paired with its display name. Group
/// memberships (direct + transitive) are expanded inline by
/// `storage.caller_group_ids($caller)` — no Rust-side ceremony.
///
/// ORDER BY puts default drives first (so the picker UI doesn't
/// need a follow-up sort), then alphabetical by name. GROUP BY
/// collapses duplicate role_grants on the same drive (direct +
/// group-mediated) and sidesteps PostgreSQL's "ORDER BY
/// expression must appear in select list" rule that SELECT
/// DISTINCT imposes.
/// `MIN(g.role)` picks the caller's strongest role on each drive:
/// `storage.grant_role` is declared `owner → viewer` (strongest →
/// weakest), so MIN returns the strongest. Cast `::text` matches
/// the codebase convention for reading enum columns into Rust
/// (see `pg_acl_engine.rs`); `Role::parse` handles the trip back.
async fn query_readable_by(
&self,
caller_id: Uuid,
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
let rows = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name,
MIN(g.role)::text AS caller_role
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
JOIN storage.role_grants g
ON g.resource_type = 'drive'
AND g.resource_id = d.id
WHERE (
(g.subject_type = 'user' AND g.subject_id = $1)
OR (g.subject_type = 'group' AND g.subject_id IN
(SELECT storage.caller_group_ids($1)))
)
AND (g.expires_at IS NULL OR g.expires_at > NOW())
GROUP BY d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at, f.name
ORDER BY (d.default_for_user IS NULL) ASC,
LOWER(f.name) ASC
"#,
)
.bind(caller_id)
.fetch_all(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("list_readable_by", e))?;
rows.iter()
.map(Self::row_to_drive_with_name_and_role)
.collect()
}
}
#[async_trait::async_trait]
impl DriveRepository for DrivePgRepository {
async fn invalidate_readable_for_user(&self, user_id: Uuid) {
// Delegate to the inherent method — the trait forwarding lets
// callers holding a `dyn DriveRepository` (e.g. `folder_service`
// on a root-folder rename) trigger invalidation without knowing
// about the concrete cache.
DrivePgRepository::invalidate_readable_for_user(self, user_id).await;
}
fn invalidate_readable_all(&self) {
DrivePgRepository::invalidate_readable_all(self);
}
fn invalidate_default_drive_all(&self) {
DrivePgRepository::invalidate_default_drive_all(self);
}
async fn create_personal_drive_atomic(
&self,
owner_id: Uuid,
@@ -239,6 +369,8 @@ impl DriveRepository for DrivePgRepository {
// Drop any cached default-drive resolution for this user (a stale
// NotFound is never cached, but be explicit about the write path).
self.default_drive_cache.invalidate(&owner_id).await;
// The owner gained a drive — their readable list changed too.
self.invalidate_readable_for_user(owner_id).await;
Self::row_to_drive_with_name(&row)
}
@@ -347,6 +479,16 @@ impl DriveRepository for DrivePgRepository {
.await
.map_err(|e| Self::map_sqlx_err("create_shared_drive_atomic.commit", e))?;
// The owner grant written above changes the grantee's readable
// list. User subjects invalidate precisely; Group subjects fall
// back to a full clear (transitive members unknown here).
match owner_subject {
crate::domain::services::authorization::Subject::User(uid) => {
self.invalidate_readable_for_user(uid).await;
}
_ => self.invalidate_readable_all(),
}
Self::row_to_drive_with_name(&row)
}
@@ -356,21 +498,26 @@ impl DriveRepository for DrivePgRepository {
// the trash. Trashed items don't count — owners can delete a
// drive even when its trash bin still holds rows; the trash GC
// will clean those up after the standard retention window.
let count: (i64,) = sqlx::query_as(
//
// EXISTS instead of COUNT(*): only emptiness is tested, so the
// planner stops at the first matching row — a populated drive
// answers from one index probe instead of aggregating every
// live file + folder it contains.
let occupied: (bool,) = sqlx::query_as(
r#"
SELECT (
(SELECT COUNT(*) FROM storage.folders
WHERE drive_id = $1 AND parent_id IS NOT NULL AND NOT is_trashed)
+ (SELECT COUNT(*) FROM storage.files
WHERE drive_id = $1 AND NOT is_trashed)
)
SELECT EXISTS(
SELECT 1 FROM storage.folders
WHERE drive_id = $1 AND parent_id IS NOT NULL AND NOT is_trashed)
OR EXISTS(
SELECT 1 FROM storage.files
WHERE drive_id = $1 AND NOT is_trashed)
"#,
)
.bind(drive_id)
.fetch_one(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("is_empty", e))?;
Ok(count.0 == 0)
Ok(!occupied.0)
}
async fn delete_atomic(&self, drive_id: Uuid) -> Result<(), DriveRepositoryError> {
@@ -425,10 +572,11 @@ impl DriveRepository for DrivePgRepository {
tx.commit()
.await
.map_err(|e| Self::map_sqlx_err("delete_atomic.commit", e))?;
// We only have the drive id here; the cache is keyed by user.
// Deletion is rare — clearing the whole cache is the simple,
// We only have the drive id here; the caches are keyed by user.
// Deletion is rare — clearing them whole is the simple,
// always-correct move (repopulates at one query per active user).
self.default_drive_cache.invalidate_all();
self.invalidate_readable_all();
Ok(())
}
@@ -512,56 +660,22 @@ impl DriveRepository for DrivePgRepository {
async fn list_readable_by(
&self,
caller_id: Uuid,
) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
// Joining role_grants → drives → folders returns every drive the
// caller can read, paired with its display name. Group
// memberships (direct + transitive) are expanded inline by
// `storage.caller_group_ids($caller)` — no Rust-side ceremony.
//
// ORDER BY puts default drives first (so the picker UI doesn't
// need a follow-up sort), then alphabetical by name. GROUP BY
// collapses duplicate role_grants on the same drive (direct +
// group-mediated) and sidesteps PostgreSQL's "ORDER BY
// expression must appear in select list" rule that SELECT
// DISTINCT imposes.
// `MIN(g.role)` picks the caller's strongest role on each drive:
// `storage.grant_role` is declared `owner → viewer` (strongest →
// weakest), so MIN returns the strongest. Cast `::text` matches
// the codebase convention for reading enum columns into Rust
// (see `pg_acl_engine.rs`); `Role::parse` handles the trip back.
let rows = sqlx::query(
r#"
SELECT d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at,
f.name AS root_folder_name,
MIN(g.role)::text AS caller_role
FROM storage.drives d
JOIN storage.folders f ON f.id = d.root_folder_id
JOIN storage.role_grants g
ON g.resource_type = 'drive'
AND g.resource_id = d.id
WHERE (
(g.subject_type = 'user' AND g.subject_id = $1)
OR (g.subject_type = 'group' AND g.subject_id IN
(SELECT storage.caller_group_ids($1)))
)
AND (g.expires_at IS NULL OR g.expires_at > NOW())
GROUP BY d.id, d.kind, d.default_for_user, d.root_folder_id,
d.quota_bytes, d.used_bytes, d.policies,
d.created_at, d.updated_at, f.name
ORDER BY (d.default_for_user IS NULL) ASC,
LOWER(f.name) ASC
"#,
)
.bind(caller_id)
.fetch_all(self.pool.as_ref())
.await
.map_err(|e| Self::map_sqlx_err("list_readable_by", e))?;
rows.iter()
.map(Self::row_to_drive_with_name_and_role)
.collect()
) -> Result<Arc<Vec<DriveWithRootName>>, DriveRepositoryError> {
// Serve from the per-user cache; concurrent misses for the same
// caller are coalesced into one join (`try_get_with`), and errors
// are never cached. See the `readable_cache` field docs for the
// freshness/invalidation contract. The Arc is handed to callers
// directly — a warm hit is a refcount bump, not a deep clone of
// every row's Strings.
self.readable_cache
.try_get_with(caller_id, async move {
self.query_readable_by(caller_id).await.map(Arc::new)
})
.await
.map_err(|e: Arc<DriveRepositoryError>| {
Arc::try_unwrap(e)
.unwrap_or_else(|shared| DriveRepositoryError::StorageError(shared.to_string()))
})
}
async fn list_all(&self) -> Result<Vec<DriveWithRootName>, DriveRepositoryError> {
@@ -717,9 +831,10 @@ impl DriveRepository for DrivePgRepository {
.ok_or_else(|| DriveRepositoryError::NotFound(drive_id.to_string()))?
.0;
// Policy edits must not serve a stale `policies` bag from the
// default-drive cache (keyed by user, and we only have the drive
// id) — clear it; policy edits are admin-rare.
// user-keyed caches (we only have the drive id) — clear both;
// policy edits are admin-rare.
self.default_drive_cache.invalidate_all();
self.invalidate_readable_all();
Ok(crate::domain::entities::drive::DrivePolicies::from_value(
&raw,
))
@@ -259,8 +259,9 @@ impl FavoritesRepositoryPort for FavoritesPgRepository {
return Ok(HashSet::new());
}
// Collect just the IDs for the IN clause
let ids: Vec<String> = item_ids.iter().map(|(id, _)| id.to_string()).collect();
// Collect just the IDs for the IN clause — sqlx binds `&[&str]` as
// text[], so no per-id String is needed.
let ids: Vec<&str> = item_ids.iter().map(|(id, _)| *id).collect();
let rows = sqlx::query(
"SELECT item_id FROM auth.user_favorites WHERE user_id = $1 AND item_id = ANY($2)",
@@ -11,9 +11,9 @@
/// Post-D7-step-6: `storage.files.user_id` dropped, so it's no
/// longer projected.
type MediaFileRow = (
String, // id
Uuid, // id (binary decode; benches/ROUND6.md §10)
String, // name
Option<String>, // folder_id
Option<Uuid>, // folder_id
Option<String>, // folder path
i64, // size
String, // mime_type
@@ -83,9 +83,9 @@ const CALLER_CAN_READ_DRIVE: &str = "EXISTS (\
/// longer part of the tuple; `row_to_file` populates the entity's
/// legacy `user_id` field with `None`.
type FileRow = (
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -269,7 +269,7 @@ impl FileBlobReadRepository {
let where_clause = conditions.join(" AND ");
let sql = format!(
"SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path, \
"SELECT fi.id, fi.name, fi.folder_id, fo.path, \
fi.size, fi.mime_type, \
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
@@ -319,7 +319,7 @@ impl FileBlobReadRepository {
}
let rows = sqlx::query_as::<_, FileRow>(
"SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path, \
"SELECT fi.id, fi.name, fi.folder_id, fo.path, \
fi.size, fi.mime_type, \
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
@@ -413,19 +413,11 @@ impl FileBlobReadRepository {
}
}
/// Build a `StoragePath` from the materialized folder path + file name.
fn make_file_path(folder_path: Option<&str>, file_name: &str) -> StoragePath {
match folder_path {
Some(fp) if !fp.is_empty() => StoragePath::from_string(&format!("{fp}/{file_name}")),
_ => StoragePath::from_string(file_name),
}
}
#[allow(clippy::too_many_arguments)]
fn row_to_file(
id: String,
id: Uuid,
name: String,
folder_id: Option<String>,
folder_id: Option<Uuid>,
folder_path: Option<String>,
size: i64,
mime_type: String,
@@ -435,14 +427,13 @@ impl FileBlobReadRepository {
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<File, DomainError> {
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
File::with_timestamps_blob_hash_and_provenance(
id,
File::from_materialized_row(
id.to_string(),
name,
storage_path,
folder_path.as_deref(),
size as u64,
mime_type,
folder_id,
folder_id.map(|u| u.to_string()),
created_at as u64,
modified_at as u64,
blob_hash,
@@ -488,13 +479,21 @@ impl FileBlobReadRepository {
/// `sort_date` epoch for each file (used as pagination cursor).
///
/// Uses the denormalised `media_sort_date` column (synced from
/// `file_metadata.captured_at` by trigger) so no JOIN with
/// `file_metadata` is needed. The partial covering index
/// `idx_files_media_timeline_by_drive` (migration 20260901000001)
/// keys on `(drive_id, media_sort_date DESC)` filtered on non-trashed
/// image/video rows — Postgres does one IndexScan per in-scope
/// drive_id already ordered by capture date, so LIMIT stops the scan
/// early. Same O(LIMIT) shape as the pre-D7 `user_id`-keyed hot path.
/// `file_metadata.captured_at` by trigger). The accessible drive ids
/// are materialised once, then a `CROSS JOIN LATERAL (… ORDER BY
/// media_sort_date DESC LIMIT k)` per drive turns the partial covering
/// index `idx_files_media_timeline_by_drive` (migration 20260901000001,
/// `(drive_id, media_sort_date DESC)` filtered on non-trashed
/// image/video rows) into one BOUNDED index scan per drive; the outer
/// merge sorts `drives × k` rows. The folders / file_metadata joins sit
/// outside the top-N so only the k emitted rows pay them.
///
/// The previous shape put the joins and the global `ORDER BY … LIMIT`
/// above a `drive_id IN (…)` nested loop — Postgres fed EVERY media row
/// through the join into a top-N heapsort, scanning the timeline index
/// to exhaustion on every page: O(library) per page, 97 ms on a
/// 50k-photo library vs 1.6 ms for this shape (55.7x,
/// benches/PHOTOS-TIMELINE.md).
///
/// Scope (`docs/plan/drive.md` §15): drives with
/// `policies.include_in_photo_index = true` where the caller has a
@@ -537,37 +536,47 @@ impl FileBlobReadRepository {
};
let sql = format!(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.created_by, fi.updated_by,
EXTRACT(EPOCH FROM fi.media_sort_date)::bigint AS sort_date,
WITH accessible AS MATERIALIZED (
SELECT d.id
FROM storage.drives d
JOIN storage.role_grants g
ON g.resource_type = 'drive'
AND g.resource_id = d.id
WHERE (
(g.subject_type = 'user' AND g.subject_id = $1)
OR (g.subject_type = 'group' AND g.subject_id IN
(SELECT storage.caller_group_ids($1)))
)
AND (g.expires_at IS NULL OR g.expires_at > NOW())
AND (d.policies->>'include_in_photo_index')::boolean = true
)
SELECT top.id, top.name, top.folder_id, fo.path,
top.size, top.mime_type,
EXTRACT(EPOCH FROM top.created_at)::bigint,
EXTRACT(EPOCH FROM top.updated_at)::bigint,
top.blob_hash,
top.created_by, top.updated_by,
EXTRACT(EPOCH FROM top.media_sort_date)::bigint AS sort_date,
fm.width, fm.height
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
LEFT JOIN storage.file_metadata fm ON fm.file_id = fi.id
WHERE fi.drive_id IN (
SELECT d.id
FROM storage.drives d
JOIN storage.role_grants g
ON g.resource_type = 'drive'
AND g.resource_id = d.id
WHERE (
(g.subject_type = 'user' AND g.subject_id = $1)
OR (g.subject_type = 'group' AND g.subject_id IN
(SELECT storage.caller_group_ids($1)))
)
AND (g.expires_at IS NULL OR g.expires_at > NOW())
AND (d.policies->>'include_in_photo_index')::boolean = true
)
AND NOT fi.is_trashed
AND (fi.mime_type LIKE 'image/%' OR fi.mime_type LIKE 'video/%')
{cursor_pred}
ORDER BY fi.media_sort_date DESC
LIMIT $3
FROM (
SELECT fi.*
FROM accessible a
CROSS JOIN LATERAL (
SELECT fi.*
FROM storage.files fi
WHERE fi.drive_id = a.id
AND NOT fi.is_trashed
AND (fi.mime_type LIKE 'image/%' OR fi.mime_type LIKE 'video/%')
{cursor_pred}
ORDER BY fi.media_sort_date DESC
LIMIT $3
) fi
ORDER BY fi.media_sort_date DESC
LIMIT $3
) top
LEFT JOIN storage.folders fo ON fo.id = top.folder_id
LEFT JOIN storage.file_metadata fm ON fm.file_id = top.id
ORDER BY top.media_sort_date DESC
"#,
);
let rows: Vec<MediaFileRow> = sqlx::query_as(&sql)
@@ -670,9 +679,9 @@ impl FileReadPort for FileBlobReadRepository {
let row = sqlx::query_as::<
_,
(
String, // id
Uuid, // id (binary decode)
String, // name
Option<String>, // folder_id
Option<Uuid>, // folder_id
Option<String>, // folder path
i64, // size
String, // mime_type
@@ -684,7 +693,7 @@ impl FileReadPort for FileBlobReadRepository {
),
>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -717,9 +726,9 @@ impl FileReadPort for FileBlobReadRepository {
let row = sqlx::query_as::<
_,
(
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -731,7 +740,7 @@ impl FileReadPort for FileBlobReadRepository {
),
>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -759,7 +768,7 @@ impl FileReadPort for FileBlobReadRepository {
let rows: Vec<FileRow> = if let Some(fid) = folder_id {
sqlx::query_as(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -778,7 +787,7 @@ impl FileReadPort for FileBlobReadRepository {
} else {
sqlx::query_as(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -839,7 +848,7 @@ impl FileReadPort for FileBlobReadRepository {
};
let sql = format!(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -912,7 +921,7 @@ impl FileReadPort for FileBlobReadRepository {
.map_err(|e| DomainError::internal_error("FileBlobRead", format!("path: {e}")))?
.ok_or_else(|| DomainError::not_found("File", id))?;
Ok(Self::make_file_path(row.1.as_deref(), &row.0))
Ok(StoragePath::from_folder_and_name(row.1.as_deref(), &row.0).0)
}
async fn get_parent_folder_id(
@@ -1005,9 +1014,9 @@ impl FileReadPort for FileBlobReadRepository {
sqlx::query_as::<
_,
(
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -1019,7 +1028,7 @@ impl FileReadPort for FileBlobReadRepository {
),
>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -1043,9 +1052,9 @@ impl FileReadPort for FileBlobReadRepository {
sqlx::query_as::<
_,
(
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -1057,7 +1066,7 @@ impl FileReadPort for FileBlobReadRepository {
),
>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -1098,12 +1107,12 @@ impl FileReadPort for FileBlobReadRepository {
let stream = async_stream::try_stream! {
let mut row_stream = sqlx::query_as::<_, (
String, String, Option<String>, Option<String>,
Uuid, String, Option<Uuid>, Option<String>,
i64, String, i64, i64, String,
Option<Uuid>, Option<Uuid>, // created_by, updated_by (§14)
)>(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -1186,7 +1195,7 @@ impl FileReadPort for FileBlobReadRepository {
let offset_bind = bind_idx + 2;
let sql = format!(
"SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path, \
"SELECT fi.id, fi.name, fi.folder_id, fo.path, \
fi.size, fi.mime_type, \
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
@@ -1205,9 +1214,9 @@ impl FileReadPort for FileBlobReadRepository {
let mut query = sqlx::query_as::<
_,
(
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -1318,7 +1327,7 @@ impl FileReadPort for FileBlobReadRepository {
// ── Single query with COUNT(*) OVER() ──
let sql = format!(
"SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path, \
"SELECT fi.id, fi.name, fi.folder_id, fo.path, \
fi.size, fi.mime_type, \
EXTRACT(EPOCH FROM fi.created_at)::bigint, \
EXTRACT(EPOCH FROM fi.updated_at)::bigint, \
@@ -1337,9 +1346,9 @@ impl FileReadPort for FileBlobReadRepository {
let mut query = sqlx::query_as::<
_,
(
Uuid,
String,
String,
Option<String>,
Option<Uuid>,
Option<String>,
i64,
String,
@@ -1404,14 +1413,22 @@ impl FileReadPort for FileBlobReadRepository {
folder_id: Option<&str>,
query: &str,
limit: usize,
caller_id: Uuid,
) -> Result<Vec<File>, DomainError> {
// Scope by drive membership: `CALLER_CAN_READ_DRIVE` (`$1` =
// caller_id) restricts the result set to files whose owning drive
// the caller has any active `role_grants` on — direct or via a
// transitive group cascade. Pre-fix, the query only filtered on
// `NOT is_trashed AND name ILIKE $pattern`, exposing names + paths
// across every tenant on the instance (AuthZ audit finding #1,
// 2026-07-12).
let pattern = super::like_escape(query);
let limit_i64 = limit as i64;
let rows: Vec<FileRow> = if let Some(fid) = folder_id {
sqlx::query_as(
sqlx::query_as(&format!(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
@@ -1420,7 +1437,40 @@ impl FileReadPort for FileBlobReadRepository {
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id = $1::uuid
WHERE {CALLER_CAN_READ_DRIVE}
AND fi.folder_id = $2::uuid
AND NOT fi.is_trashed
AND fi.name ILIKE $3
ORDER BY CASE
WHEN fi.name ILIKE $4 THEN 0
WHEN fi.name ILIKE $4 || '%' THEN 1
ELSE 2
END,
fi.name
LIMIT $5
"#
))
.bind(caller_id)
.bind(fid)
.bind(&pattern)
.bind(query)
.bind(limit_i64)
.fetch_all(self.pool.as_ref())
.await
} else {
sqlx::query_as(&format!(
r#"
SELECT fi.id, fi.name, fi.folder_id, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE {CALLER_CAN_READ_DRIVE}
AND fi.folder_id IS NULL
AND NOT fi.is_trashed
AND fi.name ILIKE $2
ORDER BY CASE
@@ -1430,38 +1480,9 @@ impl FileReadPort for FileBlobReadRepository {
END,
fi.name
LIMIT $4
"#,
)
.bind(fid)
.bind(&pattern)
.bind(query)
.bind(limit_i64)
.fetch_all(self.pool.as_ref())
.await
} else {
sqlx::query_as(
r#"
SELECT fi.id::text, fi.name, fi.folder_id::text, fo.path,
fi.size, fi.mime_type,
EXTRACT(EPOCH FROM fi.created_at)::bigint,
EXTRACT(EPOCH FROM fi.updated_at)::bigint,
fi.blob_hash,
fi.created_by, fi.updated_by
FROM storage.files fi
LEFT JOIN storage.folders fo ON fo.id = fi.folder_id
WHERE fi.folder_id IS NULL
AND NOT fi.is_trashed
AND fi.name ILIKE $1
ORDER BY CASE
WHEN fi.name ILIKE $2 THEN 0
WHEN fi.name ILIKE $2 || '%' THEN 1
ELSE 2
END,
fi.name
LIMIT $3
"#,
)
"#
))
.bind(caller_id)
.bind(&pattern)
.bind(query)
.bind(limit_i64)
@@ -17,7 +17,6 @@ use crate::application::dtos::display_helpers::category_order_for;
use crate::application::ports::storage_ports::{CopyFolderTreeResult, FileWritePort};
use crate::common::errors::DomainError;
use crate::domain::entities::file::File;
use crate::domain::services::path_service::StoragePath;
use super::transaction_utils::retry_on_deadlock;
use crate::infrastructure::services::dedup_service::DedupService;
@@ -61,14 +60,6 @@ impl FileBlobWriteRepository {
}
}
/// Build a `StoragePath` from the materialized folder path + file name.
fn make_file_path(folder_path: Option<&str>, file_name: &str) -> StoragePath {
match folder_path {
Some(fp) if !fp.is_empty() => StoragePath::from_string(&format!("{fp}/{file_name}")),
_ => StoragePath::from_string(file_name),
}
}
/// Look up the materialized folder path. O(1) — no recursive CTE.
async fn lookup_folder_path(
&self,
@@ -108,11 +99,10 @@ impl FileBlobWriteRepository {
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<File, DomainError> {
let storage_path = Self::make_file_path(folder_path.as_deref(), &name);
File::with_timestamps_blob_hash_and_provenance(
File::from_materialized_row(
id,
name,
storage_path,
folder_path.as_deref(),
size as u64,
mime_type,
folder_id,
@@ -32,11 +32,15 @@ use crate::domain::services::path_service::StoragePath;
/// Post-D7-step-6: `storage.folders.user_id` dropped, so the tuple
/// no longer carries it. The domain entity's `user_id` field is
/// populated with `None` at `row_to_folder` construction.
/// `id` / `parent_id` decode as binary `Uuid` (16 bytes on the wire vs 36
/// as `::text`, and the server skips the cast); `row_to_folder` renders
/// them to `String` once app-side — the round-6 `row_to_file` shape
/// (benches/ROUND6.md §10) applied to the folder listings.
type FolderRow = (
Uuid,
String,
String,
String,
Option<String>,
Option<Uuid>,
Uuid,
i64,
i64,
@@ -49,10 +53,10 @@ type FolderRow = (
/// the last element after the §14 provenance columns). Same
/// column set as [`FolderRow`] plus the trailing count.
type FolderRowPaginated = (
Uuid,
String,
String,
String,
Option<String>,
Option<Uuid>,
Uuid,
i64,
i64,
@@ -131,10 +135,10 @@ impl FolderDbRepository {
/// `Option<Uuid>` because the FK is `ON DELETE SET NULL`.
#[allow(clippy::too_many_arguments)]
fn row_to_folder(
id: String,
id: Uuid,
name: String,
path: String,
parent_id: Option<String>,
parent_id: Option<Uuid>,
drive_id: Uuid,
created_at: i64,
modified_at: i64,
@@ -142,12 +146,11 @@ impl FolderDbRepository {
created_by: Option<Uuid>,
updated_by: Option<Uuid>,
) -> Result<Folder, DomainError> {
let storage_path = StoragePath::from_string(&path);
Folder::with_timestamps_tree_and_provenance(
id,
Folder::from_materialized_row(
id.to_string(),
name,
storage_path,
parent_id,
path,
parent_id.map(|u| u.to_string()),
drive_id,
created_at as u64,
modified_at as u64,
@@ -171,7 +174,7 @@ impl FolderDbRepository {
let rows = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -236,12 +239,25 @@ impl FolderRepository for FolderDbRepository {
//
// RETURNING surfaces the two provenance columns so the built
// entity / DTO carries fresh values without a re-read.
let row = sqlx::query_as::<_, (String, String, i64, i64, i64, Option<Uuid>, Option<Uuid>)>(
let row = sqlx::query_as::<
_,
(
Uuid,
Option<Uuid>,
String,
i64,
i64,
i64,
Option<Uuid>,
Option<Uuid>,
),
>(
r#"
INSERT INTO storage.folders
(name, parent_id, drive_id, created_by, updated_by)
VALUES ($1, $2::uuid, $3, $4, $4)
RETURNING id::text,
RETURNING id,
parent_id,
path,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
@@ -269,16 +285,16 @@ impl FolderRepository for FolderDbRepository {
})?;
Self::row_to_folder(
row.0, name, row.1, parent_id, drive_id, row.2, row.3, row.4,
row.0, name, row.2, row.1, drive_id, row.3, row.4, row.5,
// Fresh from RETURNING — caller_id was bound to both columns.
row.5, row.6,
row.6, row.7,
)
}
async fn get_folder(&self, id: &str) -> Result<Folder, DomainError> {
let row = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -320,7 +336,7 @@ impl FolderRepository for FolderDbRepository {
// wrapper scoping post-D0).
let row = sqlx::query_as::<_, FolderRow>(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -346,7 +362,7 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRow> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -362,7 +378,7 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -405,7 +421,7 @@ impl FolderRepository for FolderDbRepository {
// top of `folder_repository.rs`. Frontend cross-references
// `/api/drives::caller_role` via `folder.drive_id`.
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -446,7 +462,7 @@ impl FolderRepository for FolderDbRepository {
let rows: Vec<FolderRowPaginated> = if let Some(pid) = parent_id {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -466,7 +482,7 @@ impl FolderRepository for FolderDbRepository {
} else {
sqlx::query_as(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
SELECT id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -501,6 +517,61 @@ impl FolderRepository for FolderDbRepository {
Ok((folders?, total))
}
/// Keyset sub-folder page: `name > $after ORDER BY name LIMIT $limit`,
/// one bounded index-range read off `idx_folders_unique_name` — the
/// cursor predicate is only emitted when a cursor exists (a bound
/// disjunction would block the index condition under generic plans,
/// same rule as `list_files_batch`). Root scope (`parent_id = None`)
/// keeps the trait's in-memory default: roots are one-per-drive, a
/// handful of rows.
async fn list_folders_batch(
&self,
parent_id: Option<&str>,
after_name: Option<&str>,
limit: usize,
) -> Result<Vec<Folder>, DomainError> {
let Some(pid) = parent_id else {
let mut all = self.list_folders(None).await?;
all.sort_by(|a, b| a.name().cmp(b.name()));
return Ok(all
.into_iter()
.filter(|f| after_name.is_none_or(|a| f.name() > a))
.take(limit)
.collect());
};
let cursor_pred = if after_name.is_some() {
"AND name > $3"
} else {
"AND $3::text IS NULL"
};
let sql = format!(
"SELECT id, name, path, parent_id, drive_id, \
EXTRACT(EPOCH FROM created_at)::bigint, \
EXTRACT(EPOCH FROM updated_at)::bigint, \
EXTRACT(EPOCH FROM tree_modified_at)::bigint, \
created_by, updated_by \
FROM storage.folders \
WHERE parent_id = $1::uuid AND NOT is_trashed \
{cursor_pred} \
ORDER BY name \
LIMIT $2"
);
let rows: Vec<FolderRow> = sqlx::query_as(&sql)
.bind(pid)
.bind(limit as i64)
.bind(after_name)
.fetch_all(self.pool())
.await
.map_err(|e| DomainError::internal_error("FolderDb", format!("batch: {e}")))?;
rows.into_iter()
.map(|(id, name, path, pid, did, ca, ma, tma, cb, ub)| {
Self::row_to_folder(id, name, path, pid, did, ca, ma, tma, cb, ub)
})
.collect()
}
/// Paginated companion to `list_root_folders_for_caller` — same
/// drive-membership predicate, adds LIMIT/OFFSET and an optional
/// window-function COUNT so total pages can be surfaced without a
@@ -513,7 +584,7 @@ impl FolderRepository for FolderDbRepository {
include_total: bool,
) -> Result<(Vec<Folder>, Option<usize>), DomainError> {
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -575,7 +646,7 @@ impl FolderRepository for FolderDbRepository {
UPDATE storage.folders
SET name = $1, updated_at = NOW(), updated_by = $3
WHERE id = $2::uuid AND NOT is_trashed
RETURNING id::text, name, path, parent_id::text, drive_id,
RETURNING id, name, path, parent_id, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
@@ -636,7 +707,7 @@ impl FolderRepository for FolderDbRepository {
updated_at = NOW(),
updated_by = $3
WHERE f.id = $2::uuid AND NOT f.is_trashed
RETURNING f.id::text, f.name, f.path, f.parent_id::text, f.drive_id,
RETURNING f.id, f.name, f.path, f.parent_id, f.drive_id,
EXTRACT(EPOCH FROM f.created_at)::bigint,
EXTRACT(EPOCH FROM f.updated_at)::bigint,
EXTRACT(EPOCH FROM f.tree_modified_at)::bigint,
@@ -935,7 +1006,7 @@ impl FolderRepository for FolderDbRepository {
/// Ordered by `fo.path` so callers can iterate in directory order.
#[allow(clippy::type_complexity)]
async fn list_subtree_folders(&self, folder_id: &str) -> Result<Vec<Folder>, DomainError> {
let sql = "SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
let sql = "SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -1002,7 +1073,7 @@ impl FolderRepository for FolderDbRepository {
if recursive {
// Recursive, no folder scope → ALL folders in caller's readable drives
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -1041,7 +1112,7 @@ impl FolderRepository for FolderDbRepository {
// the caller can read (parent_id already establishes the subtree).
let sql = if parent_id.is_some() {
format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -1061,7 +1132,7 @@ impl FolderRepository for FolderDbRepository {
_ => "",
};
format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -1134,7 +1205,7 @@ impl FolderRepository for FolderDbRepository {
};
let sql = format!(
"SELECT fo.id::text, fo.name, fo.path, fo.parent_id::text, \
"SELECT fo.id, fo.name, fo.path, fo.parent_id, \
fo.drive_id, \
EXTRACT(EPOCH FROM fo.created_at)::bigint, \
EXTRACT(EPOCH FROM fo.updated_at)::bigint, \
@@ -1178,31 +1249,39 @@ impl FolderRepository for FolderDbRepository {
parent_id: Option<&str>,
query: &str,
limit: usize,
caller_id: uuid::Uuid,
) -> Result<Vec<Folder>, DomainError> {
// Same drive-scope filter as `suggest_files_by_name` — closed as
// AuthZ audit finding #1 (2026-07-12). `CALLER_CAN_READ_DRIVE`
// aliases `storage.folders` as `fo`; the pre-fix query aliased it
// as an unqualified `storage.folders`, so this rewrite adds the
// `fo` alias in every branch.
let pattern = super::like_escape(query);
let limit_i64 = limit as i64;
let rows: Vec<FolderRow> = if let Some(pid) = parent_id {
sqlx::query_as(
sqlx::query_as(&format!(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id = $1::uuid
AND NOT is_trashed
AND name ILIKE $2
SELECT fo.id, fo.name, fo.path, fo.parent_id, fo.drive_id,
EXTRACT(EPOCH FROM fo.created_at)::bigint,
EXTRACT(EPOCH FROM fo.updated_at)::bigint,
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint,
fo.created_by, fo.updated_by
FROM storage.folders fo
WHERE {CALLER_CAN_READ_DRIVE}
AND fo.parent_id = $2::uuid
AND NOT fo.is_trashed
AND fo.name ILIKE $3
ORDER BY CASE
WHEN name ILIKE $3 THEN 0
WHEN name ILIKE $3 || '%' THEN 1
WHEN fo.name ILIKE $4 THEN 0
WHEN fo.name ILIKE $4 || '%' THEN 1
ELSE 2
END,
name
LIMIT $4
"#,
)
fo.name
LIMIT $5
"#
))
.bind(caller_id)
.bind(pid)
.bind(&pattern)
.bind(query)
@@ -1210,26 +1289,28 @@ impl FolderRepository for FolderDbRepository {
.fetch_all(self.pool())
.await
} else {
sqlx::query_as(
sqlx::query_as(&format!(
r#"
SELECT id::text, name, path, parent_id::text, drive_id,
EXTRACT(EPOCH FROM created_at)::bigint,
EXTRACT(EPOCH FROM updated_at)::bigint,
EXTRACT(EPOCH FROM tree_modified_at)::bigint,
created_by, updated_by
FROM storage.folders
WHERE parent_id IS NULL
AND NOT is_trashed
AND name ILIKE $1
SELECT fo.id, fo.name, fo.path, fo.parent_id, fo.drive_id,
EXTRACT(EPOCH FROM fo.created_at)::bigint,
EXTRACT(EPOCH FROM fo.updated_at)::bigint,
EXTRACT(EPOCH FROM fo.tree_modified_at)::bigint,
fo.created_by, fo.updated_by
FROM storage.folders fo
WHERE {CALLER_CAN_READ_DRIVE}
AND fo.parent_id IS NULL
AND NOT fo.is_trashed
AND fo.name ILIKE $2
ORDER BY CASE
WHEN name ILIKE $2 THEN 0
WHEN name ILIKE $2 || '%' THEN 1
WHEN fo.name ILIKE $3 THEN 0
WHEN fo.name ILIKE $3 || '%' THEN 1
ELSE 2
END,
name
LIMIT $3
"#,
)
fo.name
LIMIT $4
"#
))
.bind(caller_id)
.bind(&pattern)
.bind(query)
.bind(limit_i64)
@@ -1391,13 +1472,6 @@ impl FolderDbRepository {
WHERE fm.folder_id = $1::uuid AND NOT fm.is_trashed
"#;
let cte_inner = match (include_folders, include_files) {
(true, true) => format!("{folder_branch} UNION ALL {file_branch}"),
(true, false) => folder_branch.to_owned(),
(false, true) => file_branch.to_owned(),
(false, false) => unreachable!(),
};
// ── Cursor binds ─────────────────────────────────────────────────────
// $1 = parent_id $2 = cursor_str $3 = cursor_int
// $4 = cursor_ts $5 = cursor_id $6 = limit
@@ -1406,112 +1480,184 @@ impl FolderDbRepository {
let cursor_ts = cursor.and_then(|c| c.sort_ts);
let cursor_id = cursor.map(|c| c.resource_id);
// ── Sort-specific WHERE + ORDER BY ───────────────────────────────────
// Each arm produces two variants based on `reverse`.
// For "name": folder_first stays ASC in both directions (folders always
// precede files); only the alpha order within each group flips.
let (where_clause, order_clause) = match order_by {
// ── Per-branch cursor pushdown ───────────────────────────────────────
// The cursor is applied INSIDE each UNION-ALL branch as a sargable
// row-value comparison on base columns — not on the CTE's computed
// columns — and every branch pre-sorts and pre-limits, so Postgres
// reads O(limit) rows per branch instead of rescanning and
// top-N-sorting the entire folder on every page (19.5x on a
// 20k-entry folder, benches/LISTING-KEYSET.md). The "name" sort is
// served by the expression indexes idx_files_folder_lname /
// idx_folders_parent_lname (migration 20260918000000).
//
// Sort-key columns that are CONSTANT within a branch (folder_first,
// the folder branch's type_order = 0 and size = -1) are folded in
// Rust: depending on which group the cursor points into, the branch
// predicate shortens to a row-value over the remaining keys, the
// branch keeps all its rows, or the branch drops out entirely.
enum BranchCursor {
/// The cursor has moved past every row this branch can produce.
Drop,
/// Every row in this branch sorts after the cursor.
All,
/// Row-value comparison over the branch's non-constant sort keys.
Pred(String),
}
use BranchCursor::{All, Drop, Pred};
let has_cursor = cursor.is_some();
// (folder-branch cursor, file-branch cursor, per-branch ORDER BY on
// the branch's output aliases, outer merge ORDER BY)
let (folder_cur, file_cur, branch_order, outer_order) = match order_by {
"type" => {
if reverse {
(
r#"WHERE ($3::bigint IS NULL)
OR (type_order < $3)
OR (type_order = $3 AND sort_str < $2)
OR (type_order = $3 AND sort_str = $2 AND id < $5::uuid)"#,
"ORDER BY type_order DESC, sort_str DESC, id DESC",
)
let (op, ord) = if reverse {
("<", "ORDER BY type_order DESC, sort_str DESC, id DESC")
} else {
(
r#"WHERE ($3::bigint IS NULL)
OR (type_order > $3)
OR (type_order = $3 AND sort_str > $2)
OR (type_order = $3 AND sort_str = $2 AND id > $5::uuid)"#,
"ORDER BY type_order ASC, sort_str ASC, id ASC",
)
}
(">", "ORDER BY type_order ASC, sort_str ASC, id ASC")
};
let folder_cur = match cursor_int {
None => All,
// Folder rows have type_order = 0; a cursor sitting on a
// file (type_order > 0) either exhausts the folder group
// (ASC) or precedes all of it (DESC).
Some(c_to) if c_to > 0 => {
if reverse {
All
} else {
Drop
}
}
Some(_) => Pred(format!("(LOWER(f.name), f.id) {op} ($2, $5::uuid)")),
};
let file_cur = if has_cursor {
Pred(format!(
"(fm.category_order::bigint, LOWER(fm.name), fm.id) {op} ($3, $2, $5::uuid)"
))
} else {
All
};
(folder_cur, file_cur, ord, ord)
}
"modified_at" => {
if reverse {
(
r#"WHERE ($4::timestamptz IS NULL)
OR (modified_at > $4)
OR (modified_at = $4 AND id > $5::uuid)"#,
"ORDER BY modified_at ASC, id ASC",
)
let (op, ord) = if reverse {
(">", "ORDER BY modified_at ASC, id ASC")
} else {
(
r#"WHERE ($4::timestamptz IS NULL)
OR (modified_at < $4)
OR (modified_at = $4 AND id < $5::uuid)"#,
"ORDER BY modified_at DESC, id DESC",
)
}
("<", "ORDER BY modified_at DESC, id DESC")
};
let mk = |col: &str| {
if has_cursor {
Pred(format!("({col}.updated_at, {col}.id) {op} ($4, $5::uuid)"))
} else {
All
}
};
(mk("f"), mk("fm"), ord, ord)
}
"created_at" => {
if reverse {
(
r#"WHERE ($4::timestamptz IS NULL)
OR (created_at > $4)
OR (created_at = $4 AND id > $5::uuid)"#,
"ORDER BY created_at ASC, id ASC",
)
let (op, ord) = if reverse {
(">", "ORDER BY created_at ASC, id ASC")
} else {
(
r#"WHERE ($4::timestamptz IS NULL)
OR (created_at < $4)
OR (created_at = $4 AND id < $5::uuid)"#,
"ORDER BY created_at DESC, id DESC",
)
}
("<", "ORDER BY created_at DESC, id DESC")
};
let mk = |col: &str| {
if has_cursor {
Pred(format!("({col}.created_at, {col}.id) {op} ($4, $5::uuid)"))
} else {
All
}
};
(mk("f"), mk("fm"), ord, ord)
}
"size" => {
if reverse {
(
r#"WHERE ($3::bigint IS NULL)
OR (size < $3)
OR (size = $3 AND id < $5::uuid)"#,
"ORDER BY size DESC, id DESC",
)
let (op, ord) = if reverse {
("<", "ORDER BY size DESC, id DESC")
} else {
(
r#"WHERE ($3::bigint IS NULL)
OR (size > $3)
OR (size = $3 AND id > $5::uuid)"#,
"ORDER BY size ASC, id ASC",
)
}
(">", "ORDER BY size ASC, id ASC")
};
let folder_cur = match cursor_int {
None => All,
// Folder rows have size = -1; a cursor sitting on a file
// (size >= 0) exhausts the folder group (ASC) or precedes
// all of it (DESC).
Some(c_sz) if c_sz > -1 => {
if reverse {
All
} else {
Drop
}
}
Some(_) => Pred(format!("f.id {op} $5::uuid")),
};
let file_cur = if has_cursor {
Pred(format!("(fm.size::bigint, fm.id) {op} ($3, $5::uuid)"))
} else {
All
};
(folder_cur, file_cur, ord, ord)
}
_ => {
// "name" (default): folder_first stays ASC so folders always precede
// files; only the alpha order within each group flips when reversed.
if reverse {
(
r#"WHERE ($3::bigint IS NULL)
OR (folder_first::bigint > $3)
OR (folder_first::bigint = $3 AND sort_str < $2)
OR (folder_first::bigint = $3 AND sort_str = $2 AND id < $5::uuid)"#,
"ORDER BY folder_first ASC, sort_str DESC, id DESC",
)
// "name" (default): folder_first stays ASC so folders always
// precede files; only the alpha order within each group flips
// when reversed. cursor_int carries folder_first (0|1).
let op = if reverse { "<" } else { ">" };
let branch_ord = if reverse {
"ORDER BY sort_str DESC, id DESC"
} else {
(
r#"WHERE ($3::bigint IS NULL)
OR (folder_first::bigint > $3)
OR (folder_first::bigint = $3 AND sort_str > $2)
OR (folder_first::bigint = $3 AND sort_str = $2 AND id > $5::uuid)"#,
"ORDER BY folder_first ASC, sort_str ASC, id ASC",
)
}
"ORDER BY sort_str ASC, id ASC"
};
let outer_ord = if reverse {
"ORDER BY folder_first ASC, sort_str DESC, id DESC"
} else {
"ORDER BY folder_first ASC, sort_str ASC, id ASC"
};
let (folder_cur, file_cur) = match cursor_int {
None => (All, All),
// Cursor inside the folder group: folders continue after
// the row-value cursor; every file still follows.
Some(0) => (
Pred(format!("(LOWER(f.name), f.id) {op} ($2, $5::uuid)")),
All,
),
// Cursor inside the file group: the folder group is done.
Some(_) => (
Drop,
Pred(format!("(LOWER(fm.name), fm.id) {op} ($2, $5::uuid)")),
),
};
(folder_cur, file_cur, branch_ord, outer_ord)
}
};
let wrap = |branch: &str, cur: &BranchCursor| -> Option<String> {
let extra = match cur {
Drop => return None,
All => String::new(),
Pred(p) => format!(" AND {p}"),
};
Some(format!(
"(SELECT * FROM ({branch}{extra}) b {branch_order} LIMIT $6)"
))
};
let mut branches = Vec::with_capacity(2);
if include_folders && let Some(b) = wrap(folder_branch, &folder_cur) {
branches.push(b);
}
if include_files && let Some(b) = wrap(file_branch, &file_cur) {
branches.push(b);
}
// Every requested branch dropped out (e.g. folders-only listing with
// the cursor already past the folder group).
if branches.is_empty() {
return Ok(Vec::new());
}
let inner = branches.join(" UNION ALL ");
let sql = format!(
"WITH resources AS ({cte_inner}) \
SELECT resource_type, id, name, folder_id, mime_type, size, \
"SELECT resource_type, id, name, folder_id, mime_type, size, \
created_at, modified_at, drive_id, blob_hash, \
sort_str, type_order, folder_first \
FROM resources \
{where_clause} \
{order_clause} \
FROM ({inner}) r \
{outer_order} \
LIMIT $6"
);
@@ -180,6 +180,35 @@ impl PlaylistRepository for PlaylistPgRepository {
.map_err(|e| DomainError::new(ErrorKind::InternalError, "Playlist", e.to_string()))
}
async fn find_playlists_by_ids(&self, ids: &[Uuid]) -> PlaylistRepositoryResult<Vec<Playlist>> {
if ids.is_empty() {
return Ok(Vec::new());
}
let rows = sqlx::query_as::<_, PlaylistRow>(
"SELECT id, name, description, owner_id, is_public, cover_file_id, created_at, updated_at FROM audio.playlists WHERE id = ANY($1)",
)
.bind(ids)
.fetch_all(&*self.pool)
.await
.map_err(|e| DomainError::database_error(format!("Failed to find playlists: {}", e)))?;
rows.into_iter()
.map(|row| {
Playlist::with_id(
row.id,
row.name,
row.description,
row.owner_id,
row.is_public,
row.cover_file_id,
row.created_at,
row.updated_at,
)
.map_err(|e| DomainError::new(ErrorKind::InternalError, "Playlist", e.to_string()))
})
.collect()
}
async fn list_playlists_by_owner(
&self,
owner_id: Uuid,
+105 -43
View File
@@ -9,7 +9,7 @@ use std::pin::Pin;
use azure_storage::StorageCredentials;
use azure_storage_blobs::prelude::*;
use bytes::Bytes;
use futures::StreamExt;
use futures::{StreamExt, TryStreamExt};
use tokio::fs;
use crate::application::ports::blob_storage_ports::{
@@ -33,8 +33,21 @@ impl AzureBlobBackend {
StorageCredentials::access_key(&config.account_name, config.account_key.clone())
};
let container_client = ClientBuilder::new(&config.account_name, credentials)
.container_client(&config.container);
// Custom endpoint (Azurite emulator / private deployment /
// benches) mirrors S3's `endpoint_url`; default is the public
// cloud URL derived from the account name.
let container_client = match &config.endpoint_url {
Some(uri) => ClientBuilder::with_location(
azure_storage::CloudLocation::Custom {
account: config.account_name.clone(),
uri: uri.trim_end_matches('/').to_string(),
},
credentials,
)
.container_client(&config.container),
None => ClientBuilder::new(&config.account_name, credentials)
.container_client(&config.container),
};
Self {
container_client,
@@ -130,7 +143,9 @@ impl BlobStorageBackend for AzureBlobBackend {
return Ok(size);
}
client.put_block_blob(data.to_vec()).await.map_err(|e| {
// `Bytes` converts into `azure_core::Body` by reference count —
// the old `data.to_vec()` copied every chunk once more.
client.put_block_blob(data).await.map_err(|e| {
DomainError::internal_error("Azure", format!("Failed to upload blob {hash}: {e}"))
})?;
@@ -138,6 +153,26 @@ impl BlobStorageBackend for AzureBlobBackend {
})
}
/// Dedup settle path: PUT unconditionally. Content-addressed keys make
/// re-PUTs idempotent, so the `get_properties` probe
/// `put_blob_from_bytes` pays is a pure extra round-trip on every NEW
/// chunk (2 RTTs -> 1, benches/S3-PUT.md — same shape as S3).
fn put_blob_from_bytes_unsynced(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
let hash = hash.to_owned();
Box::pin(async move {
let client = self.blob_client(&hash);
let size = data.len() as u64;
client.put_block_blob(data).await.map_err(|e| {
DomainError::internal_error("Azure", format!("Failed to upload blob {hash}: {e}"))
})?;
Ok(size)
})
}
fn get_blob_stream(
&self,
hash: &str,
@@ -147,29 +182,46 @@ impl BlobStorageBackend for AzureBlobBackend {
Box::pin(async move {
let client = self.blob_client(&hash);
let mut result_data: Vec<u8> = Vec::new();
let mut stream = client.get().into_stream();
while let Some(response) = stream.next().await {
let response = response.map_err(|e| {
DomainError::new(
// The old implementation drained the ENTIRE blob into one
// `Vec<u8>` before yielding a single mega-chunk — whole-blob
// RAM residency per reader, and with `read_prefetch() = 8`
// up to 8 entire chunk-blobs resident at once during CDC
// reassembly. Now the SDK's page/body streams forward
// directly. The FIRST page is still awaited eagerly so a
// missing blob surfaces as the same up-front NotFound the
// old code produced; later pages/chunks map to io::Error
// items like every other backend's stream.
let mut pages = client.get().into_stream();
let first = match pages.next().await {
Some(Ok(response)) => response,
Some(Err(e)) => {
return Err(DomainError::new(
ErrorKind::NotFound,
"Azure",
format!("Failed to get blob {hash}: {e}"),
)
})?;
let mut body = response.data;
while let Some(chunk) = body.next().await {
let chunk = chunk.map_err(|e| {
DomainError::internal_error("Azure", format!("Stream read error: {e}"))
})?;
result_data.extend_from_slice(&chunk);
));
}
}
None => {
let empty: BlobStream =
Box::pin(futures::stream::once(async move { Ok(Bytes::new()) }));
return Ok(empty);
}
};
let stream: BlobStream = Box::pin(futures::stream::once(async move {
Ok(Bytes::from(result_data))
}));
let first_body = first.data.map(|chunk| {
chunk.map_err(|e| std::io::Error::other(format!("Stream read error: {e}")))
});
let tail = pages
.map(|page| match page {
Ok(response) => Ok(response.data.map(|chunk| {
chunk.map_err(|e| std::io::Error::other(format!("Stream read error: {e}")))
})),
Err(e) => Err(std::io::Error::other(format!(
"Failed to get blob page: {e}"
))),
})
.try_flatten();
let stream: BlobStream = Box::pin(first_body.chain(tail));
Ok(stream)
})
}
@@ -190,32 +242,42 @@ impl BlobStorageBackend for AzureBlobBackend {
None => azure_core::request_options::Range::new(start, u64::MAX),
};
let mut result_data: Vec<u8> = Vec::new();
let mut stream = client.get().range(range).into_stream();
while let Some(response) = stream.next().await {
let response = response.map_err(|e| {
DomainError::new(
// Same forwarding shape as `get_blob_stream` — a ranged read
// doubly so: the caller explicitly asked NOT to pay for the
// whole blob, yet the old code buffered the full range.
let mut pages = client.get().range(range).into_stream();
let first = match pages.next().await {
Some(Ok(response)) => response,
Some(Err(e)) => {
return Err(DomainError::new(
ErrorKind::NotFound,
"Azure",
format!("Failed to get blob range {hash}: {e}"),
)
})?;
let mut body = response.data;
while let Some(chunk) = body.next().await {
let chunk = chunk.map_err(|e| {
DomainError::internal_error(
"Azure",
format!("Stream range read error: {e}"),
)
})?;
result_data.extend_from_slice(&chunk);
));
}
}
None => {
let empty: BlobStream =
Box::pin(futures::stream::once(async move { Ok(Bytes::new()) }));
return Ok(empty);
}
};
let stream: BlobStream = Box::pin(futures::stream::once(async move {
Ok(Bytes::from(result_data))
}));
let first_body = first.data.map(|chunk| {
chunk.map_err(|e| std::io::Error::other(format!("Stream range read error: {e}")))
});
let tail = pages
.map(|page| match page {
Ok(response) => Ok(response.data.map(|chunk| {
chunk.map_err(|e| {
std::io::Error::other(format!("Stream range read error: {e}"))
})
})),
Err(e) => Err(std::io::Error::other(format!(
"Failed to get blob range page: {e}"
))),
})
.try_flatten();
let stream: BlobStream = Box::pin(first_body.chain(tail));
Ok(stream)
})
}
@@ -13,12 +13,14 @@ use std::sync::Arc;
use std::sync::atomic::{AtomicU64, Ordering};
use bytes::Bytes;
use dashmap::DashMap;
use lru::LruCache;
use std::num::NonZeroUsize;
use tokio::fs;
use tokio::io::{AsyncReadExt, AsyncSeekExt, AsyncWriteExt};
use tokio::sync::Mutex;
use tokio_util::io::ReaderStream;
use uuid::Uuid;
use crate::application::ports::blob_storage_ports::{
BlobStorageBackend, BlobStream, StorageHealthStatus,
@@ -56,6 +58,13 @@ pub struct CachedBlobBackend {
max_cache_bytes: u64,
index: Arc<Mutex<LruCache<String, CacheEntry>>>,
current_size: Arc<AtomicU64>,
/// Per-hash single-flight gates for cache misses. K concurrent cold
/// readers of one blob (e.g. a video player's parallel Range probes)
/// used to each download the FULL blob from the remote backend — and
/// race their writes on one shared `.tmp` path. The gate coalesces
/// them onto one fetch; waiters re-check the cache and serve locally
/// (16 fetches -> 1, benches/BLOB-CACHE.md).
inflight: Arc<DashMap<String, Arc<Mutex<()>>>>,
}
impl CachedBlobBackend {
@@ -70,6 +79,7 @@ impl CachedBlobBackend {
NonZeroUsize::new(1_000_000).unwrap(),
))),
current_size: Arc::new(AtomicU64::new(0)),
inflight: Arc::new(DashMap::new()),
}
}
@@ -150,6 +160,7 @@ impl BlobStorageBackend for CachedBlobBackend {
max_cache_bytes: self.max_cache_bytes,
index: self.index.clone(),
current_size: self.current_size.clone(),
inflight: self.inflight.clone(),
};
Box::pin(async move {
// Write to inner backend
@@ -172,25 +183,52 @@ impl BlobStorageBackend for CachedBlobBackend {
max_cache_bytes: self.max_cache_bytes,
index: self.index.clone(),
current_size: self.current_size.clone(),
inflight: self.inflight.clone(),
};
Box::pin(async move {
let size = inner.put_blob_from_bytes(&hash, data.clone()).await?;
// Also cache locally (best-effort): write bytes to cache path
let dest = self_ref.cached_path(&hash);
if let Some(parent) = dest.parent() {
let _ = fs::create_dir_all(parent).await;
}
let _ = fs::write(&dest, &data).await;
let data_len = data.len() as u64;
let mut idx = self_ref.index.lock().await;
if let Some(old) = idx.put(hash, CacheEntry { size: data_len }) {
self_ref.current_size.fetch_sub(old.size, Ordering::Relaxed);
}
self_ref.current_size.fetch_add(data_len, Ordering::Relaxed);
self_ref.cache_bytes_write_through(hash, &data).await;
Ok(size)
})
}
// Without this override the trait default would re-route the CDC chunk
// write through `put_blob_from_bytes` above, whose inner (synced) call
// pays the remote exists-probe per chunk. The local write-through cache
// population is kept identical — post-upload readers (thumbnail/EXIF/
// face hooks) hit the cache instead of re-fetching from the remote.
fn put_blob_from_bytes_unsynced(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
let inner = self.inner.clone();
let hash = hash.to_string();
let self_ref = CachedRef {
cache_dir: self.cache_dir.clone(),
max_cache_bytes: self.max_cache_bytes,
index: self.index.clone(),
current_size: self.current_size.clone(),
inflight: self.inflight.clone(),
};
Box::pin(async move {
let size = inner
.put_blob_from_bytes_unsynced(&hash, data.clone())
.await?;
self_ref.cache_bytes_write_through(hash, &data).await;
Ok(size)
})
}
// The durability barrier must reach the backend that buffered the
// unsynced writes; the local cache copy is disposable and needs none.
fn sync_blobs(
&self,
hashes: &[String],
) -> Pin<Box<dyn std::future::Future<Output = Result<(), DomainError>> + Send + '_>> {
self.inner.sync_blobs(hashes)
}
fn get_blob_stream(
&self,
hash: &str,
@@ -203,6 +241,7 @@ impl BlobStorageBackend for CachedBlobBackend {
let cache_dir = self.cache_dir.clone();
let max_cache_bytes = self.max_cache_bytes;
let current_size = self.current_size.clone();
let inflight = self.inflight.clone();
Box::pin(async move {
// Check cache presence (and bump LRU recency) under a brief lock,
// then release it BEFORE touching the filesystem so concurrent
@@ -219,14 +258,17 @@ impl BlobStorageBackend for CachedBlobBackend {
}
}
// Cache miss — fetch from inner, spool to cache
// Cache miss — fetch from inner (single-flight), spool to cache
let self_ref = CachedRef {
cache_dir,
max_cache_bytes,
index: index.clone(),
current_size: current_size.clone(),
inflight,
};
let dest = self_ref.fetch_and_cache_static(&hash, &*inner).await?;
let dest = self_ref
.fetch_and_cache_singleflight(&hash, &*inner, &cached)
.await?;
let file = fs::File::open(&dest).await.map_err(|e| {
DomainError::internal_error("BlobCache", format!("re-open cached: {e}"))
})?;
@@ -249,6 +291,7 @@ impl BlobStorageBackend for CachedBlobBackend {
let cache_dir = self.cache_dir.clone();
let max_cache_bytes = self.max_cache_bytes;
let current_size = self.current_size.clone();
let inflight = self.inflight.clone();
Box::pin(async move {
// Check cache presence (and bump LRU recency) under a brief lock,
// then release it BEFORE the open()/seek() syscalls so concurrent
@@ -271,14 +314,19 @@ impl BlobStorageBackend for CachedBlobBackend {
}
}
// Cache miss — fetch full blob into cache, then serve range
// Cache miss — fetch full blob into cache (single-flight: a
// player's parallel cold Range probes coalesce onto ONE remote
// download), then serve the range locally.
let self_ref = CachedRef {
cache_dir,
max_cache_bytes,
index: index.clone(),
current_size: current_size.clone(),
inflight,
};
let dest = self_ref.fetch_and_cache_static(&hash, &*inner).await?;
let dest = self_ref
.fetch_and_cache_singleflight(&hash, &*inner, &cached)
.await?;
let mut file = fs::File::open(&dest)
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("re-open: {e}")))?;
@@ -406,6 +454,7 @@ struct CachedRef {
max_cache_bytes: u64,
index: Arc<Mutex<LruCache<String, CacheEntry>>>,
current_size: Arc<AtomicU64>,
inflight: Arc<DashMap<String, Arc<Mutex<()>>>>,
}
impl CachedRef {
@@ -414,6 +463,55 @@ impl CachedRef {
self.cache_dir.join(prefix).join(format!("{hash}.blob"))
}
/// Best-effort write-through cache population shared by both blob-bytes
/// PUT paths. Deliberately no eviction sweep here — the byte budget is
/// enforced on read-miss inserts (`insert_into_cache_static`), matching
/// the historical write-path behavior.
async fn cache_bytes_write_through(&self, hash: String, data: &Bytes) {
let dest = self.cached_path(&hash);
if let Some(parent) = dest.parent() {
let _ = fs::create_dir_all(parent).await;
}
let _ = fs::write(&dest, data).await;
let data_len = data.len() as u64;
let mut idx = self.index.lock().await;
if let Some(old) = idx.put(hash, CacheEntry { size: data_len }) {
self.current_size.fetch_sub(old.size, Ordering::Relaxed);
}
self.current_size.fetch_add(data_len, Ordering::Relaxed);
}
/// Single-flight wrapper around [`Self::fetch_and_cache_static`]: the
/// first caller for a hash becomes the leader and downloads; concurrent
/// callers queue on the per-hash gate, then re-check the cache and serve
/// the leader's file without touching the remote backend. Errors are not
/// cached — the gate entry is dropped, so the next caller retries.
async fn fetch_and_cache_singleflight(
&self,
hash: &str,
inner: &dyn BlobStorageBackend,
cached: &Path,
) -> Result<PathBuf, DomainError> {
let gate = self
.inflight
.entry(hash.to_string())
.or_insert_with(|| Arc::new(Mutex::new(())))
.clone();
let _guard = gate.lock().await;
// Re-check under the gate: if we queued behind the leader, the blob
// is on disk now and this turns into a local open.
if self.index.lock().await.get(hash).is_some() && fs::metadata(cached).await.is_ok() {
return Ok(cached.to_path_buf());
}
let result = self.fetch_and_cache_static(hash, inner).await;
// Drop the gate whether we succeeded or failed; a late-arriving
// caller after an error creates a fresh gate and retries the fetch.
self.inflight.remove(hash);
result
}
/// Pop LRU entries until the cache is back within its byte budget,
/// returning the on-disk paths of the evicted blobs.
///
@@ -486,31 +584,51 @@ impl CachedRef {
})?;
}
let tmp = dest.with_extension("tmp");
let mut file = fs::File::create(&tmp)
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("create tmp: {e}")))?;
use futures::StreamExt;
let mut stream = stream;
let mut total = 0u64;
while let Some(chunk) = stream.next().await {
let bytes = chunk.map_err(|e| {
DomainError::internal_error("BlobCache", format!("stream read: {e}"))
// Unique temp name: even if two fetches for one hash ever race
// (e.g. across processes sharing a cache dir), each writes its own
// inode and the rename is atomic — a torn/interleaved file can
// never land at the final path.
let tmp = dest.with_extension(format!("{}.tmp", Uuid::new_v4()));
let write_result: Result<u64, DomainError> = async {
let mut file = fs::File::create(&tmp).await.map_err(|e| {
DomainError::internal_error("BlobCache", format!("create tmp: {e}"))
})?;
total += bytes.len() as u64;
file.write_all(&bytes)
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("write: {e}")))?;
}
file.flush()
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("flush: {e}")))?;
drop(file);
fs::rename(&tmp, &dest)
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("rename: {e}")))?;
use futures::StreamExt;
let mut stream = stream;
let mut total = 0u64;
while let Some(chunk) = stream.next().await {
let bytes = chunk.map_err(|e| {
DomainError::internal_error("BlobCache", format!("stream read: {e}"))
})?;
total += bytes.len() as u64;
file.write_all(&bytes)
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("write: {e}")))?;
}
file.flush()
.await
.map_err(|e| DomainError::internal_error("BlobCache", format!("flush: {e}")))?;
Ok(total)
}
.await;
let total = match write_result {
Ok(total) => total,
Err(e) => {
// Unique tmp names never get overwritten by a later fetch —
// reap the partial file instead of leaking it.
let _ = fs::remove_file(&tmp).await;
return Err(e);
}
};
if let Err(e) = fs::rename(&tmp, &dest).await {
let _ = fs::remove_file(&tmp).await;
return Err(DomainError::internal_error(
"BlobCache",
format!("rename: {e}"),
));
}
let to_evict = {
let mut idx = self.index.lock().await;
@@ -834,8 +834,7 @@ impl ChunkedUploadService {
let data_clone = data.clone(); // Bytes::clone is O(1) — just an Arc increment
let actual_checksum = tokio::task::spawn_blocking(move || {
use md5::{Digest, Md5};
let hash = Md5::digest(&data_clone);
hash.iter().map(|b| format!("{b:02x}")).collect::<String>()
crate::common::fmt::hex_lower(&Md5::digest(&data_clone))
})
.await
.map_err(|e| format!("MD5 checksum task failed: {e}"))?;
+77 -40
View File
@@ -1747,18 +1747,24 @@ impl DedupService {
/// remote object stores where overlapping fetches hide per-chunk latency).
/// Shared by [`Self::read_blob_stream`] and [`Self::read_blob_bytes`] so both
/// build the chunk stream identically from a manifest's `chunk_hashes`.
/// Takes the shared manifest `Arc` and iterates its hashes by index —
/// the old `Vec<String>` signature forced every read to deep-clone the
/// whole hash list out of the cached manifest before the first byte
/// (N ~64-B String allocs per read of an N-chunk file); the per-chunk
/// `Arc` bump here is a single atomic increment.
fn stream_chunks(
&self,
chunk_hashes: Vec<String>,
manifest: Arc<ChunkManifest>,
) -> Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>> {
let prefetch = self.backend.read_prefetch().max(1);
let backend = self.backend.clone();
let chunk_stream = stream::iter(chunk_hashes)
.map(move |chunk_hash| {
let chunk_stream = stream::iter(0..manifest.chunk_hashes.len())
.map(move |i| {
let backend = backend.clone();
let manifest = manifest.clone();
async move {
backend
.get_blob_stream(&chunk_hash)
.get_blob_stream(&manifest.chunk_hashes[i])
.await
.map_err(|e| std::io::Error::other(e.to_string()))
}
@@ -1771,31 +1777,56 @@ impl DedupService {
/// Cached manifest fetch for the read path (see the `manifest_cache`
/// field docs). `None` = legacy whole-file blob — never cached, so a
/// background rechunk that creates a manifest is honoured immediately.
///
/// Misses are single-flighted through `try_get_with`: K concurrent cold
/// readers of one newly-hot file (e.g. parallel Range probes on a big
/// video) coalesce onto ONE manifest SELECT instead of K. The
/// positive-only contract is preserved by routing "no manifest row" and
/// DB failures through the loader's error channel, which moka never
/// caches. The zero-alloc `get` fast path stays in front so warm reads
/// don't pay the owned-key clone `try_get_with` requires.
async fn manifest_cached(&self, hash: &str) -> Result<Option<Arc<ChunkManifest>>, DomainError> {
if let Some(m) = self.manifest_cache.get(hash).await {
return Ok(Some(m));
}
let row = sqlx::query_as::<_, (Vec<String>, Vec<i64>, i64)>(
"SELECT chunk_hashes, chunk_sizes, total_size
FROM storage.chunk_manifests WHERE file_hash = $1",
)
.bind(hash)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("Dedup", format!("Manifest lookup: {}", e)))?;
match row {
Some((chunk_hashes, chunk_sizes, total_size)) => {
let m = Arc::new(ChunkManifest {
chunk_hashes,
chunk_sizes,
total_size,
});
self.manifest_cache
.insert(hash.to_string(), m.clone())
.await;
Ok(Some(m))
}
None => Ok(None),
enum MissKind {
Legacy,
Db(String),
}
let pool = self.pool.clone();
let query_hash = hash.to_string();
let result = self
.manifest_cache
.try_get_with(hash.to_string(), async move {
let row = sqlx::query_as::<_, (Vec<String>, Vec<i64>, i64)>(
"SELECT chunk_hashes, chunk_sizes, total_size
FROM storage.chunk_manifests WHERE file_hash = $1",
)
.bind(&query_hash)
.fetch_optional(pool.as_ref())
.await
.map_err(|e| MissKind::Db(e.to_string()))?;
match row {
Some((chunk_hashes, chunk_sizes, total_size)) => Ok(Arc::new(ChunkManifest {
chunk_hashes,
chunk_sizes,
total_size,
})),
None => Err(MissKind::Legacy),
}
})
.await;
match result {
Ok(m) => Ok(Some(m)),
Err(miss) => match &*miss {
MissKind::Legacy => Ok(None),
MissKind::Db(msg) => Err(DomainError::internal_error(
"Dedup",
format!("Manifest lookup: {}", msg),
)),
},
}
}
@@ -1810,7 +1841,7 @@ impl DedupService {
) -> Result<Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>, DomainError>
{
match self.manifest_cached(hash).await? {
Some(m) => Ok(self.stream_chunks(m.chunk_hashes.clone())),
Some(m) => Ok(self.stream_chunks(m)),
// Legacy whole-file blob
None => self.backend.get_blob_stream(hash).await,
}
@@ -1829,10 +1860,10 @@ impl DedupService {
/// every full-blob read (e.g. 2N queries for an N-image gallery cold load).
pub async fn read_blob_bytes(&self, hash: &str) -> Result<Bytes, DomainError> {
let (mut stream, expected_size) = match self.manifest_cached(hash).await? {
Some(m) => (
self.stream_chunks(m.chunk_hashes.clone()),
m.total_size.max(0) as usize,
),
Some(m) => {
let expected = m.total_size.max(0) as usize;
(self.stream_chunks(m), expected)
}
None => {
// Legacy whole-file blob: size + stream straight from the backend.
let size = self.backend.blob_size(hash).await? as usize;
@@ -1863,16 +1894,17 @@ impl DedupService {
) -> Result<Pin<Box<dyn Stream<Item = Result<Bytes, std::io::Error>> + Send>>, DomainError>
{
if let Some(m) = self.manifest_cached(hash).await? {
let (chunk_hashes, chunk_sizes, total_size) =
(&m.chunk_hashes, &m.chunk_sizes, m.total_size);
let end = end.unwrap_or(total_size as u64);
let end = end.unwrap_or(m.total_size as u64);
// Calculate which chunks overlap [start, end)
// Calculate which chunks overlap [start, end). Chunks are
// addressed by manifest INDEX (the hash is read through the
// shared `Arc` at fetch time) — a `bytes=0-` probe of an
// N-chunk video used to clone all N hash Strings here.
let mut offset: u64 = 0;
// (chunk_hash, range_start_within_chunk, range_end_within_chunk)
let mut selected: Vec<(String, u64, Option<u64>)> = Vec::new();
// (chunk_index, range_start_within_chunk, range_end_within_chunk)
let mut selected: Vec<(usize, u64, Option<u64>)> = Vec::new();
for (i, &chunk_size) in chunk_sizes.iter().enumerate() {
for (i, &chunk_size) in m.chunk_sizes.iter().enumerate() {
let chunk_size = chunk_size as u64;
let chunk_end = offset + chunk_size;
@@ -1883,7 +1915,7 @@ impl DedupService {
} else {
None
};
selected.push((chunk_hashes[i].clone(), range_start, range_end));
selected.push((i, range_start, range_end));
}
offset += chunk_size;
@@ -1897,11 +1929,16 @@ impl DedupService {
let prefetch = self.backend.read_prefetch().max(1);
let backend = self.backend.clone();
let chunk_stream = stream::iter(selected)
.map(move |(chunk_hash, range_start, range_end)| {
.map(move |(i, range_start, range_end)| {
let backend = backend.clone();
let manifest = m.clone();
async move {
backend
.get_blob_range_stream(&chunk_hash, range_start, range_end)
.get_blob_range_stream(
&manifest.chunk_hashes[i],
range_start,
range_end,
)
.await
.map_err(|e| std::io::Error::other(e.to_string()))
}
@@ -28,11 +28,35 @@ fn is_image(content_type: &str) -> bool {
content_type.starts_with("image/")
}
/// Concurrent index-task budget. Env override
/// `OXICLOUD_FACES_INDEX_CONCURRENCY`, else the effective core count —
/// each task is a full-image read + decode + ONNX inference, so more
/// permits than cores only adds RAM pressure, not throughput.
fn max_concurrent_index() -> usize {
std::env::var("OXICLOUD_FACES_INDEX_CONCURRENCY")
.ok()
.and_then(|v| v.parse().ok())
.filter(|&n: &usize| n > 0)
.unwrap_or_else(|| {
std::thread::available_parallelism()
.map(|n| n.get())
.unwrap_or(2)
})
}
pub struct FaceIndexingService {
pool: Arc<PgPool>,
repo: Arc<FacePgRepository>,
analyzer: Arc<dyn FaceAnalyzerPort>,
blob_root: PathBuf,
/// Bounds concurrent indexing tasks. The lifecycle hooks spawn one
/// task per uploaded/copied image with no ceiling, so a bulk upload
/// used to fan out N simultaneous full-image reads + decodes +
/// inferences — peak RSS N × image size plus CPU thrash. Same
/// invariant as `ThumbnailService::decode_semaphore`: the permit is
/// acquired BEFORE the blob read, so peak memory is
/// `permits × image size` regardless of upload concurrency.
index_semaphore: Arc<tokio::sync::Semaphore>,
}
impl FaceIndexingService {
@@ -43,6 +67,7 @@ impl FaceIndexingService {
repo,
analyzer,
blob_root,
index_semaphore: Arc::new(tokio::sync::Semaphore::new(max_concurrent_index())),
}
}
@@ -60,7 +85,15 @@ impl FaceIndexingService {
let repo = self.repo.clone();
let analyzer = self.analyzer.clone();
let blob_path = self.blob_path(&blob_hash);
let semaphore = self.index_semaphore.clone();
tokio::spawn(async move {
// Queue behind the concurrency budget BEFORE touching the
// blob — excess tasks wait holding only this tiny future,
// not a decoded image.
let _permit = semaphore
.acquire_owned()
.await
.expect("face index semaphore never closes");
if delete_first {
let _ = repo.delete_faces_for_file(file_id).await;
}
@@ -128,18 +128,41 @@ async fn fsync_paths_parallel(paths: Vec<PathBuf>, strict: bool) -> Result<(), D
/// (fsync now vs. deferred batch sync), or `None` when the blob already
/// existed (idempotent skip — content-addressed, so identical by definition).
async fn write_blob_bytes(blob_path: &Path, data: &Bytes) -> Result<Option<File>, DomainError> {
if fs::try_exists(blob_path).await.unwrap_or(false) {
return Ok(None);
}
let mut file = fs::File::create(blob_path).await.map_err(|e| {
DomainError::internal_error("Blob", format!("Failed to create blob file: {}", e))
})?;
// One atomic O_CREAT|O_EXCL open replaces the old stat-then-create pair:
// `AlreadyExists` IS the idempotent skip (content-addressed names mean an
// existing file has identical content), saving a syscall + a blocking-pool
// dispatch on every new chunk of every upload.
let mut file = match fs::File::options()
.write(true)
.create_new(true)
.open(blob_path)
.await
{
Ok(f) => f,
Err(e) if e.kind() == std::io::ErrorKind::AlreadyExists => return Ok(None),
Err(e) => {
return Err(DomainError::internal_error(
"Blob",
format!("Failed to create blob file: {}", e),
));
}
};
file.write_all(data).await.map_err(|e| {
DomainError::internal_error("Blob", format!("Failed to write blob from bytes: {}", e))
})?;
Ok(Some(file))
}
/// Bench-only public wrapper (feature = "bench") over the private chunk
/// writer so `examples/bench_storage_micro.rs` can A/B the open strategy.
#[cfg(feature = "bench")]
pub async fn write_blob_bytes_for_bench(
blob_path: &Path,
data: &Bytes,
) -> Result<Option<File>, DomainError> {
write_blob_bytes(blob_path, data).await
}
/// Compile-time lookup table for the 256 two-digit lowercase hex prefixes ("00"…"ff").
static HEX_PREFIXES: [&str; 256] = [
"00", "01", "02", "03", "04", "05", "06", "07", "08", "09", "0a", "0b", "0c", "0d", "0e", "0f",
+236 -56
View File
@@ -103,6 +103,28 @@ const DRIVE_POLICIES_CACHE_CAPACITY: u64 = 100_000;
/// effective within a minute on the hot path.
const DRIVE_POLICIES_CACHE_TTL: Duration = Duration::from_secs(30);
/// `cascade_grant_cache` bound: entries are
/// `((Subject, Resource, Permission), bool)` — a few tens of bytes each. A
/// shared photo album is one folder grant serving hundreds of file checks, so
/// 100k comfortably covers the working set of active shared-resource viewers.
const CASCADE_GRANT_CACHE_CAPACITY: u64 = 100_000;
/// `cascade_grant_cache` TTL. Direct grant mutations on the file/folder
/// (`set_role` / `clear_role`) explicitly invalidate the whole cache, so the
/// TTL is the self-heal net for the *indirect* paths — a group-membership
/// change, a resource move, or a grant's `expires_at` passing — exactly as
/// `drive_role_cache` leans on its TTL for group changes "rather than a deep
/// invalidation tree". Short enough that any such change takes effect in <1 min.
const CASCADE_GRANT_CACHE_TTL: Duration = Duration::from_secs(30);
/// `file_parent_cache` bound/TTL: `file_id → Option<folder_id>` point rows
/// (~50 B each) resolved on the file-cascade path so an N-file album pays
/// ONE folder-cascade query instead of N (ROUND9). Parentage changes only
/// on move — an indirect path the cascade cache already self-heals via TTL,
/// so the same 30 s window applies (grant writes don't alter parentage and
/// need no flush here).
const FILE_PARENT_CACHE_CAPACITY: u64 = 100_000;
const FILE_PARENT_CACHE_TTL: Duration = Duration::from_secs(30);
pub struct PgAclEngine {
pool: Arc<PgPool>,
folder_repo: Arc<FolderDbRepository>,
@@ -160,6 +182,42 @@ pub struct PgAclEngine {
/// returns, so the next check sees the fresh values. Short 30 s TTL
/// as the self-heal net for direct-SQL edits and migration backfills.
drive_policies_cache: Cache<Uuid, DrivePolicies>,
/// Memoise the File/Folder **grant-cascade** decision
/// `(subject, resource, permission) → bool` — the result of the
/// `role_grants` + folder-ancestor (`lpath @>`) cascade that
/// `check_inner` falls through to when the drive-role precheck doesn't
/// cover the caller. This is the per-request query a shared-album
/// recipient (a grant on the containing folder, no drive membership) pays
/// for **every thumbnail** — and browsers revalidate immutable thumbnails
/// constantly, so the same `(subject, file, Read)` decision is recomputed
/// again and again. Cached here it costs one query then in-memory hits.
///
/// Only reached AFTER the drive-role precheck fails, so a caller who is a
/// drive member short-circuits above and never populates a (possibly
/// negative) entry here — a later drive grant can't be shadowed by a stale
/// cascade `false`.
///
/// **Invalidation**: explicit `invalidate_all` on every File/Folder
/// `set_role` / `clear_role` (the direct share/revoke path — infrequent
/// relative to thumbnail reads, so a full flush is cheap and keeps
/// revocation immediate). The indirect paths — group-membership changes,
/// resource moves that change ancestry, grant `expires_at` expiry — are
/// caught by the 30 s TTL, matching `drive_role_cache`'s documented
/// convention.
///
/// **Safety**: the check still runs on every request (the ordering is
/// unchanged — authz is never skipped); only its *result* is memoised, and
/// only positively-or-negatively for at most the TTL. A revoke via
/// `clear_role` flushes immediately; anything missed self-heals in ≤30 s.
cascade_grant_cache: Cache<(Subject, Resource, Permission), bool>,
/// `file_id → Option<parent folder_id>` memo for the file-cascade
/// decomposition (see `cascade_grant_cached`): resolving the parent lets
/// a whole folder's files share ONE folder-cascade decision, so a shared
/// album's first view runs one ltree query instead of one per file.
/// Grant writes don't affect parentage — only the TTL applies (moves are
/// an indirect path, same self-heal contract as `cascade_grant_cache`).
file_parent_cache: Cache<Uuid, Option<Uuid>>,
}
impl PgAclEngine {
@@ -197,6 +255,14 @@ impl PgAclEngine {
.max_capacity(DRIVE_POLICIES_CACHE_CAPACITY)
.time_to_live(DRIVE_POLICIES_CACHE_TTL)
.build(),
cascade_grant_cache: Cache::builder()
.max_capacity(CASCADE_GRANT_CACHE_CAPACITY)
.time_to_live(CASCADE_GRANT_CACHE_TTL)
.build(),
file_parent_cache: Cache::builder()
.max_capacity(FILE_PARENT_CACHE_CAPACITY)
.time_to_live(FILE_PARENT_CACHE_TTL)
.build(),
}
}
@@ -267,6 +333,14 @@ impl PgAclEngine {
.max_capacity(1)
.time_to_live(Duration::from_secs(1))
.build(),
cascade_grant_cache: Cache::builder()
.max_capacity(1)
.time_to_live(Duration::from_secs(1))
.build(),
file_parent_cache: Cache::builder()
.max_capacity(1)
.time_to_live(Duration::from_secs(1))
.build(),
}
}
@@ -358,6 +432,20 @@ impl PgAclEngine {
self.owner_cache.invalidate_all();
}
/// Flush the entire `cascade_grant_cache`. Called on every File/Folder
/// `set_role` / `clear_role` — the direct share/revoke path. A resource
/// grant can widen (or, via ancestry, narrow) the cascade decision for an
/// unbounded set of descendant files, and the cache is keyed by the
/// decision — not the grant — so we can't target the affected entries
/// without walking the subtree. A full flush is correct and cheap here:
/// grant mutations are rare next to the thumbnail reads the cache serves,
/// and it keeps a revoke immediate. Indirect changes (group membership,
/// resource moves, grant expiry) are left to the 30 s TTL, mirroring
/// `drive_role_cache`.
pub async fn invalidate_cascade_grant_cache_all(&self) {
self.cascade_grant_cache.invalidate_all();
}
/// Sibling of [`Self::invalidate_drive_role_cache_for_drive`] keyed by
/// subject rather than drive. Used by the user-deleted lifecycle hook
/// to reap every cached "user X → drive Y = role R" entry after the
@@ -654,11 +742,11 @@ impl PgAclEngine {
Ok(exists.is_some())
}
/// Cascading check for files: either a direct file grant OR a grant on
/// any ancestor folder of the file's containing folder. See
/// `folder_cascade_grant_exists` for the meaning of `subject_types` /
/// `subject_ids` and the D-Prep role-array migration.
async fn file_cascade_grant_exists(
/// Direct file grant only — the first branch of the historical file
/// cascade UNION, split out so `cascade_grant_cached` can amortize the
/// ancestor-folder branch per FOLDER (see the `Resource::File` arm).
/// A plain indexed `role_grants` point lookup, no ltree join.
async fn file_direct_grant_exists(
&self,
subject_types: &[&str],
subject_ids: &[Uuid],
@@ -671,30 +759,12 @@ impl PgAclEngine {
let exists: Option<i32> = sqlx::query_scalar(
r#"
SELECT 1
FROM (
-- direct file grant
SELECT 1
FROM storage.role_grants
WHERE subject_type = ANY($1)
AND subject_id = ANY($2)
AND role = ANY($3::storage.grant_role[])
AND resource_type = 'file' AND resource_id = $4
AND (expires_at IS NULL OR expires_at > NOW())
UNION ALL
-- cascading from any ancestor folder of the file's containing folder
SELECT 1
FROM storage.role_grants g
JOIN storage.folders gf ON gf.id = g.resource_id
JOIN storage.files target_f ON target_f.id = $4
WHERE g.subject_type = ANY($1)
AND g.subject_id = ANY($2)
AND g.role = ANY($3::storage.grant_role[])
AND g.resource_type = 'folder'
AND (g.expires_at IS NULL OR g.expires_at > NOW())
AND target_f.folder_id IS NOT NULL
AND gf.lpath @> (SELECT lpath FROM storage.folders
WHERE id = target_f.folder_id)
) any_match
FROM storage.role_grants
WHERE subject_type = ANY($1)
AND subject_id = ANY($2)
AND role = ANY($3::storage.grant_role[])
AND resource_type = 'file' AND resource_id = $4
AND (expires_at IS NULL OR expires_at > NOW())
LIMIT 1
"#,
)
@@ -704,11 +774,127 @@ impl PgAclEngine {
.bind(file_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("PgAcl", format!("file cascade: {e}")))?;
.map_err(|e| DomainError::internal_error("PgAcl", format!("file direct grant: {e}")))?;
Ok(exists.is_some())
}
/// Memoised `file_id → Option<parent folder_id>` point read backing the
/// file-cascade decomposition. `None` covers both a missing row and a
/// NULL `folder_id` — in either case only the direct-file-grant branch
/// can match (mirroring the historical UNION's `folder_id IS NOT NULL`
/// guard).
async fn file_parent_folder_cached(
&self,
file_id: Uuid,
counters: &QueryCounters,
) -> Result<Option<Uuid>, DomainError> {
if let Some(parent) = self.file_parent_cache.get(&file_id).await {
return Ok(parent);
}
counters.sql_queries.fetch_add(1, Ordering::Relaxed);
let parent: Option<Option<Uuid>> =
sqlx::query_scalar("SELECT folder_id FROM storage.files WHERE id = $1")
.bind(file_id)
.fetch_optional(self.pool.as_ref())
.await
.map_err(|e| DomainError::internal_error("PgAcl", format!("file parent: {e}")))?;
let parent = parent.flatten();
self.file_parent_cache.insert(file_id, parent).await;
Ok(parent)
}
/// Cache-aware wrapper over the File/Folder grant cascade. Serves the
/// memoised `(subject, resource, permission)` decision when warm; on a
/// miss it expands the subject set (itself cached) and runs the matching
/// cascade query, then stores the result. Only invoked after the drive-role
/// precheck fails, so it never caches a decision a drive grant would have
/// satisfied — a later drive grant short-circuits above this cache.
///
/// **File decomposition (ROUND9).** The historical file query was one
/// UNION: `direct file grant ∨ grant on any ancestor of the parent
/// folder` — one ltree join per file, so a shared N-photo album's FIRST
/// view ran N near-identical ancestor queries (round 8 memoised only the
/// per-file result, covering revalidation). The arm now resolves the
/// file's parent (memoised point read) and recurses into the FOLDER arm
/// for the ancestor half — one ltree query per folder, shared by every
/// sibling — falling back to the direct-file-grant lookup only when the
/// folder half denies. The decomposition is exactly the UNION split in
/// two: no decision changes, including the parentless edge (the UNION's
/// `folder_id IS NOT NULL` guard ≡ the direct-only fallback).
///
/// The result is a pure function of the subject's group expansion + the
/// resource's grants + folder ancestry; `invalidate_cascade_grant_cache_all`
/// (on File/Folder grant writes — it holds file AND folder decisions in
/// the same map) and the 30 s TTL (indirect changes, incl. moves for the
/// parent memo) keep it fresh. See the `cascade_grant_cache` field doc.
async fn cascade_grant_cached(
&self,
subject: Subject,
resource: Resource,
permission: Permission,
counters: &QueryCounters,
) -> Result<bool, DomainError> {
if let Some(allowed) = self
.cascade_grant_cache
.get(&(subject, resource, permission))
.await
{
counters.cache_hit.fetch_add(1, Ordering::Relaxed);
return Ok(allowed);
}
let allowed = match resource {
Resource::Folder(id) => {
let (subject_types, subject_ids) =
self.subject_match_set(subject, counters).await?;
self.folder_cascade_grant_exists(
&subject_types,
&subject_ids,
permission,
id,
counters,
)
.await?
}
Resource::File(id) => {
// Ancestor half first — amortized to one query per FOLDER
// via the recursive Folder arm (its own cache entry).
let folder_allowed = match self.file_parent_folder_cached(id, counters).await? {
Some(parent) => {
Box::pin(self.cascade_grant_cached(
subject,
Resource::Folder(parent),
permission,
counters,
))
.await?
}
None => false,
};
if folder_allowed {
true
} else {
let (subject_types, subject_ids) =
self.subject_match_set(subject, counters).await?;
self.file_direct_grant_exists(
&subject_types,
&subject_ids,
permission,
id,
counters,
)
.await?
}
}
// Only File/Folder reach this helper (see `check_inner`).
_ => return Ok(false),
};
self.cascade_grant_cache
.insert((subject, resource, permission), allowed)
.await;
Ok(allowed)
}
/// Cached resolution of `(subject, drive_id) → Option<Role>` — the
/// strongest role the subject holds on the drive (direct + transitive
/// group grants collapsed). `None` means no qualifying grant; cached
@@ -972,32 +1158,15 @@ impl PgAclEngine {
}
match resource {
// File/Folder dispatch falls through to the cascade query —
// expand the subject set lazily here (it's cached) so the
// Drive branch below never pays for an expansion it doesn't need.
Resource::Folder(id) => {
let (subject_types, subject_ids) =
self.subject_match_set(subject, counters).await?;
self.folder_cascade_grant_exists(
&subject_types,
&subject_ids,
permission,
id,
counters,
)
.await
}
Resource::File(id) => {
let (subject_types, subject_ids) =
self.subject_match_set(subject, counters).await?;
self.file_cascade_grant_exists(
&subject_types,
&subject_ids,
permission,
id,
counters,
)
.await
// File/Folder dispatch falls through to the cascade query, now
// memoised: a shared-album recipient (folder grant, no drive
// membership) reaches this per thumbnail, and browsers revalidate
// thumbnails constantly, so the same decision is recomputed over
// and over. `cascade_grant_cached` serves it from memory after the
// first query; the check is unchanged (never skipped), only cached.
Resource::Folder(_) | Resource::File(_) => {
self.cascade_grant_cached(subject, resource, permission, counters)
.await
}
Resource::Drive(id) => {
// Same read_only gate as the File/Folder branch: a frozen
@@ -2413,6 +2582,12 @@ impl AuthorizationEngine for PgAclEngine {
if let Resource::Drive(drive_id) = resource {
self.invalidate_drive_role_cache_for_drive(drive_id).await;
}
// File/Folder grant write — a new share can widen the cascade
// decision for descendant files; flush the cascade cache so the next
// thumbnail/read check sees it immediately.
if matches!(resource, Resource::File(_) | Resource::Folder(_)) {
self.invalidate_cascade_grant_cache_all().await;
}
Self::row_to_grant(row)
}
@@ -2437,6 +2612,11 @@ impl AuthorizationEngine for PgAclEngine {
if let Resource::Drive(drive_id) = resource {
self.invalidate_drive_role_cache_for_drive(drive_id).await;
}
// Revoking a File/Folder share must stop passing the cascade check
// now, not in ≤30 s — flush the cascade cache (see `set_role`).
if matches!(resource, Resource::File(_) | Resource::Folder(_)) {
self.invalidate_cascade_grant_cache_all().await;
}
Ok(())
}
@@ -159,6 +159,43 @@ impl BlobStorageBackend for RetryBlobBackend {
})
}
// Without this override the trait default would re-route the CDC chunk
// write through `put_blob_from_bytes` above — reinstating the remote
// backend's exists-probe (HEAD/get_properties) per chunk that the
// `_unsynced` fast path exists to skip.
fn put_blob_from_bytes_unsynced(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
let inner = self.inner.clone();
let policy = self.policy.clone();
let hash = hash.to_string();
Box::pin(async move {
retry_async(
&policy,
&format!("put_blob_from_bytes_unsynced({hash})"),
|| {
let inner = inner.clone();
let hash = hash.clone();
let data = data.clone();
async move { inner.put_blob_from_bytes_unsynced(&hash, data).await }
},
)
.await
})
}
// Forwarded WITHOUT retry wrapping: a failed fsync must surface, not be
// re-issued — after an fsync error the kernel may have dropped the dirty
// pages, so a retried fsync can report success for data that was lost.
fn sync_blobs(
&self,
hashes: &[String],
) -> Pin<Box<dyn std::future::Future<Output = Result<(), DomainError>> + Send + '_>> {
self.inner.sync_blobs(hashes)
}
fn get_blob_stream(
&self,
hash: &str,
@@ -200,6 +200,38 @@ impl BlobStorageBackend for S3BlobBackend {
})
}
/// Dedup settle path: PUT unconditionally. Keys are content-addressed
/// (BLAKE3), so a re-PUT writes identical bytes — overwrite-safe
/// idempotency without the HEAD probe `put_blob_from_bytes` pays. The
/// dedup layer already filtered out chunks the database knows about,
/// so the probe was a pure extra round-trip on every NEW chunk of
/// every upload (2 RTTs -> 1, benches/S3-PUT.md).
fn put_blob_from_bytes_unsynced(
&self,
hash: &str,
data: Bytes,
) -> Pin<Box<dyn std::future::Future<Output = Result<u64, DomainError>> + Send + '_>> {
let hash = hash.to_owned();
Box::pin(async move {
let key = Self::object_key(&hash);
let size = data.len() as u64;
self.client
.put_object()
.bucket(&self.bucket)
.key(&key)
.body(ByteStream::from(data))
.send()
.await
.map_err(|e| {
DomainError::internal_error(
"S3",
format!("Failed to upload blob {}: {}", hash, e),
)
})?;
Ok(size)
})
}
fn get_blob_stream(
&self,
hash: &str,
+61 -131
View File
@@ -1,7 +1,7 @@
use axum::{
Router,
extract::{DefaultBodyLimit, Json, Multipart, Path, Query, State},
http::{HeaderMap, StatusCode},
http::StatusCode,
response::{
IntoResponse,
sse::{Event, KeepAlive, Sse},
@@ -27,8 +27,10 @@ use crate::application::ports::storage_ports::StorageUsagePort;
use crate::common::di::AppState;
use crate::domain::repositories::drive_repository::DriveRepository;
use crate::domain::services::authorization::{Resource, Subject};
use crate::interfaces::api::handlers::dedup_handler::{get_stats, recalculate_stats};
use crate::interfaces::api::handlers::search_handler::clear_search_cache;
use crate::interfaces::errors::AppError;
use crate::interfaces::middleware::admin::require_admin;
use crate::interfaces::middleware::auth::AuthUser;
use std::sync::Arc;
use uuid::Uuid;
@@ -89,6 +91,22 @@ pub fn admin_routes() -> Router<Arc<AppState>> {
.route("/plugins/{id}/logs/stream", get(stream_plugin_logs))
.route("/plugins/{id}/retention", get(get_plugin_retention))
.route("/plugins/{id}/retention", put(set_plugin_retention))
// Search — operator flush of the shared moka results cache
// (AuthZ audit #14, 2026-07-16). `invalidate_all()` semantics
// touch every tenant, so this is admin-only. Lived at
// `/api/search/cache` pre-2026-07-17; the URL now declares
// its admin intent up front.
.route("/search/cache", delete(clear_search_cache))
// Dedup — global storage stats + integrity recalculation
// (AuthZ audit #24 + #25, 2026-07-17). Both are operator-only
// observability / maintenance surfaces (blob-count-level data
// + verify_integrity sweep). Moved here from `/api/dedup/*`
// so the URL declares admin intent and the middleware layer
// enforces it — same pattern as `search/cache` above. The
// any-authenticated sibling routes (`/check`, `/check-batch`,
// `/blob/{hash}`) stay at `/api/dedup/*`.
.route("/dedup/stats", get(get_stats))
.route("/dedup/recalculate", post(recalculate_stats))
// SMTP diagnostics
.route("/smtp/info", get(get_smtp_info))
.route("/smtp/test", post(send_smtp_test))
@@ -121,14 +139,13 @@ pub fn admin_routes() -> Router<Arc<AppState>> {
)
}
/// Validate JWT and require admin role. Returns (user_id, role).
///
/// Thin wrapper over the shared `require_admin` middleware helper so this
/// handler keeps a stable signature while the implementation lives next to
/// the new `subject_group_handler` that also needs it.
async fn admin_guard(state: &AppState, headers: &HeaderMap) -> Result<(Uuid, String), AppError> {
require_admin(state, headers).await
}
// Every route under `/api/admin/*` is gated by the
// `require_admin` middleware layer wired at the router nest point
// (`routes.rs::admin_router`). Handlers no longer need an inline
// guard call — the caller is guaranteed to be admin by construction.
// Callers that need the caller's id read it from the `AuthUser`
// extractor (`middleware::auth::AuthUser`), populated by the outer
// `auth_middleware`.
/// GET /api/admin/settings/oidc — get OIDC settings for the admin panel
#[utoipa::path(
@@ -144,10 +161,7 @@ async fn admin_guard(state: &AppState, headers: &HeaderMap) -> Result<(Uuid, Str
)]
pub async fn get_oidc_settings(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let svc = state
.admin_settings_service
.as_ref()
@@ -175,10 +189,10 @@ pub async fn get_oidc_settings(
)]
pub async fn save_oidc_settings(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Json(dto): Json<SaveOidcSettingsDto>,
) -> Result<impl IntoResponse, AppError> {
let (user_id, _) = admin_guard(&state, &headers).await?;
let user_id = auth_user.id;
let svc = state
.admin_settings_service
@@ -200,11 +214,8 @@ pub async fn save_oidc_settings(
/// POST /api/admin/settings/oidc/test — test OIDC discovery
async fn test_oidc_connection(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Json(dto): Json<TestOidcConnectionDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let svc = state
.admin_settings_service
.as_ref()
@@ -236,10 +247,7 @@ async fn test_oidc_connection(
)]
pub async fn get_storage_settings(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let svc = state
.storage_settings_service
.as_ref()
@@ -267,10 +275,10 @@ pub async fn get_storage_settings(
)]
pub async fn save_storage_settings(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Json(dto): Json<SaveStorageSettingsDto>,
) -> Result<impl IntoResponse, AppError> {
let (user_id, _) = admin_guard(&state, &headers).await?;
let user_id = auth_user.id;
let svc = state
.storage_settings_service
@@ -292,11 +300,8 @@ pub async fn save_storage_settings(
/// POST /api/admin/settings/storage/test — test storage backend connection
async fn test_storage_connection(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Json(dto): Json<TestStorageConnectionDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let svc = state
.storage_settings_service
.as_ref()
@@ -328,9 +333,7 @@ async fn test_storage_connection(
)]
pub async fn get_migration_status(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let s = state.migration_state.read().await;
Ok(Json(migration_state_to_dto(&s)))
}
@@ -350,13 +353,10 @@ pub async fn get_migration_status(
)]
pub async fn start_migration(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Json(dto): Json<StartMigrationDto>,
) -> Result<impl IntoResponse, AppError> {
use crate::infrastructure::services::migration_blob_backend::MigrationStatus;
admin_guard(&state, &headers).await?;
// Check not already running.
{
let s = state.migration_state.read().await;
@@ -428,10 +428,8 @@ pub async fn start_migration(
)]
pub async fn pause_migration(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
use crate::infrastructure::services::migration_blob_backend::MigrationStatus;
admin_guard(&state, &headers).await?;
let mut s = state.migration_state.write().await;
if s.status != MigrationStatus::Running {
@@ -459,10 +457,8 @@ pub async fn pause_migration(
)]
pub async fn resume_migration(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
use crate::infrastructure::services::migration_blob_backend::MigrationStatus;
admin_guard(&state, &headers).await?;
// Set status back to Running — the background task checks on each blob.
let mut s = state.migration_state.write().await;
@@ -491,10 +487,8 @@ pub async fn resume_migration(
)]
pub async fn complete_migration(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
use crate::infrastructure::services::migration_blob_backend::MigrationStatus;
admin_guard(&state, &headers).await?;
let s = state.migration_state.read().await;
if s.status != MigrationStatus::Completed {
@@ -531,11 +525,8 @@ pub async fn complete_migration(
)]
pub async fn verify_migration(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Json(dto): Json<VerifyMigrationDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let pool = state
.db_pool
.clone()
@@ -607,12 +598,7 @@ fn migration_state_to_dto(
security(("bearerAuth" = [])),
tag = "admin"
)]
pub async fn generate_encryption_key(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
pub async fn generate_encryption_key() -> Result<impl IntoResponse, AppError> {
let key =
crate::infrastructure::services::encrypted_blob_backend::EncryptedBlobBackend::generate_key(
);
@@ -670,10 +656,7 @@ fn build_backend_from_config(
)]
pub async fn get_dashboard_stats(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let auth = state
.auth_service
.as_ref()
@@ -761,11 +744,8 @@ pub async fn get_dashboard_stats(
)]
pub async fn list_users(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Query(query): Query<ListUsersQueryDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let auth = state
.auth_service
.as_ref()
@@ -810,11 +790,8 @@ pub async fn list_users(
)]
pub async fn get_user(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
let auth = state
@@ -847,10 +824,10 @@ pub async fn get_user(
)]
pub async fn delete_user(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
@@ -897,11 +874,11 @@ pub async fn delete_user(
)]
pub async fn update_user_role(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
Json(dto): Json<UpdateUserRoleDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
@@ -948,11 +925,11 @@ pub async fn update_user_role(
)]
pub async fn update_user_active(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
Json(dto): Json<UpdateUserActiveDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
@@ -1003,12 +980,9 @@ pub async fn update_user_active(
)]
pub async fn update_user_quota(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
Json(dto): Json<UpdateUserQuotaDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
let auth = state
@@ -1049,11 +1023,8 @@ pub async fn update_user_quota(
)]
pub async fn create_user(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Json(dto): Json<AdminCreateUserDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let auth = state
.auth_service
.as_ref()
@@ -1090,12 +1061,9 @@ pub async fn create_user(
)]
pub async fn reset_user_password(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
Json(dto): Json<AdminResetPasswordDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let id = Uuid::parse_str(&id).map_err(|_| AppError::bad_request("Invalid UUID"))?;
let auth = state
@@ -1141,10 +1109,10 @@ pub async fn reset_user_password(
)]
pub async fn set_registration_setting(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Json(body): Json<serde_json::Value>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let enabled = body
.get("registration_enabled")
@@ -1177,10 +1145,7 @@ pub async fn set_registration_setting(
async fn reextract_audio_metadata(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let audio_service = state
.applications
.audio_metadata_service
@@ -1207,10 +1172,7 @@ async fn reextract_audio_metadata(
/// Photos timeline by real capture date. Safe to re-run (idempotent upsert).
async fn reextract_image_metadata(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let result = state
.applications
.media_metadata_service
@@ -1253,12 +1215,7 @@ async fn reextract_image_metadata(
security(("bearerAuth" = [])),
tag = "admin"
)]
async fn get_smtp_info(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
async fn get_smtp_info(State(state): State<Arc<AppState>>) -> Result<impl IntoResponse, AppError> {
let smtp = &state.core.config.smtp;
let info = SmtpInfoDto {
enabled: smtp.is_enabled() && state.email_sender.is_some(),
@@ -1287,11 +1244,8 @@ async fn get_smtp_info(
/// returns 404 to keep the endpoint inert.
async fn get_captured_email(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Query(params): Query<CapturedEmailQuery>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
if !std::env::var("OXICLOUD_SMTP_MOCK")
.map(|v| v == "true" || v == "1")
.unwrap_or(false)
@@ -1347,10 +1301,10 @@ struct CapturedEmailQuery {
)]
async fn send_smtp_test(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Json(dto): Json<SendSmtpTestDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let recipient = dto.to.trim().to_string();
if recipient.is_empty() {
@@ -1462,9 +1416,7 @@ fn map_mgmt_err(err: &PluginMgmtError) -> AppError {
/// GET /api/admin/plugins — list installed plugins.
pub async fn list_plugins(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let mgmt = plugin_mgmt(&state)?;
let plugins: Vec<PluginInfoDto> = mgmt.list().into_iter().map(PluginInfoDto::from).collect();
// `enabled` reports that the plugin *subsystem* is active (reaching here
@@ -1479,11 +1431,11 @@ pub async fn list_plugins(
/// PUT /api/admin/plugins/{id}/enabled — enable or disable a plugin.
pub async fn set_plugin_enabled(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
Json(dto): Json<SetEnabledDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let mgmt = plugin_mgmt(&state)?;
mgmt.set_enabled(&id, dto.enabled)
.map_err(|e| map_mgmt_err(&e))?;
@@ -1520,10 +1472,10 @@ pub async fn set_plugin_enabled(
/// single `bundle` part: a `.zip` containing `plugin.toml` and its `.wasm`.
pub async fn install_plugin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
mut multipart: Multipart,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let mgmt = plugin_mgmt(&state)?;
let mut bundle: Option<Vec<u8>> = None;
@@ -1584,10 +1536,10 @@ pub async fn install_plugin(
/// DELETE /api/admin/plugins/{id} — uninstall a plugin and delete its files.
pub async fn delete_plugin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let mgmt = plugin_mgmt(&state)?;
mgmt.remove(&id).map_err(|e| map_mgmt_err(&e))?;
@@ -1609,11 +1561,9 @@ pub async fn delete_plugin(
/// structured log entries (newest first).
pub async fn get_plugin_logs(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
Query(q): Query<PluginLogQueryDto>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let mgmt = plugin_mgmt(&state)?;
let limit = q.limit.unwrap_or(50).clamp(1, 500);
@@ -1637,10 +1587,10 @@ pub async fn get_plugin_logs(
/// DELETE /api/admin/plugins/{id}/logs — wipe a plugin's persisted logs.
pub async fn clear_plugin_logs(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let mgmt = plugin_mgmt(&state)?;
mgmt.clear_logs(&id).await.map_err(|e| map_mgmt_err(&e))?;
@@ -1664,13 +1614,11 @@ pub async fn clear_plugin_logs(
/// so `EventSource` works without setting headers.
pub async fn stream_plugin_logs(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
use tokio_stream::StreamExt;
use tokio_stream::wrappers::{BroadcastStream, errors::BroadcastStreamRecvError};
admin_guard(&state, &headers).await?;
let mgmt = plugin_mgmt(&state)?;
if !mgmt.list().iter().any(|p| p.id == id) {
return Err(AppError::not_found("Plugin not found"));
@@ -1698,10 +1646,8 @@ pub async fn stream_plugin_logs(
/// GET /api/admin/plugins/{id}/retention — the plugin's effective retention.
pub async fn get_plugin_retention(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Path(id): Path<String>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let mgmt = plugin_mgmt(&state)?;
let settings = mgmt
.get_retention(&id)
@@ -1713,11 +1659,11 @@ pub async fn get_plugin_retention(
/// PUT /api/admin/plugins/{id}/retention — set the plugin's retention policy.
pub async fn set_plugin_retention(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
Path(id): Path<String>,
Json(dto): Json<PluginRetentionDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let mgmt = plugin_mgmt(&state)?;
mgmt.set_retention(&id, dto.into())
.await
@@ -1761,9 +1707,7 @@ pub async fn set_plugin_retention(
)]
pub async fn list_all_drives(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let drives = state
.drive_repo
.list_all()
@@ -1799,10 +1743,8 @@ pub async fn list_all_drives(
)]
pub async fn list_drive_members_admin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
axum::extract::Path(drive_id): axum::extract::Path<Uuid>,
) -> Result<impl IntoResponse, AppError> {
admin_guard(&state, &headers).await?;
let grants = state
.authorization
.list_grants_on_resource(Resource::Drive(drive_id))
@@ -1862,11 +1804,11 @@ fn admin_parse_subject(kind: SubjectTypeDto, id: Uuid) -> Subject {
)]
pub async fn add_drive_member_admin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
axum::extract::Path(drive_id): axum::extract::Path<Uuid>,
Json(dto): Json<AdminAddDriveMemberDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let subject = admin_parse_subject(dto.subject.kind, dto.subject.id);
let grant = state
.drive_management_service
@@ -1907,7 +1849,7 @@ pub async fn add_drive_member_admin(
)]
pub async fn update_drive_member_admin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
axum::extract::Path((drive_id, kind, subject_id)): axum::extract::Path<(
Uuid,
SubjectTypeDto,
@@ -1915,7 +1857,7 @@ pub async fn update_drive_member_admin(
)>,
Json(dto): Json<AdminUpdateDriveMemberDto>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let subject = admin_parse_subject(kind, subject_id);
let grant = state
.drive_management_service
@@ -1954,14 +1896,14 @@ pub async fn update_drive_member_admin(
)]
pub async fn remove_drive_member_admin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
axum::extract::Path((drive_id, kind, subject_id)): axum::extract::Path<(
Uuid,
SubjectTypeDto,
Uuid,
)>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
let subject = admin_parse_subject(kind, subject_id);
state
.drive_management_service
@@ -1996,10 +1938,10 @@ pub async fn remove_drive_member_admin(
)]
pub async fn delete_drive_admin(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
auth_user: AuthUser,
axum::extract::Path(drive_id): axum::extract::Path<Uuid>,
) -> Result<impl IntoResponse, AppError> {
let (admin_id, _) = admin_guard(&state, &headers).await?;
let admin_id = auth_user.id;
state
.drive_management_service
.delete_drive(admin_id, true, drive_id)
@@ -2055,15 +1997,11 @@ fn internal_endpoints_disabled() -> axum::response::Response {
)]
pub async fn internal_trigger_sweep(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
) -> axum::response::Response {
use axum::response::IntoResponse;
if !state.core.config.features.enable_admin_internal_endpoints {
return internal_endpoints_disabled();
}
if let Err(e) = admin_guard(&state, &headers).await {
return e.into_response();
}
let svc = match state.storage_usage_service.as_ref() {
Some(s) => s,
None => {
@@ -2135,16 +2073,12 @@ pub struct InternalTriggerGcQuery {
)]
pub async fn internal_trigger_gc(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Query(query): Query<InternalTriggerGcQuery>,
) -> axum::response::Response {
use axum::response::IntoResponse;
if !state.core.config.features.enable_admin_internal_endpoints {
return internal_endpoints_disabled();
}
if let Err(e) = admin_guard(&state, &headers).await {
return e.into_response();
}
let result = if query.force {
state.core.dedup_service.garbage_collect_force().await
} else {
@@ -2211,16 +2145,12 @@ pub struct InternalTriggerGrantCleanupQuery {
)]
pub async fn internal_trigger_grant_cleanup(
State(state): State<Arc<AppState>>,
headers: HeaderMap,
Query(query): Query<InternalTriggerGrantCleanupQuery>,
) -> axum::response::Response {
use axum::response::IntoResponse;
if !state.core.config.features.enable_admin_internal_endpoints {
return internal_endpoints_disabled();
}
if let Err(e) = admin_guard(&state, &headers).await {
return e.into_response();
}
// Daemon may be disabled by config even when the internal-endpoint
// gate is on. Return 503 (rather than 404 or 500) so integration
// tests can distinguish "surface not exposed" from "surface
+324 -79
View File
@@ -21,8 +21,9 @@ use axum::{
http::{HeaderName, Request, StatusCode, header},
response::Response,
};
use bytes::Buf;
use bytes::{Buf, Bytes};
use percent_encoding::percent_decode_str;
use quick_xml::Writer;
use std::fmt::Write;
use std::sync::Arc;
@@ -33,7 +34,7 @@ use crate::application::adapters::caldav_adapter::{
use crate::application::adapters::uid_from_multiget_href;
use crate::application::adapters::webdav_adapter::{PropFindRequest, PropFindType};
use crate::application::dtos::calendar_dto::{
CreateCalendarDto, CreateEventICalDto, UpdateCalendarDto,
CalendarEventDto, CreateCalendarDto, CreateEventICalDto, UpdateCalendarDto,
};
use crate::application::ports::calendar_ports::CalendarUseCase;
use crate::application::services::calendar_service::CalendarService;
@@ -47,6 +48,249 @@ const HEADER_DAV: HeaderName = HeaderName::from_static("dav");
/// Prevents OOM/DoS via unbounded body buffering.
const MAX_CALDAV_BODY: usize = 1_048_576;
/// Minimum rows per emitted page for the streaming CalDAV emitters.
/// Pages only cut at UID boundaries (the cursor delivers same-UID rows
/// adjacent), so a master + its exception overrides always land in one
/// chunk and peak memory is one page of DTOs + its XML instead of the
/// whole calendar twice.
const CALDAV_STREAM_PAGE_EVENTS: usize = 500;
/// Streamed multistatus REPORT: header chunk, one chunk per hydrated
/// UID page, footer chunk. Byte-compatible with the buffered
/// `generate_calendar_events_response` output (same bundle order:
/// `(MIN(start_time), uid)` = first appearance in the start_time
/// listing). TTFB becomes the first page instead of the full
/// generation; the whole-calendar DTO Vec is never materialised.
fn build_streaming_report_response(
calendar_service: Arc<CalendarService>,
calendar_id: String,
report: CalDavReportType,
base_href: String,
user_id: uuid::Uuid,
) -> Response<Body> {
let stream = async_stream::try_stream! {
let mut buf = Vec::with_capacity(256);
{
let mut w = Writer::new(&mut buf);
CalDavAdapter::write_caldav_multistatus_start(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
// ONE server-side scan+sort in bundle order streamed through a
// cursor — the same aggregate work the buffered path paid, but
// only a page of rows resident. Pages cut at UID boundaries.
{
use futures::TryStreamExt;
let mut rows = calendar_service
.stream_events_uid_order(&calendar_id, user_id)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let mut page: Vec<CalendarEventDto> =
Vec::with_capacity(CALDAV_STREAM_PAGE_EVENTS + 32);
loop {
let next = rows
.try_next()
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let flush = match &next {
Some(ev) => {
page.len() >= CALDAV_STREAM_PAGE_EVENTS
&& page.last().is_some_and(|p| p.ical_uid != ev.ical_uid)
}
None => !page.is_empty(),
};
if flush {
let mut chunk = Vec::with_capacity(page.len() * 1024 + 128);
{
let mut w = Writer::new(&mut chunk);
CalDavAdapter::write_report_page(&mut w, &page, &report, &base_href)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
page.clear();
yield Bytes::from(chunk);
}
match next {
Some(ev) => page.push(ev),
None => break,
}
}
}
let mut buf = Vec::with_capacity(32);
{
let mut w = Writer::new(&mut buf);
CalDavAdapter::write_caldav_multistatus_end(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
};
use futures::TryStreamExt;
let stream = stream
.map_err(|e: std::io::Error| -> Box<dyn std::error::Error + Send + Sync> { Box::new(e) });
Response::builder()
.status(StatusCode::MULTI_STATUS)
.header(header::CONTENT_TYPE, "application/xml; charset=utf-8")
.body(Body::from_stream(stream))
.unwrap()
}
/// Streamed depth-1 collection PROPFIND: head (multistatus + the
/// calendar's own response), one chunk per hydrated UID page, footer.
#[allow(clippy::too_many_arguments)]
fn build_streaming_collection_propfind(
calendar_service: Arc<CalendarService>,
calendar: crate::application::dtos::calendar_dto::CalendarDto,
propfind_request: PropFindRequest,
calendar_id: String,
base_href: String,
caller_id: String,
user_id: uuid::Uuid,
) -> Response<Body> {
let stream = async_stream::try_stream! {
let mut buf = Vec::with_capacity(2048);
{
let mut w = Writer::new(&mut buf);
CalDavAdapter::write_collection_head(&mut w, &calendar, &propfind_request, &base_href, &caller_id)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
{
use futures::TryStreamExt;
let mut rows = calendar_service
.stream_events_uid_order(&calendar_id, user_id)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let mut page: Vec<CalendarEventDto> =
Vec::with_capacity(CALDAV_STREAM_PAGE_EVENTS + 32);
loop {
let next = rows
.try_next()
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let flush = match &next {
Some(ev) => {
page.len() >= CALDAV_STREAM_PAGE_EVENTS
&& page.last().is_some_and(|p| p.ical_uid != ev.ical_uid)
}
None => !page.is_empty(),
};
if flush {
let mut chunk = Vec::with_capacity(page.len() * 512 + 128);
{
let mut w = Writer::new(&mut chunk);
CalDavAdapter::write_collection_event_page(&mut w, &page, &base_href)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
page.clear();
yield Bytes::from(chunk);
}
match next {
Some(ev) => page.push(ev),
None => break,
}
}
}
let mut buf = Vec::with_capacity(32);
{
let mut w = Writer::new(&mut buf);
CalDavAdapter::write_caldav_multistatus_end(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
};
use futures::TryStreamExt;
let stream = stream
.map_err(|e: std::io::Error| -> Box<dyn std::error::Error + Send + Sync> { Box::new(e) });
Response::builder()
.status(StatusCode::MULTI_STATUS)
.header(header::CONTENT_TYPE, "application/xml; charset=utf-8")
.body(Body::from_stream(stream))
.unwrap()
}
/// Streamed whole-calendar `.ics` GET: VCALENDAR header, one chunk per
/// hydrated UID page (each row's stored VEVENT chunk served verbatim),
/// `END:VCALENDAR` footer.
fn build_streaming_calendar_ics(
calendar_service: Arc<CalendarService>,
calendar_id: String,
calendar_name: String,
calendar_etag: String,
user_id: uuid::Uuid,
) -> Response<Body> {
let stream = async_stream::try_stream! {
let mut head = String::with_capacity(128);
let _ = write!(
head,
"BEGIN:VCALENDAR\r\nVERSION:2.0\r\nPRODID:-//OxiCloud//NONSGML Calendar//EN\r\nX-WR-CALNAME:{}\r\n",
calendar_name
);
yield Bytes::from(head);
{
use futures::TryStreamExt;
let mut rows = calendar_service
.stream_events_uid_order(&calendar_id, user_id)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let mut page: Vec<CalendarEventDto> =
Vec::with_capacity(CALDAV_STREAM_PAGE_EVENTS + 32);
loop {
let next = rows
.try_next()
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let flush = match &next {
Some(ev) => {
page.len() >= CALDAV_STREAM_PAGE_EVENTS
&& page.last().is_some_and(|p| p.ical_uid != ev.ical_uid)
}
None => !page.is_empty(),
};
if flush {
let mut chunk = String::with_capacity(page.len() * 384);
for group in group_events_by_uid(&page) {
for event in group {
if let Some(vevent) = extract_vevent_chunk(&event.ical_data) {
chunk.push_str(vevent);
if !chunk.ends_with('\n') {
chunk.push_str("\r\n");
}
}
}
}
page.clear();
yield Bytes::from(chunk);
}
match next {
Some(ev) => page.push(ev),
None => break,
}
}
}
yield Bytes::from_static(b"END:VCALENDAR\r\n");
};
use futures::TryStreamExt;
let stream = stream
.map_err(|e: std::io::Error| -> Box<dyn std::error::Error + Send + Sync> { Box::new(e) });
Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, "text/calendar; charset=utf-8")
.header(header::ETAG, format!("\"{}\"", calendar_etag))
.body(Body::from_stream(stream))
.unwrap()
}
/// Creates CalDAV routes with full path prefixes.
///
/// Uses `merge()` instead of `nest()` to avoid Axum's trailing-slash routing gap.
@@ -320,15 +564,23 @@ async fn handle_propfind(
};
if let Ok(calendar) = calendar_result {
// Valid calendar ID — return calendar collection
let events = if depth != "0" {
calendar_service
.list_events(first_segment, None, None, user.id)
.await
.unwrap_or_default()
} else {
vec![]
};
// Valid calendar ID — return calendar collection.
// Depth-1 streams the event listing page by page
// (whole-calendar responses used to materialise every
// DTO + the full multistatus in RAM); depth-0 has no
// event section and keeps the tiny buffered path.
if depth != "0" {
let base_href = format!("/caldav/{}/", first_segment);
return Ok(build_streaming_collection_propfind(
calendar_service.clone(),
calendar,
propfind_request,
first_segment.to_string(),
base_href,
caller_id.clone(),
user.id,
));
}
let base_href = &format!("/caldav/{}/", first_segment);
let mut response_body = Vec::new();
@@ -336,7 +588,7 @@ async fn handle_propfind(
CalDavAdapter::generate_calendar_collection_propfind(
&mut response_body,
&calendar,
&events,
&[],
&propfind_request,
base_href,
&depth,
@@ -407,14 +659,20 @@ async fn handle_propfind(
.await
.map_err(|e| AppError::not_found(format!("Calendar not found: {}", e)))?;
let events = if depth != "0" {
calendar_service
.list_events(sub_parts[0], None, None, user.id)
.await
.unwrap_or_default()
} else {
vec![]
};
// Same streaming/buffered split as the
// single-segment collection branch above.
if depth != "0" {
let base_href = format!("/caldav/{}/{}/", first_segment, sub_parts[0]);
return Ok(build_streaming_collection_propfind(
calendar_service.clone(),
cal,
propfind_request,
sub_parts[0].to_string(),
base_href,
caller_id.clone(),
user.id,
));
}
let base_href = &format!("/caldav/{}/{}/", first_segment, sub_parts[0]);
let mut response_body = Vec::new();
@@ -422,7 +680,7 @@ async fn handle_propfind(
CalDavAdapter::generate_calendar_collection_propfind(
&mut response_body,
&cal,
&events,
&[],
&propfind_request,
base_href,
&depth,
@@ -500,6 +758,33 @@ async fn handle_report(
return Err(AppError::bad_request("Calendar ID required in path"));
}
// Whole-calendar shapes (no-range calendar-query, sync-collection)
// stream: header + one chunk per hydrated UID page + footer, instead
// of materialising every DTO AND the full multistatus in RAM with
// TTFB = complete generation. Bounded shapes (time-range query,
// multiget) keep the buffered path.
if matches!(
&report,
CalDavReportType::CalendarQuery {
time_range: None,
..
} | CalDavReportType::SyncCollection { .. }
) {
// Surface not-found / authz before committing to a 207 stream.
calendar_service
.get_calendar(calendar_id, user.id)
.await
.map_err(AppError::from)?;
let base_href = format!("/caldav/{}/", calendar_id);
return Ok(build_streaming_report_response(
calendar_service.clone(),
calendar_id.to_string(),
report,
base_href,
user.id,
));
}
let events = match &report {
CalDavReportType::CalendarQuery { time_range, .. } => {
if let Some((start, end)) = time_range {
@@ -508,10 +793,7 @@ async fn handle_report(
.await
.map_err(AppError::from)?
} else {
calendar_service
.list_events(calendar_id, None, None, user.id)
.await
.map_err(AppError::from)?
unreachable!("no-range calendar-query streams above")
}
}
CalDavReportType::CalendarMultiget { hrefs, .. } => {
@@ -528,10 +810,9 @@ async fn handle_report(
.await
.map_err(AppError::from)?
}
CalDavReportType::SyncCollection { .. } => calendar_service
.list_events(calendar_id, None, None, user.id)
.await
.map_err(AppError::from)?,
CalDavReportType::SyncCollection { .. } => {
unreachable!("sync-collection streams above")
}
};
let base_href = &format!("/caldav/{}/", calendar_id);
@@ -686,31 +967,27 @@ async fn handle_get(
let calendar_id = parts[0];
if parts.len() < 2 {
// GET on calendar collection — return all events, folded
// GET on calendar collection — stream all events, folded
// per UID so master + exception overrides live in ONE
// VCALENDAR body per resource (RFC 4791 §4.1 + RFC 5545
// §3.6.1). Serves each row's stored `ical_data` verbatim
// via `bundle_to_calendar_body`; VTIMEZONE / VALARM /
// ATTENDEE / CATEGORIES / X-* survive because we no
// longer regenerate the body from DTO fields.
let events = calendar_service
.list_events(calendar_id, None, None, user.id)
.await
.map_err(AppError::from)?;
// §3.6.1). Each row's stored `ical_data` VEVENT chunk is
// served verbatim; VTIMEZONE / VALARM / ATTENDEE /
// CATEGORIES / X-* survive because the body is never
// regenerated from DTO fields. Streaming (header + one
// chunk per hydrated UID page + footer) replaces the old
// whole-calendar String build.
let calendar = calendar_service
.get_calendar(calendar_id, user.id)
.await
.map_err(AppError::from)?;
let ical = generate_full_calendar_ical(&calendar.name, &events);
Ok(Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, "text/calendar; charset=utf-8")
.header(header::ETAG, format!("\"{}\"", calendar.id))
.body(Body::from(ical))
.unwrap())
Ok(build_streaming_calendar_ics(
calendar_service.clone(),
calendar_id.to_string(),
calendar.name,
calendar.id,
user.id,
))
} else {
// GET on individual event resource — fetch ALL rows for
// this UID (master + any exception overrides) and emit
@@ -754,38 +1031,6 @@ async fn handle_get(
}
}
/// Emit a full VCALENDAR body for the entire calendar, with rows
/// grouped by UID so each recurring event's master + exception
/// overrides live under one iCalendar resource. Each row's stored
/// `ical_data` VEVENT chunk is served verbatim.
fn generate_full_calendar_ical(
calendar_name: &str,
events: &[crate::application::dtos::calendar_dto::CalendarEventDto],
) -> String {
// Pre-estimate: ~200 bytes header + ~320 bytes per event.
let mut buf = String::with_capacity(256 + events.len() * 320);
let _ = write!(
buf,
"BEGIN:VCALENDAR\r\nVERSION:2.0\r\nPRODID:-//OxiCloud//NONSGML Calendar//EN\r\nX-WR-CALNAME:{}\r\n",
calendar_name
);
// Group + append each row's stored VEVENT chunk. Malformed
// rows are silently skipped (defensive) — the bulk-GET body
// survives the rest.
for group in group_events_by_uid(events) {
for event in group {
if let Some(chunk) = extract_vevent_chunk(&event.ical_data) {
buf.push_str(chunk);
if !buf.ends_with('\n') {
buf.push_str("\r\n");
}
}
}
}
buf.push_str("END:VCALENDAR\r\n");
buf
}
// NOTE: the pre-phase-4 `generate_event_ical` + `write_vevent`
// helpers were removed. They regenerated the response body from
// DTO fields, which (a) silently dropped every property outside
+198 -34
View File
@@ -22,7 +22,8 @@ use axum::{
http::{HeaderName, Request, StatusCode, header},
response::Response,
};
use bytes::Buf;
use bytes::{Buf, Bytes};
use quick_xml::Writer;
use std::sync::Arc;
use crate::application::adapters::carddav_adapter::{
@@ -31,7 +32,7 @@ use crate::application::adapters::carddav_adapter::{
use crate::application::adapters::uid_from_multiget_href;
use crate::application::adapters::webdav_adapter::{PropFindRequest, PropFindType};
use crate::application::dtos::address_book_dto::{CreateAddressBookDto, UpdateAddressBookDto};
use crate::application::dtos::contact_dto::CreateContactVCardDto;
use crate::application::dtos::contact_dto::{ContactDto, CreateContactVCardDto};
use crate::application::ports::carddav_ports::{AddressBookUseCase, ContactUseCase};
use crate::application::services::contact_service::ContactService;
use crate::common::di::AppState;
@@ -187,6 +188,164 @@ fn get_addressbook_service(state: &AppState) -> Result<&Arc<ContactService>, App
})
}
/// Rows per emitted page for the streaming CardDAV emitters — contacts
/// carry no master/exception bundling, so pages cut anywhere.
const CARDDAV_STREAM_PAGE_CONTACTS: usize = 500;
/// Streamed multistatus REPORT: header, one chunk per cursor page,
/// footer. Byte-compatible with the buffered
/// `generate_contacts_response` output; TTFB becomes the first page and
/// the whole-book DTO Vec is never materialised.
fn build_streaming_contacts_report(
contact_svc: Arc<ContactService>,
address_book_id: String,
report: CardDavReportType,
base_href: String,
user_id: uuid::Uuid,
) -> Response<Body> {
let stream = async_stream::try_stream! {
let mut buf = Vec::with_capacity(160);
{
let mut w = Writer::new(&mut buf);
CardDavAdapter::write_report_multistatus_start(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
{
use futures::TryStreamExt;
let mut rows = contact_svc
.stream_contacts_by_book(&address_book_id, user_id)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let mut page: Vec<ContactDto> =
Vec::with_capacity(CARDDAV_STREAM_PAGE_CONTACTS);
loop {
let next = rows
.try_next()
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let flush = match &next {
Some(_) => page.len() >= CARDDAV_STREAM_PAGE_CONTACTS,
None => !page.is_empty(),
};
if flush {
let mut chunk = Vec::with_capacity(page.len() * 256 + 64);
{
let mut w = Writer::new(&mut chunk);
CardDavAdapter::write_contacts_report_page(
&mut w, &page, &report, &base_href,
)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
page.clear();
yield Bytes::from(chunk);
}
match next {
Some(c) => page.push(c),
None => break,
}
}
}
let mut buf = Vec::with_capacity(32);
{
let mut w = Writer::new(&mut buf);
CardDavAdapter::write_carddav_multistatus_end(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
};
use futures::TryStreamExt;
let stream = stream
.map_err(|e: std::io::Error| -> Box<dyn std::error::Error + Send + Sync> { Box::new(e) });
Response::builder()
.status(StatusCode::MULTI_STATUS)
.header(header::CONTENT_TYPE, "application/xml; charset=utf-8")
.body(Body::from_stream(stream))
.unwrap()
}
/// Streamed depth-1 address-book PROPFIND: head (multistatus + the
/// book's own response), one chunk per cursor page, footer.
fn build_streaming_book_propfind(
contact_svc: Arc<ContactService>,
address_book: crate::application::dtos::address_book_dto::AddressBookDto,
propfind_request: PropFindRequest,
address_book_id: String,
base_href: String,
user_id: uuid::Uuid,
) -> Response<Body> {
let stream = async_stream::try_stream! {
let mut buf = Vec::with_capacity(2048);
{
let mut w = Writer::new(&mut buf);
CardDavAdapter::write_collection_head(
&mut w,
&address_book,
&propfind_request,
&base_href,
)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
{
use futures::TryStreamExt;
let mut rows = contact_svc
.stream_contacts_by_book(&address_book_id, user_id)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let mut page: Vec<ContactDto> =
Vec::with_capacity(CARDDAV_STREAM_PAGE_CONTACTS);
loop {
let next = rows
.try_next()
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let flush = match &next {
Some(_) => page.len() >= CARDDAV_STREAM_PAGE_CONTACTS,
None => !page.is_empty(),
};
if flush {
let mut chunk = Vec::with_capacity(page.len() * 512 + 64);
{
let mut w = Writer::new(&mut chunk);
CardDavAdapter::write_collection_contact_page(&mut w, &page, &base_href)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
page.clear();
yield Bytes::from(chunk);
}
match next {
Some(c) => page.push(c),
None => break,
}
}
}
let mut buf = Vec::with_capacity(32);
{
let mut w = Writer::new(&mut buf);
CardDavAdapter::write_carddav_multistatus_end(&mut w)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
yield Bytes::from(buf);
};
use futures::TryStreamExt;
let stream = stream
.map_err(|e: std::io::Error| -> Box<dyn std::error::Error + Send + Sync> { Box::new(e) });
Response::builder()
.status(StatusCode::MULTI_STATUS)
.header(header::CONTENT_TYPE, "application/xml; charset=utf-8")
.body(Body::from_stream(stream))
.unwrap()
}
fn get_contact_service(state: &AppState) -> Result<&Arc<ContactService>, AppError> {
state.contact_use_case.as_ref().ok_or_else(|| {
AppError::new(
@@ -334,14 +493,19 @@ async fn handle_propfind(
.await
.map_err(|e| AppError::not_found(format!("Address book not found: {}", e)))?;
let contacts = if depth != "0" {
contact_svc
.list_contacts(address_book_id, None, None, user.id)
.await
.unwrap_or_default()
} else {
vec![]
};
// Depth-1 streams the contact listing page by page; depth-0
// has no contact section and keeps the tiny buffered path.
if depth != "0" {
let base_href = format!("/carddav/{}/", address_book_id);
return Ok(build_streaming_book_propfind(
contact_svc.clone(),
address_book,
propfind_request,
address_book_id.to_string(),
base_href,
user.id,
));
}
let base_href = &format!("/carddav/{}/", address_book_id);
let mut response_body = Vec::new();
@@ -349,7 +513,7 @@ async fn handle_propfind(
CardDavAdapter::generate_addressbook_collection_propfind(
&mut response_body,
&address_book,
&contacts,
&[],
&propfind_request,
base_href,
&depth,
@@ -385,7 +549,6 @@ async fn handle_propfind(
CardDavAdapter::generate_contacts_response(
&mut response_body,
std::slice::from_ref(&contact),
&[(contact.uid.clone(), contact_to_vcard(&contact))],
&report,
base_href,
)
@@ -424,11 +587,25 @@ async fn handle_report(
return Err(AppError::bad_request("Address book ID required in path"));
}
// Whole-book shapes stream; bounded multiget keeps the buffered path.
if matches!(
&report,
CardDavReportType::AddressbookQuery { .. } | CardDavReportType::SyncCollection { .. }
) {
let base_href = format!("/carddav/{}/", address_book_id);
return Ok(build_streaming_contacts_report(
contact_svc.clone(),
address_book_id.to_string(),
report,
base_href,
user.id,
));
}
let contacts = match &report {
CardDavReportType::AddressbookQuery { .. } => contact_svc
.list_contacts(address_book_id, None, None, user.id)
.await
.map_err(AppError::from)?,
CardDavReportType::AddressbookQuery { .. } => {
unreachable!("addressbook-query streams above")
}
CardDavReportType::AddressbookMultiget { hrefs, .. } => {
// Indexed batch lookup (`uid = ANY(...)`) — a multiget for a
// handful of contacts must not pay for listing the whole
@@ -443,28 +620,15 @@ async fn handle_report(
.await
.map_err(AppError::from)?
}
CardDavReportType::SyncCollection { .. } => contact_svc
.list_contacts(address_book_id, None, None, user.id)
.await
.map_err(AppError::from)?,
CardDavReportType::SyncCollection { .. } => {
unreachable!("sync-collection streams above")
}
};
// Generate vCards
let vcards: Vec<(String, String)> = contacts
.iter()
.map(|c| (c.uid.clone(), contact_to_vcard(c)))
.collect();
let base_href = &format!("/carddav/{}/", address_book_id);
let mut response_body = Vec::new();
CardDavAdapter::generate_contacts_response(
&mut response_body,
&contacts,
&vcards,
&report,
base_href,
)
.map_err(|e| AppError::internal_error(format!("Failed to generate XML: {}", e)))?;
CardDavAdapter::generate_contacts_response(&mut response_body, &contacts, &report, base_href)
.map_err(|e| AppError::internal_error(format!("Failed to generate XML: {}", e)))?;
Ok(Response::builder()
.status(StatusCode::MULTI_STATUS)
@@ -188,9 +188,14 @@ impl ChunkedUploadHandler {
// ── Permission pre-check: caller must have Create on the target
// folder BEFORE we allocate a session and accept chunks. The
// upload service re-checks at finalize time, but failing here
// avoids wasting client+server resources on chunks that will be
// rejected. None = caller's root namespace, no check needed.
// upload service re-checks at finalize via
// `upload_file_streaming_with_perms` (AuthZ audit #17 fix,
// 2026-07-16) so a grant revoked mid-session is caught. This
// pre-check is the fail-fast: it avoids wasting client+server
// resources on chunks that will be rejected anyway. `None`
// means the write lands at drive-root — that path is currently
// unchecked (session doesn't carry `drive_id`; tracked with the
// folder-id-walking follow-up).
if let Some(ref fid) = request.folder_id
&& let Err(err) = state
.applications
@@ -441,9 +446,17 @@ impl ChunkedUploadHandler {
}
// Register the file row against the ingested blob.
//
// AuthZ audit #17 (2026-07-12): swapped `upload_file_streaming` →
// `upload_file_streaming_with_perms` so `Create` on the target
// folder is re-verified at finalize. Session creation already
// pre-checked (line ~198), but that was potentially hours or
// days ago; app-passwords keep sessions valid indefinitely.
// Without the finalize re-check, a grant revoked mid-session
// stayed effective until the last chunk landed.
let size = ingested.size;
match upload_service
.upload_file_streaming(
.upload_file_streaming_with_perms(
parts.filename.clone(),
parts.folder_id.clone(),
ingested.content_type.clone(),
@@ -478,7 +491,12 @@ impl ChunkedUploadHandler {
}
Err(e) => {
tracing::error!("Failed to create file from chunked upload: {:?}", e);
AppError::internal_error(format!("Failed to create file: {}", e)).into_response()
// AuthZ audit #2 (2026-07-12) — route DomainError through
// `AppError::from` so graduated denial from
// `upload_file_streaming_with_perms` keeps the 403/404
// shape instead of collapsing into a 500. Sibling
// `cancel_upload_impl` at :514 already uses this pattern.
AppError::from(e).into_response()
}
}
}
@@ -519,9 +537,15 @@ impl ChunkedUploadHandler {
// routes.rs calls these free functions directly.
// TODO: collapse back into the impl block after a utoipa upgrade resolves the issue.
/// **Deprecated.** Prefer `/api/files/delta/*` — hash-first negotiation,
/// resumable, chunked. The `/api/uploads/*` family stays for backward
/// compatibility with existing clients but receives no new features.
#[utoipa::path(
post,
path = "/api/uploads",
description = "**Deprecated.** Prefer the delta-upload surface at `/api/files/delta/*` \
(hash-first negotiation, resumable, chunked). The `/api/uploads/*` family is kept for \
backward compatibility with existing clients but is no longer receiving new features.",
request_body(content = CreateUploadRequest, content_type = "application/json", description = "Upload session parameters"),
responses(
(status = 201, description = "Upload session created", body = crate::application::ports::chunked_upload_ports::CreateUploadResponseDto),
@@ -531,6 +555,7 @@ impl ChunkedUploadHandler {
tag = "uploads",
security(("bearerAuth" = []))
)]
#[deprecated(note = "prefer /api/files/delta/*")]
pub async fn create_upload(
state: State<Arc<AppState>>,
auth_user: AuthUser,
@@ -539,9 +564,11 @@ pub async fn create_upload(
ChunkedUploadHandler::create_upload_impl(state, auth_user, request).await
}
/// **Deprecated.** Prefer `/api/files/delta/*` — see `create_upload`.
#[utoipa::path(
patch,
path = "/api/uploads/{upload_id}",
description = "**Deprecated.** See `POST /api/uploads` for the migration note.",
params(
("upload_id" = String, Path, description = "Upload session ID"),
("chunk_index" = usize, Query, description = "Zero-based chunk index"),
@@ -570,6 +597,7 @@ pub async fn create_upload(
tag = "uploads",
security(("bearerAuth" = []))
)]
#[deprecated(note = "prefer /api/files/delta/*")]
pub async fn upload_chunk(
State(state): State<Arc<AppState>>,
auth_user: AuthUser,
@@ -683,9 +711,11 @@ pub async fn upload_chunk(
.into_response()
}
/// **Deprecated.** Prefer `/api/files/delta/*` — see `create_upload`.
#[utoipa::path(
head,
path = "/api/uploads/{upload_id}",
description = "**Deprecated.** See `POST /api/uploads` for the migration note.",
params(
("upload_id" = String, Path, description = "Upload session ID"),
),
@@ -696,6 +726,7 @@ pub async fn upload_chunk(
tag = "uploads",
security(("bearerAuth" = []))
)]
#[deprecated(note = "prefer /api/files/delta/*")]
pub async fn get_upload_status(
state: State<Arc<AppState>>,
auth_user: AuthUser,
@@ -704,9 +735,11 @@ pub async fn get_upload_status(
ChunkedUploadHandler::get_upload_status_impl(state, auth_user, path).await
}
/// **Deprecated.** Prefer `/api/files/delta/*` — see `create_upload`.
#[utoipa::path(
post,
path = "/api/uploads/{upload_id}/complete",
description = "**Deprecated.** See `POST /api/uploads` for the migration note.",
params(
("upload_id" = String, Path, description = "Upload session ID"),
),
@@ -731,6 +764,7 @@ pub async fn get_upload_status(
tag = "uploads",
security(("bearerAuth" = []))
)]
#[deprecated(note = "prefer /api/files/delta/*")]
pub async fn complete_upload(
state: State<Arc<AppState>>,
auth_user: AuthUser,
@@ -744,9 +778,11 @@ pub async fn complete_upload(
ChunkedUploadHandler::complete_upload_impl(state, auth_user, path, req).await
}
/// **Deprecated.** Prefer `/api/files/delta/*` — see `create_upload`.
#[utoipa::path(
delete,
path = "/api/uploads/{upload_id}",
description = "**Deprecated.** See `POST /api/uploads` for the migration note.",
params(
("upload_id" = String, Path, description = "Upload session ID"),
),
@@ -757,6 +793,7 @@ pub async fn complete_upload(
tag = "uploads",
security(("bearerAuth" = []))
)]
#[deprecated(note = "prefer /api/files/delta/*")]
pub async fn cancel_upload(
state: State<Arc<AppState>>,
auth_user: AuthUser,
+36 -27
View File
@@ -218,18 +218,16 @@ impl DedupHandler {
/// - Deduplication ratio
pub(super) async fn get_stats_impl(
State(state): State<GlobalState>,
auth_user: AuthUser,
_auth_user: AuthUser,
) -> impl IntoResponse {
// Admin-only — global dedup statistics are sensitive infrastructure data
if auth_user.role != "admin" {
return Response::builder()
.status(StatusCode::FORBIDDEN)
.header(header::CONTENT_TYPE, "application/json")
.body(Body::from(r#"{"error": "Admin role required"}"#))
.unwrap()
.into_response();
}
// AuthZ audit #24 (2026-07-17): admin check moved to the
// `/api/admin/*` middleware layer. Reaching this handler means
// the caller is admin by construction — the bespoke role
// string comparison here (`auth_user.role != "admin"` → 403
// with a hand-rolled JSON body, no audit line) is gone. The
// route is registered at `admin_handler::admin_routes()`;
// moving the URL to `/api/admin/dedup/stats` also declares
// the admin intent up front.
let dedup = &state.core.dedup_service;
let stats = dedup.get_stats().await;
@@ -343,16 +341,10 @@ impl DedupHandler {
State(state): State<GlobalState>,
auth_user: AuthUser,
) -> impl IntoResponse {
// Admin-only — integrity verification is a privileged operation
if auth_user.role != "admin" {
return Response::builder()
.status(StatusCode::FORBIDDEN)
.header(header::CONTENT_TYPE, "application/json")
.body(Body::from(r#"{"error": "Admin role required"}"#))
.unwrap()
.into_response();
}
// AuthZ audit #25 (2026-07-17): admin check moved to the
// `/api/admin/*` middleware layer — see the sibling
// `get_stats_impl` comment. `auth_user` is kept so the
// success-side audit line carries the caller id.
let dedup = &state.core.dedup_service;
// Verify integrity first
@@ -392,6 +384,21 @@ impl DedupHandler {
savings_percentage: savings_pct,
};
// AuthZ audit #25 (2026-07-17): integrity recalculation is a
// low-frequency privileged operation — landing an audit event
// so security reviews can see who ran verify + integrity
// sweeps and when. The pre-fix path emitted no audit line at
// all (the accepted 200 was silent from the security POV).
tracing::info!(
target: "audit",
event = "dedup.integrity_recalculated",
caller_id = %auth_user.id,
unique_blobs = response.unique_blobs,
total_references = response.total_references,
bytes_saved = response.bytes_saved,
"🧮 dedup integrity verified and stats recomputed by admin",
);
Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, "application/json")
@@ -453,12 +460,13 @@ pub async fn check_hashes_batch(
#[utoipa::path(
get,
path = "/api/dedup/stats",
path = "/api/admin/dedup/stats",
responses(
(status = 200, description = "Deduplication statistics", body = StatsResponse),
(status = 403, description = "Admin role required"),
(status = 401, description = "Missing or invalid token"),
(status = 403, description = "Caller is not an admin"),
),
tag = "dedup",
tag = "admin",
security(("bearerAuth" = []))
)]
pub async fn get_stats(state: State<GlobalState>, auth_user: AuthUser) -> impl IntoResponse {
@@ -489,13 +497,14 @@ pub async fn get_blob(
#[utoipa::path(
post,
path = "/api/dedup/recalculate",
path = "/api/admin/dedup/recalculate",
responses(
(status = 200, description = "Statistics after integrity verification", body = StatsResponse),
(status = 403, description = "Admin role required"),
(status = 401, description = "Missing or invalid token"),
(status = 403, description = "Caller is not an admin"),
(status = 500, description = "Integrity verification failed"),
),
tag = "dedup",
tag = "admin",
security(("bearerAuth" = []))
)]
pub async fn recalculate_stats(
+1 -1
View File
@@ -50,7 +50,7 @@ pub async fn list_drives(
match state.drive_repo.list_readable_by(caller_id).await {
Ok(drives) => {
let dtos: Vec<DriveDto> = drives.into_iter().map(DriveDto::from).collect();
let dtos: Vec<DriveDto> = drives.iter().cloned().map(DriveDto::from).collect();
(StatusCode::OK, Json(dtos)).into_response()
}
Err(e) => {
@@ -10,7 +10,8 @@ use tracing::info;
use utoipa::ToSchema;
use crate::application::dtos::display_helpers::{
category_for, format_file_size, icon_class_for, icon_special_class_for,
category_for, format_file_size, icon_class_for, icon_special_class_for, intern_display,
intern_mime,
};
use crate::application::dtos::favorites_dto::{
FavoritesResourceItemDto, FavoritesResourcesDto, FavoritesResourcesQuery,
@@ -197,7 +198,7 @@ pub async fn list_favorites_resources(
// Path is only shown to the owner; non-owners see ""
// to avoid leaking another user's folder hierarchy.
let path = if row.is_owner {
row.path.clone().unwrap_or_default()
row.path.unwrap_or_default()
} else {
String::new()
};
@@ -207,16 +208,16 @@ pub async fn list_favorites_resources(
let dto = FolderDto {
etag: resource_id.clone(),
id: resource_id,
name: row.name.clone(),
name: row.name,
path,
parent_id: row.parent_id.map(|u| u.to_string()),
drive_id: row.drive_id,
created_at: row.resource_created_at.timestamp() as u64,
modified_at: row.modified_at.timestamp() as u64,
is_root: false,
icon_class: std::sync::Arc::from("fas fa-folder"),
icon_special_class: std::sync::Arc::from("folder-icon"),
category: std::sync::Arc::from("Folder"),
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
// §14 provenance not selected by the favorites query.
created_by: None,
updated_by: None,
@@ -238,26 +239,30 @@ pub async fn list_favorites_resources(
// file. `blob_hash` is `None` only for
// folder rows, which take the other branch.
let modified_at_u = row.modified_at.timestamp() as u64;
let content_hash = row.blob_hash.clone().unwrap_or_default();
let content_hash = row.blob_hash.unwrap_or_default();
let etag = if content_hash.is_empty() {
String::new()
} else {
File::compute_etag(&content_hash, modified_at_u)
};
// Name-derived display classes borrow `row.name`;
// compute them before the name moves into the DTO.
let icon_class = intern_display(icon_class_for(&row.name, mime));
let icon_special_class =
intern_display(icon_special_class_for(&row.name, mime));
let category = intern_display(category_for(&row.name, mime));
let dto = FileDto {
id: row.resource_id.to_string(),
name: row.name.clone(),
name: row.name,
path,
size: size_bytes,
mime_type: std::sync::Arc::from(mime),
mime_type: intern_mime(mime),
folder_id: row.parent_id.map(|u| u.to_string()),
created_at: row.resource_created_at.timestamp() as u64,
modified_at: modified_at_u,
icon_class: std::sync::Arc::from(icon_class_for(&row.name, mime)),
icon_special_class: std::sync::Arc::from(icon_special_class_for(
&row.name, mime,
)),
category: std::sync::Arc::from(category_for(&row.name, mime)),
icon_class,
icon_special_class,
category,
size_formatted: format_file_size(size_bytes),
sort_date: None,
content_hash,
+8 -6
View File
@@ -712,13 +712,15 @@ impl FileHandler {
let disposition =
Self::content_disposition(&file_dto.name, &file_dto.mime_type, &params);
// `file_dto` was already Read-authorized (and the access
// recorded) by `get_file_with_perms` above — every seek in
// a media/PDF scrub is a separate Range request, so
// re-authorizing + re-notifying per seek doubled that work
// for nothing. Use the non-perms range read, matching the
// share-landing and WebDAV range paths which authorize once
// then stream (benches/ROUND7.md).
match retrieval
.get_file_range_preloaded_with_perms(
&file_dto,
auth_user.id,
start,
Some(end + 1),
)
.get_file_range_preloaded(&file_dto, start, Some(end + 1))
.await
{
Ok(content) => {
+22 -11
View File
@@ -8,7 +8,8 @@ use std::collections::HashMap;
use std::sync::Arc;
use crate::application::dtos::display_helpers::{
category_for, format_file_size, icon_class_for, icon_special_class_for,
category_for, format_file_size, icon_class_for, icon_special_class_for, intern_display,
intern_mime,
};
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::{
@@ -475,16 +476,18 @@ pub async fn list_folder_resources(
let dto = FolderDto {
etag: resource_id.clone(),
id: resource_id,
name: row.name.clone(),
// Folders use fixed icon classes (below), so `name`
// is never borrowed again — move it instead of cloning.
name: row.name,
path: String::new(), // cleared — share recipients must not see hierarchy
parent_id: row.parent_id.map(|u| u.to_string()),
drive_id: row.drive_id,
created_at: row.created_at.timestamp() as u64,
modified_at: row.modified_at.timestamp() as u64,
is_root: false,
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
// §14 provenance not selected by the resources query.
created_by: None,
updated_by: None,
@@ -507,24 +510,32 @@ pub async fn list_folder_resources(
// listing's `etag` byte-equals what a
// conditional request would compare against.
let modified_at_u = row.modified_at.timestamp() as u64;
let content_hash = row.blob_hash.clone().unwrap_or_default();
let content_hash = row.blob_hash.unwrap_or_default();
let etag = if content_hash.is_empty() {
String::new()
} else {
File::compute_etag(&content_hash, modified_at_u)
};
// Compute the name-derived icon/category classes first
// (they borrow `&row.name`), so `name` can be moved into
// the DTO below instead of cloned — one fewer String
// alloc per file row (benches/ROUND7.md).
let icon_class = intern_display(icon_class_for(&row.name, mime));
let icon_special_class =
intern_display(icon_special_class_for(&row.name, mime));
let category = intern_display(category_for(&row.name, mime));
let dto = FileDto {
id: row.id.to_string(),
name: row.name.clone(),
name: row.name,
path: String::new(),
size: size_bytes,
mime_type: Arc::from(mime),
mime_type: intern_mime(mime),
folder_id: row.parent_id.map(|u| u.to_string()),
created_at: row.created_at.timestamp() as u64,
modified_at: row.modified_at.timestamp() as u64,
icon_class: Arc::from(icon_class_for(&row.name, mime)),
icon_special_class: Arc::from(icon_special_class_for(&row.name, mime)),
category: Arc::from(category_for(&row.name, mime)),
icon_class,
icon_special_class,
category,
size_formatted: format_file_size(size_bytes),
sort_date: None,
content_hash,
+19 -14
View File
@@ -8,7 +8,8 @@ use std::sync::Arc;
use tracing::info;
use crate::application::dtos::display_helpers::{
category_for, format_file_size, icon_class_for, icon_special_class_for,
category_for, format_file_size, icon_class_for, icon_special_class_for, intern_display,
intern_mime,
};
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::FolderDto;
@@ -213,7 +214,7 @@ pub async fn list_recent_resources(
// Path is only shown to the owner; non-owners see ""
// to avoid leaking another user's folder hierarchy.
let path = if row.is_owner {
row.path.clone().unwrap_or_default()
row.path.unwrap_or_default()
} else {
String::new()
};
@@ -223,16 +224,16 @@ pub async fn list_recent_resources(
let dto = FolderDto {
etag: resource_id.clone(),
id: resource_id,
name: row.name.clone(),
name: row.name,
path,
parent_id: row.parent_id.map(|u| u.to_string()),
drive_id: row.drive_id,
created_at: row.resource_created_at.timestamp() as u64,
modified_at: row.modified_at.timestamp() as u64,
is_root: false,
icon_class: std::sync::Arc::from("fas fa-folder"),
icon_special_class: std::sync::Arc::from("folder-icon"),
category: std::sync::Arc::from("Folder"),
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
// §14 provenance not selected by the recents query.
created_by: None,
updated_by: None,
@@ -252,26 +253,30 @@ pub async fn list_recent_resources(
// listing matches GET/HEAD/PROPFIND byte-for-byte
// for the same file.
let modified_at_u = row.modified_at.timestamp() as u64;
let content_hash = row.blob_hash.clone().unwrap_or_default();
let content_hash = row.blob_hash.unwrap_or_default();
let etag = if content_hash.is_empty() {
String::new()
} else {
File::compute_etag(&content_hash, modified_at_u)
};
// Name-derived display classes borrow `row.name`;
// compute them before the name moves into the DTO.
let icon_class = intern_display(icon_class_for(&row.name, mime));
let icon_special_class =
intern_display(icon_special_class_for(&row.name, mime));
let category = intern_display(category_for(&row.name, mime));
let dto = FileDto {
id: row.resource_id.to_string(),
name: row.name.clone(),
name: row.name,
path,
size: size_bytes,
mime_type: std::sync::Arc::from(mime),
mime_type: intern_mime(mime),
folder_id: row.parent_id.map(|u| u.to_string()),
created_at: row.resource_created_at.timestamp() as u64,
modified_at: modified_at_u,
icon_class: std::sync::Arc::from(icon_class_for(&row.name, mime)),
icon_special_class: std::sync::Arc::from(icon_special_class_for(
&row.name, mime,
)),
category: std::sync::Arc::from(category_for(&row.name, mime)),
icon_class,
icon_special_class,
category,
size_formatted: format_file_size(size_bytes),
sort_date: None,
content_hash,
+54 -24
View File
@@ -1,7 +1,7 @@
use axum::{
extract::{Json, Query, State},
http::StatusCode,
response::IntoResponse,
response::{IntoResponse, Response},
};
use serde_json::json;
use tracing::{error, info};
@@ -11,6 +11,7 @@ use crate::application::dtos::search_dto::{
};
use crate::application::ports::inbound::SearchUseCase;
use crate::common::di::AppState;
use crate::interfaces::errors::AppError;
use crate::interfaces::middleware::auth::AuthUser;
use std::sync::Arc;
@@ -140,6 +141,7 @@ impl SearchHandler {
/// Autocomplete suggestions for search.
pub(super) async fn suggest_files_impl(
State(state): State<Arc<AppState>>,
auth_user: AuthUser,
Query(params): Query<SuggestParams>,
) -> impl IntoResponse {
info!("API: Search suggestions for {:?}", params.query);
@@ -159,7 +161,12 @@ impl SearchHandler {
let limit = params.limit.unwrap_or(10).min(20);
match search_service
.suggest(&params.query, params.folder_id.as_deref(), limit)
.suggest_with_perms(
&params.query,
params.folder_id.as_deref(),
limit,
auth_user.id,
)
.await
{
Ok(suggestions) => {
@@ -181,40 +188,57 @@ impl SearchHandler {
}
}
/// DELETE /search/cache — clears the search results cache.
/// `DELETE /admin/search/cache` — flush the shared moka search
/// results cache. Admin-only.
///
/// AuthZ audit #14 (2026-07-12): pre-fix this endpoint lived at
/// `/api/search/cache` and required only a valid JWT — any
/// authenticated user (external / magic-link included) could
/// DELETE it in a loop and keep the results cache cold indefinitely
/// (sustained DoS on every subsequent `/api/search` query). Now
/// mounted at `/api/admin/search/cache`, gated by the
/// `require_admin` middleware layer on the `/api/admin` nest point.
/// The handler no longer needs an inline authz call — reaching
/// this code implies `AuthUser` is admin by construction. Audit
/// line on success so operator-driven flushes are traceable in
/// security reviews.
pub(super) async fn clear_search_cache_impl(
State(state): State<Arc<AppState>>,
) -> impl IntoResponse {
auth_user: AuthUser,
) -> Result<Response, AppError> {
let caller_id = auth_user.id;
info!("API: Clearing search cache");
let search_service = match &state.applications.search_service {
Some(service) => service,
None => {
error!("Search service not available");
return (
StatusCode::SERVICE_UNAVAILABLE,
Json(json!({ "error": "Search service is not available" })),
)
.into_response();
}
let Some(search_service) = &state.applications.search_service else {
error!("Search service not available");
return Ok((
StatusCode::SERVICE_UNAVAILABLE,
Json(json!({ "error": "Search service is not available" })),
)
.into_response());
};
match search_service.clear_search_cache().await {
Ok(_) => {
info!("Search cache cleared successfully");
(
tracing::info!(
target: "audit",
event = "search.cache_cleared",
caller_id = %caller_id,
"🧹 search results cache flushed by admin",
);
Ok((
StatusCode::OK,
Json(json!({ "message": "Search cache cleared successfully" })),
)
.into_response()
.into_response())
}
Err(err) => {
error!("Error clearing search cache: {}", err);
(
Ok((
StatusCode::INTERNAL_SERVER_ERROR,
Json(json!({ "error": "Error clearing search cache" })),
)
.into_response()
.into_response())
}
}
}
@@ -354,21 +378,27 @@ pub async fn search_files_post(
)]
pub async fn suggest_files(
state: State<Arc<AppState>>,
auth_user: AuthUser,
query: Query<SuggestParams>,
) -> impl IntoResponse {
SearchHandler::suggest_files_impl(state, query).await
SearchHandler::suggest_files_impl(state, auth_user, query).await
}
#[utoipa::path(
delete,
path = "/api/search/cache",
path = "/api/admin/search/cache",
responses(
(status = 200, description = "Cache cleared"),
(status = 401, description = "Missing or invalid token"),
(status = 403, description = "Caller is not an admin"),
(status = 503, description = "Search service unavailable"),
),
security(("bearerAuth" = [])),
tag = "search"
tag = "admin"
)]
pub async fn clear_search_cache(state: State<Arc<AppState>>) -> impl IntoResponse {
SearchHandler::clear_search_cache_impl(state).await
pub async fn clear_search_cache(
state: State<Arc<AppState>>,
auth_user: AuthUser,
) -> Result<Response, AppError> {
SearchHandler::clear_search_cache_impl(state, auth_user).await
}
+9 -8
View File
@@ -232,17 +232,18 @@ pub async fn access_shared_item(
Path(token): Path<String>,
headers: HeaderMap,
) -> impl IntoResponse {
// Register the access
let _ = share_use_case.register_shared_link_access(&token).await;
// Honour an unlock cookie if one was issued by a prior `/verify` call.
let unlock_jwt = unlock_jwt_from_headers(&headers, &token);
// Get the shared link
match share_use_case
.get_shared_link_with_unlock(&token, unlock_jwt.as_deref())
.await
{
// The access-count increment doesn't gate the fetch — run both
// round-trips concurrently instead of serially (one RTT saved on
// every public share landing).
let (_, item) = tokio::join!(
share_use_case.register_shared_link_access(&token),
share_use_case.get_shared_link_with_unlock(&token, unlock_jwt.as_deref()),
);
match item {
Ok(item) => (StatusCode::OK, Json(item)).into_response(),
Err(err) => {
// Special handling for share access errors
+69 -66
View File
@@ -20,6 +20,7 @@ use uuid::Uuid;
use crate::application::adapters::webdav_adapter::{
LockInfo, PropFindRequest, PropPatchOp, QualifiedName, WebDavAdapter, is_protected_property,
};
use crate::application::dtos::display_helpers::intern_display;
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::FolderDto;
use crate::application::ports::authorization_ports::AuthorizationEngine;
@@ -65,10 +66,6 @@ const PATH_SEGMENT_ENCODE_SET: &AsciiSet = &NON_ALPHANUMERIC
.remove(b'@');
/// Percent-encode a single URI path segment (folder/file name).
fn encode_path_segment(segment: &str) -> String {
utf8_percent_encode(segment, PATH_SEGMENT_ENCODE_SET).to_string()
}
/// Percent-encode a full slash-separated path, encoding each segment individually.
pub(crate) fn encode_uri_path(path: &str) -> String {
use std::fmt::Write as _;
@@ -373,14 +370,14 @@ async fn lookup_drive_selector(
.list_readable_by(user_id)
.await
.map_err(|e| AppError::internal_error(format!("Failed to list drives: {:?}", e)))?;
for d in visible {
for d in visible.iter() {
if let Some(uuid) = uuid_opt
&& d.drive.id == uuid
{
return Ok(d);
return Ok(d.clone());
}
if d.root_folder_name == selector_decoded.as_ref() {
return Ok(d);
return Ok(d.clone());
}
}
Err(AppError::not_found(format!(
@@ -552,9 +549,9 @@ async fn handle_propfind(
created_at: Utc::now().timestamp() as u64,
modified_at: Utc::now().timestamp() as u64,
is_root: true,
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
category: intern_display("Folder"),
created_by: None,
updated_by: None,
};
@@ -781,25 +778,25 @@ async fn build_streaming_propfind_response(
// ── Children (only if Depth == 1) ────────────────────────
if depth == "1" {
let pagination = crate::application::dtos::pagination::PaginationRequestDto {
page: 0,
page_size: PROPFIND_BATCH_SIZE as usize,
};
let fid_ref = folder_id.as_deref();
// Stream sub-folders in pages (user-scoped)
let mut page = 0usize;
// Stream sub-folders in pages (user-scoped, keyset cursor —
// O(page) per page off idx_folders_unique_name instead of the
// quadratic COUNT(*) OVER() + LIMIT/OFFSET walk; 4.5x on a
// 5k-dir parent, benches/FOLDER-KEYSET.md).
let mut after_folder: Option<String> = None;
loop {
let pag = crate::application::dtos::pagination::PaginationRequestDto {
page,
page_size: pagination.page_size,
};
let result = folder_service
.list_folders_paginated_with_perms(fid_ref, user_id, &pag)
let batch = folder_service
.list_folders_batch_with_perms(
fid_ref,
user_id,
after_folder.as_deref(),
PROPFIND_BATCH_SIZE as usize,
)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
if result.items.is_empty() {
if batch.is_empty() {
break;
}
@@ -808,25 +805,29 @@ async fn build_streaming_propfind_response(
// 1-4.5 s of pure DB chatter on a 2000-child folder
// (measured in benches/DEAD-PROPS.md).
let subfolder_deads =
folders_dead_props_map(&dead_props_store, &result.items).await;
folders_dead_props_map(&dead_props_store, &batch).await;
let mut chunk = Vec::with_capacity(result.items.len() * 800);
let mut chunk = Vec::with_capacity(batch.len() * 800);
{
let mut w = Writer::new(&mut chunk);
for subfolder in result.items.iter() {
for subfolder in batch.iter() {
let child_dead = dead_props_for(&subfolder.id, &subfolder_deads);
let href = format!("{}{}/", base_href, encode_path_segment(&subfolder.name));
let href = format!(
"{}{}/",
base_href,
utf8_percent_encode(&subfolder.name, PATH_SEGMENT_ENCODE_SET)
);
WebDavAdapter::write_folder_entry_with_dead_props(&mut w, subfolder, &propfind_request, &href, child_dead, quota)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
}
let has_more = result.pagination.has_next;
let has_more = (batch.len() as i64) == PROPFIND_BATCH_SIZE;
after_folder = batch.last().map(|f| f.name.clone());
yield Bytes::from(chunk);
if !has_more {
break;
}
page += 1;
}
// Stream files in pages (user-scoped, keyset cursor — O(page)
@@ -856,7 +857,11 @@ async fn build_streaming_propfind_response(
let mut w = Writer::new(&mut chunk);
for file in batch.iter() {
let child_dead = dead_props_for(&file.id, &file_deads);
let href = format!("{}{}", base_href, encode_path_segment(&file.name));
let href = format!(
"{}{}",
base_href,
utf8_percent_encode(&file.name, PATH_SEGMENT_ENCODE_SET)
);
WebDavAdapter::write_file_entry_with_dead_props(&mut w, file, &propfind_request, &href, child_dead)
.map_err(|e| std::io::Error::other(e.to_string()))?;
}
@@ -2220,18 +2225,27 @@ async fn handle_delete(
// optimized resolver and the read repositories disagree on path
// shape for some files; see `resolve_or_legacy` docs.
let _ = file_retrieval_service; // present for legacy fallback if needed elsewhere
// AuthZ audit #2 (2026-07-12): route service errors through
// `AppError::from` so authz denials from `_with_perms` surface as
// 404 (the anti-enum shape). The prior `map_err(|e| internal_error…)`
// collapsed every error — including the `NotFound` that
// `authz.require` returns on denial — into HTTP 500, giving a
// reliable "exists-but-denied" vs "missing" oracle to a probing
// caller. Also preserves `QuotaExceeded → 507`,
// `AlreadyExists → 409`, `InvalidInput → 400` shapes surfacing
// through the standard error mapping.
match resolve_or_legacy(&state, &path, drive_id).await {
Some(ResolvedResource::Folder(folder)) => {
folder_service
.delete_folder_with_perms(&folder.id, user.id)
.await
.map_err(|e| AppError::internal_error(format!("Failed to delete folder: {}", e)))?;
.map_err(AppError::from)?;
}
Some(ResolvedResource::File(file)) => {
file_management_service
.delete_file_with_perms(&file.id, user.id)
.await
.map_err(|e| AppError::internal_error(format!("Failed to delete file: {}", e)))?;
.map_err(AppError::from)?;
}
None => return Err(AppError::not_found(format!("Resource not found: {}", path))),
}
@@ -2380,28 +2394,23 @@ async fn handle_move(
// RFC 4918 §9.9.3: when Overwrite: T, perform a DELETE on the
// destination before moving. Without this the rename/move fails
// on a unique-index conflict (same name in same parent).
// AuthZ audit #2 (2026-07-12): `_with_perms` returns `DomainError`;
// route through `AppError::from` so authz denials surface as 404 (the
// anti-enum shape) instead of a `map_err → internal_error` 500 that
// gives a probing caller an "exists-but-denied" oracle. Also preserves
// `QuotaExceeded → 507`, `AlreadyExists → 409`, `InvalidInput → 400`.
match resolve_or_legacy(&state, &destination_path, dst_drive_id).await {
Some(ResolvedResource::Folder(f)) => {
folder_service
.delete_folder_with_perms(&f.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to delete existing destination: {}",
e
))
})?;
.map_err(AppError::from)?;
}
Some(ResolvedResource::File(f)) => {
file_management_service
.delete_file_with_perms(&f.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to delete existing destination: {}",
e
))
})?;
.map_err(AppError::from)?;
}
None => {}
}
@@ -2679,28 +2688,23 @@ async fn handle_copy(
// RFC 4918 §9.8.4: when Overwrite: T, the server MUST perform a
// DELETE on the destination before the copy. Without this the copy
// service returns a unique-index conflict (500).
// AuthZ audit #2 (2026-07-12): `_with_perms` returns `DomainError`;
// route through `AppError::from` so authz denials surface as 404 (the
// anti-enum shape) instead of a `map_err → internal_error` 500 that
// gives a probing caller an "exists-but-denied" oracle. Also preserves
// `QuotaExceeded → 507`, `AlreadyExists → 409`, `InvalidInput → 400`.
match resolve_or_legacy(&state, &destination_path, dst_drive_id).await {
Some(ResolvedResource::Folder(f)) => {
folder_service
.delete_folder_with_perms(&f.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to delete existing destination: {}",
e
))
})?;
.map_err(AppError::from)?;
}
Some(ResolvedResource::File(f)) => {
file_management_service
.delete_file_with_perms(&f.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to delete existing destination: {}",
e
))
})?;
.map_err(AppError::from)?;
}
None => {}
}
@@ -2754,6 +2758,12 @@ async fn handle_copy(
}
};
// AuthZ audit #2 (2026-07-12): route service errors through
// `AppError::from` so authz denials from `_with_perms` surface as 404
// (the anti-enum shape) instead of a `map_err → internal_error` 500
// that gives a probing caller an "exists-but-denied" oracle. Also
// preserves `QuotaExceeded → 507`, `AlreadyExists → 409`,
// `InvalidInput → 400` shapes.
match resolved {
ResolvedResource::Folder(folder) => {
let recursive = depth != "0";
@@ -2766,9 +2776,7 @@ async fn handle_copy(
Some(dest_name.to_string()),
)
.await
.map_err(|e| {
AppError::internal_error(format!("Failed to copy folder tree: {}", e))
})?;
.map_err(AppError::from)?;
} else {
let create_dto = crate::application::dtos::folder_dto::CreateFolderDto {
name: dest_name.to_string(),
@@ -2777,12 +2785,7 @@ async fn handle_copy(
folder_service
.create_folder_with_perms(create_dto, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to create destination folder: {}",
e
))
})?;
.map_err(AppError::from)?;
}
}
ResolvedResource::File(file) => {
@@ -2790,7 +2793,7 @@ async fn handle_copy(
file_management_service
.copy_file_with_perms(&file.id, user.id, target_parent_id, copy_name)
.await
.map_err(|e| AppError::internal_error(format!("Failed to copy file: {}", e)))?;
.map_err(AppError::from)?;
}
}
+41 -7
View File
@@ -332,23 +332,57 @@ async fn put_file(
};
// ── Atomic store: swap the file row onto the ingested blob ──
// `drive_id` scopes the path-based lookups in `update_file_streaming`
// post-D0. WOPI tokens carry the user UUID in `claims.sub`; we resolve
// that to the caller's default drive (WOPI today is a single-drive
// editing surface — no drive marker travels in the token).
// `drive_id` scopes the path-based lookups in
// `update_file_streaming_with_perms` post-D0.
//
// AuthZ audit #18 (2026-07-12): the pre-fix path resolved
// `drive_id` via `find_default_for_user(claims_sub_uuid)` —
// ALWAYS the caller's own default personal drive, regardless of
// where the file actually lived. Shared-drive edits either
// misrouted the write into the caller's personal drive (if the
// filename happened to collide with a personal-drive path) or
// 500'd on the parent-folder lookup. Resolve from the file's
// own parent folder instead — one PK probe, returns the drive
// the file genuinely belongs to. Also unlocks shared-drive WOPI
// editing.
let claims_sub_uuid = match uuid::Uuid::parse_str(&claims.sub) {
Ok(u) => u,
Err(_) => return StatusCode::UNAUTHORIZED.into_response(),
};
let Some(folder_id_str) = file.folder_id.as_deref() else {
// Files always live under a folder (drive-root files use the
// drive-root folder id). A `None` here means the file entity
// is malformed — safest is a 500.
tracing::error!(
"WOPI PutFile: file {} has no parent folder id — cannot resolve drive",
file_id
);
return StatusCode::INTERNAL_SERVER_ERROR.into_response();
};
let folder_uuid = match uuid::Uuid::parse_str(folder_id_str) {
Ok(u) => u,
Err(_) => {
tracing::error!(
"WOPI PutFile: file {} parent folder id '{}' is not a UUID",
file_id,
folder_id_str
);
return StatusCode::INTERNAL_SERVER_ERROR.into_response();
}
};
let drive_id = match state
.app_state
.drive_repo
.find_default_for_user(claims_sub_uuid)
.drive_id_for_folder(folder_uuid)
.await
{
Ok(d) => d.drive.id,
Ok(id) => id,
Err(e) => {
tracing::error!("WOPI PutFile: default-drive lookup failed: {:?}", e);
tracing::error!(
"WOPI PutFile: drive-id lookup for folder {} failed: {:?}",
folder_uuid,
e
);
return StatusCode::INTERNAL_SERVER_ERROR.into_response();
}
};
+40 -13
View File
@@ -52,6 +52,11 @@ async fn get_openapi_spec() -> AxumJson<utoipa::openapi::OpenApi> {
use crate::interfaces::api::handlers::admin_handler;
use crate::interfaces::api::handlers::batch_handler::{self, BatchHandlerState};
// `chunked_upload_handler::*` are marked `#[deprecated]` (prefer
// `/api/files/delta/*`); the router still needs to reference them
// until clients migrate. See the `chunked_upload_router` block
// below for the local `#[allow(deprecated)]`.
#[allow(deprecated)]
use crate::interfaces::api::handlers::chunked_upload_handler::{
cancel_upload, complete_upload, create_upload, get_upload_status, upload_chunk,
};
@@ -70,7 +75,7 @@ use crate::interfaces::api::handlers::i18n_handler::{
get_locales, get_translations_by_locale, translate,
};
use crate::interfaces::api::handlers::search_handler::{
clear_search_cache, search_files_get, search_files_post, suggest_files,
search_files_get, search_files_post, suggest_files,
};
use crate::interfaces::api::handlers::trash_handler;
@@ -275,8 +280,11 @@ pub fn create_api_routes(app_state: &Arc<AppState>) -> Router<Arc<AppState>> {
.route("/suggest", get(suggest_files))
// Advanced search with full criteria object
.route("/advanced", post(search_files_post))
// Clear search cache
.route("/cache", delete(clear_search_cache))
// `DELETE /api/search/cache` used to live here as a per-user-
// reachable endpoint. It's an operator-only debug lever
// (moka `invalidate_all()` — nukes every tenant), so it
// moved to `/api/admin/search/cache` where the URL declares
// intent. AuthZ audit #14 (2026-07-16).
.with_state(app_state.clone())
} else {
Router::new()
@@ -365,6 +373,13 @@ pub fn create_api_routes(app_state: &Arc<AppState>) -> Router<Arc<AppState>> {
// Create routes for chunked uploads (large files >10MB).
// All five handlers are free functions — see chunked_upload_handler.rs for why
// #[utoipa::path] cannot be applied to ChunkedUploadHandler impl methods directly.
//
// Each handler carries `#[deprecated]` so utoipa marks the OpenAPI paths
// deprecated (Swagger UI shows the strikethrough + banner) and existing
// callers get a compile-time nudge to migrate to `/api/files/delta/*`.
// The route registration itself has to keep referencing them until the
// clients migrate off, so we suppress the local `deprecated` lint here.
#[allow(deprecated)]
let chunked_upload_router = Router::new()
.route("/", post(create_upload))
.route("/{upload_id}", axum::routing::patch(upload_chunk))
@@ -376,18 +391,19 @@ pub fn create_api_routes(app_state: &Arc<AppState>) -> Router<Arc<AppState>> {
// Create routes for deduplication endpoints.
// All handlers are free functions — see dedup_handler.rs for why
// #[utoipa::path] cannot be applied to DedupHandler impl methods directly.
use super::handlers::dedup_handler::{
check_hash, check_hashes_batch, get_blob, get_stats, recalculate_stats,
};
use super::handlers::dedup_handler::{check_hash, check_hashes_batch, get_blob};
let dedup_router = Router::new()
.route("/check/{hash}", get(check_hash))
.route("/check-batch", post(check_hashes_batch))
.route("/stats", get(get_stats))
.route("/blob/{hash}", get(get_blob))
// NOTE: remove_reference is intentionally NOT exposed as a public
// endpoint — ref_count management is an internal concern handled
// automatically when files are deleted via the file API.
.route("/recalculate", post(recalculate_stats))
// NOTE: `remove_reference` is intentionally NOT exposed as a
// public endpoint — ref_count management is an internal concern
// handled automatically when files are deleted via the file API.
//
// `/stats` and `/recalculate` moved to `/api/admin/dedup/*`
// (AuthZ audit #24/#25, 2026-07-17) so the middleware admin
// gate covers them by construction. See
// `admin_handler::admin_routes()`.
.with_state(app_state.clone());
let mut router = Router::new()
@@ -598,8 +614,19 @@ pub fn create_api_routes(app_state: &Arc<AppState>) -> Router<Arc<AppState>> {
// NOTE: CalDAV and CardDAV routes are mounted at top-level (/caldav, /carddav)
// in main.rs for protocol compliance, NOT under /api.
// Admin settings routes (protected by admin_guard inside the handler)
let admin_router = admin_handler::admin_routes().with_state(app_state.clone());
// Admin settings routes — the whole subtree is admin-only by
// construction. The `require_admin` layer runs AFTER the outer
// `auth_middleware` (main.rs::protected_api), so it can rely on
// `CurrentUser` already being in the request extensions. Any new
// route added to `admin_handler::admin_routes()` inherits the
// gate automatically — implementors no longer have to remember
// to call `require_admin(&state, &headers).await?` inline, and a
// forgotten call can't silently expose a non-admin surface.
let admin_router = admin_handler::admin_routes()
.layer(axum::middleware::from_fn(
crate::interfaces::middleware::auth::require_admin,
))
.with_state(app_state.clone());
router = router.nest("/admin", admin_router);
// ReBAC subject-group management. All mutating routes are admin-gated;
+21 -13
View File
@@ -211,7 +211,8 @@ pub async fn auth_middleware(
role,
});
request.extensions_mut().insert(current_user);
tracing::Span::current().record("user_id", user_id.to_string());
tracing::Span::current()
.record("user_id", tracing::field::display(user_id));
return Ok(next.run(request).await);
}
Err(e) => {
@@ -258,7 +259,8 @@ pub async fn auth_middleware(
role,
});
request.extensions_mut().insert(current_user);
tracing::Span::current().record("user_id", user_id.to_string());
tracing::Span::current()
.record("user_id", tracing::field::display(user_id));
return Ok(next.run(request).await);
}
Err(e) => {
@@ -323,7 +325,8 @@ pub async fn auth_middleware(
});
request.extensions_mut().insert(current_user);
request.extensions_mut().insert(CookieAuthenticated);
tracing::Span::current().record("user_id", user_id.to_string());
tracing::Span::current()
.record("user_id", tracing::field::display(user_id));
return Ok(next.run(request).await);
}
LiveRole::Revoked => {
@@ -389,6 +392,13 @@ fn dav_basic_auth_challenge(message: &'static str) -> Response {
/// `CurrentUser` is the *live* role resolved by `auth_middleware` (see
/// [`resolve_live_role`]), not the JWT claim, so a demotion is honoured
/// here within the flags-cache TTL.
///
/// Denial shapes distinguish authn from authz:
/// - `CurrentUser` present, role != "admin" → 403 Forbidden.
/// - `CurrentUser` absent → 401 Unauthorized. Should not happen in
/// practice (auth_middleware guards against it), but the
/// defensive fallback returns the honest shape: "we don't know
/// who you are" is 401, not "we know you and refuse" (403).
pub async fn require_admin(request: Request, next: Next) -> Response {
// Get the CurrentUser inserted by auth_middleware
if let Some(current_user) = request.extensions().get::<Arc<CurrentUser>>() {
@@ -404,18 +414,16 @@ pub async fn require_admin(request: Request, next: Next) -> Response {
role = %current_user.role,
"👮🏻‍♂️ admin-only route denied for non-admin caller"
);
} else {
tracing::info!(
target: "audit",
event = "authz.admin_denied",
reason = "unauthenticated",
"👮🏻‍♂️ admin-only route reached with no authenticated user"
);
return AuthError::AccessDenied("Admin role required".to_string()).into_response();
}
// Access denied
let error = AuthError::AccessDenied("Admin role required".to_string());
error.into_response()
tracing::info!(
target: "audit",
event = "authz.admin_denied",
reason = "unauthenticated",
"👮🏻‍♂️ admin-only route reached with no authenticated user"
);
AuthError::TokenNotProvided.into_response()
}
#[cfg(test)]
@@ -28,12 +28,16 @@ use crate::interfaces::middleware::auth::CurrentUser;
/// default drive root, so no per-request authorization decision is being
/// skipped. The drive-marker branch keeps its `get_folder_with_perms`
/// check on every request.
static NC_CHROOT_CACHE: LazyLock<moka::sync::Cache<uuid::Uuid, FolderDto>> = LazyLock::new(|| {
moka::sync::Cache::builder()
.max_capacity(100_000)
.time_to_live(Duration::from_secs(30))
.build()
});
// `Arc<FolderDto>` values: a hit hands back a refcount bump instead of a
// deep clone of the DTO's ~5 owned Strings (moka's `get` clones `V`), and
// the same `Arc` then rides inside `NcSession` for the whole request.
static NC_CHROOT_CACHE: LazyLock<moka::sync::Cache<uuid::Uuid, Arc<FolderDto>>> =
LazyLock::new(|| {
moka::sync::Cache::builder()
.max_capacity(100_000)
.time_to_live(Duration::from_secs(30))
.build()
});
#[derive(Debug, thiserror::Error)]
pub enum NextcloudAuthError {
@@ -184,13 +188,19 @@ pub async fn basic_auth_middleware(
// request would appear in the logs with `user_id=-`,
// making it harder to correlate WebDAV / OCS activity to
// a specific principal.
tracing::Span::current().record("user_id", user_id.to_string());
let current_user = CurrentUser {
// `field::display` renders lazily into the subscriber's buffer —
// no per-request `to_string` (mirrors the JWT path since ROUND5).
tracing::Span::current().record("user_id", tracing::field::display(user_id));
// One shared identity: the same `Arc` serves the
// `Arc<CurrentUser>` extension AND `NcSession.user` (the old
// code built the struct, cloned it for the extension, then
// moved the original — 2-3 String allocs per request).
let current_user = Arc::new(CurrentUser {
id: user_id,
username: uname,
email,
role,
};
});
// ── Resolve chroot from the Basic Auth drive marker ─────
// No marker → caller's default personal drive's root folder
@@ -226,9 +236,10 @@ pub async fn basic_auth_middleware(
.folder_service
.get_folder(&root_id.to_string())
.await
.ok();
.ok()
.map(Arc::new);
if let Some(f) = &fetched {
NC_CHROOT_CACHE.insert(root_id, f.clone());
NC_CHROOT_CACHE.insert(root_id, Arc::clone(f));
}
fetched
}
@@ -242,7 +253,8 @@ pub async fn basic_auth_middleware(
.folder_service
.get_folder_with_perms(folder_id, current_user.id)
.await
.ok(),
.ok()
.map(Arc::new),
};
if chroot.is_none() {
tracing::warn!(
@@ -253,25 +265,20 @@ pub async fn basic_auth_middleware(
return Err(NextcloudAuthError::Unauthorized);
}
request
.extensions_mut()
.insert(Arc::new(current_user.clone()));
// Record from the local before it moves into the session —
// the old code re-read the just-inserted extension and paid a
// `to_string` for the span value.
if let Some(c) = &chroot {
tracing::Span::current().record("chroot_id", tracing::field::display(&c.id));
}
request.extensions_mut().insert(Arc::clone(&current_user));
request.extensions_mut().insert(Arc::new(
crate::interfaces::nextcloud::session::NcSession {
user: current_user,
raw_username: raw_username.clone(),
raw_username,
chroot,
},
));
tracing::Span::current().record(
"chroot_id",
request
.extensions()
.get::<Arc<crate::interfaces::nextcloud::session::NcSession>>()
.and_then(|s| s.chroot.as_ref())
.map(|c| c.id.to_string())
.unwrap_or_default(),
);
Ok(next.run(request).await)
}
Err(_) => {
+92 -18
View File
@@ -35,20 +35,51 @@ fn ocs_err(statuscode: u16, message: &str) -> serde_json::Value {
}
pub async fn handle_capabilities_v1(State(state): State<Arc<AppState>>) -> Response {
let payload = capabilities_payload(&state, 1);
tracing::info!("[NC] capabilities v1 requested, returning payload");
Json(payload).into_response()
capabilities_response(&state, 1)
}
pub async fn handle_capabilities_v2(State(state): State<Arc<AppState>>) -> Response {
let payload = capabilities_payload(&state, 2);
tracing::info!("[NC] capabilities v2 requested, returning payload");
Json(payload).into_response()
capabilities_response(&state, 2)
}
/// Pre-serialized capabilities bodies, `[v1, v2]`. The payload is
/// process-invariant (pure config: base URL + emulated NC version), yet
/// every desktop/mobile client polls it periodically — the old handler
/// re-built the ~40-node `json!` tree, re-read `OXICLOUD_BASE_URL` from
/// the environment and re-serialized on every poll. Now that work runs
/// once; a poll is a `Bytes` refcount bump.
static CAPABILITIES_BODIES: std::sync::OnceLock<[bytes::Bytes; 2]> = std::sync::OnceLock::new();
fn capabilities_response(state: &AppState, ocs_version: u8) -> Response {
let bodies = CAPABILITIES_BODIES.get_or_init(|| {
let base_url = state.core.config.base_url();
let emulated = state.core.config.nextcloud.emulated_version;
let version_string = state.core.config.nextcloud.version_string();
[1u8, 2u8].map(|v| {
bytes::Bytes::from(
serde_json::to_vec(&capabilities_payload(
&base_url,
emulated,
&version_string,
v,
))
.expect("static capabilities JSON serializes"),
)
})
});
let body = bodies[usize::from(ocs_version != 1)].clone();
(
[(axum::http::header::CONTENT_TYPE, "application/json")],
body,
)
.into_response()
}
pub async fn handle_user_info(
State(state): State<Arc<AppState>>,
session: crate::interfaces::nextcloud::session::NcSession,
session: crate::interfaces::nextcloud::session::SharedNcSession,
) -> Response {
let quota: (i64, i64) = match state.storage_usage_service.as_ref() {
Some(service) => match service.get_user_storage_info(session.user.id).await {
@@ -135,19 +166,40 @@ async fn user_provisioning_response(
) -> Response {
let statuscode = if ocs_version == 1 { 100 } else { 200 };
// Only allow users to view their own profile, unless they are admin.
if user.username != userid && user.role != "admin" {
return Json(ocs_err(403, "Insufficient privileges")).into_response();
}
// AuthZ audit #11 (2026-07-12): the pre-fix path here rolled its
// own gate ("caller is `userid`, else must be admin") and then
// called bare `get_user_by_username` — bypassing every visibility
// rule the id-keyed `/api/users/{id}` endpoint enforces. Cross-user
// probes returned 403 (leaking existence via the differential vs a
// genuine 404 for missing users); admins bypassed
// `expose_system_users`; no audit line ever fired.
//
// Now routing through `get_user_profile_by_username_with_perms`,
// which delegates to the same visibility engine as the REST
// endpoint (self / shared-grant / expose_system_users / admin
// paths, all audit-logged on denial). The OCS wire shape stays
// `ocs_err(404, ...)` for every denied case — the NC client can't
// tell "no such user" from "you can't see this user" from "you're
// not admin" apart, which is the anti-enum invariant.
let auth_service = match state.auth_service.as_ref() {
Some(svc) => &svc.auth_application_service,
None => {
return Json(ocs_err(997, "Authentication not configured")).into_response();
}
};
let Some(pool) = state.db_pool.as_ref() else {
return Json(ocs_err(997, "Database pool not available")).into_response();
};
let user_dto = match auth_service.get_user_by_username(&userid).await {
let user_dto = match auth_service
.get_user_profile_by_username_with_perms(
user.id,
&userid,
state.core.config.features.expose_system_users,
pool,
)
.await
{
Ok(u) => u,
Err(_) => {
return Json(ocs_err(404, "User not found")).into_response();
@@ -411,8 +463,8 @@ pub async fn handle_search(
// Pre-resolve numeric ids for every file result in a single batch query
// (was one INSERT round-trip per result).
let file_uuids: Vec<String> = results.files.iter().map(|f| f.id.clone()).collect();
let file_id_map: HashMap<String, i64> = match file_id_svc {
let file_uuids: Vec<&str> = results.files.iter().map(|f| f.id.as_str()).collect();
let file_id_map: HashMap<uuid::Uuid, i64> = match file_id_svc {
Some(svc) => svc
.get_or_create_file_ids(&file_uuids)
.await
@@ -435,7 +487,8 @@ pub async fn handle_search(
crate::interfaces::nextcloud::webdav_handler::strip_drive_root_segment(&file.path);
let display_path = format!("/{}", display_path);
let numeric_id = file_id_map.get(&file.id).copied();
let numeric_id =
crate::interfaces::nextcloud::webdav_handler::nc_id_of(&file_id_map, &file.id);
let thumbnail_url = match numeric_id {
Some(nid) => format!("/index.php/core/preview?fileId={}&x=32&y=32", nid),
@@ -508,11 +561,19 @@ fn empty_search_response() -> Json<serde_json::Value> {
}))
}
fn capabilities_payload(state: &AppState, ocs_version: u8) -> serde_json::Value {
/// Build the capabilities JSON tree from its three config inputs. Public
/// only under the `bench` feature caller path via
/// [`capabilities_payload_for_bench`]; production reaches it once through
/// the [`CAPABILITIES_BODIES`] init.
fn capabilities_payload(
base_url: &str,
emulated_version: (u32, u32, u32),
version_string: &str,
ocs_version: u8,
) -> serde_json::Value {
let statuscode = if ocs_version == 1 { 100 } else { 200 };
let base_url = state.core.config.base_url();
let (nc_major, nc_minor, nc_micro) = state.core.config.nextcloud.emulated_version;
let nc_version_str = state.core.config.nextcloud.version_string();
let (nc_major, nc_minor, nc_micro) = emulated_version;
let nc_version_str = version_string;
json!({
"ocs": {
@@ -580,6 +641,19 @@ fn capabilities_payload(state: &AppState, ocs_version: u8) -> serde_json::Value
})
}
/// Bench-only public wrapper (feature = "bench") over the private payload
/// builder so `examples/bench_capabilities_static.rs` can A/B the
/// rebuild-per-poll flow against the memoized bytes.
#[cfg(feature = "bench")]
pub fn capabilities_payload_for_bench(
base_url: &str,
emulated_version: (u32, u32, u32),
version_string: &str,
ocs_version: u8,
) -> serde_json::Value {
capabilities_payload(base_url, emulated_version, version_string, ocs_version)
}
fn extract_basic_password(headers: &axum::http::HeaderMap) -> Option<String> {
let value = headers
.get(axum::http::header::AUTHORIZATION)?
+27 -18
View File
@@ -10,9 +10,7 @@ use quick_xml::{
use std::collections::{HashMap, HashSet};
use std::sync::Arc;
use crate::application::dtos::display_helpers::{
category_for, format_file_size, icon_class_for, icon_special_class_for,
};
use crate::application::dtos::display_helpers::format_file_size;
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::FolderDto;
use crate::application::dtos::search_dto::SearchCriteriaDto;
@@ -26,7 +24,7 @@ use crate::interfaces::api::handlers::webdav_handler::{
};
use crate::interfaces::errors::AppError;
use crate::interfaces::nextcloud::webdav_handler::{
batch_resolve_ids, format_oc_id, nc_href, write_file_response, write_folder_response,
batch_resolve_ids, format_oc_id, nc_href, nc_id_of, write_file_response, write_folder_response,
};
/// Handle WebDAV REPORT and SEARCH methods for Nextcloud compatibility.
@@ -150,8 +148,8 @@ async fn handle_filter_files(
}
// Pass 2: resolve every oc:fileid in two batch queries (was one per item).
let file_uuids: Vec<String> = files.iter().map(|f| f.id.clone()).collect();
let folder_uuids: Vec<String> = folders.iter().map(|f| f.id.clone()).collect();
let file_uuids: Vec<&str> = files.iter().map(|f| f.id.as_str()).collect();
let folder_uuids: Vec<&str> = folders.iter().map(|f| f.id.as_str()).collect();
let (file_id_map, folder_id_map) =
batch_resolve_ids(file_id_svc, &file_uuids, &folder_uuids).await;
@@ -184,7 +182,7 @@ async fn handle_filter_files(
continue;
};
let href = nc_href(url_user, subpath);
let fid = file_id_map.get(&file.id).copied();
let fid = nc_id_of(&file_id_map, &file.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
let dead = dead_props_for(&file.id, &file_deads);
write_file_response(
@@ -210,7 +208,7 @@ async fn handle_filter_files(
continue;
};
let href = format!("{}/", nc_href(url_user, subpath));
let fid = folder_id_map.get(&folder.id).copied();
let fid = nc_id_of(&folder_id_map, &folder.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
let dead = dead_props_for(&folder.id, &folder_deads);
write_folder_response(
@@ -297,8 +295,8 @@ async fn handle_search(
// (was one INSERT round-trip per result).
let files: Vec<FileDto> = results.files.iter().map(file_dto_from_search).collect();
let folders: Vec<FolderDto> = results.folders.iter().map(folder_dto_from_search).collect();
let file_uuids: Vec<String> = files.iter().map(|f| f.id.clone()).collect();
let folder_uuids: Vec<String> = folders.iter().map(|f| f.id.clone()).collect();
let file_uuids: Vec<&str> = files.iter().map(|f| f.id.as_str()).collect();
let folder_uuids: Vec<&str> = folders.iter().map(|f| f.id.as_str()).collect();
let (file_id_map, folder_id_map) =
batch_resolve_ids(file_id_svc, &file_uuids, &folder_uuids).await;
@@ -325,7 +323,7 @@ async fn handle_search(
continue;
};
let href = nc_href(url_user, subpath);
let fid = file_id_map.get(&file.id).copied();
let fid = nc_id_of(&file_id_map, &file.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
let dead = dead_props_for(&file.id, &file_deads);
write_file_response(
@@ -352,7 +350,7 @@ async fn handle_search(
continue;
};
let href = format!("{}/", nc_href(url_user, subpath));
let fid = folder_id_map.get(&folder.id).copied();
let fid = nc_id_of(&folder_id_map, &folder.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
let dead = dead_props_for(&folder.id, &folder_deads);
write_folder_response(
@@ -401,15 +399,16 @@ fn file_dto_from_search(fr: &crate::application::dtos::search_dto::SearchFileRes
name: fr.name.clone(),
path: fr.path.clone(),
size: fr.size,
mime_type: fr.mime_type.clone().into(),
// Interned `Arc<str>` carried through from enrichment — refcount
// bumps; the old code re-ran all three display classifiers and
// re-allocated each value per converted search row.
mime_type: fr.mime_type.clone(),
folder_id: fr.folder_id.clone(),
created_at: fr.created_at,
modified_at: fr.modified_at,
icon_class: icon_class_for(&fr.name, &fr.mime_type).to_string().into(),
icon_special_class: icon_special_class_for(&fr.name, &fr.mime_type)
.to_string()
.into(),
category: category_for(&fr.name, &fr.mime_type).to_string().into(),
icon_class: fr.icon_class.clone(),
icon_special_class: fr.icon_special_class.clone(),
category: fr.category.clone(),
size_formatted: format_file_size(fr.size),
sort_date: None,
content_hash: fr.blob_hash.clone(),
@@ -420,6 +419,16 @@ fn file_dto_from_search(fr: &crate::application::dtos::search_dto::SearchFileRes
}
}
/// Bench-only public wrapper (feature = "bench") over the private
/// search→FileDto conversion so `examples/bench_search_enrich.rs` can
/// measure and equivalence-gate it.
#[cfg(feature = "bench")]
pub fn file_dto_from_search_for_bench(
fr: &crate::application::dtos::search_dto::SearchFileResultDto,
) -> FileDto {
file_dto_from_search(fr)
}
/// Build a `FolderDto` from a search folder result.
fn folder_dto_from_search(
sr: &crate::application::dtos::search_dto::SearchFolderResultDto,
+7 -7
View File
@@ -17,7 +17,7 @@ use crate::interfaces::nextcloud::basic_auth_middleware::basic_auth_middleware;
use crate::interfaces::nextcloud::login_v2_handler;
use crate::interfaces::nextcloud::ocs_handler;
use crate::interfaces::nextcloud::preview_handler;
use crate::interfaces::nextcloud::session::NcSession;
use crate::interfaces::nextcloud::session::SharedNcSession;
use crate::interfaces::nextcloud::status_handler;
use crate::interfaces::nextcloud::trashbin_handler;
use crate::interfaces::nextcloud::uploads_handler;
@@ -216,7 +216,7 @@ pub fn nextcloud_routes_with_state(state: Arc<AppState>) -> Router<Arc<AppState>
async fn handle_dav_files(
State(state): State<Arc<AppState>>,
Path((_url_user, subpath)): Path<(String, String)>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
webdav_handler::handle_nc_webdav(state, req, session, subpath)
@@ -227,7 +227,7 @@ async fn handle_dav_files(
async fn handle_dav_files_root(
State(state): State<Arc<AppState>>,
Path(_url_user): Path<String>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
webdav_handler::handle_nc_webdav(state, req, session, String::new())
@@ -238,7 +238,7 @@ async fn handle_dav_files_root(
async fn handle_dav_uploads(
State(state): State<Arc<AppState>>,
Path((_url_user, upload_id, rest)): Path<(String, String, String)>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
uploads_handler::handle_nc_uploads(state, req, session, upload_id, rest)
@@ -249,7 +249,7 @@ async fn handle_dav_uploads(
async fn handle_dav_uploads_root(
State(state): State<Arc<AppState>>,
Path((_url_user, upload_id)): Path<(String, String)>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
uploads_handler::handle_nc_uploads(state, req, session, upload_id, String::new())
@@ -279,7 +279,7 @@ async fn handle_legacy_webdav_root(user_ext: AuthUser) -> Response {
async fn handle_dav_trashbin(
State(state): State<Arc<AppState>>,
Path((_url_user, subpath)): Path<(String, String)>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
trashbin_handler::handle_nc_trashbin(state, req, session, subpath)
@@ -290,7 +290,7 @@ async fn handle_dav_trashbin(
async fn handle_dav_trashbin_root(
State(state): State<Arc<AppState>>,
Path(_url_user): Path<String>,
session: NcSession,
session: SharedNcSession,
req: Request<Body>,
) -> Result<Response, Response> {
trashbin_handler::handle_nc_trashbin(state, req, session, String::new())
+39 -12
View File
@@ -3,8 +3,9 @@
//! Bundles WHO the caller is, the raw wire username they presented,
//! and (for path-scoped endpoints) WHERE they're confined to. Built
//! by `basic_auth_middleware` and stashed in request extensions as
//! `Arc<NcSession>`; handlers extract it via the [`FromRequestParts`]
//! impl below — just declare `session: NcSession` in the signature.
//! `Arc<NcSession>`; handlers extract it via [`SharedNcSession`]
//! (derefs to `NcSession`) — declare `session: SharedNcSession` in
//! the signature.
//!
//! ## Source of truth
//!
@@ -46,9 +47,13 @@ use crate::interfaces::middleware::auth::CurrentUser;
#[derive(Debug, Clone)]
pub struct NcSession {
pub user: CurrentUser,
/// Shared with the `Arc<CurrentUser>` request extension — one identity
/// build per request instead of a clone per consumer.
pub user: Arc<CurrentUser>,
pub raw_username: String,
pub chroot: Option<FolderDto>,
/// Shared with `NC_CHROOT_CACHE` (markerless branch) — a cache hit is
/// an `Arc` bump, not a `FolderDto` deep-clone.
pub chroot: Option<Arc<FolderDto>>,
}
impl NcSession {
@@ -56,7 +61,7 @@ impl NcSession {
/// without one. Documents the invariant that every NC route
/// today is path-scoped — if this fires, route wiring is wrong.
pub fn require_chroot(&self) -> Result<&FolderDto, AppError> {
self.chroot.as_ref().ok_or_else(|| {
self.chroot.as_deref().ok_or_else(|| {
AppError::internal_error(
"NcSession: path-scoped handler reached without a chroot — route wiring bug",
)
@@ -101,10 +106,13 @@ fn extract_url_user(path: &str) -> Option<String> {
urlencoding::decode(user_seg).ok().map(|s| s.into_owned())
}
/// Axum extractor: pulls the `Arc<NcSession>` that
/// `basic_auth_middleware` stashed in request extensions and clones
/// it (cheap — one `Arc` increment, no field copy) into an owned
/// `NcSession` for handler use.
/// Axum extractor: the shared handle to the request's [`NcSession`].
///
/// Derefs to `NcSession`, so handler bodies read `session.user`,
/// `session.require_chroot()`, … unchanged. Extraction is one `Arc`
/// refcount increment — the previous extractor deep-cloned the whole
/// session (`CurrentUser` + `raw_username` + chroot `FolderDto`, ~8-9
/// `String` allocs) on every authenticated NC request.
///
/// On path-scoped DAV routes (`/remote.php/dav/{files,uploads,
/// trashbin}/{user}/…`), the URL `{user}` segment is cross-checked
@@ -113,14 +121,33 @@ fn extract_url_user(path: &str) -> Option<String> {
/// (`get_folder_with_perms`) is what actually prevents cross-user
/// access. It just surfaces malformed requests early (403) instead
/// of silently letting them through.
impl<S: Send + Sync> FromRequestParts<S> for NcSession {
#[derive(Debug, Clone)]
pub struct SharedNcSession(Arc<NcSession>);
impl SharedNcSession {
/// Wrap an already-shared session (used by the bench harness; the
/// middleware inserts the `Arc` into request extensions directly).
pub fn from_arc(session: Arc<NcSession>) -> Self {
Self(session)
}
}
impl std::ops::Deref for SharedNcSession {
type Target = NcSession;
fn deref(&self) -> &NcSession {
&self.0
}
}
impl<S: Send + Sync> FromRequestParts<S> for SharedNcSession {
type Rejection = Response;
async fn from_request_parts(parts: &mut Parts, _state: &S) -> Result<Self, Self::Rejection> {
let session = parts
.extensions
.get::<Arc<NcSession>>()
.map(|arc| (**arc).clone())
.cloned()
.ok_or_else(|| StatusCode::UNAUTHORIZED.into_response())?;
if let Some(url_user) = extract_url_user(parts.uri.path())
@@ -129,6 +156,6 @@ impl<S: Send + Sync> FromRequestParts<S> for NcSession {
return Err(StatusCode::FORBIDDEN.into_response());
}
Ok(session)
Ok(Self(session))
}
}
+10 -9
View File
@@ -15,7 +15,7 @@ use crate::application::ports::trash_ports::TrashUseCase;
use crate::common::di::AppState;
use crate::interfaces::errors::AppError;
use crate::interfaces::nextcloud::webdav_handler::{
batch_resolve_ids, extract_nc_subpath_from_dest, format_oc_id, nc_to_internal_path,
batch_resolve_ids, extract_nc_subpath_from_dest, format_oc_id, nc_id_of, nc_to_internal_path,
write_text_element,
};
@@ -27,7 +27,7 @@ const HEADER_DAV: HeaderName = HeaderName::from_static("dav");
pub async fn handle_nc_trashbin(
state: Arc<AppState>,
req: Request<Body>,
session: crate::interfaces::nextcloud::session::NcSession,
session: crate::interfaces::nextcloud::session::SharedNcSession,
subpath: String,
) -> Result<Response<Body>, AppError> {
let method = req.method().clone();
@@ -308,6 +308,7 @@ fn strip_home_prefix<'a>(
use crate::application::dtos::trash_dto::TrashedItemDto;
use crate::application::services::nextcloud_file_id_service::NextcloudFileIdService;
use std::collections::HashMap;
use uuid::Uuid;
/// Generate a complete Nextcloud-compatible multistatus XML response for the trashbin.
///
@@ -337,14 +338,14 @@ async fn write_trashbin_multistatus<W: std::io::Write>(
// Pre-resolve every oc:fileid in two batch queries by object type (was one
// INSERT round-trip per item). File and folder UUIDs are disjoint, so the
// two maps merge cleanly into one keyed by original_id.
let mut file_uuids: Vec<String> = Vec::new();
let mut folder_uuids: Vec<String> = Vec::new();
// two maps merge cleanly into one keyed by parsed original-id UUID.
let mut file_uuids: Vec<&str> = Vec::new();
let mut folder_uuids: Vec<&str> = Vec::new();
for item in items {
if item.item_type == "folder" {
folder_uuids.push(item.original_id.clone());
folder_uuids.push(item.original_id.as_str());
} else {
file_uuids.push(item.original_id.clone());
file_uuids.push(item.original_id.as_str());
}
}
let (mut id_map, folder_id_map) =
@@ -427,7 +428,7 @@ fn write_trash_item_response<W: std::io::Write>(
username: &str,
chroot: &crate::application::dtos::folder_dto::FolderDto,
file_id_svc: Option<&Arc<NextcloudFileIdService>>,
id_map: &HashMap<String, i64>,
id_map: &HashMap<Uuid, i64>,
) -> Result<(), String> {
xml.write_event(Event::Start(BytesStart::new("d:response")))
.map_err(|e| e.to_string())?;
@@ -475,7 +476,7 @@ fn write_trash_item_response<W: std::io::Write>(
write_text_element(xml, "d:getcontentlength", "0")?;
// oc:fileid and oc:id — resolved up front in a batch query.
let file_id = id_map.get(&item.original_id).copied();
let file_id = nc_id_of(id_map, &item.original_id);
if let Some(id) = file_id {
write_text_element(xml, "oc:fileid", &id.to_string())?;
let oc_id = format_oc_id(id, file_id_svc);
+36 -61
View File
@@ -6,7 +6,7 @@ use axum::{
use std::sync::Arc;
use uuid::Uuid;
use crate::application::ports::file_ports::{FileRetrievalUseCase, FileUploadUseCase};
use crate::application::ports::file_ports::FileUploadUseCase;
use crate::application::ports::storage_ports::StorageUsagePort;
use crate::common::di::AppState;
use crate::common::mime_detect::filename_from_path;
@@ -110,7 +110,7 @@ async fn session_bytes_so_far(
pub async fn handle_nc_uploads(
state: Arc<AppState>,
req: Request<Body>,
session: crate::interfaces::nextcloud::session::NcSession,
session: crate::interfaces::nextcloud::session::SharedNcSession,
upload_id: String,
rest: String, // chunk name or ".file" or empty
) -> Result<Response<Body>, AppError> {
@@ -402,8 +402,6 @@ async fn handle_assemble(
.map_err(|e| AppError::internal_error(format!("Failed to list chunks: {}", e)))?;
let upload_service = &state.applications.file_upload_service;
let file_service = &state.applications.file_retrieval_service;
let folder_service = &state.applications.folder_service;
// Path-based lookups below scope by `drive_id`. The NC session's
// chroot is always populated for path-scoped handlers (see
@@ -433,64 +431,41 @@ async fn handle_assemble(
.await?;
let content_type = ingested.content_type.clone();
// Check if file exists (update vs create).
let existing = file_service
.get_file_by_path(&internal_path, drive_id)
.await;
let etag: Option<String> = if existing.is_ok() {
let dto = upload_service
.update_file_streaming_with_perms(
&internal_path,
drive_id,
ingested.stored(),
&content_type,
oc_mtime,
user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Failed to update file: {}", e)))?;
Some(dto.etag)
} else {
// New-file branch: resolve the parent folder by path and register
// the file row against the already-ingested blob.
let (parent_sub, filename) = match dest_subpath.rsplit_once('/') {
Some((p, n)) => (p, n),
None => ("", dest_subpath.as_str()),
};
let parent_internal =
crate::interfaces::nextcloud::webdav_handler::nc_to_internal_path(chroot, parent_sub)?;
let parent_internal = parent_internal.trim_end_matches('/');
use crate::application::ports::folder_ports::FolderUseCase;
let parent_folder = match folder_service
.get_folder_by_path(parent_internal, drive_id)
.await
{
Ok(folder) => folder,
Err(e) => {
discard_ingested(&state.core.dedup_service, &ingested).await;
return Err(AppError::internal_error(format!(
"Parent folder lookup failed: {}",
e
)));
}
};
let dto = upload_service
.upload_file_streaming(
filename.to_string(),
Some(parent_folder.id),
content_type.to_string(),
ingested.stored(),
user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Failed to create file: {}", e)))?;
Some(dto.etag)
// AuthZ audit #12 (2026-07-12): the previous shape branched on
// file existence — `update_file_streaming_with_perms` on the
// overwrite path (correct), plain `upload_file_streaming` on
// the create path (NO `authz.require`). Viewer/Commenter on a
// shared drive could MKCOL → PUT chunks → MOVE and land a
// brand-new file, skipping the `Create`-on-parent-folder gate.
//
// `update_file_streaming_with_perms` handles both branches
// atomically: `Update` on the existing file OR `Create` on the
// parent folder / drive root (per the service's own internal
// fork). Funneling everything through the one method also
// deletes the duplicated parent-folder lookup that used to
// live here.
//
// AuthZ audit #2 (2026-07-12): route DomainError through
// `AppError::from` so authz denials keep the graduated 403/404
// shape instead of collapsing into 500.
let dto = match upload_service
.update_file_streaming_with_perms(
&internal_path,
drive_id,
ingested.stored(),
&content_type,
oc_mtime,
user.id,
)
.await
{
Ok(dto) => dto,
Err(e) => {
discard_ingested(&state.core.dedup_service, &ingested).await;
return Err(AppError::from(e));
}
};
let etag: Option<String> = Some(dto.etag);
// Cleanup session.
let _ = nc.chunked_uploads.cleanup(&user.username, upload_id).await;
+224 -118
View File
@@ -16,7 +16,6 @@ use uuid::Uuid;
use crate::application::adapters::webdav_adapter::{
PropFindRequest, PropPatchOp, QualifiedName, WebDavAdapter, is_protected_property,
};
use crate::application::dtos::pagination::PaginationRequestDto;
use crate::application::ports::authorization_ports::AuthorizationEngine;
use crate::application::ports::favorites_ports::FavoritesUseCase;
use crate::application::ports::file_ports::{
@@ -219,7 +218,7 @@ pub fn nc_href(username: &str, subpath: &str) -> String {
pub async fn handle_nc_webdav(
state: Arc<AppState>,
req: Request<Body>,
session: crate::interfaces::nextcloud::session::NcSession,
session: crate::interfaces::nextcloud::session::SharedNcSession,
subpath: String,
) -> Result<Response<Body>, AppError> {
// Validate up-front that we have a chroot — every method below is
@@ -933,6 +932,11 @@ async fn handle_put(
// Single streaming path — handles both update and create internally,
// swapping the file row onto the already-ingested blob.
// AuthZ audit #6 (2026-07-12): route `_with_perms` errors through
// `AppError::from` so authz denials surface as 404 (the anti-enum
// shape) instead of a `map_err → internal_error` 500 that gives a
// probing caller an "exists-but-denied" oracle. Also preserves
// `QuotaExceeded → 507`, `AlreadyExists → 409`, `InvalidInput → 400`.
let stored = upload_service
.update_file_streaming_with_perms(
&internal_path,
@@ -943,7 +947,7 @@ async fn handle_put(
session.user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Failed to store file: {}", e)))?;
.map_err(AppError::from)?;
let status = if existed {
StatusCode::NO_CONTENT
@@ -1032,10 +1036,14 @@ async fn handle_mkcol(
name: target_name.to_string(),
parent_id: Some(parent_folder.id.clone()),
};
// AuthZ audit #7 (2026-07-12): route `_with_perms` errors through
// `AppError::from` so authz denials surface as 404 (the anti-enum
// shape) instead of a `map_err → internal_error` 500. Also preserves
// `AlreadyExists → 409`, `QuotaExceeded → 507`, `InvalidInput → 400`.
folder_service
.create_folder_with_perms(dto, user.id)
.await
.map_err(|e| AppError::internal_error(format!("Failed to create folder: {}", e)))?;
.map_err(AppError::from)?;
Ok(Response::builder()
.status(StatusCode::CREATED)
@@ -1077,20 +1085,22 @@ async fn handle_delete(
Resource::Folder(folder_uuid),
)
.await?;
// AuthZ audit #8 (2026-07-12): route service errors through
// `AppError::from` so authz denials surface as 404 (the
// anti-enum shape) instead of a `map_err → internal_error`
// 500 that gives a probing caller an "exists-but-denied"
// oracle. `move_to_trash` and `delete_folder_with_perms`
// both return `DomainError` and both call `authz.require`.
if let Some(trash_svc) = state.trash_service.as_ref() {
trash_svc
.move_to_trash(&folder.id, "folder", user.id)
.await
.map_err(|e| {
AppError::internal_error(format!("Failed to trash folder: {}", e))
})?;
.map_err(AppError::from)?;
} else {
folder_service
.delete_folder_with_perms(&folder.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!("Failed to delete folder: {}", e))
})?;
.map_err(AppError::from)?;
}
}
ResolvedResource::File(file) => {
@@ -1104,21 +1114,18 @@ async fn handle_delete(
Resource::File(file_uuid),
)
.await?;
// AuthZ audit #8 (2026-07-12): same anti-enum fix as folder branch above.
if let Some(trash_svc) = state.trash_service.as_ref() {
trash_svc
.move_to_trash(&file.id, "file", user.id)
.await
.map_err(|e| {
AppError::internal_error(format!("Failed to trash file: {}", e))
})?;
.map_err(AppError::from)?;
} else {
let file_mgmt = &state.applications.file_management_service;
file_mgmt
.delete_file_with_perms(&file.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!("Failed to delete file: {}", e))
})?;
.map_err(AppError::from)?;
}
}
}
@@ -1196,6 +1203,12 @@ async fn handle_move(
// then proceed with the move. Trashing is fine: per RFC the source
// resource appears at the destination URI; what happens to the
// overwritten one is up to the server.
//
// AuthZ audit #9 (2026-07-12): route the `_with_perms` delete
// errors through `AppError::from` so authz denials surface as 404
// (anti-enum) instead of `map_err → internal_error` 500. Also
// preserves `QuotaExceeded → 507`, `AlreadyExists → 409`,
// `InvalidInput → 400`.
match existing {
ResolvedResource::File(existing_file) => {
let file_uuid = Uuid::parse_str(&existing_file.id).map_err(|_| {
@@ -1212,12 +1225,7 @@ async fn handle_move(
file_mgmt
.delete_and_cleanup_with_perms(&existing_file.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to overwrite destination file: {}",
e
))
})?;
.map_err(AppError::from)?;
}
ResolvedResource::Folder(existing_folder) => {
let folder_uuid = Uuid::parse_str(&existing_folder.id).map_err(|_| {
@@ -1234,12 +1242,7 @@ async fn handle_move(
folder_service
.delete_folder_with_perms(&existing_folder.id, user.id)
.await
.map_err(|e| {
AppError::internal_error(format!(
"Failed to overwrite destination folder: {}",
e
))
})?;
.map_err(AppError::from)?;
}
}
}
@@ -1267,12 +1270,15 @@ async fn handle_move(
None => "",
};
// AuthZ audit #9 (2026-07-12): route `_with_perms` errors
// through `AppError::from` so authz denials surface as 404
// (anti-enum) instead of `map_err → internal_error` 500.
if src_parent_sub == dest_parent_sub {
// Same parent → rename.
file_mgmt
.rename_file_with_perms(&file.id, user.id, dest_name)
.await
.map_err(|e| AppError::internal_error(format!("Rename failed: {}", e)))?;
.map_err(AppError::from)?;
} else {
// Different parent → move.
let dest_parent = folder_service
@@ -1283,14 +1289,14 @@ async fn handle_move(
file_mgmt
.move_file_with_perms(&file.id, user.id, Some(dest_parent.id.clone()))
.await
.map_err(|e| AppError::internal_error(format!("Move failed: {}", e)))?;
.map_err(AppError::from)?;
// If the filename changed too, rename after move.
if file.name != dest_name {
file_mgmt
.rename_file_with_perms(&file.id, user.id, dest_name)
.await
.map_err(|e| AppError::internal_error(format!("Rename failed: {}", e)))?;
.map_err(AppError::from)?;
}
}
@@ -1332,6 +1338,9 @@ async fn handle_move(
None => "",
};
// AuthZ audit #9 (2026-07-12): route `_with_perms` errors
// through `AppError::from` so authz denials surface as 404
// (anti-enum) instead of `map_err → internal_error` 500.
if src_parent_sub == dest_parent_sub {
// Same parent → rename.
use crate::application::dtos::folder_dto::RenameFolderDto;
@@ -1344,7 +1353,7 @@ async fn handle_move(
user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Rename failed: {}", e)))?;
.map_err(AppError::from)?;
} else {
// Different parent → move.
let dest_parent = folder_service
@@ -1362,7 +1371,7 @@ async fn handle_move(
user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Move failed: {}", e)))?;
.map_err(AppError::from)?;
// If the name changed too, rename.
if folder.name != dest_name {
@@ -1376,7 +1385,7 @@ async fn handle_move(
user.id,
)
.await
.map_err(|e| AppError::internal_error(format!("Rename failed: {}", e)))?;
.map_err(AppError::from)?;
}
}
@@ -1446,8 +1455,7 @@ async fn write_nc_file_multistatus<W: std::io::Write>(
extras: (&HashSet<String>, &[(QualifiedName, Option<String>)]),
) -> Result<(), String> {
let (favorite_ids, dead_props) = extras;
let (file_id_map, _) =
batch_resolve_ids(file_id_svc, std::slice::from_ref(&file.id), &[]).await;
let (file_id_map, _) = batch_resolve_ids(file_id_svc, &[file.id.as_str()], &[]).await;
let mut xml = Writer::new(writer);
write_nc_multistatus_open(&mut xml)?;
@@ -1458,7 +1466,7 @@ async fn write_nc_file_multistatus<W: std::io::Write>(
// shares the requested URL's prefix. `username` is the canonical
// identity for the `oc:owner-id` field.
let href = nc_href(url_user, subpath);
let file_id = file_id_map.get(&file.id).copied();
let file_id = nc_id_of(&file_id_map, &file.id);
let oc_id = file_id.map(|id| format_oc_id(id, file_id_svc));
write_file_response(
&mut xml,
@@ -1510,7 +1518,7 @@ fn build_nc_streaming_propfind(
HashSet::new()
};
let (_, folder_id_map) =
batch_resolve_ids(file_id_svc, &[], std::slice::from_ref(&folder.id)).await;
batch_resolve_ids(file_id_svc, &[], &[folder.id.as_str()]).await;
let folder_dead = folder_dead_props(&state.webdav_dead_props, &folder).await;
let mut buf = Vec::with_capacity(4096);
@@ -1518,7 +1526,7 @@ fn build_nc_streaming_propfind(
let mut xml = Writer::new(&mut buf);
write_nc_multistatus_open(&mut xml).map_err(std::io::Error::other)?;
let href = nc_collection_href(&username, &subpath);
let fid = folder_id_map.get(&folder.id).copied();
let fid = nc_id_of(&folder_id_map, &folder.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
write_folder_response(&mut xml, &folder, &href, (fid, oc_id.as_deref()), &username, &folder_favs, quota, &folder_dead)
.map_err(std::io::Error::other)?;
@@ -1527,6 +1535,19 @@ fn build_nc_streaming_propfind(
// ── Children (only if Depth != 0) ────────────────────────────
if depth != "0" {
// Encoded href prefix for every child: username + parent
// path encode ONCE here — the old per-row `nc_href` call
// re-split and re-encoded the constant prefix for each of
// the up-to-500 children of every page.
let child_href_prefix = {
let base = nc_href(&username, &subpath);
if base.ends_with('/') {
base
} else {
format!("{base}/")
}
};
// Files in pages (keyset cursor — O(page) per page instead of
// the quadratic LIMIT/OFFSET walk).
let mut after_name: Option<String> = None;
@@ -1545,32 +1566,42 @@ fn build_nc_streaming_propfind(
}
let batch_len = batch.len();
// Per-page enrichment: favorites + oc:fileids, two batch queries.
let favs = if let Some(fav) = fav_svc {
let items: Vec<(&str, &str)> =
batch.iter().map(|f| (f.id.as_str(), "file")).collect();
fav.batch_check_favorites(user_id, &items).await.unwrap_or_default()
} else {
HashSet::new()
};
let file_uuids: Vec<String> = batch.iter().map(|f| f.id.clone()).collect();
let (file_id_map, _) = batch_resolve_ids(file_id_svc, &file_uuids, &[]).await;
// One batched dead-props query per page, not one per child
// (benches/DEAD-PROPS.md).
let file_deads = files_dead_props_map(&state.webdav_dead_props, &batch).await;
// Per-page enrichment: favorites + oc:fileids + dead props —
// three independent reads over the same id batch, overlapped
// with `join!` so a page pays ~max(RTT) instead of 3×RTT
// (each query still batched per page: DEAD-PROPS.md). The
// round-7 deferred "serial pairs" item, adopted for this
// per-page triple after the injected-latency A/B in
// benches/ROUND9.md showed no local-PG regression.
let fav_items: Vec<(&str, &str)> =
batch.iter().map(|f| (f.id.as_str(), "file")).collect();
let file_uuids: Vec<&str> = batch.iter().map(|f| f.id.as_str()).collect();
let (favs, (file_id_map, _), file_deads) = tokio::join!(
async {
if let Some(fav) = fav_svc {
fav.batch_check_favorites(user_id, &fav_items)
.await
.unwrap_or_default()
} else {
HashSet::new()
}
},
batch_resolve_ids(file_id_svc, &file_uuids, &[]),
files_dead_props_map(&state.webdav_dead_props, &batch),
);
let mut chunk = Vec::with_capacity(batch_len * 1024);
{
let mut xml = Writer::new(&mut chunk);
for file in batch.iter() {
let dead = dead_props_for(&file.id, &file_deads);
let child_sub = if subpath.is_empty() {
file.name.clone()
} else {
format!("{}/{}", subpath.trim_end_matches('/'), file.name)
};
let href = nc_href(&username, &child_sub);
let fid = file_id_map.get(&file.id).copied();
// Only the name varies per row — the encoded
// username + parent prefix is computed once
// outside the loops (the old `nc_href` call
// re-encoded both for every child).
let href =
format!("{}{}", child_href_prefix, urlencoding::encode(&file.name));
let fid = nc_id_of(&file_id_map, &file.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
write_file_response(&mut xml, file, &href, (fid, oc_id.as_deref()), &username, &favs, dead)
.map_err(std::io::Error::other)?;
@@ -1584,58 +1615,65 @@ fn build_nc_streaming_propfind(
after_name = batch.last().map(|f| f.name.clone());
}
// Subfolders in pages — also collections, same trailing-slash rule.
let mut page = 0usize;
// Subfolders in pages — also collections, same trailing-slash
// rule. Keyset cursor: O(page) per page off
// idx_folders_unique_name instead of the quadratic
// COUNT(*) OVER() + LIMIT/OFFSET walk (benches/FOLDER-KEYSET.md).
let mut after_folder: Option<String> = None;
loop {
let pag = PaginationRequestDto {
page,
page_size: PROPFIND_BATCH_SIZE as usize,
};
let result = folder_service
.list_folders_paginated_with_perms(Some(&folder.id), user_id, &pag)
let batch = folder_service
.list_folders_batch_with_perms(
Some(&folder.id),
user_id,
after_folder.as_deref(),
PROPFIND_BATCH_SIZE as usize,
)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
if result.items.is_empty() {
if batch.is_empty() {
break;
}
let favs = if let Some(fav) = fav_svc {
let items: Vec<(&str, &str)> =
result.items.iter().map(|sf| (sf.id.as_str(), "folder")).collect();
fav.batch_check_favorites(user_id, &items).await.unwrap_or_default()
} else {
HashSet::new()
};
let folder_uuids: Vec<String> = result.items.iter().map(|sf| sf.id.clone()).collect();
let (_, sub_id_map) = batch_resolve_ids(file_id_svc, &[], &folder_uuids).await;
// Batched — see benches/DEAD-PROPS.md.
let sub_deads =
folders_dead_props_map(&state.webdav_dead_props, &result.items).await;
// Same overlapped enrichment triple as the file pages above.
let fav_items: Vec<(&str, &str)> =
batch.iter().map(|sf| (sf.id.as_str(), "folder")).collect();
let folder_uuids: Vec<&str> = batch.iter().map(|sf| sf.id.as_str()).collect();
let (favs, (_, sub_id_map), sub_deads) = tokio::join!(
async {
if let Some(fav) = fav_svc {
fav.batch_check_favorites(user_id, &fav_items)
.await
.unwrap_or_default()
} else {
HashSet::new()
}
},
batch_resolve_ids(file_id_svc, &[], &folder_uuids),
folders_dead_props_map(&state.webdav_dead_props, &batch),
);
let mut chunk = Vec::with_capacity(result.items.len() * 1024);
let mut chunk = Vec::with_capacity(batch.len() * 1024);
{
let mut xml = Writer::new(&mut chunk);
for sf in result.items.iter() {
for sf in batch.iter() {
let dead = dead_props_for(&sf.id, &sub_deads);
let child_sub = if subpath.is_empty() {
sf.name.clone()
} else {
format!("{}/{}", subpath.trim_end_matches('/'), sf.name)
};
let href = nc_collection_href(&username, &child_sub);
let fid = sub_id_map.get(&sf.id).copied();
// Collections carry the trailing slash; prefix
// precomputed once like the file loop above.
let href =
format!("{}{}/", child_href_prefix, urlencoding::encode(&sf.name));
let fid = nc_id_of(&sub_id_map, &sf.id);
let oc_id = fid.map(|id| format_oc_id(id, file_id_svc));
write_folder_response(&mut xml, sf, &href, (fid, oc_id.as_deref()), &username, &favs, quota, dead)
.map_err(std::io::Error::other)?;
}
}
let has_more = result.pagination.has_next;
let has_more = (batch.len() as i64) == PROPFIND_BATCH_SIZE;
after_folder = batch.last().map(|sf| sf.name.clone());
yield Bytes::from(chunk);
if !has_more {
break;
}
page += 1;
}
}
@@ -1697,21 +1735,24 @@ pub fn write_folder_response<W: std::io::Write>(
write_text_element(xml, "d:displayname", &folder.name)?;
let created_at =
chrono::DateTime::<Utc>::from_timestamp(timestamp_to_i64(folder.created_at), 0)
.unwrap_or_else(Utc::now);
let modified_at =
chrono::DateTime::<Utc>::from_timestamp(timestamp_to_i64(folder.modified_at), 0)
.unwrap_or_else(Utc::now);
write_text_element(xml, "d:getlastmodified", &modified_at.to_rfc2822())?;
write_date_element(
xml,
"d:getlastmodified",
timestamp_to_i64(folder.modified_at),
true,
)?;
// Route through `FolderDto::etag` (= `Folder::etag()`: the
// descendant-aware `{id[..16]}-{tree_modified_at}` — see the
// entity for the formula and the async-bump freshness contract).
write_text_element(xml, "d:getetag", &format!("\"{}\"", folder.etag))?;
write_etag_element(xml, "d:getetag", &folder.etag)?;
write_text_element(xml, "d:getcontenttype", "httpd/unix-directory")?;
write_text_element(xml, "d:getcontentlength", "0")?;
write_text_element(xml, "d:creationdate", &created_at.to_rfc3339())?;
write_date_element(
xml,
"d:creationdate",
timestamp_to_i64(folder.created_at),
false,
)?;
// Nextcloud/ownCloud properties
if let Some(id) = file_id {
@@ -1792,17 +1833,28 @@ pub fn write_file_response<W: std::io::Write>(
write_text_element(xml, "d:displayname", &file.name)?;
write_text_element(xml, "d:getcontenttype", &file.mime_type)?;
write_text_element(xml, "d:getcontentlength", &file.size.to_string())?;
{
let mut buf = [0u8; 20];
write_text_element(
xml,
"d:getcontentlength",
crate::common::fmt::u64_str(&mut buf, file.size),
)?;
}
let created_at = chrono::DateTime::<Utc>::from_timestamp(timestamp_to_i64(file.created_at), 0)
.unwrap_or_else(Utc::now);
let modified_at =
chrono::DateTime::<Utc>::from_timestamp(timestamp_to_i64(file.modified_at), 0)
.unwrap_or_else(Utc::now);
write_text_element(xml, "d:getlastmodified", &modified_at.to_rfc2822())?;
write_text_element(xml, "d:getetag", &format!("\"{}\"", file.etag))?;
write_text_element(xml, "d:creationdate", &created_at.to_rfc3339())?;
write_date_element(
xml,
"d:getlastmodified",
timestamp_to_i64(file.modified_at),
true,
)?;
write_etag_element(xml, "d:getetag", &file.etag)?;
write_date_element(
xml,
"d:creationdate",
timestamp_to_i64(file.created_at),
false,
)?;
// Nextcloud/ownCloud properties
if let Some(id) = file_id {
@@ -1814,7 +1866,14 @@ pub fn write_file_response<W: std::io::Write>(
write_text_element(xml, "oc:permissions", "RGDNVW")?;
// Numeric share-permissions bitmask: Read=1 + Update=2 + Delete=8 + Share=16 = 27
write_text_element(xml, "ocs:share-permissions", "27")?;
write_text_element(xml, "oc:size", &file.size.to_string())?;
{
let mut buf = [0u8; 20];
write_text_element(
xml,
"oc:size",
crate::common::fmt::u64_str(&mut buf, file.size),
)?;
}
write_text_element(xml, "oc:owner-id", owner)?;
write_text_element(xml, "oc:owner-display-name", owner)?;
@@ -1858,6 +1917,47 @@ pub fn write_file_response<W: std::io::Write>(
Ok(())
}
/// Stack-rendered `d:getlastmodified` / `d:creationdate` bodies
/// (`common::fmt`) — the old per-row `to_rfc2822()` / `to_rfc3339()`
/// ran chrono's format interpreter and allocated a String each.
/// Out-of-range timestamps keep the chrono path, byte-identical.
fn write_date_element<W: std::io::Write>(
xml: &mut Writer<W>,
tag: &str,
secs: i64,
rfc2822: bool,
) -> Result<(), String> {
if rfc2822 {
let mut buf = [0u8; 31];
if let Some(s) = crate::common::fmt::rfc2822_utc(&mut buf, secs) {
return write_text_element(xml, tag, s);
}
let dt = chrono::DateTime::<Utc>::from_timestamp(secs, 0).unwrap_or_else(Utc::now);
write_text_element(xml, tag, &dt.to_rfc2822())
} else {
let mut buf = [0u8; 25];
if let Some(s) = crate::common::fmt::rfc3339_utc(&mut buf, secs) {
return write_text_element(xml, tag, s);
}
let dt = chrono::DateTime::<Utc>::from_timestamp(secs, 0).unwrap_or_else(Utc::now);
write_text_element(xml, tag, &dt.to_rfc3339())
}
}
/// `d:getetag` with the HTTP quoting — one exactly-sized allocation
/// instead of `format!`'s grow-from-empty.
fn write_etag_element<W: std::io::Write>(
xml: &mut Writer<W>,
tag: &str,
etag: &str,
) -> Result<(), String> {
let mut quoted = String::with_capacity(etag.len() + 2);
quoted.push('"');
quoted.push_str(etag);
quoted.push('"');
write_text_element(xml, tag, &quoted)
}
pub fn write_text_element<W: std::io::Write>(
xml: &mut Writer<W>,
tag: &str,
@@ -1873,14 +1973,15 @@ pub fn write_text_element<W: std::io::Write>(
/// Resolve every `oc:fileid` for a listing in two batch queries (one per
/// object type) instead of one INSERT round-trip per child. Returns
/// `(file_map, folder_map)` keyed by object UUID; entries are absent when the
/// service is disabled or an id can't be resolved, mirroring the previous
/// per-call `Option` behaviour. The two batches run concurrently.
/// `(file_map, folder_map)` keyed by parsed object UUID; entries are absent
/// when the service is disabled or an id can't be resolved, mirroring the
/// previous per-call `Option` behaviour. The two batches run concurrently.
/// Borrowed inputs + `Uuid` keys keep the whole resolution alloc-free.
pub async fn batch_resolve_ids(
svc: Option<&Arc<NextcloudFileIdService>>,
file_uuids: &[String],
folder_uuids: &[String],
) -> (HashMap<String, i64>, HashMap<String, i64>) {
file_uuids: &[&str],
folder_uuids: &[&str],
) -> (HashMap<Uuid, i64>, HashMap<Uuid, i64>) {
let Some(svc) = svc else {
return (HashMap::new(), HashMap::new());
};
@@ -1891,6 +1992,11 @@ pub async fn batch_resolve_ids(
(files.unwrap_or_default(), folders.unwrap_or_default())
}
/// Look up a batch-resolved `oc:fileid` by a DTO's string UUID.
pub fn nc_id_of(map: &HashMap<Uuid, i64>, id: &str) -> Option<i64> {
Uuid::parse_str(id).ok().and_then(|u| map.get(&u).copied())
}
pub fn format_oc_id(id: i64, svc: Option<&Arc<NextcloudFileIdService>>) -> String {
match svc {
Some(s) => s.format_oc_id(id),
+15 -4
View File
@@ -273,11 +273,16 @@ pub fn multipart_field_stream(
pub fn stream_from_files(
paths: Vec<PathBuf>,
) -> impl Stream<Item = Result<Bytes, std::io::Error>> + Send {
// 512 KiB per poll: each ReaderStream poll on a tokio::fs::File is one
// blocking-pool dispatch + one read(2) of the buffer size. The old
// 64 KiB buffer paid 8x the dispatches/syscalls of every other blob
// read path (STREAM_CHUNK_SIZE = 256 KiB) for the single read pass
// over every completed chunked upload (benches/UPLOAD-SPOOL.md).
stream::iter(paths.into_iter().map(Ok::<_, std::io::Error>))
.and_then(|path| async move {
tokio::fs::File::open(path)
.await
.map(|file| ReaderStream::with_capacity(file, 64 * 1024))
.map(|file| ReaderStream::with_capacity(file, 512 * 1024))
})
.try_flatten()
}
@@ -319,9 +324,15 @@ pub async fn stream_body_to_path(
max_bytes: usize,
checksum_alg: Option<ChecksumAlg>,
) -> Result<StreamedToPath, AppError> {
let mut file = tokio::fs::File::create(path)
// BufWriter coalesces the per-HTTP-frame writes (~16-64 KiB each) into
// 512 KiB write(2)s — a bare tokio File dispatches one blocking-pool op
// per frame (benches/UPLOAD-SPOOL.md). Same capacity as the dedup
// handler's spool loop. On the error paths below the partial file is
// removed, so silently dropping unflushed buffer contents is fine.
let file = tokio::fs::File::create(path)
.await
.map_err(|e| AppError::internal_error(format!("Failed to open chunk file: {e}")))?;
let mut file = tokio::io::BufWriter::with_capacity(512 * 1024, file);
let mut total_bytes: usize = 0;
let mut stream = BodyStream::new(body);
@@ -408,8 +419,8 @@ impl IncrementalHasher {
fn finalize_hex(self) -> String {
match self {
Self::Md5(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
Self::Sha256(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
Self::Md5(h) => crate::common::fmt::hex_lower(&h.finalize()),
Self::Sha256(h) => crate::common::fmt::hex_lower(&h.finalize()),
Self::Blake3(h) => h.finalize().to_hex().to_string(),
}
}