Files
Oxicloud/src/application/services/search_service.rs
T

1014 lines
41 KiB
Rust
Raw Normal View History

use std::cmp::Reverse;
2026-02-14 01:29:34 +01:00
use std::sync::Arc;
use std::time::{Duration, Instant};
2025-03-27 01:13:34 +01:00
use crate::application::dtos::display_helpers::intern_display;
2025-03-27 01:13:34 +01:00
use crate::application::dtos::file_dto::FileDto;
use crate::application::dtos::folder_dto::FolderDto;
use crate::application::dtos::search_dto::{
SearchCriteriaDto, SearchFileResultDto, SearchFolderResultDto, SearchResultsDto,
SearchSuggestionItem, SearchSuggestionsDto,
};
use crate::application::ports::content_index_ports::{ContentHitDto, ContentIndexPort};
2025-03-27 01:13:34 +01:00
use crate::application::ports::inbound::SearchUseCase;
use crate::application::ports::storage_ports::FileReadPort;
2026-02-14 01:29:34 +01:00
use crate::common::errors::Result;
use crate::domain::entities::folder::Folder;
use crate::domain::repositories::folder_repository::FolderRepository;
use crate::infrastructure::repositories::pg::file_blob_read_repository::FileBlobReadRepository;
use crate::infrastructure::repositories::pg::folder_db_repository::FolderDbRepository;
use std::hash::{Hash, Hasher};
use uuid::Uuid;
2025-03-27 01:13:34 +01:00
/**
* High-performance search service implementation for files and folders.
2026-02-14 01:29:34 +01:00
*
* All search processing (filtering, scoring, sorting, categorization,
* formatting) is performed server-side in Rust for maximum efficiency.
* The frontend acts as a thin rendering client only.
*
* Features:
* - Single-query recursive subtree search via PostgreSQL ltree
* - Relevance scoring (exact match > starts-with > contains)
* - Content categorization and icon mapping
* - Multiple sort options (relevance, name, date, size)
* - Server-side formatted file sizes
* - Quick suggestions endpoint for autocomplete
* - TTL-based result caching
2025-03-27 01:13:34 +01:00
*/
pub struct SearchService {
/// Repository for file operations
file_repository: Arc<FileBlobReadRepository>,
2026-02-14 01:29:34 +01:00
/// Repository for folder operations
folder_repository: Arc<FolderDbRepository>,
2026-02-14 01:29:34 +01:00
/// Optional full-text content index (embedded Tantivy). When present,
/// query-bearing searches additionally surface files whose CONTENT
/// matches; hits are hydrated and re-filtered through SQL before use.
content_index: Option<Arc<dyn ContentIndexPort>>,
2026-06-18 13:29:41 +02:00
/// Optional authorization engine — needed to resolve the caller's
/// accessible drive set before querying the content index, and to
/// re-verify each Tantivy hit against `engine.check(Read, File(id))`
/// as a defense-in-depth measure (catches index staleness and
/// per-file grants that the drive-only Tantivy filter misses; see
/// `docs/plan/drive.md` §11). `None` short-circuits the content
/// index (the cheapest safe degradation).
authorization: Option<Arc<crate::infrastructure::services::pg_acl_engine::PgAclEngine>>,
/// Optional drive repository — used in tandem with the authorization
/// engine to resolve the caller's accessible drives for the Tantivy
/// filter. `None` short-circuits the content index.
drive_repo: Option<Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>>,
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
///
/// **Byte-bounded**, not entry-bounded: entries are weighed by
/// [`search_results_entry_weight`] and `max_capacity` is a byte budget.
/// Keys span user × query × offset × limit, and each page holds up to 500
/// enriched rows (~500–900 B of owned Strings each) — an entry-count bound
/// let hundreds of MB of result pages accumulate invisibly.
search_cache: moka::future::Cache<u64, Arc<SearchResultsDto>>,
2025-03-27 01:13:34 +01:00
}
// ─── Search-results cache (byte-bounded) ─────────────────────────────────
/// Approximate heap bytes retained by one cached search page.
///
/// With a `weigher` installed, moka's `max_capacity` is the sum of entry
/// *weights*, so this converts the cache bound from "number of entries" to
/// real bytes: the length of every owned `String` in each file/folder row,
/// plus a fixed per-row and per-entry overhead for struct fields, the 24-B
/// `String` headers, `Vec` slots and allocator slop. Same pattern as the
/// file-content cache and the dedup manifest cache.
///
/// `pub` so `examples/bench_search_cache_mem.rs` can recompute retained
/// bytes with the exact production formula.
pub fn search_results_entry_weight(_key: &u64, value: &Arc<SearchResultsDto>) -> u32 {
/// Fixed per-row overhead: struct scalars + one 24-B header per `String`
/// field (12 on a file row, 4 on a folder row) + `Vec` slot + allocator
/// slop. Deliberately a round upper-ish estimate — under-weighing is the
/// failure mode that re-opens the memory hole.
const ROW_OVERHEAD: usize = 200;
/// Fixed per-entry overhead: `Arc` + `SearchResultsDto` scalars + `Vec`
/// headers + moka's own bookkeeping per entry.
const ENTRY_OVERHEAD: usize = 256;
fn opt_len(s: &Option<String>) -> usize {
s.as_deref().map_or(0, str::len)
}
let mut bytes = ENTRY_OVERHEAD + value.sort_by.len();
for f in &value.files {
bytes += ROW_OVERHEAD
+ f.id.len()
+ f.name.len()
+ f.path.len()
+ f.mime_type.len()
+ opt_len(&f.folder_id)
+ f.size_formatted.len()
+ f.icon_class.len()
+ f.icon_special_class.len()
+ f.category.len()
+ f.blob_hash.len()
+ opt_len(&f.snippet)
+ opt_len(&f.match_source);
}
for d in &value.folders {
bytes += ROW_OVERHEAD + d.id.len() + d.name.len() + d.path.len() + opt_len(&d.parent_id);
}
bytes.min(u32::MAX as usize) as u32
}
/// Build the search-results cache exactly as production wires it: a byte
/// budget enforced through [`search_results_entry_weight`], plus TTL.
///
/// Shared with `examples/bench_search_cache_mem.rs` so the benchmark
/// measures the identical cache configuration that serves requests.
pub fn build_search_results_cache(
cache_ttl_secs: u64,
max_bytes: u64,
) -> moka::future::Cache<u64, Arc<SearchResultsDto>> {
moka::future::Cache::builder()
.max_capacity(max_bytes)
.weigher(search_results_entry_weight)
.time_to_live(Duration::from_secs(cache_ttl_secs))
.build()
}
// ─── Utility functions (pure, no self — computed on the server) ─────────
/// Compute relevance score (0–100) for a name against a query.
/// Exact match = 100, starts-with = 80, contains = 50, no match = 0.
///
/// `query_lower` **must** already be lowercased by the caller so that the
/// allocation happens once per search, not once per result.
fn compute_relevance(name: &str, query_lower: &str) -> u32 {
let name_lower = name.to_lowercase();
if name_lower == query_lower {
100
} else if name_lower.starts_with(query_lower) {
80
} else if name_lower.contains(query_lower) {
// Bonus for shorter names (more specific match)
let ratio = query_lower.len() as f64 / name_lower.len() as f64;
50 + (ratio * 20.0) as u32
} else {
0
}
}
/// Max content-index candidates fetched per search. Hydration re-filters
/// them in ONE SQL round-trip, so this bounds both index and DB work.
const CONTENT_HITS_LIMIT: usize = 200;
/// Map a BM25 score into the 10–45 relevance band, normalized against the
/// best score of the result set. Deliberately below the weakest name match
/// (contains = 50): a filename hit is more specific than a body mention.
fn content_relevance(score: f32, max_score: f32) -> u32 {
if !score.is_finite() || max_score <= 0.0 {
return 10;
}
let ratio = (score / max_score).clamp(0.0, 1.0);
10 + (ratio * 35.0).round() as u32
}
/// Re-sort the merged file list with the same semantics the folder list
/// uses. Only invoked when content hits were merged into a SQL-ordered page.
fn sort_enriched_files(files: &mut [SearchFileResultDto], sort_by: &str) {
match sort_by {
"name" => files.sort_by_cached_key(|f| f.name.to_lowercase()),
"name_desc" => files.sort_by_cached_key(|f| Reverse(f.name.to_lowercase())),
"date" => files.sort_by_key(|f| f.modified_at),
"date_desc" => files.sort_by_key(|f| Reverse(f.modified_at)),
"size" => files.sort_by_key(|f| f.size),
"size_desc" => files.sort_by_key(|f| Reverse(f.size)),
_ => files.sort_by_key(|f| Reverse(f.relevance_score)),
}
}
/// Format bytes into a human-readable string (e.g. "2.5 MB").
fn format_bytes(bytes: u64) -> String {
const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
if bytes == 0 {
return "0 B".to_string();
}
let exp = (bytes as f64).log(1024.0).floor() as usize;
let exp = exp.min(UNITS.len() - 1);
let value = bytes as f64 / 1024_f64.powi(exp as i32);
if exp == 0 {
format!("{} B", bytes)
} else {
format!("{:.1} {}", value, UNITS[exp])
}
}
// ─── SearchService implementation ───────────────────────────────────────
2025-03-27 01:13:34 +01:00
impl SearchService {
/**
* Creates a new instance of the search service.
*
* `max_cache_bytes` is the byte budget for the results cache (weigher-
* bounded, see [`search_results_entry_weight`]) — it replaced the old
* entry-count capacity, which was blind to how big each cached page is.
2025-03-27 01:13:34 +01:00
*/
pub fn new(
file_repository: Arc<FileBlobReadRepository>,
folder_repository: Arc<FolderDbRepository>,
content_index: Option<Arc<dyn ContentIndexPort>>,
2026-06-18 13:29:41 +02:00
authorization: Option<Arc<crate::infrastructure::services::pg_acl_engine::PgAclEngine>>,
drive_repo: Option<Arc<dyn crate::domain::repositories::drive_repository::DriveRepository>>,
2025-03-27 01:13:34 +01:00
cache_ttl: u64,
max_cache_bytes: u64,
2025-03-27 01:13:34 +01:00
) -> Self {
let search_cache = build_search_results_cache(cache_ttl, max_cache_bytes);
Self {
2025-03-27 01:13:34 +01:00
file_repository,
folder_repository,
content_index,
2026-06-18 13:29:41 +02:00
authorization,
drive_repo,
search_cache,
2025-03-27 01:13:34 +01:00
}
}
2026-02-14 01:29:34 +01:00
/// Creates a cache key from the search criteria using zero-allocation hashing.
fn create_cache_key(criteria: &SearchCriteriaDto, user_id: &str) -> u64 {
let mut hasher = std::collections::hash_map::DefaultHasher::new();
criteria.hash(&mut hasher);
user_id.hash(&mut hasher);
hasher.finish()
2025-03-27 01:13:34 +01:00
}
2026-02-14 01:29:34 +01:00
/// Enrich a FileDto → SearchFileResultDto with server-computed metadata.
///
/// Consumes the DTO: every `String` moves and the interned display
/// fields (`mime_type`/`icon_class`/`icon_special_class`/`category`,
/// already computed once in `FileDto::from`) transfer as refcount
/// bumps — the old borrow-based version cloned all of them AND re-ran
/// the three display classifiers per result row.
///
/// `query_lower` must already be lowercased (empty string when no query).
fn enrich_file(file: FileDto, query_lower: &str) -> SearchFileResultDto {
let relevance = if query_lower.is_empty() {
50
} else {
compute_relevance(&file.name, query_lower)
};
2026-02-14 01:29:34 +01:00
SearchFileResultDto {
id: file.id,
name: file.name,
path: file.path,
size: file.size,
mime_type: file.mime_type,
folder_id: file.folder_id,
created_at: file.created_at,
modified_at: file.modified_at,
relevance_score: relevance,
size_formatted: format_bytes(file.size),
icon_class: file.icon_class,
icon_special_class: file.icon_special_class,
category: file.category,
// Carry the content hash through so REPORT/SEARCH
// responses on the NC surface can emit the same ETag
// (`File::compute_etag`) as PROPFIND/GET would.
blob_hash: file.content_hash,
snippet: None,
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
}
}
2026-02-14 01:29:34 +01:00
/// Enrich a FolderDto → SearchFolderResultDto with server-computed metadata.
///
/// Consumes the DTO so the owned strings move instead of cloning.
///
/// `query_lower` must already be lowercased (empty string when no query).
fn enrich_folder(folder: FolderDto, query_lower: &str) -> SearchFolderResultDto {
let relevance = if query_lower.is_empty() {
50
} else {
compute_relevance(&folder.name, query_lower)
};
2026-02-14 01:29:34 +01:00
SearchFolderResultDto {
id: folder.id,
name: folder.name,
path: folder.path,
parent_id: folder.parent_id,
drive_id: folder.drive_id,
created_at: folder.created_at,
modified_at: folder.modified_at,
is_root: folder.is_root,
relevance_score: relevance,
}
2025-03-27 01:13:34 +01:00
}
2026-02-14 01:29:34 +01:00
/// Query the content index for files matching by CONTENT (when the index
/// is enabled). First page only — content hits have no stable
/// interleaving with SQL pagination beyond it, and page one is where
/// search UX lives. Index failures degrade to name-only results, never
/// to a failed search.
async fn lookup_content_hits(
&self,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Vec<ContentHitDto> {
2026-06-18 13:29:41 +02:00
use crate::application::ports::authorization_ports::AuthorizationEngine;
use crate::domain::services::authorization::Subject;
2026-06-18 13:29:41 +02:00
let Some(index) = &self.content_index else {
return Vec::new();
};
2026-06-18 13:29:41 +02:00
let Some(authz) = &self.authorization else {
return Vec::new();
};
let Some(drive_repo) = &self.drive_repo else {
return Vec::new();
};
if criteria.offset != 0 {
return Vec::new();
}
let Some(query) = criteria
.name_contains
.as_deref()
.map(str::trim)
.filter(|q| q.len() >= 2)
else {
return Vec::new();
};
2026-07-02 00:01:15 +02:00
// Resolve the caller's accessible drive set. Group-mediated
// grants are honoured inline by `storage.caller_group_ids` on
// the SQL side, so no Rust-side subject expansion here.
let accessible_drives: Vec<Uuid> = match drive_repo.list_readable_by(user_id).await {
Ok(drives) => drives.iter().map(|d| d.drive.id).collect(),
2026-06-18 13:29:41 +02:00
Err(e) => {
tracing::warn!("Content-index: drive lookup failed — degrading to empty: {e}");
return Vec::new();
}
};
// Tantivy filter (Must drive_id ∈ accessible_drives) handles
// the cross-drive isolation. Empty drive list short-circuits
// inside `search_content`.
let hits = match index
.search_content(&accessible_drives, query, CONTENT_HITS_LIMIT)
.await
{
Ok(hits) => hits,
Err(e) => {
tracing::warn!("Content-index lookup failed — returning name-only results: {e}");
2026-06-18 13:29:41 +02:00
return Vec::new();
}
};
// Defense in depth: re-verify each hit through the engine.
// Catches two cases the drive_id filter can't:
// * Index staleness — the file just moved drives and the
// worker hasn't caught up.
// * Per-file grants — ReBAC can grant a single file inside a
// drive the caller doesn't otherwise have. The Tantivy
// filter is drive-only; this re-check restores per-file
// resolution.
// Failures degrade conservatively (drop the hit / the page,
// log it) — never leak. Batched: one drive-resolution query for
// the whole page instead of up to CONTENT_HITS_LIMIT sequential
// point SELECTs (benches/SEARCH-REBAC.md).
let mut hit_ids = Vec::with_capacity(hits.len());
for hit in &hits {
match Uuid::parse_str(&hit.file_id) {
Ok(u) => hit_ids.push(u),
2026-06-18 13:29:41 +02:00
Err(_) => {
tracing::warn!("Content-index hit had non-UUID file_id: {}", hit.file_id);
}
}
}
let allowed = match authz
.check_files_read_batch(Subject::User(user_id), &hit_ids)
.await
{
Ok(set) => set,
Err(e) => {
tracing::warn!("ReBAC re-check failed for content hits: {e}");
return Vec::new();
}
};
let mut verified = Vec::with_capacity(hits.len());
for hit in hits {
let Ok(file_uuid) = Uuid::parse_str(&hit.file_id) else {
continue; // already warned above
2026-06-18 13:29:41 +02:00
};
if allowed.contains(&file_uuid) {
verified.push(hit);
} else {
tracing::debug!(
target: "oxicloud::search",
file_id = %file_uuid,
"dropping content-index hit: ReBAC denies Read after Tantivy filter",
);
}
}
2026-06-18 13:29:41 +02:00
verified
}
/// Merge content-index hits into the name-search result page:
/// * files the name search already found just gain their `snippet`;
/// * content-only candidates are hydrated through SQL in one round-trip
/// (re-applying user scope, trash state and every active filter — a
/// stale index id silently drops out), enriched, scored into the
/// content relevance band and appended;
/// * the merged page is re-sorted with the caller's `sort_by`.
///
/// Returns how many files were added (callers bump their totals by it).
async fn merge_content_hits(
&self,
hits: Vec<ContentHitDto>,
enriched_files: &mut Vec<SearchFileResultDto>,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Result<usize> {
if hits.is_empty() {
return Ok(0);
}
let mut by_id: std::collections::HashMap<&str, &ContentHitDto> =
hits.iter().map(|h| (h.file_id.as_str(), h)).collect();
for file in enriched_files.iter_mut() {
if let Some(hit) = by_id.remove(file.id.as_str()) {
file.snippet = hit.snippet.clone();
}
}
if by_id.is_empty() {
return Ok(0);
}
// Preserve the index's score order when collecting the leftovers.
let candidate_ids: Vec<String> = hits
.iter()
.filter(|h| by_id.contains_key(h.file_id.as_str()))
.map(|h| h.file_id.clone())
.collect();
let files = self
.file_repository
.fetch_files_by_ids_filtered(&candidate_ids, criteria, user_id)
.await?;
if files.is_empty() {
return Ok(0);
}
let max_score = hits.iter().map(|h| h.score).fold(0.0_f32, f32::max);
let mut added = 0usize;
for file in files {
let dto = FileDto::from(file);
let Some(hit) = by_id.get(dto.id.as_str()) else {
continue;
};
let (score, snippet) = (hit.score, hit.snippet.clone());
let mut enriched = Self::enrich_file(dto, "");
enriched.relevance_score = content_relevance(score, max_score);
enriched.snippet = snippet;
enriched.match_source = Some("content".to_string());
enriched_files.push(enriched);
added += 1;
}
if added > 0 {
sort_enriched_files(enriched_files, &criteria.sort_by);
}
Ok(added)
}
/// Quick suggestions search — returns up to `limit` name suggestions
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
/// so only a handful of rows cross the DB→app boundary.
///
/// `caller_id` scopes the underlying repo queries to drives the caller
/// can Read. Without it (the pre-fix shape) any authenticated user —
/// including external magic-link recipients — could autocomplete both
/// names and full paths across every tenant on the instance (AuthZ
/// audit finding #1, 2026-07-12). Named `_with_perms` per the
/// AGENTS.md AuthZ convention.
pub async fn suggest_with_perms(
2025-03-27 01:13:34 +01:00
&self,
query: &str,
folder_id: Option<&str>,
limit: usize,
caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
let start = Instant::now();
// Ask SQL for at most `limit` best-matching files and folders
let (files, folders) = tokio::join!(
self.file_repository
.suggest_files_by_name(folder_id, query, limit, caller_id),
self.folder_repository
.suggest_folders_by_name(folder_id, query, limit, caller_id),
);
let files = files?;
let folders = folders?;
let mut suggestions: Vec<SearchSuggestionItem> =
Vec::with_capacity(files.len() + folders.len());
// Pre-compute once — avoids N heap allocations inside the loops.
let query_lower = query.to_lowercase();
// Consume the entities: the old loop deep-cloned every File into
// the DTO conversion and then cloned name/id/path AGAIN into the
// suggestion — 3 field clones + a full entity clone per row on
// an every-keystroke path.
for file in files {
let file_dto = FileDto::from(file);
let score = compute_relevance(&file_dto.name, &query_lower);
suggestions.push(SearchSuggestionItem {
name: file_dto.name,
item_type: "file".to_string(),
id: file_dto.id,
path: file_dto.path,
// Interned in `FileDto::from` — reuse instead of re-running
// the display classifiers per keystroke suggestion.
icon_class: file_dto.icon_class,
icon_special_class: file_dto.icon_special_class,
relevance_score: score,
});
}
2026-02-14 01:29:34 +01:00
for folder in folders {
let folder_dto = FolderDto::from(folder);
let score = compute_relevance(&folder_dto.name, &query_lower);
suggestions.push(SearchSuggestionItem {
name: folder_dto.name,
item_type: "folder".to_string(),
id: folder_dto.id,
path: folder_dto.path,
icon_class: intern_display("fas fa-folder"),
icon_special_class: intern_display("folder-icon"),
relevance_score: score,
});
}
// Merge files + folders by relevance and truncate to the final limit
2026-05-04 12:46:57 +02:00
suggestions.sort_by_key(|f| Reverse(f.relevance_score));
suggestions.truncate(limit);
let elapsed = start.elapsed().as_millis() as u64;
Ok(SearchSuggestionsDto {
suggestions,
query_time_ms: elapsed,
2026-02-14 01:29:34 +01:00
})
2025-03-27 01:13:34 +01:00
}
}
// ─── Bench-only public wrappers (feature = "bench") ──────────────────────
#[cfg(feature = "bench")]
impl SearchService {
/// Public wrapper over the private `enrich_file` so
/// `examples/bench_search_enrich.rs` can measure it.
pub fn enrich_file_for_bench(file: FileDto, query_lower: &str) -> SearchFileResultDto {
Self::enrich_file(file, query_lower)
}
/// Public wrapper over the private `enrich_folder` for the same bench.
pub fn enrich_folder_for_bench(folder: FolderDto, query_lower: &str) -> SearchFolderResultDto {
Self::enrich_folder(folder, query_lower)
}
}
// ─── SearchUseCase trait implementation ──────────────────────────────────
2025-03-27 01:13:34 +01:00
impl SearchUseCase for SearchService {
/**
* Performs a search based on the specified criteria.
2026-02-14 01:29:34 +01:00
*
* Optimization: For non-recursive searches, uses database-level pagination
* for better performance. For recursive searches, uses the parallel approach.
*
* All processing happens server-side:
* - Database-level pagination for non-recursive searches
* - Parallel recursive traversal for recursive searches
* - Filtering by name, type, dates, size
* - Relevance scoring
* - Sorting (relevance, name, date, size)
* - Content categorization & icon mapping
* - Human-readable size formatting
* - Pagination
2025-03-27 01:13:34 +01:00
*/
2026-02-26 00:47:32 +01:00
async fn search(
&self,
criteria: SearchCriteriaDto,
user_id: Uuid,
2026-02-26 00:47:32 +01:00
) -> Result<Arc<SearchResultsDto>> {
let user_id_str = user_id.to_string();
let cache_key = Self::create_cache_key(&criteria, &user_id_str);
// Single-flight: collapse N identical concurrent searches into ONE
// execution. `try_get_with` serves the cached result on a hit and, on a
// miss, runs the closure exactly once while the other callers await it
// — so a burst of identical queries no longer floods Postgres or drains
// the connection pool (the old get-from-cache fast path is subsumed).
self.search_cache
.try_get_with(cache_key, async move {
let start = Instant::now();
let query = criteria.name_contains.as_deref().unwrap_or("");
// Pre-compute once — avoids N heap allocations inside enrich_file/enrich_folder.
let query_lower = query.to_lowercase();
// Content-index candidates (first page only). Feature-off or an
// index failure yields an empty set — the search stays name-only.
let content_hits = self.lookup_content_hits(&criteria, user_id).await;
// For non-recursive searches, use efficient database-level pagination
// This avoids loading all files into memory
if !criteria.recursive {
// Use database-level pagination
let (files, total_file_count) = self
.file_repository
.search_files_paginated(criteria.folder_id.as_deref(), &criteria, user_id)
.await?;
// Convert to DTOs and enrich with metadata — one fused
// pass, no intermediate Vec<FileDto> materialization.
let mut enriched_files: Vec<SearchFileResultDto> = files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.collect();
// Get folders for this folder (non-recursive, filtered in SQL)
let folders = self
.folder_repository
.search_folders(
criteria.folder_id.as_deref(),
criteria.name_contains.as_deref(),
user_id,
false,
)
.await?;
// For folders, apply sorting and pagination in memory (usually fewer folders)
let mut enriched_folders: Vec<SearchFolderResultDto> = folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.collect();
// Sort folders (cached_key avoids O(N log N) temporary String allocations)
match criteria.sort_by.as_str() {
"name" => {
enriched_folders.sort_by_cached_key(|f| f.name.to_lowercase());
}
"name_desc" => {
enriched_folders.sort_by_cached_key(|f| Reverse(f.name.to_lowercase()));
}
"date" => {
enriched_folders.sort_by_key(|f| f.modified_at);
}
"date_desc" => {
enriched_folders.sort_by_key(|f| Reverse(f.modified_at));
}
_ => {
enriched_folders.sort_by_key(|f| Reverse(f.relevance_score));
}
}
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
// Combine and paginate (folders first, then files)
let start_idx = criteria.offset.min(total_count);
let end_idx = (criteria.offset + criteria.limit).min(total_count);
let folder_start = start_idx.min(folder_count);
let folder_end = end_idx.min(folder_count);
let paginated_folders = enriched_folders[folder_start..folder_end].to_vec();
let file_start = start_idx.saturating_sub(folder_count);
let file_end = end_idx
.saturating_sub(folder_count)
.min(enriched_files.len());
let paginated_files = enriched_files[file_start..file_end].to_vec();
let elapsed_ms = start.elapsed().as_millis() as u64;
let search_results = Arc::new(SearchResultsDto::new(
paginated_files,
paginated_folders,
criteria.limit,
criteria.offset,
Some(total_count),
elapsed_ms,
criteria.sort_by.clone(),
));
return Ok(search_results);
}
// ── Recursive search via ltree (single SQL query per entity type) ──
// Uses PostgreSQL ltree GiST index to find all files and folders
// in the subtree in O(1) queries, replacing the O(N) spawn-per-folder
// approach that could saturate the connection pool.
let (found_files, total_file_count) = self
.file_repository
.search_files_in_subtree(criteria.folder_id.as_deref(), &criteria, user_id)
.await?;
// Get folders (SQL-filtered, user-scoped, recursive when applicable)
let found_folders: Vec<Folder> = self
.folder_repository
.search_folders(
criteria.folder_id.as_deref(),
criteria.name_contains.as_deref(),
user_id,
true,
)
.await?;
// ── Convert to DTOs and enrich with server-computed metadata ──
// Fused single pass: no intermediate DTO Vec materialization.
let mut enriched_files: Vec<SearchFileResultDto> = found_files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.collect();
let mut enriched_folders: Vec<SearchFolderResultDto> = found_folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.collect();
// ── Sort folders (cached_key avoids O(N log N) temporary String allocations) ──
match criteria.sort_by.as_str() {
"name" => {
enriched_folders.sort_by_cached_key(|f| f.name.to_lowercase());
}
"name_desc" => {
enriched_folders.sort_by_cached_key(|f| Reverse(f.name.to_lowercase()));
}
"date" => {
enriched_folders.sort_by_key(|f| f.modified_at);
}
"date_desc" => {
enriched_folders.sort_by_key(|f| Reverse(f.modified_at));
}
_ => {
enriched_folders.sort_by_key(|f| Reverse(f.relevance_score));
}
}
2026-02-14 01:29:34 +01:00
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
// ── Pagination (folders first, then files) ──
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
let start_idx = criteria.offset.min(total_count);
let end_idx = (criteria.offset + criteria.limit).min(total_count);
let folder_start = start_idx.min(folder_count);
let folder_end = end_idx.min(folder_count);
let paginated_folders = enriched_folders[folder_start..folder_end].to_vec();
let file_start = start_idx.saturating_sub(folder_count);
let file_end = end_idx
.saturating_sub(folder_count)
.min(enriched_files.len());
let paginated_files = enriched_files[file_start..file_end].to_vec();
let elapsed_ms = start.elapsed().as_millis() as u64;
let search_results = Arc::new(SearchResultsDto::new(
paginated_files,
paginated_folders,
criteria.limit,
criteria.offset,
Some(total_count),
elapsed_ms,
criteria.sort_by.clone(),
));
Ok(search_results)
})
.await
.map_err(|shared: Arc<crate::common::errors::DomainError>| {
crate::common::errors::DomainError::new(
shared.kind,
shared.entity_type,
shared.message.clone(),
)
})
2025-03-27 01:13:34 +01:00
}
2026-02-14 01:29:34 +01:00
/// Returns quick suggestions for autocomplete. Delegates to the
/// inherent `suggest_with_perms` — the trait method is preserved as
/// the polymorphic entry point (e.g. for `StubSearchUseCase` in
/// tests); production callers can equivalently call the inherent
/// method directly.
async fn suggest(
&self,
query: &str,
folder_id: Option<&str>,
limit: usize,
caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
self.suggest_with_perms(query, folder_id, limit, caller_id)
.await
}
/// Clears the search results cache.
2025-03-27 01:13:34 +01:00
async fn clear_search_cache(&self) -> Result<()> {
self.search_cache.invalidate_all();
self.search_cache.run_pending_tasks().await;
2025-03-27 01:13:34 +01:00
Ok(())
}
}
// ─── Stub for testing ────────────────────────────────────────────────────
2025-03-27 01:13:34 +01:00
impl SearchService {
/// Creates a stub version of the service for testing
2025-03-27 01:13:34 +01:00
pub fn new_stub() -> impl SearchUseCase {
struct SearchServiceStub;
2026-02-14 01:29:34 +01:00
2025-03-27 01:13:34 +01:00
impl SearchUseCase for SearchServiceStub {
async fn search(
&self,
_criteria: SearchCriteriaDto,
_user_id: Uuid,
) -> Result<Arc<SearchResultsDto>> {
Ok(Arc::new(SearchResultsDto::empty()))
2025-03-27 01:13:34 +01:00
}
2026-02-14 01:29:34 +01:00
async fn suggest(
&self,
_query: &str,
_folder_id: Option<&str>,
_limit: usize,
_caller_id: Uuid,
) -> Result<SearchSuggestionsDto> {
Ok(SearchSuggestionsDto {
suggestions: Vec::new(),
query_time_ms: 0,
})
}
2025-03-27 01:13:34 +01:00
async fn clear_search_cache(&self) -> Result<()> {
Ok(())
}
}
2026-02-14 01:29:34 +01:00
2025-03-27 01:13:34 +01:00
SearchServiceStub
}
2026-02-14 01:29:34 +01:00
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn content_relevance_stays_below_name_contains_band() {
// Best hit of the set caps at 45 — always under contains (50).
assert_eq!(content_relevance(8.0, 8.0), 45);
assert_eq!(content_relevance(4.0, 8.0), 28);
// Degenerate inputs fall to the floor instead of panicking.
assert_eq!(content_relevance(1.0, 0.0), 10);
assert_eq!(content_relevance(f32::NAN, 8.0), 10);
assert!(content_relevance(0.0, 8.0) >= 10);
}
fn dto(name: &str, relevance: u32, size: u64, modified_at: u64) -> SearchFileResultDto {
SearchFileResultDto {
id: name.to_string(),
name: name.to_string(),
path: format!("/{name}"),
size,
mime_type: "text/plain".into(),
folder_id: None,
created_at: 0,
modified_at,
relevance_score: relevance,
size_formatted: String::new(),
icon_class: "".into(),
icon_special_class: "".into(),
category: "".into(),
blob_hash: String::new(),
snippet: None,
match_source: None,
}
}
#[test]
fn entry_weight_counts_every_owned_string_plus_overheads() {
// Empty page: entry overhead + sort_by ("relevance" = 9 bytes).
let empty = Arc::new(SearchResultsDto::empty());
let base = search_results_entry_weight(&0, &empty) as usize;
assert_eq!(base, 256 + 9);
// One file row: base + row overhead + its owned string bytes
// (id 7 + name 7 + path 8 + mime 10; the rest are empty/None).
let one_file = Arc::new(SearchResultsDto::new(
vec![dto("abc.txt", 50, 10, 1)],
Vec::new(),
100,
0,
Some(1),
0,
"relevance".to_string(),
));
let w = search_results_entry_weight(&0, &one_file) as usize;
assert_eq!(w, base + 200 + 7 + 7 + 8 + 10);
// Folder rows weigh too (id 2 + name 4 + path 5 + parent 6 = 17).
let one_folder = Arc::new(SearchResultsDto::new(
Vec::new(),
vec![SearchFolderResultDto {
id: "f1".to_string(),
name: "docs".to_string(),
path: "/docs".to_string(),
parent_id: Some("parent".to_string()),
drive_id: Uuid::nil(),
created_at: 0,
modified_at: 0,
is_root: false,
relevance_score: 50,
}],
100,
0,
Some(1),
0,
"relevance".to_string(),
));
let w = search_results_entry_weight(&0, &one_folder) as usize;
assert_eq!(w, base + 200 + 2 + 4 + 5 + 6);
}
#[tokio::test]
async fn cache_evicts_down_to_the_byte_budget() {
// Budget fits ~2 of these entries; inserting 20 must never let the
// weighted size settle above the budget.
let entry = |i: usize| {
Arc::new(SearchResultsDto::new(
(0..50)
.map(|r| dto(&format!("file_{i}_{r}_{}", "x".repeat(100)), 50, 1, 1))
.collect(),
Vec::new(),
50,
0,
Some(50),
0,
"relevance".to_string(),
))
};
let per_entry = search_results_entry_weight(&0, &entry(0)) as u64;
let budget = per_entry * 2 + per_entry / 2;
let cache = build_search_results_cache(300, budget);
for i in 0..20u64 {
cache.insert(i, entry(i as usize)).await;
}
cache.run_pending_tasks().await;
let retained: u64 = cache
.iter()
.map(|(k, v)| search_results_entry_weight(&k, &v) as u64)
.sum();
assert!(
retained <= budget,
"retained {retained} B exceeds budget {budget} B"
);
assert!(cache.entry_count() <= 2);
}
#[test]
fn merged_files_resort_by_relevance_and_by_column() {
let mut files = vec![
dto("b-content.txt", 30, 10, 200),
dto("a-name.txt", 80, 99, 100),
];
sort_enriched_files(&mut files, "relevance");
assert_eq!(
files[0].name, "a-name.txt",
"name match must outrank content match"
);
sort_enriched_files(&mut files, "size_desc");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "date");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "name_desc");
assert_eq!(files[0].name, "b-content.txt");
}
}