Merge origin/main (Tantivy content search) into delta-sync branch

Both sides added a parameter to create_application_services and a
setup step before it: this branch's storage-usage/quota service (for
the instant-upload path) and main's Tantivy content index (for
SearchService). The resolution keeps both — the signature takes both
arguments and the build runs storage usage as step 3c and the content
index as 3d.

https://claude.ai/code/session_01WdNenpnujNR2sc32XVvwfS
This commit is contained in:
Claude
2026-06-11 18:32:27 +00:00
19 changed files with 2807 additions and 68 deletions
+8
View File
@@ -134,6 +134,14 @@ pub struct SearchFileResultDto {
/// that pre-date the column.
#[serde(default)]
pub blob_hash: String,
/// Plain-text fragment around the first content match, present only for
/// hits discovered through the full-text content index.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub snippet: Option<String>,
/// Where the match came from: "name" (filename matched the query) or
/// "content" (discovered via the full-text content index).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub match_source: Option<String>,
}
/// A folder search result enriched with server-computed metadata
@@ -0,0 +1,49 @@
//! Content Index Port - Application layer abstraction for full-text search
//! over file names and extracted file content.
//!
//! The implementation (an embedded Tantivy BM25 index) lives in the
//! infrastructure layer; `SearchService` only sees this port. The index is a
//! DERIVED artifact fed asynchronously by a background worker — it never sits
//! on a request path, and PostgreSQL remains the source of truth (hits are
//! re-validated and hydrated through SQL before they reach the caller, so a
//! stale index can only ever produce a dropped candidate, never a leak).
use async_trait::async_trait;
use uuid::Uuid;
use crate::common::errors::DomainError;
/// One content-index hit: a candidate file id with its BM25 score and an
/// optional plain-text snippet around the first match.
///
/// `file_id` is a CANDIDATE — callers must hydrate it through the metadata
/// repository (which re-applies user scoping, trash state and the active
/// search filters) before exposing it.
#[derive(Debug, Clone)]
pub struct ContentHitDto {
/// File UUID as string (matches `storage.files.id`).
pub file_id: String,
/// BM25 relevance score (positive, unbounded — normalize per result set).
pub score: f32,
/// Plain-text fragment around the first matched term, when available.
pub snippet: Option<String>,
}
/// Port for querying the full-text content index.
///
/// `#[async_trait]` is used so the trait is dyn-compatible — `SearchService`
/// holds an `Option<Arc<dyn ContentIndexPort>>` (the feature is toggleable).
#[async_trait]
pub trait ContentIndexPort: Send + Sync + 'static {
/// Search indexed file names + content for `query`, scoped to `user_id`.
///
/// Returns up to `limit` hits sorted by BM25 score descending. Matching is
/// tokenized (not substring): exact terms, typo-tolerant fuzzy terms
/// (edit distance 1) and prefix expansion on the last query token.
async fn search_content(
&self,
user_id: Uuid,
query: &str,
limit: usize,
) -> Result<Vec<ContentHitDto>, DomainError>;
}
+1
View File
@@ -7,6 +7,7 @@ pub mod calendar_ports;
pub mod carddav_ports;
pub mod chunked_upload_ports;
pub mod compression_ports;
pub mod content_index_ports;
pub mod dedup_ports;
pub mod email_sender;
pub mod favorites_ports;
+216 -2
View File
@@ -11,6 +11,7 @@ use crate::application::dtos::search_dto::{
SearchCriteriaDto, SearchFileResultDto, SearchFolderResultDto, SearchResultsDto,
SearchSuggestionItem, SearchSuggestionsDto,
};
use crate::application::ports::content_index_ports::{ContentHitDto, ContentIndexPort};
use crate::application::ports::inbound::SearchUseCase;
use crate::application::ports::storage_ports::FileReadPort;
use crate::common::errors::Result;
@@ -44,6 +45,11 @@ pub struct SearchService {
/// Repository for folder operations
folder_repository: Arc<FolderDbRepository>,
/// Optional full-text content index (embedded Tantivy). When present,
/// query-bearing searches additionally surface files whose CONTENT
/// matches; hits are hydrated and re-filtered through SQL before use.
content_index: Option<Arc<dyn ContentIndexPort>>,
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
@@ -73,6 +79,35 @@ fn compute_relevance(name: &str, query_lower: &str) -> u32 {
}
}
/// Max content-index candidates fetched per search. Hydration re-filters
/// them in ONE SQL round-trip, so this bounds both index and DB work.
const CONTENT_HITS_LIMIT: usize = 200;
/// Map a BM25 score into the 10–45 relevance band, normalized against the
/// best score of the result set. Deliberately below the weakest name match
/// (contains = 50): a filename hit is more specific than a body mention.
fn content_relevance(score: f32, max_score: f32) -> u32 {
if !score.is_finite() || max_score <= 0.0 {
return 10;
}
let ratio = (score / max_score).clamp(0.0, 1.0);
10 + (ratio * 35.0).round() as u32
}
/// Re-sort the merged file list with the same semantics the folder list
/// uses. Only invoked when content hits were merged into a SQL-ordered page.
fn sort_enriched_files(files: &mut [SearchFileResultDto], sort_by: &str) {
match sort_by {
"name" => files.sort_by_cached_key(|f| f.name.to_lowercase()),
"name_desc" => files.sort_by_cached_key(|f| Reverse(f.name.to_lowercase())),
"date" => files.sort_by_key(|f| f.modified_at),
"date_desc" => files.sort_by_key(|f| Reverse(f.modified_at)),
"size" => files.sort_by_key(|f| f.size),
"size_desc" => files.sort_by_key(|f| Reverse(f.size)),
_ => files.sort_by_key(|f| Reverse(f.relevance_score)),
}
}
/// Format bytes into a human-readable string (e.g. "2.5 MB").
fn format_bytes(bytes: u64) -> String {
const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
@@ -115,6 +150,7 @@ impl SearchService {
pub fn new(
file_repository: Arc<FileBlobReadRepository>,
folder_repository: Arc<FolderDbRepository>,
content_index: Option<Arc<dyn ContentIndexPort>>,
cache_ttl: u64,
max_cache_size: usize,
) -> Self {
@@ -126,6 +162,7 @@ impl SearchService {
Self {
file_repository,
folder_repository,
content_index,
search_cache,
}
}
@@ -176,6 +213,8 @@ impl SearchService {
// responses on the NC surface can emit the same ETag
// (`File::compute_etag`) as PROPFIND/GET would.
blob_hash: file.content_hash.clone(),
snippet: None,
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
}
}
@@ -201,6 +240,108 @@ impl SearchService {
}
}
/// Query the content index for files matching by CONTENT (when the index
/// is enabled). First page only — content hits have no stable
/// interleaving with SQL pagination beyond it, and page one is where
/// search UX lives. Index failures degrade to name-only results, never
/// to a failed search.
async fn lookup_content_hits(
&self,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Vec<ContentHitDto> {
let Some(index) = &self.content_index else {
return Vec::new();
};
if criteria.offset != 0 {
return Vec::new();
}
let Some(query) = criteria
.name_contains
.as_deref()
.map(str::trim)
.filter(|q| q.len() >= 2)
else {
return Vec::new();
};
match index
.search_content(user_id, query, CONTENT_HITS_LIMIT)
.await
{
Ok(hits) => hits,
Err(e) => {
tracing::warn!("Content-index lookup failed — returning name-only results: {e}");
Vec::new()
}
}
}
/// Merge content-index hits into the name-search result page:
/// * files the name search already found just gain their `snippet`;
/// * content-only candidates are hydrated through SQL in one round-trip
/// (re-applying user scope, trash state and every active filter — a
/// stale index id silently drops out), enriched, scored into the
/// content relevance band and appended;
/// * the merged page is re-sorted with the caller's `sort_by`.
///
/// Returns how many files were added (callers bump their totals by it).
async fn merge_content_hits(
&self,
hits: Vec<ContentHitDto>,
enriched_files: &mut Vec<SearchFileResultDto>,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Result<usize> {
if hits.is_empty() {
return Ok(0);
}
let mut by_id: std::collections::HashMap<&str, &ContentHitDto> =
hits.iter().map(|h| (h.file_id.as_str(), h)).collect();
for file in enriched_files.iter_mut() {
if let Some(hit) = by_id.remove(file.id.as_str()) {
file.snippet = hit.snippet.clone();
}
}
if by_id.is_empty() {
return Ok(0);
}
// Preserve the index's score order when collecting the leftovers.
let candidate_ids: Vec<String> = hits
.iter()
.filter(|h| by_id.contains_key(h.file_id.as_str()))
.map(|h| h.file_id.clone())
.collect();
let files = self
.file_repository
.fetch_files_by_ids_filtered(&candidate_ids, criteria, user_id)
.await?;
if files.is_empty() {
return Ok(0);
}
let max_score = hits.iter().map(|h| h.score).fold(0.0_f32, f32::max);
let mut added = 0usize;
for file in files {
let dto = FileDto::from(file);
let Some(hit) = by_id.get(dto.id.as_str()) else {
continue;
};
let mut enriched = Self::enrich_file(&dto, "");
enriched.relevance_score = content_relevance(hit.score, max_score);
enriched.snippet = hit.snippet.clone();
enriched.match_source = Some("content".to_string());
enriched_files.push(enriched);
added += 1;
}
if added > 0 {
sort_enriched_files(enriched_files, &criteria.sort_by);
}
Ok(added)
}
/// Quick suggestions search — returns up to `limit` name suggestions
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
/// so only a handful of rows cross the DB→app boundary.
@@ -305,6 +446,10 @@ impl SearchUseCase for SearchService {
// Pre-compute once — avoids N heap allocations inside enrich_file/enrich_folder.
let query_lower = query.to_lowercase();
// Content-index candidates (first page only). Feature-off or an
// index failure yields an empty set — the search stays name-only.
let content_hits = self.lookup_content_hits(&criteria, user_id).await;
// For non-recursive searches, use efficient database-level pagination
// This avoids loading all files into memory
if !criteria.recursive {
@@ -316,7 +461,7 @@ impl SearchUseCase for SearchService {
// Convert to DTOs and enrich with metadata
let file_dtos: Vec<FileDto> = files.into_iter().map(FileDto::from).collect();
let enriched_files: Vec<SearchFileResultDto> = file_dtos
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
.collect();
@@ -360,6 +505,12 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
@@ -416,7 +567,7 @@ impl SearchUseCase for SearchService {
// ── Convert to DTOs and enrich with server-computed metadata ──
let file_dtos: Vec<FileDto> = found_files.into_iter().map(FileDto::from).collect();
let enriched_files: Vec<SearchFileResultDto> = file_dtos
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
.collect();
@@ -446,6 +597,12 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
// ── Pagination (folders first, then files) ──
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
@@ -535,3 +692,60 @@ impl SearchService {
SearchServiceStub
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn content_relevance_stays_below_name_contains_band() {
// Best hit of the set caps at 45 — always under contains (50).
assert_eq!(content_relevance(8.0, 8.0), 45);
assert_eq!(content_relevance(4.0, 8.0), 28);
// Degenerate inputs fall to the floor instead of panicking.
assert_eq!(content_relevance(1.0, 0.0), 10);
assert_eq!(content_relevance(f32::NAN, 8.0), 10);
assert!(content_relevance(0.0, 8.0) >= 10);
}
fn dto(name: &str, relevance: u32, size: u64, modified_at: u64) -> SearchFileResultDto {
SearchFileResultDto {
id: name.to_string(),
name: name.to_string(),
path: format!("/{name}"),
size,
mime_type: "text/plain".to_string(),
folder_id: None,
created_at: 0,
modified_at,
relevance_score: relevance,
size_formatted: String::new(),
icon_class: String::new(),
icon_special_class: String::new(),
category: String::new(),
blob_hash: String::new(),
snippet: None,
match_source: None,
}
}
#[test]
fn merged_files_resort_by_relevance_and_by_column() {
let mut files = vec![
dto("b-content.txt", 30, 10, 200),
dto("a-name.txt", 80, 99, 100),
];
sort_enriched_files(&mut files, "relevance");
assert_eq!(
files[0].name, "a-name.txt",
"name match must outrank content match"
);
sort_enriched_files(&mut files, "size_desc");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "date");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "name_desc");
assert_eq!(files[0].name, "b-content.txt");
}
}