Merge origin/main (Tantivy content search) into delta-sync branch
Both sides added a parameter to create_application_services and a setup step before it: this branch's storage-usage/quota service (for the instant-upload path) and main's Tantivy content index (for SearchService). The resolution keeps both — the signature takes both arguments and the build runs storage usage as step 3c and the content index as 3d. https://claude.ai/code/session_01WdNenpnujNR2sc32XVvwfS
This commit is contained in:
@@ -11,6 +11,7 @@ use crate::application::dtos::search_dto::{
|
||||
SearchCriteriaDto, SearchFileResultDto, SearchFolderResultDto, SearchResultsDto,
|
||||
SearchSuggestionItem, SearchSuggestionsDto,
|
||||
};
|
||||
use crate::application::ports::content_index_ports::{ContentHitDto, ContentIndexPort};
|
||||
use crate::application::ports::inbound::SearchUseCase;
|
||||
use crate::application::ports::storage_ports::FileReadPort;
|
||||
use crate::common::errors::Result;
|
||||
@@ -44,6 +45,11 @@ pub struct SearchService {
|
||||
/// Repository for folder operations
|
||||
folder_repository: Arc<FolderDbRepository>,
|
||||
|
||||
/// Optional full-text content index (embedded Tantivy). When present,
|
||||
/// query-bearing searches additionally surface files whose CONTENT
|
||||
/// matches; hits are hydrated and re-filtered through SQL before use.
|
||||
content_index: Option<Arc<dyn ContentIndexPort>>,
|
||||
|
||||
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
|
||||
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
|
||||
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
|
||||
@@ -73,6 +79,35 @@ fn compute_relevance(name: &str, query_lower: &str) -> u32 {
|
||||
}
|
||||
}
|
||||
|
||||
/// Max content-index candidates fetched per search. Hydration re-filters
|
||||
/// them in ONE SQL round-trip, so this bounds both index and DB work.
|
||||
const CONTENT_HITS_LIMIT: usize = 200;
|
||||
|
||||
/// Map a BM25 score into the 10–45 relevance band, normalized against the
|
||||
/// best score of the result set. Deliberately below the weakest name match
|
||||
/// (contains = 50): a filename hit is more specific than a body mention.
|
||||
fn content_relevance(score: f32, max_score: f32) -> u32 {
|
||||
if !score.is_finite() || max_score <= 0.0 {
|
||||
return 10;
|
||||
}
|
||||
let ratio = (score / max_score).clamp(0.0, 1.0);
|
||||
10 + (ratio * 35.0).round() as u32
|
||||
}
|
||||
|
||||
/// Re-sort the merged file list with the same semantics the folder list
|
||||
/// uses. Only invoked when content hits were merged into a SQL-ordered page.
|
||||
fn sort_enriched_files(files: &mut [SearchFileResultDto], sort_by: &str) {
|
||||
match sort_by {
|
||||
"name" => files.sort_by_cached_key(|f| f.name.to_lowercase()),
|
||||
"name_desc" => files.sort_by_cached_key(|f| Reverse(f.name.to_lowercase())),
|
||||
"date" => files.sort_by_key(|f| f.modified_at),
|
||||
"date_desc" => files.sort_by_key(|f| Reverse(f.modified_at)),
|
||||
"size" => files.sort_by_key(|f| f.size),
|
||||
"size_desc" => files.sort_by_key(|f| Reverse(f.size)),
|
||||
_ => files.sort_by_key(|f| Reverse(f.relevance_score)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Format bytes into a human-readable string (e.g. "2.5 MB").
|
||||
fn format_bytes(bytes: u64) -> String {
|
||||
const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
|
||||
@@ -115,6 +150,7 @@ impl SearchService {
|
||||
pub fn new(
|
||||
file_repository: Arc<FileBlobReadRepository>,
|
||||
folder_repository: Arc<FolderDbRepository>,
|
||||
content_index: Option<Arc<dyn ContentIndexPort>>,
|
||||
cache_ttl: u64,
|
||||
max_cache_size: usize,
|
||||
) -> Self {
|
||||
@@ -126,6 +162,7 @@ impl SearchService {
|
||||
Self {
|
||||
file_repository,
|
||||
folder_repository,
|
||||
content_index,
|
||||
search_cache,
|
||||
}
|
||||
}
|
||||
@@ -176,6 +213,8 @@ impl SearchService {
|
||||
// responses on the NC surface can emit the same ETag
|
||||
// (`File::compute_etag`) as PROPFIND/GET would.
|
||||
blob_hash: file.content_hash.clone(),
|
||||
snippet: None,
|
||||
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -201,6 +240,108 @@ impl SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
/// Query the content index for files matching by CONTENT (when the index
|
||||
/// is enabled). First page only — content hits have no stable
|
||||
/// interleaving with SQL pagination beyond it, and page one is where
|
||||
/// search UX lives. Index failures degrade to name-only results, never
|
||||
/// to a failed search.
|
||||
async fn lookup_content_hits(
|
||||
&self,
|
||||
criteria: &SearchCriteriaDto,
|
||||
user_id: Uuid,
|
||||
) -> Vec<ContentHitDto> {
|
||||
let Some(index) = &self.content_index else {
|
||||
return Vec::new();
|
||||
};
|
||||
if criteria.offset != 0 {
|
||||
return Vec::new();
|
||||
}
|
||||
let Some(query) = criteria
|
||||
.name_contains
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|q| q.len() >= 2)
|
||||
else {
|
||||
return Vec::new();
|
||||
};
|
||||
|
||||
match index
|
||||
.search_content(user_id, query, CONTENT_HITS_LIMIT)
|
||||
.await
|
||||
{
|
||||
Ok(hits) => hits,
|
||||
Err(e) => {
|
||||
tracing::warn!("Content-index lookup failed — returning name-only results: {e}");
|
||||
Vec::new()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Merge content-index hits into the name-search result page:
|
||||
/// * files the name search already found just gain their `snippet`;
|
||||
/// * content-only candidates are hydrated through SQL in one round-trip
|
||||
/// (re-applying user scope, trash state and every active filter — a
|
||||
/// stale index id silently drops out), enriched, scored into the
|
||||
/// content relevance band and appended;
|
||||
/// * the merged page is re-sorted with the caller's `sort_by`.
|
||||
///
|
||||
/// Returns how many files were added (callers bump their totals by it).
|
||||
async fn merge_content_hits(
|
||||
&self,
|
||||
hits: Vec<ContentHitDto>,
|
||||
enriched_files: &mut Vec<SearchFileResultDto>,
|
||||
criteria: &SearchCriteriaDto,
|
||||
user_id: Uuid,
|
||||
) -> Result<usize> {
|
||||
if hits.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
let mut by_id: std::collections::HashMap<&str, &ContentHitDto> =
|
||||
hits.iter().map(|h| (h.file_id.as_str(), h)).collect();
|
||||
for file in enriched_files.iter_mut() {
|
||||
if let Some(hit) = by_id.remove(file.id.as_str()) {
|
||||
file.snippet = hit.snippet.clone();
|
||||
}
|
||||
}
|
||||
if by_id.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
// Preserve the index's score order when collecting the leftovers.
|
||||
let candidate_ids: Vec<String> = hits
|
||||
.iter()
|
||||
.filter(|h| by_id.contains_key(h.file_id.as_str()))
|
||||
.map(|h| h.file_id.clone())
|
||||
.collect();
|
||||
let files = self
|
||||
.file_repository
|
||||
.fetch_files_by_ids_filtered(&candidate_ids, criteria, user_id)
|
||||
.await?;
|
||||
if files.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
let max_score = hits.iter().map(|h| h.score).fold(0.0_f32, f32::max);
|
||||
let mut added = 0usize;
|
||||
for file in files {
|
||||
let dto = FileDto::from(file);
|
||||
let Some(hit) = by_id.get(dto.id.as_str()) else {
|
||||
continue;
|
||||
};
|
||||
let mut enriched = Self::enrich_file(&dto, "");
|
||||
enriched.relevance_score = content_relevance(hit.score, max_score);
|
||||
enriched.snippet = hit.snippet.clone();
|
||||
enriched.match_source = Some("content".to_string());
|
||||
enriched_files.push(enriched);
|
||||
added += 1;
|
||||
}
|
||||
if added > 0 {
|
||||
sort_enriched_files(enriched_files, &criteria.sort_by);
|
||||
}
|
||||
Ok(added)
|
||||
}
|
||||
|
||||
/// Quick suggestions search — returns up to `limit` name suggestions
|
||||
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
|
||||
/// so only a handful of rows cross the DB→app boundary.
|
||||
@@ -305,6 +446,10 @@ impl SearchUseCase for SearchService {
|
||||
// Pre-compute once — avoids N heap allocations inside enrich_file/enrich_folder.
|
||||
let query_lower = query.to_lowercase();
|
||||
|
||||
// Content-index candidates (first page only). Feature-off or an
|
||||
// index failure yields an empty set — the search stays name-only.
|
||||
let content_hits = self.lookup_content_hits(&criteria, user_id).await;
|
||||
|
||||
// For non-recursive searches, use efficient database-level pagination
|
||||
// This avoids loading all files into memory
|
||||
if !criteria.recursive {
|
||||
@@ -316,7 +461,7 @@ impl SearchUseCase for SearchService {
|
||||
|
||||
// Convert to DTOs and enrich with metadata
|
||||
let file_dtos: Vec<FileDto> = files.into_iter().map(FileDto::from).collect();
|
||||
let enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
.iter()
|
||||
.map(|f| Self::enrich_file(f, &query_lower))
|
||||
.collect();
|
||||
@@ -360,6 +505,12 @@ impl SearchUseCase for SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
// Blend in content-discovered files before the pagination math.
|
||||
let added = self
|
||||
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
|
||||
.await?;
|
||||
let total_file_count = total_file_count + added;
|
||||
|
||||
let folder_count = enriched_folders.len();
|
||||
let total_count = total_file_count + folder_count;
|
||||
|
||||
@@ -416,7 +567,7 @@ impl SearchUseCase for SearchService {
|
||||
|
||||
// ── Convert to DTOs and enrich with server-computed metadata ──
|
||||
let file_dtos: Vec<FileDto> = found_files.into_iter().map(FileDto::from).collect();
|
||||
let enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
.iter()
|
||||
.map(|f| Self::enrich_file(f, &query_lower))
|
||||
.collect();
|
||||
@@ -446,6 +597,12 @@ impl SearchUseCase for SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
// Blend in content-discovered files before the pagination math.
|
||||
let added = self
|
||||
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
|
||||
.await?;
|
||||
let total_file_count = total_file_count + added;
|
||||
|
||||
// ── Pagination (folders first, then files) ──
|
||||
let folder_count = enriched_folders.len();
|
||||
let total_count = total_file_count + folder_count;
|
||||
@@ -535,3 +692,60 @@ impl SearchService {
|
||||
SearchServiceStub
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn content_relevance_stays_below_name_contains_band() {
|
||||
// Best hit of the set caps at 45 — always under contains (50).
|
||||
assert_eq!(content_relevance(8.0, 8.0), 45);
|
||||
assert_eq!(content_relevance(4.0, 8.0), 28);
|
||||
// Degenerate inputs fall to the floor instead of panicking.
|
||||
assert_eq!(content_relevance(1.0, 0.0), 10);
|
||||
assert_eq!(content_relevance(f32::NAN, 8.0), 10);
|
||||
assert!(content_relevance(0.0, 8.0) >= 10);
|
||||
}
|
||||
|
||||
fn dto(name: &str, relevance: u32, size: u64, modified_at: u64) -> SearchFileResultDto {
|
||||
SearchFileResultDto {
|
||||
id: name.to_string(),
|
||||
name: name.to_string(),
|
||||
path: format!("/{name}"),
|
||||
size,
|
||||
mime_type: "text/plain".to_string(),
|
||||
folder_id: None,
|
||||
created_at: 0,
|
||||
modified_at,
|
||||
relevance_score: relevance,
|
||||
size_formatted: String::new(),
|
||||
icon_class: String::new(),
|
||||
icon_special_class: String::new(),
|
||||
category: String::new(),
|
||||
blob_hash: String::new(),
|
||||
snippet: None,
|
||||
match_source: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merged_files_resort_by_relevance_and_by_column() {
|
||||
let mut files = vec![
|
||||
dto("b-content.txt", 30, 10, 200),
|
||||
dto("a-name.txt", 80, 99, 100),
|
||||
];
|
||||
sort_enriched_files(&mut files, "relevance");
|
||||
assert_eq!(
|
||||
files[0].name, "a-name.txt",
|
||||
"name match must outrank content match"
|
||||
);
|
||||
|
||||
sort_enriched_files(&mut files, "size_desc");
|
||||
assert_eq!(files[0].name, "a-name.txt");
|
||||
sort_enriched_files(&mut files, "date");
|
||||
assert_eq!(files[0].name, "a-name.txt");
|
||||
sort_enriched_files(&mut files, "name_desc");
|
||||
assert_eq!(files[0].name, "b-content.txt");
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user