Add embedded Tantivy full-text content search

/api/search now finds files by CONTENT as well as by name: BM25-ranked
matches over extracted text (PDF, Office OOXML/ODF, plain text/code)
with typo-tolerant fuzzy terms and search-as-you-type prefix matching,
served from an embedded Tantivy index at {storage}/.search-index.

Pipeline (all off the request path, mirroring tree-etag + thumbnails):
- statement triggers on storage.files append to a durable dirty queue
  (storage.search_index_dirty) - every write surface (REST, WebDAV,
  NextCloud, WOPI, trash) is covered, crash-safe by construction
- ContentIndexWorker drains the queue on the maintenance pool, extracts
  text once per unique BLAKE3 blob (storage.blob_extracted_text cache:
  N copies = 1 extraction, renames/moves = 0 re-extraction) and applies
  batched single-writer Tantivy commits; queue rows are deleted only
  after the commit succeeds (at-least-once, idempotent upserts)
- the index is a derived artifact: a version-marker mismatch wipes and
  reseeds it from Postgres, which remains the single source of truth

SearchService merges content hits into the existing name search: hits
are hydrated through ONE SQL round-trip that re-applies user scope,
trash state and every active filter (a stale index id can never leak),
scored below name matches, and returned with a plain-text snippet and
a match_source field. Index failure or
OXICLOUD_ENABLE_CONTENT_SEARCH=false degrades to name-only search; a
discard-only janitor keeps the trigger-fed queue bounded while disabled.

The frontend renders the snippet under the file name in list view.

New dependencies: tantivy 0.26, zip 8.6 (deflate only), pdf-extract 0.10.

https://claude.ai/code/session_01Sc7F4xbo83YbFAQ4xEeDrX
This commit is contained in:
Claude
2026-06-11 15:16:03 +00:00
parent 7157454afd
commit 8dab135090
19 changed files with 2813 additions and 70 deletions
+216 -2
View File
@@ -11,6 +11,7 @@ use crate::application::dtos::search_dto::{
SearchCriteriaDto, SearchFileResultDto, SearchFolderResultDto, SearchResultsDto,
SearchSuggestionItem, SearchSuggestionsDto,
};
use crate::application::ports::content_index_ports::{ContentHitDto, ContentIndexPort};
use crate::application::ports::inbound::SearchUseCase;
use crate::application::ports::storage_ports::FileReadPort;
use crate::common::errors::Result;
@@ -44,6 +45,11 @@ pub struct SearchService {
/// Repository for folder operations
folder_repository: Arc<FolderDbRepository>,
/// Optional full-text content index (embedded Tantivy). When present,
/// query-bearing searches additionally surface files whose CONTENT
/// matches; hits are hydrated and re-filtered through SQL before use.
content_index: Option<Arc<dyn ContentIndexPort>>,
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
@@ -73,6 +79,35 @@ fn compute_relevance(name: &str, query_lower: &str) -> u32 {
}
}
/// Max content-index candidates fetched per search. Hydration re-filters
/// them in ONE SQL round-trip, so this bounds both index and DB work.
const CONTENT_HITS_LIMIT: usize = 200;
/// Map a BM25 score into the 10–45 relevance band, normalized against the
/// best score of the result set. Deliberately below the weakest name match
/// (contains = 50): a filename hit is more specific than a body mention.
fn content_relevance(score: f32, max_score: f32) -> u32 {
if !score.is_finite() || max_score <= 0.0 {
return 10;
}
let ratio = (score / max_score).clamp(0.0, 1.0);
10 + (ratio * 35.0).round() as u32
}
/// Re-sort the merged file list with the same semantics the folder list
/// uses. Only invoked when content hits were merged into a SQL-ordered page.
fn sort_enriched_files(files: &mut [SearchFileResultDto], sort_by: &str) {
match sort_by {
"name" => files.sort_by_cached_key(|f| f.name.to_lowercase()),
"name_desc" => files.sort_by_cached_key(|f| Reverse(f.name.to_lowercase())),
"date" => files.sort_by_key(|f| f.modified_at),
"date_desc" => files.sort_by_key(|f| Reverse(f.modified_at)),
"size" => files.sort_by_key(|f| f.size),
"size_desc" => files.sort_by_key(|f| Reverse(f.size)),
_ => files.sort_by_key(|f| Reverse(f.relevance_score)),
}
}
/// Format bytes into a human-readable string (e.g. "2.5 MB").
fn format_bytes(bytes: u64) -> String {
const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
@@ -115,6 +150,7 @@ impl SearchService {
pub fn new(
file_repository: Arc<FileBlobReadRepository>,
folder_repository: Arc<FolderDbRepository>,
content_index: Option<Arc<dyn ContentIndexPort>>,
cache_ttl: u64,
max_cache_size: usize,
) -> Self {
@@ -126,6 +162,7 @@ impl SearchService {
Self {
file_repository,
folder_repository,
content_index,
search_cache,
}
}
@@ -176,6 +213,8 @@ impl SearchService {
// responses on the NC surface can emit the same ETag
// (`File::compute_etag`) as PROPFIND/GET would.
blob_hash: file.content_hash.clone(),
snippet: None,
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
}
}
@@ -201,6 +240,108 @@ impl SearchService {
}
}
/// Query the content index for files matching by CONTENT (when the index
/// is enabled). First page only — content hits have no stable
/// interleaving with SQL pagination beyond it, and page one is where
/// search UX lives. Index failures degrade to name-only results, never
/// to a failed search.
async fn lookup_content_hits(
&self,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Vec<ContentHitDto> {
let Some(index) = &self.content_index else {
return Vec::new();
};
if criteria.offset != 0 {
return Vec::new();
}
let Some(query) = criteria
.name_contains
.as_deref()
.map(str::trim)
.filter(|q| q.len() >= 2)
else {
return Vec::new();
};
match index
.search_content(user_id, query, CONTENT_HITS_LIMIT)
.await
{
Ok(hits) => hits,
Err(e) => {
tracing::warn!("Content-index lookup failed — returning name-only results: {e}");
Vec::new()
}
}
}
/// Merge content-index hits into the name-search result page:
/// * files the name search already found just gain their `snippet`;
/// * content-only candidates are hydrated through SQL in one round-trip
/// (re-applying user scope, trash state and every active filter — a
/// stale index id silently drops out), enriched, scored into the
/// content relevance band and appended;
/// * the merged page is re-sorted with the caller's `sort_by`.
///
/// Returns how many files were added (callers bump their totals by it).
async fn merge_content_hits(
&self,
hits: Vec<ContentHitDto>,
enriched_files: &mut Vec<SearchFileResultDto>,
criteria: &SearchCriteriaDto,
user_id: Uuid,
) -> Result<usize> {
if hits.is_empty() {
return Ok(0);
}
let mut by_id: std::collections::HashMap<&str, &ContentHitDto> =
hits.iter().map(|h| (h.file_id.as_str(), h)).collect();
for file in enriched_files.iter_mut() {
if let Some(hit) = by_id.remove(file.id.as_str()) {
file.snippet = hit.snippet.clone();
}
}
if by_id.is_empty() {
return Ok(0);
}
// Preserve the index's score order when collecting the leftovers.
let candidate_ids: Vec<String> = hits
.iter()
.filter(|h| by_id.contains_key(h.file_id.as_str()))
.map(|h| h.file_id.clone())
.collect();
let files = self
.file_repository
.fetch_files_by_ids_filtered(&candidate_ids, criteria, user_id)
.await?;
if files.is_empty() {
return Ok(0);
}
let max_score = hits.iter().map(|h| h.score).fold(0.0_f32, f32::max);
let mut added = 0usize;
for file in files {
let dto = FileDto::from(file);
let Some(hit) = by_id.get(dto.id.as_str()) else {
continue;
};
let mut enriched = Self::enrich_file(&dto, "");
enriched.relevance_score = content_relevance(hit.score, max_score);
enriched.snippet = hit.snippet.clone();
enriched.match_source = Some("content".to_string());
enriched_files.push(enriched);
added += 1;
}
if added > 0 {
sort_enriched_files(enriched_files, &criteria.sort_by);
}
Ok(added)
}
/// Quick suggestions search — returns up to `limit` name suggestions
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
/// so only a handful of rows cross the DB→app boundary.
@@ -305,6 +446,10 @@ impl SearchUseCase for SearchService {
// Pre-compute once — avoids N heap allocations inside enrich_file/enrich_folder.
let query_lower = query.to_lowercase();
// Content-index candidates (first page only). Feature-off or an
// index failure yields an empty set — the search stays name-only.
let content_hits = self.lookup_content_hits(&criteria, user_id).await;
// For non-recursive searches, use efficient database-level pagination
// This avoids loading all files into memory
if !criteria.recursive {
@@ -316,7 +461,7 @@ impl SearchUseCase for SearchService {
// Convert to DTOs and enrich with metadata
let file_dtos: Vec<FileDto> = files.into_iter().map(FileDto::from).collect();
let enriched_files: Vec<SearchFileResultDto> = file_dtos
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
.collect();
@@ -360,6 +505,12 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
@@ -416,7 +567,7 @@ impl SearchUseCase for SearchService {
// ── Convert to DTOs and enrich with server-computed metadata ──
let file_dtos: Vec<FileDto> = found_files.into_iter().map(FileDto::from).collect();
let enriched_files: Vec<SearchFileResultDto> = file_dtos
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
.iter()
.map(|f| Self::enrich_file(f, &query_lower))
.collect();
@@ -446,6 +597,12 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
// ── Pagination (folders first, then files) ──
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
@@ -535,3 +692,60 @@ impl SearchService {
SearchServiceStub
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn content_relevance_stays_below_name_contains_band() {
// Best hit of the set caps at 45 — always under contains (50).
assert_eq!(content_relevance(8.0, 8.0), 45);
assert_eq!(content_relevance(4.0, 8.0), 28);
// Degenerate inputs fall to the floor instead of panicking.
assert_eq!(content_relevance(1.0, 0.0), 10);
assert_eq!(content_relevance(f32::NAN, 8.0), 10);
assert!(content_relevance(0.0, 8.0) >= 10);
}
fn dto(name: &str, relevance: u32, size: u64, modified_at: u64) -> SearchFileResultDto {
SearchFileResultDto {
id: name.to_string(),
name: name.to_string(),
path: format!("/{name}"),
size,
mime_type: "text/plain".to_string(),
folder_id: None,
created_at: 0,
modified_at,
relevance_score: relevance,
size_formatted: String::new(),
icon_class: String::new(),
icon_special_class: String::new(),
category: String::new(),
blob_hash: String::new(),
snippet: None,
match_source: None,
}
}
#[test]
fn merged_files_resort_by_relevance_and_by_column() {
let mut files = vec![
dto("b-content.txt", 30, 10, 200),
dto("a-name.txt", 80, 99, 100),
];
sort_enriched_files(&mut files, "relevance");
assert_eq!(
files[0].name, "a-name.txt",
"name match must outrank content match"
);
sort_enriched_files(&mut files, "size_desc");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "date");
assert_eq!(files[0].name, "a-name.txt");
sort_enriched_files(&mut files, "name_desc");
assert_eq!(files[0].name, "b-content.txt");
}
}