Add embedded Tantivy full-text content search
/api/search now finds files by CONTENT as well as by name: BM25-ranked
matches over extracted text (PDF, Office OOXML/ODF, plain text/code)
with typo-tolerant fuzzy terms and search-as-you-type prefix matching,
served from an embedded Tantivy index at {storage}/.search-index.
Pipeline (all off the request path, mirroring tree-etag + thumbnails):
- statement triggers on storage.files append to a durable dirty queue
(storage.search_index_dirty) - every write surface (REST, WebDAV,
NextCloud, WOPI, trash) is covered, crash-safe by construction
- ContentIndexWorker drains the queue on the maintenance pool, extracts
text once per unique BLAKE3 blob (storage.blob_extracted_text cache:
N copies = 1 extraction, renames/moves = 0 re-extraction) and applies
batched single-writer Tantivy commits; queue rows are deleted only
after the commit succeeds (at-least-once, idempotent upserts)
- the index is a derived artifact: a version-marker mismatch wipes and
reseeds it from Postgres, which remains the single source of truth
SearchService merges content hits into the existing name search: hits
are hydrated through ONE SQL round-trip that re-applies user scope,
trash state and every active filter (a stale index id can never leak),
scored below name matches, and returned with a plain-text snippet and
a match_source field. Index failure or
OXICLOUD_ENABLE_CONTENT_SEARCH=false degrades to name-only search; a
discard-only janitor keeps the trigger-fed queue bounded while disabled.
The frontend renders the snippet under the file name in list view.
New dependencies: tantivy 0.26, zip 8.6 (deflate only), pdf-extract 0.10.
https://claude.ai/code/session_01Sc7F4xbo83YbFAQ4xEeDrX
This commit is contained in:
@@ -134,6 +134,14 @@ pub struct SearchFileResultDto {
|
||||
/// that pre-date the column.
|
||||
#[serde(default)]
|
||||
pub blob_hash: String,
|
||||
/// Plain-text fragment around the first content match, present only for
|
||||
/// hits discovered through the full-text content index.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub snippet: Option<String>,
|
||||
/// Where the match came from: "name" (filename matched the query) or
|
||||
/// "content" (discovered via the full-text content index).
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub match_source: Option<String>,
|
||||
}
|
||||
|
||||
/// A folder search result enriched with server-computed metadata
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
//! Content Index Port - Application layer abstraction for full-text search
|
||||
//! over file names and extracted file content.
|
||||
//!
|
||||
//! The implementation (an embedded Tantivy BM25 index) lives in the
|
||||
//! infrastructure layer; `SearchService` only sees this port. The index is a
|
||||
//! DERIVED artifact fed asynchronously by a background worker — it never sits
|
||||
//! on a request path, and PostgreSQL remains the source of truth (hits are
|
||||
//! re-validated and hydrated through SQL before they reach the caller, so a
|
||||
//! stale index can only ever produce a dropped candidate, never a leak).
|
||||
|
||||
use async_trait::async_trait;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::common::errors::DomainError;
|
||||
|
||||
/// One content-index hit: a candidate file id with its BM25 score and an
|
||||
/// optional plain-text snippet around the first match.
|
||||
///
|
||||
/// `file_id` is a CANDIDATE — callers must hydrate it through the metadata
|
||||
/// repository (which re-applies user scoping, trash state and the active
|
||||
/// search filters) before exposing it.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ContentHitDto {
|
||||
/// File UUID as string (matches `storage.files.id`).
|
||||
pub file_id: String,
|
||||
/// BM25 relevance score (positive, unbounded — normalize per result set).
|
||||
pub score: f32,
|
||||
/// Plain-text fragment around the first matched term, when available.
|
||||
pub snippet: Option<String>,
|
||||
}
|
||||
|
||||
/// Port for querying the full-text content index.
|
||||
///
|
||||
/// `#[async_trait]` is used so the trait is dyn-compatible — `SearchService`
|
||||
/// holds an `Option<Arc<dyn ContentIndexPort>>` (the feature is toggleable).
|
||||
#[async_trait]
|
||||
pub trait ContentIndexPort: Send + Sync + 'static {
|
||||
/// Search indexed file names + content for `query`, scoped to `user_id`.
|
||||
///
|
||||
/// Returns up to `limit` hits sorted by BM25 score descending. Matching is
|
||||
/// tokenized (not substring): exact terms, typo-tolerant fuzzy terms
|
||||
/// (edit distance 1) and prefix expansion on the last query token.
|
||||
async fn search_content(
|
||||
&self,
|
||||
user_id: Uuid,
|
||||
query: &str,
|
||||
limit: usize,
|
||||
) -> Result<Vec<ContentHitDto>, DomainError>;
|
||||
}
|
||||
@@ -7,6 +7,7 @@ pub mod calendar_ports;
|
||||
pub mod carddav_ports;
|
||||
pub mod chunked_upload_ports;
|
||||
pub mod compression_ports;
|
||||
pub mod content_index_ports;
|
||||
pub mod dedup_ports;
|
||||
pub mod email_sender;
|
||||
pub mod favorites_ports;
|
||||
|
||||
@@ -11,6 +11,7 @@ use crate::application::dtos::search_dto::{
|
||||
SearchCriteriaDto, SearchFileResultDto, SearchFolderResultDto, SearchResultsDto,
|
||||
SearchSuggestionItem, SearchSuggestionsDto,
|
||||
};
|
||||
use crate::application::ports::content_index_ports::{ContentHitDto, ContentIndexPort};
|
||||
use crate::application::ports::inbound::SearchUseCase;
|
||||
use crate::application::ports::storage_ports::FileReadPort;
|
||||
use crate::common::errors::Result;
|
||||
@@ -44,6 +45,11 @@ pub struct SearchService {
|
||||
/// Repository for folder operations
|
||||
folder_repository: Arc<FolderDbRepository>,
|
||||
|
||||
/// Optional full-text content index (embedded Tantivy). When present,
|
||||
/// query-bearing searches additionally surface files whose CONTENT
|
||||
/// matches; hits are hydrated and re-filtered through SQL before use.
|
||||
content_index: Option<Arc<dyn ContentIndexPort>>,
|
||||
|
||||
/// Lock-free concurrent cache with automatic TTL and LRU eviction (moka).
|
||||
/// Values are `Arc<SearchResultsDto>` so cache insert/hit is a single
|
||||
/// atomic ref-count increment (~1 ns) instead of cloning thousands of Strings.
|
||||
@@ -73,6 +79,35 @@ fn compute_relevance(name: &str, query_lower: &str) -> u32 {
|
||||
}
|
||||
}
|
||||
|
||||
/// Max content-index candidates fetched per search. Hydration re-filters
|
||||
/// them in ONE SQL round-trip, so this bounds both index and DB work.
|
||||
const CONTENT_HITS_LIMIT: usize = 200;
|
||||
|
||||
/// Map a BM25 score into the 10–45 relevance band, normalized against the
|
||||
/// best score of the result set. Deliberately below the weakest name match
|
||||
/// (contains = 50): a filename hit is more specific than a body mention.
|
||||
fn content_relevance(score: f32, max_score: f32) -> u32 {
|
||||
if !score.is_finite() || max_score <= 0.0 {
|
||||
return 10;
|
||||
}
|
||||
let ratio = (score / max_score).clamp(0.0, 1.0);
|
||||
10 + (ratio * 35.0).round() as u32
|
||||
}
|
||||
|
||||
/// Re-sort the merged file list with the same semantics the folder list
|
||||
/// uses. Only invoked when content hits were merged into a SQL-ordered page.
|
||||
fn sort_enriched_files(files: &mut [SearchFileResultDto], sort_by: &str) {
|
||||
match sort_by {
|
||||
"name" => files.sort_by_cached_key(|f| f.name.to_lowercase()),
|
||||
"name_desc" => files.sort_by_cached_key(|f| Reverse(f.name.to_lowercase())),
|
||||
"date" => files.sort_by_key(|f| f.modified_at),
|
||||
"date_desc" => files.sort_by_key(|f| Reverse(f.modified_at)),
|
||||
"size" => files.sort_by_key(|f| f.size),
|
||||
"size_desc" => files.sort_by_key(|f| Reverse(f.size)),
|
||||
_ => files.sort_by_key(|f| Reverse(f.relevance_score)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Format bytes into a human-readable string (e.g. "2.5 MB").
|
||||
fn format_bytes(bytes: u64) -> String {
|
||||
const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
|
||||
@@ -115,6 +150,7 @@ impl SearchService {
|
||||
pub fn new(
|
||||
file_repository: Arc<FileBlobReadRepository>,
|
||||
folder_repository: Arc<FolderDbRepository>,
|
||||
content_index: Option<Arc<dyn ContentIndexPort>>,
|
||||
cache_ttl: u64,
|
||||
max_cache_size: usize,
|
||||
) -> Self {
|
||||
@@ -126,6 +162,7 @@ impl SearchService {
|
||||
Self {
|
||||
file_repository,
|
||||
folder_repository,
|
||||
content_index,
|
||||
search_cache,
|
||||
}
|
||||
}
|
||||
@@ -176,6 +213,8 @@ impl SearchService {
|
||||
// responses on the NC surface can emit the same ETag
|
||||
// (`File::compute_etag`) as PROPFIND/GET would.
|
||||
blob_hash: file.content_hash.clone(),
|
||||
snippet: None,
|
||||
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -201,6 +240,108 @@ impl SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
/// Query the content index for files matching by CONTENT (when the index
|
||||
/// is enabled). First page only — content hits have no stable
|
||||
/// interleaving with SQL pagination beyond it, and page one is where
|
||||
/// search UX lives. Index failures degrade to name-only results, never
|
||||
/// to a failed search.
|
||||
async fn lookup_content_hits(
|
||||
&self,
|
||||
criteria: &SearchCriteriaDto,
|
||||
user_id: Uuid,
|
||||
) -> Vec<ContentHitDto> {
|
||||
let Some(index) = &self.content_index else {
|
||||
return Vec::new();
|
||||
};
|
||||
if criteria.offset != 0 {
|
||||
return Vec::new();
|
||||
}
|
||||
let Some(query) = criteria
|
||||
.name_contains
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|q| q.len() >= 2)
|
||||
else {
|
||||
return Vec::new();
|
||||
};
|
||||
|
||||
match index
|
||||
.search_content(user_id, query, CONTENT_HITS_LIMIT)
|
||||
.await
|
||||
{
|
||||
Ok(hits) => hits,
|
||||
Err(e) => {
|
||||
tracing::warn!("Content-index lookup failed — returning name-only results: {e}");
|
||||
Vec::new()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Merge content-index hits into the name-search result page:
|
||||
/// * files the name search already found just gain their `snippet`;
|
||||
/// * content-only candidates are hydrated through SQL in one round-trip
|
||||
/// (re-applying user scope, trash state and every active filter — a
|
||||
/// stale index id silently drops out), enriched, scored into the
|
||||
/// content relevance band and appended;
|
||||
/// * the merged page is re-sorted with the caller's `sort_by`.
|
||||
///
|
||||
/// Returns how many files were added (callers bump their totals by it).
|
||||
async fn merge_content_hits(
|
||||
&self,
|
||||
hits: Vec<ContentHitDto>,
|
||||
enriched_files: &mut Vec<SearchFileResultDto>,
|
||||
criteria: &SearchCriteriaDto,
|
||||
user_id: Uuid,
|
||||
) -> Result<usize> {
|
||||
if hits.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
let mut by_id: std::collections::HashMap<&str, &ContentHitDto> =
|
||||
hits.iter().map(|h| (h.file_id.as_str(), h)).collect();
|
||||
for file in enriched_files.iter_mut() {
|
||||
if let Some(hit) = by_id.remove(file.id.as_str()) {
|
||||
file.snippet = hit.snippet.clone();
|
||||
}
|
||||
}
|
||||
if by_id.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
// Preserve the index's score order when collecting the leftovers.
|
||||
let candidate_ids: Vec<String> = hits
|
||||
.iter()
|
||||
.filter(|h| by_id.contains_key(h.file_id.as_str()))
|
||||
.map(|h| h.file_id.clone())
|
||||
.collect();
|
||||
let files = self
|
||||
.file_repository
|
||||
.fetch_files_by_ids_filtered(&candidate_ids, criteria, user_id)
|
||||
.await?;
|
||||
if files.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
let max_score = hits.iter().map(|h| h.score).fold(0.0_f32, f32::max);
|
||||
let mut added = 0usize;
|
||||
for file in files {
|
||||
let dto = FileDto::from(file);
|
||||
let Some(hit) = by_id.get(dto.id.as_str()) else {
|
||||
continue;
|
||||
};
|
||||
let mut enriched = Self::enrich_file(&dto, "");
|
||||
enriched.relevance_score = content_relevance(hit.score, max_score);
|
||||
enriched.snippet = hit.snippet.clone();
|
||||
enriched.match_source = Some("content".to_string());
|
||||
enriched_files.push(enriched);
|
||||
added += 1;
|
||||
}
|
||||
if added > 0 {
|
||||
sort_enriched_files(enriched_files, &criteria.sort_by);
|
||||
}
|
||||
Ok(added)
|
||||
}
|
||||
|
||||
/// Quick suggestions search — returns up to `limit` name suggestions
|
||||
/// matching the query. Pushes filtering, relevance sort and LIMIT to SQL
|
||||
/// so only a handful of rows cross the DB→app boundary.
|
||||
@@ -305,6 +446,10 @@ impl SearchUseCase for SearchService {
|
||||
// Pre-compute once — avoids N heap allocations inside enrich_file/enrich_folder.
|
||||
let query_lower = query.to_lowercase();
|
||||
|
||||
// Content-index candidates (first page only). Feature-off or an
|
||||
// index failure yields an empty set — the search stays name-only.
|
||||
let content_hits = self.lookup_content_hits(&criteria, user_id).await;
|
||||
|
||||
// For non-recursive searches, use efficient database-level pagination
|
||||
// This avoids loading all files into memory
|
||||
if !criteria.recursive {
|
||||
@@ -316,7 +461,7 @@ impl SearchUseCase for SearchService {
|
||||
|
||||
// Convert to DTOs and enrich with metadata
|
||||
let file_dtos: Vec<FileDto> = files.into_iter().map(FileDto::from).collect();
|
||||
let enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
.iter()
|
||||
.map(|f| Self::enrich_file(f, &query_lower))
|
||||
.collect();
|
||||
@@ -360,6 +505,12 @@ impl SearchUseCase for SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
// Blend in content-discovered files before the pagination math.
|
||||
let added = self
|
||||
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
|
||||
.await?;
|
||||
let total_file_count = total_file_count + added;
|
||||
|
||||
let folder_count = enriched_folders.len();
|
||||
let total_count = total_file_count + folder_count;
|
||||
|
||||
@@ -416,7 +567,7 @@ impl SearchUseCase for SearchService {
|
||||
|
||||
// ── Convert to DTOs and enrich with server-computed metadata ──
|
||||
let file_dtos: Vec<FileDto> = found_files.into_iter().map(FileDto::from).collect();
|
||||
let enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
let mut enriched_files: Vec<SearchFileResultDto> = file_dtos
|
||||
.iter()
|
||||
.map(|f| Self::enrich_file(f, &query_lower))
|
||||
.collect();
|
||||
@@ -446,6 +597,12 @@ impl SearchUseCase for SearchService {
|
||||
}
|
||||
}
|
||||
|
||||
// Blend in content-discovered files before the pagination math.
|
||||
let added = self
|
||||
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
|
||||
.await?;
|
||||
let total_file_count = total_file_count + added;
|
||||
|
||||
// ── Pagination (folders first, then files) ──
|
||||
let folder_count = enriched_folders.len();
|
||||
let total_count = total_file_count + folder_count;
|
||||
@@ -535,3 +692,60 @@ impl SearchService {
|
||||
SearchServiceStub
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn content_relevance_stays_below_name_contains_band() {
|
||||
// Best hit of the set caps at 45 — always under contains (50).
|
||||
assert_eq!(content_relevance(8.0, 8.0), 45);
|
||||
assert_eq!(content_relevance(4.0, 8.0), 28);
|
||||
// Degenerate inputs fall to the floor instead of panicking.
|
||||
assert_eq!(content_relevance(1.0, 0.0), 10);
|
||||
assert_eq!(content_relevance(f32::NAN, 8.0), 10);
|
||||
assert!(content_relevance(0.0, 8.0) >= 10);
|
||||
}
|
||||
|
||||
fn dto(name: &str, relevance: u32, size: u64, modified_at: u64) -> SearchFileResultDto {
|
||||
SearchFileResultDto {
|
||||
id: name.to_string(),
|
||||
name: name.to_string(),
|
||||
path: format!("/{name}"),
|
||||
size,
|
||||
mime_type: "text/plain".to_string(),
|
||||
folder_id: None,
|
||||
created_at: 0,
|
||||
modified_at,
|
||||
relevance_score: relevance,
|
||||
size_formatted: String::new(),
|
||||
icon_class: String::new(),
|
||||
icon_special_class: String::new(),
|
||||
category: String::new(),
|
||||
blob_hash: String::new(),
|
||||
snippet: None,
|
||||
match_source: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merged_files_resort_by_relevance_and_by_column() {
|
||||
let mut files = vec![
|
||||
dto("b-content.txt", 30, 10, 200),
|
||||
dto("a-name.txt", 80, 99, 100),
|
||||
];
|
||||
sort_enriched_files(&mut files, "relevance");
|
||||
assert_eq!(
|
||||
files[0].name, "a-name.txt",
|
||||
"name match must outrank content match"
|
||||
);
|
||||
|
||||
sort_enriched_files(&mut files, "size_desc");
|
||||
assert_eq!(files[0].name, "a-name.txt");
|
||||
sort_enriched_files(&mut files, "date");
|
||||
assert_eq!(files[0].name, "a-name.txt");
|
||||
sort_enriched_files(&mut files, "name_desc");
|
||||
assert_eq!(files[0].name, "b-content.txt");
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user