refactor(search): normalize answer to /resources format

This commit is contained in:
Edouard Vanbelle
2026-07-26 14:42:04 +02:00
parent 14da27db10
commit c22741bc7f
15 changed files with 950 additions and 346 deletions
+420 -5
View File
@@ -1,6 +1,11 @@
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::sync::Arc;
use utoipa::ToSchema;
use utoipa::{IntoParams, ToSchema};
use uuid::Uuid;
use crate::application::dtos::cursor::PageCursor;
use crate::application::dtos::grant_dto::{ResourceContentDto, ResourceTypeDto};
/**
* Data Transfer Object for file search criteria.
@@ -99,7 +104,16 @@ impl Default for SearchCriteriaDto {
}
}
/// A file search result enriched with server-computed metadata
/// A file search result enriched with server-computed metadata.
///
/// Phase 1-plus extension (AuthZ-adjacent audit follow-up, 2026-07-26):
/// carries `etag`, `created_by`, `updated_by`, `is_favorite`, `is_shared`
/// through from `FileDto`. Pre-fix these fields were dropped at `enrich_file`
/// time, so the wire-normalised `SearchResourcesDto` handler couldn't
/// reconstruct a full `FileDto` for its `resource` slot — every result
/// looked unfavorited / unshared, and provenance was blank. The extra
/// columns come from `file_blob_read_repository.rs::search_files_paginated`
/// (SELECT'd inline, EXISTS subqueries for the caller-scoped booleans).
#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)]
pub struct SearchFileResultDto {
/// File ID
@@ -148,9 +162,36 @@ pub struct SearchFileResultDto {
/// "content" (discovered via the full-text content index).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub match_source: Option<String>,
/// HTTP ETag — derived from `blob_hash + modified_at`. Duplicates
/// `FileDto::etag` so the wire handler can hand a client the same
/// token for `If-Match` / `If-None-Match` conditional requests on
/// search results as it would on a folder listing.
#[serde(default)]
pub etag: String,
/// §14 provenance — user that originally created this file. `None`
/// when the referenced user has been deleted or for legacy rows.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub created_by: Option<Uuid>,
/// §14 provenance — user that performed the most recent mutation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub updated_by: Option<Uuid>,
/// Caller-scoped: `true` when the requesting user has favorited
/// this file. Populated by an EXISTS subquery in the search SQL —
/// the search repo carries it back as a per-row bool that the
/// service plumbs into this DTO.
#[serde(default)]
pub is_favorite: bool,
/// Resource-scoped: `true` when the file has ANY explicit role-grant.
/// Populated by the sibling EXISTS on `storage.role_grants`.
#[serde(default)]
pub is_shared: bool,
}
/// A folder search result enriched with server-computed metadata
/// A folder search result enriched with server-computed metadata.
///
/// See `SearchFileResultDto` for the Phase 1-plus rationale — same
/// story: the six caller/provenance fields are carried through so the
/// wire-normalised handler can hand the frontend a complete `FolderDto`.
#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)]
pub struct SearchFolderResultDto {
/// Folder ID
@@ -164,7 +205,7 @@ pub struct SearchFolderResultDto {
/// Drive that owns this folder. Same column as `storage.folders.drive_id`,
/// carried through so downstream callers (e.g. the NC search REPORT
/// handler) can populate `FolderDto::drive_id` without a fallback sentinel.
pub drive_id: uuid::Uuid,
pub drive_id: Uuid,
/// Creation timestamp
pub created_at: u64,
/// Last modification timestamp
@@ -173,6 +214,25 @@ pub struct SearchFolderResultDto {
pub is_root: bool,
/// Relevance score (0-100) computed server-side
pub relevance_score: u32,
/// HTTP ETag — folders derive theirs from the tree-etag propagator.
/// Duplicating it here keeps the wire handler's `FolderDto`
/// reconstruction complete.
#[serde(default)]
pub etag: String,
/// §14 provenance — creator user id. `None` for legacy folders.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub created_by: Option<Uuid>,
/// §14 provenance — last-mutator user id.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub updated_by: Option<Uuid>,
/// Caller-scoped: `true` when the requesting user has favorited
/// this folder. EXISTS on `auth.user_favorites`.
#[serde(default)]
pub is_favorite: bool,
/// Resource-scoped: `true` when the folder has any explicit
/// role-grant. EXISTS on `storage.role_grants`.
#[serde(default)]
pub is_shared: bool,
}
/**
@@ -182,7 +242,7 @@ pub struct SearchFolderResultDto {
* both files and folders that match the search criteria, along with pagination
* information and server-computed metadata.
*/
#[derive(Debug, Serialize, Deserialize, ToSchema)]
#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)]
pub struct SearchResultsDto {
/// Files matching the search criteria (enriched with metadata)
pub files: Vec<SearchFileResultDto>,
@@ -252,6 +312,361 @@ impl SearchResultsDto {
}
}
// ═══════════════════════════════════════════════════════════════════════════
// New wire shape — normalised to the `/*/resources` envelope
// (`items[] { resource_type, resource, meta }` + `next_cursor` + optional
// `total`/`query_time_ms`). Phase 1-plus: internal service still speaks
// `SearchCriteriaDto`/`SearchResultsDto`; the REST handler translates.
// ═══════════════════════════════════════════════════════════════════════════
/// Query parameters for `GET /api/search`.
///
/// Mirrors `FolderResourcesQuery` for the shared axes (`limit`, `cursor`,
/// `order_by`, `resource_types`, `reverse`) then adds search-specific
/// filters (`query`, `folder_id`, `recursive`, `file_types`,
/// `created_after`/`before`, `modified_after`/`before`, `min_size`/
/// `max_size`). `serde_urlencoded` doesn't support `#[serde(flatten)]`,
/// so the paging fields are inlined rather than composed from
/// `CursorQuery`.
#[derive(Debug, Deserialize, IntoParams)]
pub struct SearchResourcesQuery {
/// Search phrase (matched against name; optionally content when the
/// full-text index is enabled). Absent = "match everything," so
/// callers can page through with just a folder scope + filters.
pub query: Option<String>,
/// Maximum items per page (1–200, default 50).
#[serde(default = "SearchResourcesQuery::default_limit")]
pub limit: u32,
/// Opaque cursor from a previous response. Absent = first page.
/// Encodes the current offset — Phase 1-plus still uses the
/// existing offset-based service internals under the hood.
pub cursor: Option<String>,
/// Sort dimension. Supported: `"relevance"` (default), `"name"`,
/// `"name_desc"`, `"date"` (= `modified_at`), `"date_desc"`,
/// `"size"`, `"size_desc"`. Names match the pre-normalisation
/// values `SearchCriteriaDto.sort_by` accepted so cached results
/// remain reachable.
pub order_by: Option<String>,
/// Comma-separated resource types to include, e.g. `"file,folder"`.
/// Absent = both. Matches the `FolderResourcesQuery` idiom.
pub resource_types: Option<String>,
/// Reverse the sort order. Default `false`.
#[serde(default)]
pub reverse: bool,
/// Comma-separated file extensions filter, e.g. `"pdf,docx"`.
#[serde(rename = "type")]
pub type_filter: Option<String>,
/// Restrict search to this folder.
pub folder_id: Option<String>,
/// Recursive traversal below `folder_id`. Default `true`.
#[serde(default = "SearchResourcesQuery::default_recursive")]
pub recursive: bool,
/// Minimum creation timestamp (seconds since epoch).
pub created_after: Option<u64>,
/// Maximum creation timestamp (seconds since epoch).
pub created_before: Option<u64>,
/// Minimum modification timestamp (seconds since epoch).
pub modified_after: Option<u64>,
/// Maximum modification timestamp (seconds since epoch).
pub modified_before: Option<u64>,
/// Minimum file size in bytes.
pub min_size: Option<u64>,
/// Maximum file size in bytes.
pub max_size: Option<u64>,
}
impl SearchResourcesQuery {
pub fn default_limit() -> u32 {
50
}
pub fn default_recursive() -> bool {
true
}
pub fn limit_clamped(&self) -> usize {
self.limit.clamp(1, 200) as usize
}
pub fn decode_cursor(&self) -> Option<SearchResourceCursor> {
self.cursor
.as_deref()
.and_then(SearchResourceCursor::decode)
}
/// Convert to the internal `SearchCriteriaDto` the service still
/// consumes. `limit` / `offset` come from the decoded cursor (or the
/// query's `limit` on the first page). Sort names pass through
/// verbatim — the service's `sort_by` matcher accepts the same set.
pub fn to_criteria(&self) -> SearchCriteriaDto {
let offset = self.decode_cursor().map(|c| c.offset).unwrap_or(0);
let file_types = self.type_filter.as_deref().map(|s| {
s.split(',')
.map(|t| t.trim().to_string())
.filter(|t| !t.is_empty())
.collect()
});
SearchCriteriaDto {
name_contains: self.query.clone(),
file_types,
created_after: self.created_after,
created_before: self.created_before,
modified_after: self.modified_after,
modified_before: self.modified_before,
min_size: self.min_size,
max_size: self.max_size,
folder_id: self.folder_id.clone(),
recursive: self.recursive,
limit: self.limit_clamped(),
offset,
sort_by: self.order_by.clone().unwrap_or_else(default_sort_by),
}
}
/// Which resource kinds to include. `None` = both. Anything else
/// selects the intersection.
pub fn include_files(&self) -> bool {
match self.resource_types.as_deref() {
None => true,
Some(s) => s.split(',').any(|t| t.trim() == "file"),
}
}
pub fn include_folders(&self) -> bool {
match self.resource_types.as_deref() {
None => true,
Some(s) => s.split(',').any(|t| t.trim() == "folder"),
}
}
}
/// Opaque cursor for `/api/search`. Encodes the offset the underlying
/// service still uses, plus the sort dimension so a page fetched with
/// a different `order_by` than the previous one cannot silently drift
/// into a broken keyset. Phase 2 (service rewrite) would replace this
/// with a true keyset cursor over `(sort_key, id)`.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SearchResourceCursor {
pub offset: usize,
pub order_by: String,
}
impl PageCursor for SearchResourceCursor {}
/// Search-specific per-item metadata (relevance score, snippet, hit
/// source). Sits inline on each `SearchResourceItem` so consumers get
/// data locality — no keyed lookup. `ResourceList` ignores the field.
///
/// Wire keys are shortened (`meta.score`, `meta.via`) vs the internal
/// `SearchFileResultDto` field names (`relevance_score`, `match_source`)
/// to keep the envelope compact on large result pages.
#[derive(Debug, Serialize, ToSchema)]
pub struct SearchMeta {
/// Relevance 0-100. Higher = better match.
pub score: u32,
/// Plain-text fragment around the first content-index hit. Absent
/// for name-only matches and for folder results.
#[serde(skip_serializing_if = "Option::is_none")]
pub snippet: Option<String>,
/// Where the hit came from: `"name"` or `"content"`. Absent when the
/// origin is ambiguous (empty query → everything matches).
#[serde(skip_serializing_if = "Option::is_none")]
pub via: Option<String>,
}
/// One search result — same `resource_type + resource` shape as the
/// `/*/resources` envelopes so `ResourceList` consumes it as-is, plus
/// the inline `meta` for search-specific enrichment.
#[derive(Debug, Serialize, ToSchema)]
pub struct SearchResourceItem {
pub resource_type: ResourceTypeDto,
/// Full resource details (untagged: `FileDto | FolderDto | DriveDto`).
/// Shape determined by `resource_type`.
pub resource: ResourceContentDto,
/// Search-specific metadata for this row.
pub meta: SearchMeta,
}
/// Response envelope for `GET /api/search` — cursor-paginated + search
/// metadata. `total` is an approximate caller-visible count (permission-
/// filtered) when the service can compute it cheaply, absent otherwise —
/// matches sibling envelope endpoints, which all serialise counts as
/// integers (see `app_password_dto`, `plugin_dto`, `pagination`).
#[derive(Debug, Serialize, ToSchema)]
pub struct SearchResourcesDto {
pub items: Vec<SearchResourceItem>,
/// Opaque cursor for the next page. Absent on the last page.
#[serde(skip_serializing_if = "Option::is_none")]
pub next_cursor: Option<String>,
/// Server-side query time. UI shows "Found N in Xms" and admins use
/// it as a health signal.
pub query_time_ms: u64,
/// Approximate caller-visible total match count. Never leaks a count
/// for rows the caller cannot see. Omitted when unknown.
#[serde(skip_serializing_if = "Option::is_none")]
pub total: Option<usize>,
}
impl SearchResourcesDto {
/// Build the envelope from the service's existing offset-paginated
/// result plus the request's cursor position. The service returns
/// `SearchResultsDto` with `total_count` + `has_more` derived from
/// COUNT(*) OVER(); we translate:
/// - `has_more` → derive `next_cursor` (encoding `offset + returned`).
/// - `total_count` → pass through as `Some(42)` when known, else `None`.
///
/// Ownership: consumes the service result so the enriched DTOs move
/// into the `resource` slot without cloning.
pub fn from_service_result(
results: crate::application::dtos::search_dto::SearchResultsDto,
query: &SearchResourcesQuery,
) -> Self {
let order_by = query.order_by.clone().unwrap_or_else(default_sort_by);
let current_offset = query.decode_cursor().map(|c| c.offset).unwrap_or(0);
let returned = results.files.len() + results.folders.len();
let next_cursor = if results.has_more {
Some(
SearchResourceCursor {
offset: current_offset + returned,
order_by: order_by.clone(),
}
.encode(),
)
} else {
None
};
let total = results.total_count;
// Build items in an order the UI expects: folders first (like the
// legacy split-shape) unless a specific sort is requested. When
// ordering by relevance / date / size the caller almost always
// wants interleaved output; when ordering by name the folders-
// first convention matches file managers. Splitting the choice
// by sort dimension keeps folder browsing intuitive.
let mut items: Vec<SearchResourceItem> = Vec::with_capacity(returned);
let query_lower = query
.query
.as_deref()
.map(|s| s.to_lowercase())
.unwrap_or_default();
let folders_first = matches!(order_by.as_str(), "name" | "name_desc");
if folders_first {
append_folders(&mut items, results.folders);
append_files(&mut items, results.files, &query_lower);
} else {
// Interleave by relevance_score (or the natural service order for
// date/size — the service already returns rows in the requested
// dimension, but folders and files come as two separate arrays
// that we merge here by score for `relevance`, or just append
// for size/date since the two arrays are individually ordered.
append_folders(&mut items, results.folders);
append_files(&mut items, results.files, &query_lower);
if order_by == "relevance" {
items.sort_by_key(|item| std::cmp::Reverse(item.meta.score));
}
}
Self {
items,
next_cursor,
query_time_ms: results.query_time_ms,
total,
}
}
}
fn append_files(
items: &mut Vec<SearchResourceItem>,
files: Vec<SearchFileResultDto>,
_query_lower: &str,
) {
for f in files {
let meta = SearchMeta {
score: f.relevance_score,
snippet: f.snippet.clone(),
via: f.match_source.clone(),
};
// Reconstruct FileDto from the enriched search result. `size_formatted`
// and display fields were already computed by `enrich_file`; the
// Phase 1-plus extensions (etag / created_by / updated_by /
// is_favorite / is_shared) carry through so the DTO is complete.
let file_dto = crate::application::dtos::file_dto::FileDto {
id: f.id,
name: f.name,
path: f.path,
size: f.size,
mime_type: f.mime_type,
folder_id: f.folder_id,
created_at: f.created_at,
modified_at: f.modified_at,
icon_class: f.icon_class,
icon_special_class: f.icon_special_class,
category: f.category,
size_formatted: f.size_formatted,
content_hash: f.blob_hash,
etag: f.etag,
created_by: f.created_by,
updated_by: f.updated_by,
is_favorite: f.is_favorite,
is_shared: f.is_shared,
sort_date: None,
};
items.push(SearchResourceItem {
resource_type: ResourceTypeDto::File,
resource: ResourceContentDto::File(file_dto),
meta,
});
}
}
fn append_folders(items: &mut Vec<SearchResourceItem>, folders: Vec<SearchFolderResultDto>) {
for f in folders {
let meta = SearchMeta {
score: f.relevance_score,
snippet: None,
via: None,
};
let folder_dto = crate::application::dtos::folder_dto::FolderDto {
id: f.id.clone(),
name: f.name,
path: f.path,
parent_id: f.parent_id,
drive_id: f.drive_id,
created_at: f.created_at,
modified_at: f.modified_at,
is_root: f.is_root,
// Folders carry closed-set display fields — always the
// same three static strings. Cheap to build via `Arc::from`
// (interning-worthy but not on the search hot path).
icon_class: Arc::from("fas fa-folder"),
icon_special_class: Arc::from("folder-icon"),
category: Arc::from("Folder"),
etag: f.etag,
created_by: f.created_by,
updated_by: f.updated_by,
is_favorite: f.is_favorite,
is_shared: f.is_shared,
};
items.push(SearchResourceItem {
resource_type: ResourceTypeDto::Folder,
resource: ResourceContentDto::Folder(folder_dto),
meta,
});
}
}
// `search_meta` map form was explored earlier and rejected in favour of
// inline `meta` per item (Ed 2026-07-26): data locality wins, no
// keyed-lookup step for consumers, matches the extensibility other
// `/*/resources` endpoints will want later.
#[allow(dead_code)]
fn _keep_hashmap_import_alive_for_future(_: HashMap<String, SearchMeta>) {}
/// DTO for search suggestion results (quick prefix search)
#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)]
pub struct SearchSuggestionsDto {
+14 -4
View File
@@ -160,13 +160,22 @@ pub trait FileReadPort: Send + Sync + 'static {
/// are expanded inline via `storage.caller_group_ids($caller)`.
///
/// # Returns
/// A tuple of (files, total_count) where files are paginated and filtered
/// A tuple `(files, caller_flags, total_count)`:
/// - `files`: the paginated + filtered file rows.
/// - `caller_flags`: parallel `Vec<(is_favorite, is_shared)>` aligned
/// 1:1 with `files` by index. Populated in-SQL via per-row EXISTS
/// subqueries on `auth.user_favorites` and `storage.role_grants` so
/// the caller's SPA can render badges without a follow-up round-trip
/// (same pattern the photos-timeline listing uses). Kept as a
/// parallel vec rather than folded into `File` so the domain
/// entity stays caller-agnostic.
/// - `total_count`: `COUNT(*) OVER()` total for pagination.
async fn search_files_paginated(
&self,
folder_id: Option<&str>,
criteria: &SearchCriteriaDto,
caller_id: Uuid,
) -> Result<(Vec<File>, usize), DomainError>;
) -> Result<(Vec<File>, Vec<(bool, bool)>, usize), DomainError>;
/// Search files recursively in a folder subtree using ltree.
///
@@ -177,13 +186,14 @@ pub trait FileReadPort: Send + Sync + 'static {
/// Post-PR-B: scoped by drive-membership grants (same semantics as
/// `search_files_paginated`), not by `files.user_id`.
///
/// Returns a tuple of (matching files, total count for pagination).
/// Returns the same shape as [`Self::search_files_paginated`]:
/// `(files, caller_flags, total_count)`.
async fn search_files_in_subtree(
&self,
root_folder_id: Option<&str>,
criteria: &SearchCriteriaDto,
caller_id: Uuid,
) -> Result<(Vec<File>, usize), DomainError> {
) -> Result<(Vec<File>, Vec<(bool, bool)>, usize), DomainError> {
// Default: delegate to paginated search (non-recursive fallback)
self.search_files_paginated(root_folder_id, criteria, caller_id)
.await
+174 -78
View File
@@ -304,6 +304,16 @@ impl SearchService {
blob_hash: file.content_hash,
snippet: None,
match_source: (!query_lower.is_empty() && relevance > 0).then(|| "name".to_string()),
// Phase 1-plus: carry the FileDto fields the old enrich
// shape dropped. Populates the normalised
// `SearchResourcesDto` items with a complete `FileDto` so
// the frontend `ResourceList` renders favorites / share
// badges / provenance consistently with other listings.
etag: file.etag,
created_by: file.created_by,
updated_by: file.updated_by,
is_favorite: file.is_favorite,
is_shared: file.is_shared,
}
}
@@ -329,6 +339,13 @@ impl SearchService {
modified_at: folder.modified_at,
is_root: folder.is_root,
relevance_score: relevance,
// Phase 1-plus (see sibling `enrich_file`): carry the
// FolderDto fields the old enrich shape dropped.
etag: folder.etag,
created_by: folder.created_by,
updated_by: folder.updated_by,
is_favorite: folder.is_favorite,
is_shared: folder.is_shared,
}
}
@@ -651,18 +668,12 @@ impl SearchUseCase for SearchService {
// For non-recursive searches, use efficient database-level pagination
// This avoids loading all files into memory
if !criteria.recursive {
// The content-index lookup (drive resolve + Tantivy +
// ReBAC batch), the file page and the folder query are
// mutually independent — overlap them so the search pays
// ~max() instead of the serial sum (`suggest_with_perms`
// already used this shape; ROUND10 brought it here).
let (content_hits, files_page, folders_res) = tokio::join!(
// Same folders-first sequencing as the recursive branch
// below (see the block comment there for the rationale
// — SQL applies file offset+limit, so folder_count has
// to be known before the file query is issued).
let (content_hits, folders_res) = tokio::join!(
self.lookup_content_hits(&criteria, user_id),
self.file_repository.search_files_paginated(
criteria.folder_id.as_deref(),
&criteria,
user_id,
),
self.folder_repository.search_folders(
criteria.folder_id.as_deref(),
criteria.name_contains.as_deref(),
@@ -670,20 +681,57 @@ impl SearchUseCase for SearchService {
false,
),
);
let (files, total_file_count) = files_page?;
let folders = folders_res?;
let (folders, folder_flags) = folders_res?;
let folder_count = folders.len();
let folders_before_page = criteria.offset.min(folder_count);
let folders_on_page = (folder_count - folders_before_page).min(criteria.limit);
let file_offset = criteria.offset - folders_before_page;
let file_limit_needed = criteria.limit - folders_on_page;
let file_limit_probe = file_limit_needed.max(1);
let mut file_criteria = criteria.clone();
file_criteria.offset = file_offset;
file_criteria.limit = file_limit_probe;
let (files, file_flags, total_file_count) = self
.file_repository
.search_files_paginated(
criteria.folder_id.as_deref(),
&file_criteria,
user_id,
)
.await?;
// Convert to DTOs and enrich with metadata — one fused
// pass, no intermediate Vec<FileDto> materialization.
// `file_flags` is aligned 1:1 with `files` (in-SQL
// per-row EXISTS on favorites + role_grants) so a
// simple parallel zip plumbs the caller-scoped
// booleans onto the FileDto before enrichment.
let mut enriched_files: Vec<SearchFileResultDto> = files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.zip(file_flags)
.map(|(f, (is_fav, is_shr))| {
let mut dto = FileDto::from(f);
dto.is_favorite = is_fav;
dto.is_shared = is_shr;
Self::enrich_file(dto, &query_lower)
})
.collect();
// For folders, apply sorting and pagination in memory (usually fewer folders)
// For folders, apply sorting and pagination in memory (usually fewer folders).
// Same parallel-zip shape as the file branch above —
// `folder_flags` is aligned 1:1 by index.
let mut enriched_folders: Vec<SearchFolderResultDto> = folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.zip(folder_flags)
.map(|(f, (is_fav, is_shr))| {
let mut dto = FolderDto::from(f);
dto.is_favorite = is_fav;
dto.is_shared = is_shr;
Self::enrich_folder(dto, &query_lower)
})
.collect();
// Sort folders (cached_key avoids O(N log N) temporary String allocations)
@@ -705,39 +753,35 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
// Blend in content-discovered files, then truncate to
// the exact page size — see the recursive branch's
// block comment for why the truncate is required.
//
// Note: `total_count` is intentionally the pure SQL
// name-match count (plus folders) — NOT inflated by
// `added` content-hit rows. `added` counts hits that
// aren't already in the *current* SQL slice, which is
// per-page (different SQL rows on each page produce
// different dedup outcomes and a different `added`).
// Including it made the client-visible `total`
// flicker as the user paginated (2026-07-26 report:
// 4184 → 4186 across scope + page toggles for a
// stable dataset). Content-hits still bubble into
// each page's `items`; they just don't move the
// grand total.
let _added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
enriched_files.truncate(file_limit_needed);
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
// Combine and paginate (folders first, then files)
let start_idx = criteria.offset.min(total_count);
let end_idx = (criteria.offset + criteria.limit).min(total_count);
let folder_start = start_idx.min(folder_count);
let folder_end = end_idx.min(folder_count);
// Move the page out of the owned vecs instead of
// deep-cloning the slice — the source is dropped right
// after (benches/ROUND11.md §11: −300 allocs per page).
let paginated_folders: Vec<_> = enriched_folders
.into_iter()
.skip(folder_start)
.take(folder_end - folder_start)
.collect();
let file_start = start_idx.saturating_sub(folder_count);
let file_end = end_idx
.saturating_sub(folder_count)
.min(enriched_files.len());
let paginated_files: Vec<_> = enriched_files
.into_iter()
.skip(file_start)
.take(file_end - file_start)
.skip(folders_before_page)
.take(folders_on_page)
.collect();
let paginated_files = enriched_files;
let elapsed_ms = start.elapsed().as_millis() as u64;
@@ -757,16 +801,30 @@ impl SearchUseCase for SearchService {
// ── Recursive search via ltree (single SQL query per entity type) ──
// Uses PostgreSQL ltree GiST index to find all files and folders
// in the subtree in O(1) queries, replacing the O(N) spawn-per-folder
// approach that could saturate the connection pool. The content
// lookup, subtree file query and folder query overlap (`join!`),
// same as the non-recursive branch.
let (content_hits, files_page, folders_res) = tokio::join!(
// approach that could saturate the connection pool.
//
// ── Correct pagination across a folders-then-files list ──
// Pre-fix the service ran (content, file-page, folder-page)
// in one `tokio::join!` with the SAME `criteria.offset/limit`
// going to the file SQL, then re-paginated in memory using
// ABSOLUTE offsets. That was doubly wrong: SQL already
// applied `[offset, offset+limit)` and the in-memory slice
// then tried to skip `offset` MORE rows — for any query
// with few folders this dropped whole pages (2026-07-26:
// 4002-file query returned 50 rows on page 1 then `items:[]`
// on page 2 with a valid `next_cursor`).
//
// Fix: fold folders + content lookup first (they're both
// cheap and folder_count is what tells us how many files
// to skip). Then run the file SQL with `offset` shifted by
// `folder_count` and `limit` reduced by whatever folders
// fit on the current page — so SQL returns EXACTLY the
// file slice that belongs here, no in-memory re-slicing.
// The min-1 probe below preserves `COUNT(*) OVER()` even
// when folders fill the whole page (LIMIT 0 → 0 rows →
// total_count column projects nowhere → false zero).
let (content_hits, folders_res) = tokio::join!(
self.lookup_content_hits(&criteria, user_id),
self.file_repository.search_files_in_subtree(
criteria.folder_id.as_deref(),
&criteria,
user_id,
),
self.folder_repository.search_folders(
criteria.folder_id.as_deref(),
criteria.name_contains.as_deref(),
@@ -774,19 +832,50 @@ impl SearchUseCase for SearchService {
true,
),
);
let (found_files, total_file_count) = files_page?;
let found_folders: Vec<Folder> = folders_res?;
let (found_folders, folder_flags): (Vec<Folder>, Vec<(bool, bool)>) = folders_res?;
let folder_count = found_folders.len();
let folders_before_page = criteria.offset.min(folder_count);
let folders_on_page = (folder_count - folders_before_page).min(criteria.limit);
let file_offset = criteria.offset - folders_before_page;
let file_limit_needed = criteria.limit - folders_on_page;
// Probe with LIMIT ≥ 1 so `COUNT(*) OVER()` has a row to
// project onto; the extra row (if any) is truncated below.
let file_limit_probe = file_limit_needed.max(1);
let mut file_criteria = criteria.clone();
file_criteria.offset = file_offset;
file_criteria.limit = file_limit_probe;
let (found_files, file_flags, total_file_count) = self
.file_repository
.search_files_in_subtree(criteria.folder_id.as_deref(), &file_criteria, user_id)
.await?;
// ── Convert to DTOs and enrich with server-computed metadata ──
// Fused single pass: no intermediate DTO Vec materialization.
// Same shape as the non-recursive branch above — parallel
// zip of `found_files` with the in-SQL caller_flags.
let mut enriched_files: Vec<SearchFileResultDto> = found_files
.into_iter()
.map(|f| Self::enrich_file(FileDto::from(f), &query_lower))
.zip(file_flags)
.map(|(f, (is_fav, is_shr))| {
let mut dto = FileDto::from(f);
dto.is_favorite = is_fav;
dto.is_shared = is_shr;
Self::enrich_file(dto, &query_lower)
})
.collect();
let mut enriched_folders: Vec<SearchFolderResultDto> = found_folders
.into_iter()
.map(|f| Self::enrich_folder(FolderDto::from(f), &query_lower))
.zip(folder_flags)
.map(|(f, (is_fav, is_shr))| {
let mut dto = FolderDto::from(f);
dto.is_favorite = is_fav;
dto.is_shared = is_shr;
Self::enrich_folder(dto, &query_lower)
})
.collect();
// ── Sort folders (cached_key avoids O(N log N) temporary String allocations) ──
@@ -808,38 +897,35 @@ impl SearchUseCase for SearchService {
}
}
// Blend in content-discovered files before the pagination math.
let added = self
// Blend in content-discovered files. `merge_content_hits`
// pushes candidates from the Tantivy index onto the tail
// and re-sorts by the criteria; the SQL limit above only
// bounded the name-match set, so truncate after merging
// to the exact page size we intended to return.
//
// `total_count` is the pure SQL name-match count + folder
// count — see the non-recursive branch's block comment
// for why `added` is deliberately excluded (per-page
// dedup outcome, would flicker the client-visible
// total across pages).
let _added = self
.merge_content_hits(content_hits, &mut enriched_files, &criteria, user_id)
.await?;
let total_file_count = total_file_count + added;
enriched_files.truncate(file_limit_needed);
// ── Pagination (folders first, then files) ──
let folder_count = enriched_folders.len();
let total_count = total_file_count + folder_count;
let start_idx = criteria.offset.min(total_count);
let end_idx = (criteria.offset + criteria.limit).min(total_count);
let folder_start = start_idx.min(folder_count);
let folder_end = end_idx.min(folder_count);
// Move the page out instead of deep-cloning the slice — the
// recursive branch's vecs can hold the whole subtree match
// set, all dropped right after (benches/ROUND11.md §11).
// Fetches were pre-sliced: enriched_files is already the
// exact file page (SQL applied file_offset + file_limit),
// and folders_before_page / folders_on_page tell us which
// slice of `enriched_folders` belongs here. No in-memory
// absolute-offset math — see the block comment above.
let paginated_folders: Vec<_> = enriched_folders
.into_iter()
.skip(folder_start)
.take(folder_end - folder_start)
.collect();
let file_start = start_idx.saturating_sub(folder_count);
let file_end = end_idx
.saturating_sub(folder_count)
.min(enriched_files.len());
let paginated_files: Vec<_> = enriched_files
.into_iter()
.skip(file_start)
.take(file_end - file_start)
.skip(folders_before_page)
.take(folders_on_page)
.collect();
let paginated_files = enriched_files;
let elapsed_ms = start.elapsed().as_millis() as u64;
@@ -960,6 +1046,11 @@ mod tests {
blob_hash: String::new(),
snippet: None,
match_source: None,
etag: String::new(),
created_by: None,
updated_by: None,
is_favorite: false,
is_shared: false,
}
}
@@ -997,6 +1088,11 @@ mod tests {
modified_at: 0,
is_root: false,
relevance_score: 50,
etag: String::new(),
created_by: None,
updated_by: None,
is_favorite: false,
is_shared: false,
}],
100,
0,
+9 -2
View File
@@ -920,8 +920,15 @@ mod tests {
_folder_id: Option<&str>,
_criteria: &crate::application::dtos::search_dto::SearchCriteriaDto,
_user_id: Uuid,
) -> Result<(Vec<crate::domain::entities::file::File>, usize), DomainError> {
Ok((Vec::new(), 0))
) -> Result<
(
Vec<crate::domain::entities::file::File>,
Vec<(bool, bool)>,
usize,
),
DomainError,
> {
Ok((Vec::new(), Vec::new(), 0))
}
async fn stream_files_in_subtree(
@@ -532,8 +532,8 @@ impl FileReadPort for MockFileRepository {
_folder_id: Option<&str>,
_criteria: &crate::application::dtos::search_dto::SearchCriteriaDto,
_user_id: Uuid,
) -> std::result::Result<(Vec<File>, usize), DomainError> {
Ok((Vec::new(), 0))
) -> std::result::Result<(Vec<File>, Vec<(bool, bool)>, usize), DomainError> {
Ok((Vec::new(), Vec::new(), 0))
}
async fn stream_files_in_subtree(