From 27942a2a72f0206821591875cd6efcd6890a6176 Mon Sep 17 00:00:00 2001 From: SZCJW <792430652@qq.com> Date: Thu, 1 Oct 2026 01:10:39 +0800 Subject: [PATCH] fix(search): stop dropping the name filter for short queries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The SQL repositories gated the name ILIKE condition on name.len() >= 3 *bytes*, so a 1-2 character search (e.g. "ab") silently returned an arbitrary page of the caller's files instead of matches — while the suggest dropdown (which never had the gate) still found them, making the top-bar search feel broken. The gate existed because the pg_trgm GIN index cannot accelerate sub-trigram patterns, but wrong-but-indexed is never acceptable: caller/folder scoping still bounds the scanned set and result pages are LIMIT-bound. Add a shared name_filter_active() predicate next to like_escape() (any non-blank query filters) and use it at all 7 gate sites — 4 in file_blob_read_repository (condition/bind pairs kept in lockstep so bind indices stay aligned), 3 match guards in folder_db_repository. The Tantivy >= 2 gate on the content index is deliberately left alone: it means "too-short tokens don't enter the full-text index", not a correctness gate. Co-Authored-By: Claude Code --- .../pg/file_blob_read_repository.rs | 8 ++-- .../repositories/pg/folder_db_repository.rs | 6 +-- src/infrastructure/repositories/pg/mod.rs | 43 +++++++++++++++++++ status.md | 28 ++++++++++++ 4 files changed, 78 insertions(+), 7 deletions(-) diff --git a/src/infrastructure/repositories/pg/file_blob_read_repository.rs b/src/infrastructure/repositories/pg/file_blob_read_repository.rs index d29481ed..8288b996 100644 --- a/src/infrastructure/repositories/pg/file_blob_read_repository.rs +++ b/src/infrastructure/repositories/pg/file_blob_read_repository.rs @@ -1247,7 +1247,7 @@ impl FileReadPort for FileBlobReadRepository { } if let Some(name) = &criteria.name_contains - && name.len() >= 3 + && super::name_filter_active(name) { bind_idx += 1; conditions.push(format!("fi.name ILIKE ${bind_idx}")); @@ -1319,7 +1319,7 @@ impl FileReadPort for FileBlobReadRepository { query = query.bind(fid); } if let Some(name) = &criteria.name_contains - && name.len() >= 3 + && super::name_filter_active(name) { query = query.bind(super::like_escape(name)); } @@ -1412,7 +1412,7 @@ impl FileReadPort for FileBlobReadRepository { ); if let Some(name) = &criteria.name_contains - && name.len() >= 3 + && super::name_filter_active(name) { bind_idx += 1; conditions.push(format!("fi.name ILIKE ${bind_idx}")); @@ -1477,7 +1477,7 @@ impl FileReadPort for FileBlobReadRepository { .bind(root_id); if let Some(name) = &criteria.name_contains - && name.len() >= 3 + && super::name_filter_active(name) { query = query.bind(super::like_escape(name)); } diff --git a/src/infrastructure/repositories/pg/folder_db_repository.rs b/src/infrastructure/repositories/pg/folder_db_repository.rs index e6ce6167..39c99665 100644 --- a/src/infrastructure/repositories/pg/folder_db_repository.rs +++ b/src/infrastructure/repositories/pg/folder_db_repository.rs @@ -1147,7 +1147,7 @@ impl FolderRepository for FolderDbRepository { // Build optional name filter — use ILIKE (case-insensitive) so the // GIN trigram index idx_folders_name_trgm is used instead of a seq scan. let (name_clause, name_pattern) = match name_contains { - Some(name) if name.len() >= 3 => ( + Some(name) if super::name_filter_active(name) => ( if recursive { " AND fo.name ILIKE $2" } else { @@ -1228,7 +1228,7 @@ impl FolderRepository for FolderDbRepository { } else { // Root folders: parent_id IS NULL, params ($1=caller_id, $2=pattern) let name_clause_root = match name_contains { - Some(name) if name.len() >= 3 => " AND fo.name ILIKE $2", + Some(name) if super::name_filter_active(name) => " AND fo.name ILIKE $2", _ => "", }; format!( @@ -1295,7 +1295,7 @@ impl FolderRepository for FolderDbRepository { caller_id: Uuid, ) -> Result, DomainError> { let (where_extra, name_pattern) = match name_contains { - Some(name) if name.len() >= 3 => { + Some(name) if super::name_filter_active(name) => { (" AND fo.name ILIKE $3", Some(super::like_escape(name))) } _ => ("", None), diff --git a/src/infrastructure/repositories/pg/mod.rs b/src/infrastructure/repositories/pg/mod.rs index 857bf3d9..c9566f4c 100644 --- a/src/infrastructure/repositories/pg/mod.rs +++ b/src/infrastructure/repositories/pg/mod.rs @@ -77,3 +77,46 @@ pub fn like_escape(raw: &str) -> String { .replace('_', "\\_"); format!("%{escaped}%") } + +/// Whether a search's `name_contains` value should produce an `ILIKE` +/// predicate: any non-blank query filters, including 1–2 character ones. +/// +/// The historical `name.len() >= 3` bytes gate existed because the pg_trgm +/// GIN index cannot accelerate sub-trigram patterns — but gating on it here +/// *dropped the name condition entirely* for shorter queries, so a search +/// for `ab` returned an arbitrary page of the caller's files instead of +/// matches (and disagreed with `suggest_files_by_name`, which never had the +/// gate). Wrong-but-indexed is never acceptable in a storage product: the +/// caller/folder scoping in every caller of this predicate still bounds the +/// scanned set, and result pages are `LIMIT`-bound. +#[inline] +pub fn name_filter_active(name: &str) -> bool { + !name.trim().is_empty() +} + +#[cfg(test)] +mod name_filter_tests { + use super::name_filter_active; + + #[test] + fn one_and_two_char_queries_still_filter() { + // The old `len() >= 3` bytes gate silently dropped the name + // condition for these — this test is the regression guard. + assert!(name_filter_active("a")); + assert!(name_filter_active("ab")); + } + + #[test] + fn single_cjk_char_filters() { + // 3 bytes in UTF-8: already passed the old gate, must keep passing. + assert!(name_filter_active("合")); + } + + #[test] + fn blank_queries_do_not_filter() { + // Blank means "match everything" — no ILIKE predicate is emitted. + assert!(!name_filter_active("")); + assert!(!name_filter_active(" ")); + assert!(!name_filter_active("\t\n")); + } +} diff --git a/status.md b/status.md index 83cc1258..fca2000e 100644 --- a/status.md +++ b/status.md @@ -9,6 +9,34 @@ ## 已完成 +### [2026-10-01] 修复搜索框短关键词失效(<3 字节查询静默丢名字过滤) +- **状态**: 已完成(cargo fmt ✓;clippy --all-features --all-targets -D warnings 0 警告 ✓; + `cargo test --lib` 930 通过 0 失败,含新增 3 个回归守卫测试 ✓) +- **计划**: 修复"原先的搜索框模糊搜索能力有问题"。根因:7 处 SQL 仓库把名字 `ILIKE` + 条件门在 `name.len() >= 3`(**字节**)上——查询短于 3 字节时名字条件被**整体丢弃**, + `/api/search?q=ab` 返回任意一页文件而非匹配项;而 suggest 下拉(`suggest_files_by_name`) + 没有此门槛,所以表现为"下拉能找到、结果页找不到"。门槛初衷是 pg_trgm GIN 索引 + 无法加速 <3 字符的 pattern,但"错误但有索引"不可接受(韧性优先):调用方的 + caller/folder 作用域仍约束扫描集,结果页有 LIMIT。 +- **修法**: `pg/mod.rs` 新增共享判定 `name_filter_active`(非空白即过滤,trim 后判空) + + 3 个单测(1/2 字符、CJK 单字 3 字节、空白不过滤);7 处门槛全部换用该判定: + - `src/infrastructure/repositories/pg/file_blob_read_repository.rs` — 4 处 + (search_files_paginated 条件+bind、search_files_in_subtree 条件+bind; + 条件与 bind 成对替换,参数序号不错位) + - `src/infrastructure/repositories/pg/folder_db_repository.rs` — 3 处 + (search_folders、根目录臂、list_descendant_folders 的 match guard) + - 刻意不动:`search_service.rs` L404 Tantivy 内容索引的 `>= 2` 门槛(语义是 + "太短的词不进全文索引",非正确性门槛);`search_dto.rs` 无校验,保持原样 +- **改动文件**(均为上游文件,局部替换): + - `src/infrastructure/repositories/pg/mod.rs` — `name_filter_active` + `mod name_filter_tests` + - `src/infrastructure/repositories/pg/file_blob_read_repository.rs` — 4 处判定替换 + - `src/infrastructure/repositories/pg/folder_db_repository.rs` — 3 处判定替换 +- **仅本地文件**: 无新增(`status.md` 本身) +- **上游冲突风险**: 低 — 纯判定函数替换,无 SQL/结构改动;若上游也修此 bug 会天然收敛 +- **已知边界(未处理,记为后续)**: 词序模糊("report 2026" 匹配 "2026 report.docx") + 与分词匹配不在本次范围——`ILIKE %…%` 是子串语义,需要 tsvector/pg_trgm 相似度 + 或分词方案,属功能增强而非缺陷修复 + ### [2026-10-01] 移除文件页搜索过滤栏(顶栏统一入口) - **状态**: 已完成(`npm run check` 全绿:svelte-check 0 错 0 警 + eslint + stylelint + prettier;`vitest run` 492 通过 0 失败) - **计划**: 顶栏搜索框与筛选面板(`29d0c335`)已提供同一套筛选能力后,文件页自己的