feat(drive): remove all user_id ref in file or folder

This commit is contained in:
Edouard Vanbelle
2026-07-03 01:33:19 +02:00
parent 37467ed9d3
commit dff0e7365e
8 changed files with 264 additions and 238 deletions
@@ -78,7 +78,6 @@ impl PathResolverService {
String, // name
String, // path
Option<String>, // parent_id
Option<String>, // user_id
Uuid, // drive_id
i64, // created_at
i64, // modified_at
@@ -90,7 +89,7 @@ impl PathResolverService {
),
>(
r#"
SELECT resource_type, id, name, path, parent_id, user_id, drive_id,
SELECT resource_type, id, name, path, parent_id, drive_id,
created_at, modified_at, size, mime_type, folder_id,
blob_hash, tree_modified_at
FROM (
@@ -99,7 +98,6 @@ impl PathResolverService {
fo.name,
fo.path,
fo.parent_id::text,
fo.user_id::text,
fo.drive_id,
EXTRACT(EPOCH FROM fo.created_at)::bigint AS created_at,
EXTRACT(EPOCH FROM fo.updated_at)::bigint AS modified_at,
@@ -123,7 +121,6 @@ impl PathResolverService {
ELSE fi.name
END AS path,
NULL::text AS parent_id,
fi.user_id::text,
fi.drive_id,
EXTRACT(EPOCH FROM fi.created_at)::bigint AS created_at,
EXTRACT(EPOCH FROM fi.updated_at)::bigint AS modified_at,
@@ -160,7 +157,6 @@ impl PathResolverService {
name,
res_path,
parent_id,
_uid, // Post-D7: `user_id` column no longer flowed into the DTO.
drive_id,
created_at,
modified_at,
@@ -243,21 +243,16 @@ impl ContentIndexWorker {
// Authoritative state re-read: a queued 'upsert' whose row vanished
// or got trashed in the meantime becomes a delete.
//
// Post-D7: `fi.user_id` is nullable and new rows land NULL.
// The projected `user_id::text` therefore comes back as
// `Option<String>`; we normalise to `""` at the tuple boundary
// so the downstream indexing code doesn't have to change.
// The Tantivy `user_id` field is defence-in-depth only — every
// query is Must-scoped by `drive_id`.
// (file_id, user_id, drive_id, name, blob_hash, mime, size).
// `user_id` is `Option<String>` because post-D7 `storage.files.user_id`
// is nullable — new rows land NULL.
type FileIndexRow = (Uuid, Option<String>, String, String, String, String, i64);
// Post-D7: `fi.user_id` is dropped — no longer projected. The
// Tantivy `user_id` field survives as defence-in-depth but now
// always indexes `""`. Every query is Must-scoped by `drive_id`.
// (file_id, drive_id, name, blob_hash, mime, size).
type FileIndexRow = (Uuid, String, String, String, String, i64);
let files: Vec<FileIndexRow> = if upsert_candidates.is_empty() {
Vec::new()
} else {
sqlx::query_as(
"SELECT fi.id, fi.user_id::text, fi.drive_id::text, fi.name,
"SELECT fi.id, fi.drive_id::text, fi.name,
fi.blob_hash, fi.mime_type, fi.size
FROM storage.files fi
WHERE fi.id = ANY($1) AND NOT fi.is_trashed",
@@ -272,10 +267,10 @@ impl ContentIndexWorker {
// Per-blob text: batch-read the extraction cache, extract misses.
let wanted_hashes: Vec<String> = files
.iter()
.filter(|(_, _, _, name, _, mime, size)| {
.filter(|(_, _, name, _, mime, size)| {
text_extractor::supports(name, mime) && *size as u64 <= self.max_extract_file_bytes
})
.map(|f| f.4.clone())
.map(|f| f.3.clone())
.collect();
let mut text_by_hash: HashMap<String, Option<String>> = HashMap::new();
if !wanted_hashes.is_empty() {
@@ -292,7 +287,7 @@ impl ContentIndexWorker {
}
let mut records = Vec::with_capacity(files.len());
for (file_id, user_id, drive_id, name, blob_hash, mime, size) in files {
for (file_id, drive_id, name, blob_hash, mime, size) in files {
let supported = text_extractor::supports(&name, &mime);
let content = if !supported {
None
@@ -311,7 +306,7 @@ impl ContentIndexWorker {
.map(|t| truncate_on_char(t, PREVIEW_BYTES));
records.push(IndexDocRecord {
file_id: file_id.to_string(),
user_id: user_id.unwrap_or_default(),
user_id: String::new(),
drive_id,
name,
content,