perf: serve ranges from RAM cache, stream ZIPs, overlap ingest settle, O(1) chunk gate
Round 2 of benchmark-gated optimizations (benches/ROUND2.md; every change gated by a before/after in examples/bench_round2.rs — an AFTER that did not beat its BEFORE was to be rolled back; none needed it): - Range requests (REST/DAV/shares) answered from the moka content cache for sub-10MB files: PG resolve + open/seek/read -> Bytes::slice. 256KiB seeks: 1,730/s -> 3.7M/s (p50 552us -> 0.15us). - Streaming folder/share ZIPs via tokio duplex: TTFB no longer scales with archive size (326ms -> 0.4ms on 192MiB corpus; total also faster). Content-Length dropped (size unknown up front). - NC chunked-upload per-PUT gate: O(k) directory scan+stat -> in-RAM per-session counter (lazy rebuild on cold start). 1,000-chunk upload gate cost: 33.1s -> 0.09s cumulative. - Delta download + commit-verify now use the CDC path's buffered(read_prefetch) read-ahead: 64-chunk drain at 5ms open latency 440ms -> 51ms; order preserved. - CDC ingest settles batches on a spawned task (depth-1 pipeline) so the source stream keeps flowing during PG pin + backend writes; rollback ledger shared + lock-serialized so compensation stays exact on cancellation. 512MiB paced ingest: 60-69 -> 74-75 MB/s. OXICLOUD_INGEST_OVERLAP=0 restores inline settling (ops/bench hatch). - Frontend: instant-upload BLAKE3 hashing moved off the main thread to a bounded Web Worker pool (File handles by reference); vitest gate asserts the pool beats sequential (first gate draft posting buffers was 2.6x slower and was rewritten — copies dominated). Validation: cargo fmt + clippy -D warnings clean; 514 unit + 544 integration tests green; 270 frontend tests green. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CBK1RdtzyP6759Muqe1K1w
This commit is contained in:
@@ -44,8 +44,9 @@ impl From<ZipError> for DomainError {
|
||||
}
|
||||
}
|
||||
|
||||
/// Type alias for the fully-async ZIP writer backed by a buffered tokio file.
|
||||
type AsyncZipWriter = ZipFileWriter<Compat<BufWriter<tokio::fs::File>>>;
|
||||
/// Fully-async ZIP writer over any buffered tokio sink (temp file for the
|
||||
/// legacy path, one half of a `tokio::io::duplex` for the streaming path).
|
||||
type AsyncZipWriter<W> = ZipFileWriter<Compat<BufWriter<W>>>;
|
||||
|
||||
/// One planned archive entry, in final ZIP order.
|
||||
enum ZipPlanEntry {
|
||||
@@ -119,6 +120,110 @@ impl ZipService {
|
||||
folder_id: &str,
|
||||
folder_name: &str,
|
||||
) -> Result<NamedTempFile> {
|
||||
let plan = self.plan_archive(folder_id, folder_name).await?;
|
||||
|
||||
// ── Open the temp file + ZIP writer ──────────────────────────────
|
||||
let temp = NamedTempFile::new().map_err(ZipError::IoError)?;
|
||||
let tokio_file = tokio::fs::File::create(temp.path())
|
||||
.await
|
||||
.map_err(ZipError::IoError)?;
|
||||
|
||||
let (tx, mut rx) = tokio::sync::mpsc::channel::<Prefetched>(PREFETCH_BUFFER_CHUNKS);
|
||||
let _prefetcher = tokio::spawn(Self::prefetch_files(
|
||||
self.file_service.clone(),
|
||||
Self::planned_file_ids(&plan),
|
||||
tx,
|
||||
));
|
||||
Self::write_archive(tokio_file, &plan, &mut rx).await?;
|
||||
|
||||
Ok(temp)
|
||||
}
|
||||
|
||||
/// Streaming variant: the archive bytes are produced on a spawned task
|
||||
/// and yielded as they are written — the client's first byte arrives
|
||||
/// after the first entry starts, not after the whole archive has been
|
||||
/// built (the temp-file variant's time-to-first-byte grows with folder
|
||||
/// size; benches/ZIP-STREAM.md). The plan phase still runs inline so
|
||||
/// planning errors surface as proper HTTP errors; a blob-read error
|
||||
/// mid-archive can only truncate the stream (no central directory →
|
||||
/// clients detect the corrupt archive), which is the standard tradeoff
|
||||
/// for streamed ZIPs.
|
||||
pub async fn create_folder_zip_stream(
|
||||
&self,
|
||||
folder_id: &str,
|
||||
folder_name: &str,
|
||||
) -> Result<impl futures::Stream<Item = std::io::Result<bytes::Bytes>> + Send + use<>> {
|
||||
let plan = self.plan_archive(folder_id, folder_name).await?;
|
||||
|
||||
let (writer, reader) = tokio::io::duplex(256 * 1024);
|
||||
let (tx, mut rx) = tokio::sync::mpsc::channel::<Prefetched>(PREFETCH_BUFFER_CHUNKS);
|
||||
let _prefetcher = tokio::spawn(Self::prefetch_files(
|
||||
self.file_service.clone(),
|
||||
Self::planned_file_ids(&plan),
|
||||
tx,
|
||||
));
|
||||
tokio::spawn(async move {
|
||||
if let Err(e) = Self::write_archive(writer, &plan, &mut rx).await {
|
||||
// Dropping the writer EOFs the reader early — the truncated
|
||||
// archive has no central directory, so clients flag it.
|
||||
warn!("Streaming ZIP aborted mid-archive: {e}");
|
||||
}
|
||||
});
|
||||
|
||||
Ok(tokio_util::io::ReaderStream::new(reader))
|
||||
}
|
||||
|
||||
/// File ids of the plan, in archive order (the prefetcher's read list).
|
||||
fn planned_file_ids(plan: &[ZipPlanEntry]) -> Vec<String> {
|
||||
plan.iter()
|
||||
.filter_map(|entry| match entry {
|
||||
ZipPlanEntry::File { file_id, .. } => Some(file_id.clone()),
|
||||
ZipPlanEntry::Dir(_) => None,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Write every planned entry through a buffered ZIP writer over `sink`,
|
||||
/// then finalize (central directory + flush). Shared by the temp-file
|
||||
/// and streaming variants.
|
||||
async fn write_archive<W: tokio::io::AsyncWrite + Unpin>(
|
||||
sink: W,
|
||||
plan: &[ZipPlanEntry],
|
||||
rx: &mut tokio::sync::mpsc::Receiver<Prefetched>,
|
||||
) -> Result<()> {
|
||||
let buf_writer = BufWriter::with_capacity(256 * 1024, sink);
|
||||
let mut zip = ZipFileWriter::with_tokio(buf_writer);
|
||||
|
||||
for entry in plan {
|
||||
match entry {
|
||||
ZipPlanEntry::Dir(zip_dir) => {
|
||||
let dir_entry =
|
||||
ZipEntryBuilder::new(zip_dir.clone().into(), Compression::Stored);
|
||||
match zip.write_entry_whole(dir_entry, &[]).await {
|
||||
Ok(()) => debug!("Folder added to ZIP: {}", zip_dir),
|
||||
Err(e) => {
|
||||
warn!("Could not add folder entry (may already exist): {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
ZipPlanEntry::File {
|
||||
zip_path,
|
||||
compression,
|
||||
..
|
||||
} => {
|
||||
Self::write_prefetched_file(&mut zip, zip_path, *compression, rx).await?;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut compat_writer = zip.close().await.map_err(ZipError::AsyncZipError)?;
|
||||
compat_writer.close().await.map_err(ZipError::IoError)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Resolve the folder, fetch its subtree (2 bulk queries) and lay out
|
||||
/// the archive entries in final ZIP order.
|
||||
async fn plan_archive(&self, folder_id: &str, folder_name: &str) -> Result<Vec<ZipPlanEntry>> {
|
||||
info!(
|
||||
"Creating ZIP for folder: {} (ID: {})",
|
||||
folder_name, folder_id
|
||||
@@ -200,61 +305,7 @@ impl ZipService {
|
||||
}
|
||||
}
|
||||
|
||||
// ── 5. Open the temp file + ZIP writer ───────────────────────────
|
||||
let temp = NamedTempFile::new().map_err(ZipError::IoError)?;
|
||||
let tokio_file = tokio::fs::File::create(temp.path())
|
||||
.await
|
||||
.map_err(ZipError::IoError)?;
|
||||
let buf_writer = BufWriter::with_capacity(256 * 1024, tokio_file);
|
||||
let mut zip = ZipFileWriter::with_tokio(buf_writer);
|
||||
|
||||
// ── 6. Write entries: 2-stage pipeline ───────────────────────────
|
||||
// The prefetch task reads blob streams for the planned files, in
|
||||
// order, ahead of the writer — the next file's blob-store latency
|
||||
// overlaps the current file's deflate. If the writer bails out,
|
||||
// dropping the receiver makes the prefetcher's next send fail and
|
||||
// it stops on its own.
|
||||
let file_ids: Vec<String> = plan
|
||||
.iter()
|
||||
.filter_map(|entry| match entry {
|
||||
ZipPlanEntry::File { file_id, .. } => Some(file_id.clone()),
|
||||
ZipPlanEntry::Dir(_) => None,
|
||||
})
|
||||
.collect();
|
||||
let (tx, mut rx) = tokio::sync::mpsc::channel::<Prefetched>(PREFETCH_BUFFER_CHUNKS);
|
||||
let _prefetcher = tokio::spawn(Self::prefetch_files(
|
||||
self.file_service.clone(),
|
||||
file_ids,
|
||||
tx,
|
||||
));
|
||||
|
||||
for entry in &plan {
|
||||
match entry {
|
||||
ZipPlanEntry::Dir(zip_dir) => {
|
||||
let dir_entry =
|
||||
ZipEntryBuilder::new(zip_dir.clone().into(), Compression::Stored);
|
||||
match zip.write_entry_whole(dir_entry, &[]).await {
|
||||
Ok(()) => debug!("Folder added to ZIP: {}", zip_dir),
|
||||
Err(e) => {
|
||||
warn!("Could not add folder entry (may already exist): {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
ZipPlanEntry::File {
|
||||
zip_path,
|
||||
compression,
|
||||
..
|
||||
} => {
|
||||
Self::write_prefetched_file(&mut zip, zip_path, *compression, &mut rx).await?;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── 7. Finalize ──────────────────────────────────────────────────
|
||||
let mut compat_writer = zip.close().await.map_err(ZipError::AsyncZipError)?;
|
||||
compat_writer.close().await.map_err(ZipError::IoError)?;
|
||||
|
||||
Ok(temp)
|
||||
Ok(plan)
|
||||
}
|
||||
|
||||
/// Prefetch stage: streams each planned file's content from the blob
|
||||
@@ -302,8 +353,8 @@ impl ZipService {
|
||||
/// (`Stored` for already-compressed media, `Deflate` otherwise — see
|
||||
/// `entry_compression`). Peak memory stays bounded by the channel,
|
||||
/// independent of individual file sizes.
|
||||
async fn write_prefetched_file(
|
||||
zip: &mut AsyncZipWriter,
|
||||
async fn write_prefetched_file<W: tokio::io::AsyncWrite + Unpin>(
|
||||
zip: &mut AsyncZipWriter<W>,
|
||||
zip_path: &str,
|
||||
compression: Compression,
|
||||
rx: &mut tokio::sync::mpsc::Receiver<Prefetched>,
|
||||
|
||||
Reference in New Issue
Block a user