feat(upload): cover chunk upload + add support of different digest hash

Prefer stream storage rather using buffered (in memory)

  note: on many unix like tmpfs are in-memory, sungle PUT are sized limited

  Storage map (NC stands for Nextcloud gateway)

  ┌───────────────────────────────────────────────────────┬────────────────────────────────────────────────────────────────────┬─────────────────────────────────────────────────┐
  │                   Streaming surface                   │                            Destination                             │                Configurable via                 │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ REST chunked PUT /api/uploads/{id} chunk              │ {storage_path}/.uploads/{upload_id}/chunk_{NNNNNN}                 │ OXICLOUD_STORAGE_PATH (the .uploads subdir is   │
  │                                                       │                                                                    │ hard-wired)                                     │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ REST chunked assemble (during /complete)              │ {storage_path}/.uploads/{upload_id}/assembled                      │ same                                            │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ NC chunked PUT /dav/uploads/.../{chunk}               │ {storage_path}/.uploads/nextcloud/{user}/{upload_id}/{chunk_name}  │ same                                            │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ NC chunked assemble (during MOVE)                     │ {storage_path}/.uploads/nextcloud/{user}/{upload_id}/.assembled    │ same                                            │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ NC single-file PUT /dav/files/.../{path} (via         │ OXICLOUD_UPLOAD_TMPDIR if set, else OS default temp (/tmp on       │ OXICLOUD_UPLOAD_TMPDIR                          │
  │ spool_body_to_temp)                                   │ Linux)                                                             │                                                 │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ REST WebDAV PUT /webdav/{path} (via                   │ same as above                                                      │ OXICLOUD_UPLOAD_TMPDIR                          │
  │ spool_body_to_temp)                                   │                                                                    │                                                 │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ REST multipart upload /api/files/upload               │ {storage_path}/.dedup_temp/upload-{uuid}                           │ OXICLOUD_STORAGE_PATH (hard-wired subdir)       │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ WOPI PutFile                                          │ OS default temp via NamedTempFile::new() (no override)             │ (none — bug worth tracking)                     │
  ├───────────────────────────────────────────────────────┼────────────────────────────────────────────────────────────────────┼─────────────────────────────────────────────────┤
  │ Final blob storage (after fsync + rename)             │ {storage_path}/.blobs/{ab}/{abc…}.blob                             │ OXICLOUD_STORAGE_PATH                           │
  └───────────────────────────────────────────────────────┴────────────────────────────────────────────────────────────────────┴─────────────────────────────────────────────────┘

  one caveat: a malicious user can create many chunked upload and saturate local storage
This commit is contained in:
Edouard Vanbelle
2026-06-08 23:17:07 +02:00
parent 5e638691ad
commit 41aad26702
9 changed files with 1048 additions and 91 deletions
@@ -13,11 +13,11 @@ use axum::{
http::{HeaderMap, StatusCode, header},
response::{IntoResponse, Response},
};
use bytes::Bytes;
use serde::{Deserialize, Serialize};
use std::sync::Arc;
use utoipa::ToSchema;
use crate::application::ports::chunked_upload_ports::ChecksumAlg;
use crate::application::ports::chunked_upload_ports::ChunkedUploadPort;
use crate::application::ports::chunked_upload_ports::DEFAULT_CHUNK_SIZE;
use crate::application::ports::file_ports::FileUploadUseCase;
@@ -27,6 +27,7 @@ use crate::common::di::AppState;
use crate::domain::services::authorization::Permission;
use crate::interfaces::errors::AppError;
use crate::interfaces::middleware::auth::AuthUser;
use crate::interfaces::upload_spool::stream_body_to_path;
/// Request body for creating an upload session
#[derive(Debug, Deserialize, ToSchema)]
@@ -38,11 +39,17 @@ pub struct CreateUploadRequest {
pub chunk_size: Option<usize>,
}
/// Query params for chunk upload
/// Query params for chunk upload.
///
/// `checksumalg` is parsed via [`ChecksumAlg::parse`] and defaults to
/// `Md5` when absent — matching the legacy `Content-MD5` contract that
/// older clients rely on. Unknown algorithm names produce a 400 with the
/// offending value echoed back.
#[derive(Debug, Deserialize)]
pub struct ChunkUploadParams {
pub chunk_index: usize,
pub checksum: Option<String>,
pub checksumalg: Option<String>,
}
/// Final response after completing upload
@@ -200,58 +207,12 @@ impl ChunkedUploadHandler {
}
}
/// PATCH /api/uploads/:upload_id - Upload a chunk
///
/// Query params:
/// - chunk_index: The index of the chunk (0-based)
/// - checksum: Optional MD5 checksum for verification
///
/// Body: Raw bytes of the chunk
pub(super) async fn upload_chunk_impl(
State(state): State<Arc<AppState>>,
auth_user: AuthUser,
Path(upload_id): Path<String>,
Query(params): Query<ChunkUploadParams>,
headers: HeaderMap,
body: Bytes,
) -> impl IntoResponse {
let chunked_service = &state.core.chunked_upload_service;
// Extract checksum from header or query param
let checksum = params.checksum.or_else(|| {
headers
.get("Content-MD5")
.and_then(|v| v.to_str().ok())
.map(|s| s.to_string())
});
match chunked_service
.upload_chunk(&upload_id, auth_user.id, params.chunk_index, body, checksum)
.await
{
Ok(response) => {
let mut resp = Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, "application/json")
.header("Upload-Offset", response.bytes_received.to_string())
.header(
"Upload-Progress",
format!("{:.2}", response.progress * 100.0),
);
if response.is_complete {
resp = resp.header("Upload-Complete", "true");
}
resp.body(axum::body::Body::from(
serde_json::to_string(&response).unwrap(),
))
.unwrap()
.into_response()
}
Err(e) => AppError::from(e).into_response(),
}
}
// PATCH /api/uploads/:upload_id — moved entirely to the free
// function `upload_chunk` below so the body can be streamed
// (axum::body::Body) instead of materialised as `Bytes` here.
// The port-level `ChunkedUploadPort::upload_chunk` (Bytes-based)
// remains for tests and any future caller that genuinely has the
// bytes already in memory.
/// HEAD /api/uploads/:upload_id - Get upload status
///
@@ -448,40 +409,114 @@ pub async fn create_upload(
pub async fn upload_chunk(
State(state): State<Arc<AppState>>,
auth_user: AuthUser,
path: Path<String>,
query: Query<ChunkUploadParams>,
Path(upload_id): Path<String>,
Query(params): Query<ChunkUploadParams>,
headers: HeaderMap,
request: Request,
) -> impl IntoResponse {
// Cap the chunk body at `storage.chunk_max_bytes` (env
// `OXICLOUD_CHUNK_MAX_BYTES`, default 100 MB). Previous code used
// `usize::MAX` and `unwrap_or_default()` — two compounding bugs:
// - No upper bound → an oversized chunk OOMs the server.
// - Silent fallback to an empty body on transport error → the
// inner size check would either reject (good case) or — if the
// declared chunk_size was 0 (illegal but conceivable) — accept
// an empty upload as success. Either way the client got no
// actionable error.
let chunked_service = &state.core.chunked_upload_service;
let max_chunk = state.core.config.storage.chunk_max_bytes;
let body = match axum::body::to_bytes(request.into_body(), max_chunk).await {
Ok(b) => b,
// ── Resolve the client's checksum + algorithm ────────────────────
// Wire shape: `?checksum=<hex>&checksumalg=<name>` (or `Content-MD5`
// header for older clients). When `checksumalg` is omitted we
// default to MD5, matching the legacy contract — switching the
// default would silently break any client still relying on
// `Content-MD5` semantics.
let expected_checksum = params.checksum.clone().or_else(|| {
headers
.get("Content-MD5")
.and_then(|v| v.to_str().ok())
.map(|s| s.to_string())
});
let alg = match params.checksumalg.as_deref() {
Some(name) => match ChecksumAlg::parse(name) {
Some(a) => a,
None => {
return AppError::bad_request(format!(
"Unsupported checksumalg: {name} (supported: md5, sha256, blake3)"
))
.into_response();
}
},
None => ChecksumAlg::Md5,
};
// Only compute the hash when the client supplied an `expected_checksum`
// to verify against — saves ~30 ms per chunk for clients that don't.
let alg_to_compute = expected_checksum.as_ref().map(|_| alg);
// ── Phase 1: prepare ─────────────────────────────────────────────
// Validates session ownership + chunk index, returns the on-disk
// path and the chunk's declared size. The handler streams the body
// to that path; service finalises bookkeeping after the write.
let (chunk_path, _expected_size) = match chunked_service
.prepare_chunk(&upload_id, auth_user.id, params.chunk_index)
.await
{
Ok(p) => p,
Err(e) => return AppError::from(e).into_response(),
};
// ── Phase 2: stream the body straight to disk ────────────────────
// Peak heap ~one HTTP frame (~64 KB) regardless of chunk size or
// `chunk_max_bytes`. Optional incremental hashing happens here so
// verification doesn't require reading the chunk file back.
let streamed = match stream_body_to_path(
request.into_body(),
&chunk_path,
max_chunk,
alg_to_compute,
)
.await
{
Ok(s) => s,
Err(e) => {
tracing::warn!(
error = %e,
upload_id = %path.0,
error = ?e,
upload_id = %upload_id,
chunk_index = params.chunk_index,
max_chunk,
"Chunked upload PATCH rejected — body read failed (size cap or transport error)"
"Chunked upload PATCH rejected — streaming write failed (cap, transport, or IO)"
);
return AppError::payload_too_large(format!(
"Chunk read failed (cap {} bytes): {}",
max_chunk, e
))
.into_response();
return e.into_response();
}
};
ChunkedUploadHandler::upload_chunk_impl(State(state), auth_user, path, query, headers, body)
// ── Phase 3: commit ──────────────────────────────────────────────
// Size + checksum verification + session state update. Same RAM-only
// DashMap shard ownership pattern as the legacy `upload_chunk_inner`
// (held only for ~µs; bitmask persist done after release).
let response = match chunked_service
.commit_chunk(
&upload_id,
auth_user.id,
params.chunk_index,
streamed.bytes_written,
streamed.checksum_hex,
expected_checksum,
)
.await
.into_response()
{
Ok(r) => r,
Err(e) => return AppError::from(e).into_response(),
};
let mut resp = Response::builder()
.status(StatusCode::OK)
.header(header::CONTENT_TYPE, "application/json")
.header("Upload-Offset", response.bytes_received.to_string())
.header(
"Upload-Progress",
format!("{:.2}", response.progress * 100.0),
);
if response.is_complete {
resp = resp.header("Upload-Complete", "true");
}
resp.body(axum::body::Body::from(
serde_json::to_string(&response).unwrap(),
))
.unwrap()
.into_response()
}
#[utoipa::path(
+5 -1
View File
@@ -196,7 +196,11 @@ async fn handle_put_chunk(
.map_err(|e| AppError::bad_request(format!("Invalid chunk path: {}", e)))?;
let max_chunk = state.core.config.storage.chunk_max_bytes;
stream_body_to_path(req.into_body(), &chunk_path, max_chunk).await?;
// No client-side integrity contract on the NC chunked surface — the
// NC desktop client validates the assembled-file ETag against the
// server-side `oc:checksums` after MOVE. So we skip per-chunk
// hashing here (peak heap stays at ~one HTTP frame).
stream_body_to_path(req.into_body(), &chunk_path, max_chunk, None).await?;
Ok(Response::builder()
.status(StatusCode::CREATED)
+148 -12
View File
@@ -10,10 +10,17 @@ use std::path::{Path, PathBuf};
use axum::body::Body;
use http_body_util::BodyStream;
// The `Digest` trait (re-exported by both `md5` and `sha2` from the
// `digest` crate) gives `Md5` and `Sha256` their `new` / `update` /
// `finalize` methods. Importing once via `sha2` covers both —
// otherwise every call site would need fully-qualified
// `<md5::Md5 as md5::Digest>::…` syntax.
use sha2::Digest as _;
use tempfile::NamedTempFile;
use tokio::io::AsyncWriteExt;
use tokio_stream::StreamExt;
use crate::application::ports::chunked_upload_ports::ChecksumAlg;
use crate::common::temp::new_spool_temp_file;
use crate::interfaces::errors::AppError;
@@ -85,32 +92,50 @@ pub async fn spool_body_to_temp(
})
}
/// Result of a streamed write to a caller-supplied path.
pub struct StreamedToPath {
/// Total bytes written.
pub bytes_written: u64,
/// Lowercase hex digest, populated only when `checksum_alg=Some(_)`
/// was passed. The algorithm is identified by [`StreamedToPath::alg`].
pub checksum_hex: Option<String>,
/// Algorithm used to compute `checksum_hex`. Echoed back so the
/// caller can include it in audit logs or response headers.
pub alg: Option<ChecksumAlg>,
}
/// Stream an HTTP request body directly to a known destination file,
/// enforcing `max_bytes` as a hard size limit.
///
/// Used by the chunked-upload PUT handlers — each chunk has a deterministic
/// on-disk path (computed by `NextcloudChunkedUploadService::safe_chunk_path`
/// or the equivalent REST helper), so there's no need for a spool/move
/// dance. Peak heap is ~one HTTP frame regardless of chunk size or `max_bytes`.
/// Used by the chunked-upload PUT handlers — each chunk has a
/// deterministic on-disk path (`NextcloudChunkedUploadService::safe_chunk_path`
/// for the NC surface, `ChunkedUploadService::prepare_chunk` for the
/// REST surface), so there's no spool/move dance. Peak heap is ~one
/// HTTP frame regardless of chunk size or `max_bytes`.
///
/// **No hashing** — chunked uploads dedup at the assembled-file level, not
/// the chunk level, so computing BLAKE3 here would be wasted work.
/// `checksum_alg` is the optional client-requested integrity check
/// (default `md5` per the legacy `Content-MD5` contract; `blake3`
/// available for forward-compat). When `Some`, the hash is computed
/// incrementally during streaming — no extra disk read for verification.
///
/// On size overflow the partial file is removed before the function returns,
/// so a client retry against the same chunk name starts from a clean slate.
/// On any other I/O error the partial file is also removed and the error
/// surfaces — callers can assume the path is either fully written or absent.
/// On size overflow the partial file is removed before the function
/// returns, so a client retry against the same chunk name starts from
/// a clean slate. On any other I/O error the partial file is also
/// removed and the error surfaces — callers can assume the path is
/// either fully written or absent.
pub async fn stream_body_to_path(
body: Body,
path: &Path,
max_bytes: usize,
) -> Result<u64, AppError> {
checksum_alg: Option<ChecksumAlg>,
) -> Result<StreamedToPath, AppError> {
let mut file = tokio::fs::File::create(path)
.await
.map_err(|e| AppError::internal_error(format!("Failed to open chunk file: {e}")))?;
let mut total_bytes: usize = 0;
let mut stream = BodyStream::new(body);
let mut hasher = checksum_alg.map(IncrementalHasher::new);
while let Some(frame_result) = stream.next().await {
let frame = match frame_result {
@@ -132,6 +157,9 @@ pub async fn stream_body_to_path(
"Chunk exceeds maximum size of {max_bytes} bytes"
)));
}
if let Some(h) = hasher.as_mut() {
h.update(chunk);
}
if let Err(e) = file.write_all(chunk).await {
drop(file);
let _ = tokio::fs::remove_file(path).await;
@@ -146,5 +174,113 @@ pub async fn stream_body_to_path(
.map_err(|e| AppError::internal_error(format!("Failed to flush chunk file: {e}")))?;
drop(file);
Ok(total_bytes as u64)
Ok(StreamedToPath {
bytes_written: total_bytes as u64,
checksum_hex: hasher.map(IncrementalHasher::finalize_hex),
alg: checksum_alg,
})
}
/// Algorithm-agnostic incremental hasher used by [`stream_body_to_path`].
/// Per-frame `update` is sub-millisecond for all three algorithms at the
/// 64 KB frame sizes axum's body stream produces, so we don't need
/// `spawn_blocking` (which the old buffered path used because it hashed
/// the full multi-MB chunk in one shot).
enum IncrementalHasher {
Md5(md5::Md5),
Sha256(sha2::Sha256),
// Boxing — blake3::Hasher is ~1.7 KB on the stack while md5::Md5
// (~100 bytes) and sha2::Sha256 (~100 bytes) are tiny; boxing the
// outlier keeps the enum size proportional to the common case
// rather than the worst case.
Blake3(Box<blake3::Hasher>),
}
impl IncrementalHasher {
fn new(alg: ChecksumAlg) -> Self {
match alg {
ChecksumAlg::Md5 => Self::Md5(md5::Md5::new()),
ChecksumAlg::Sha256 => Self::Sha256(sha2::Sha256::new()),
ChecksumAlg::Blake3 => Self::Blake3(Box::new(blake3::Hasher::new())),
}
}
fn update(&mut self, bytes: &[u8]) {
match self {
Self::Md5(h) => h.update(bytes),
Self::Sha256(h) => h.update(bytes),
Self::Blake3(h) => {
h.update(bytes);
}
}
}
fn finalize_hex(self) -> String {
match self {
Self::Md5(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
Self::Sha256(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
Self::Blake3(h) => h.finalize().to_hex().to_string(),
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use bytes::Bytes;
#[tokio::test]
async fn stream_body_to_path_caps_oversized() {
let temp_dir = tempfile::tempdir().expect("tempdir");
let path = temp_dir.path().join("chunk");
// 5 MiB body, 4 MiB cap → must reject.
let body = Body::from(Bytes::from(vec![0u8; 5 * 1024 * 1024]));
let result = stream_body_to_path(body, &path, 4 * 1024 * 1024, None).await;
assert!(
result.is_err(),
"expected PayloadTooLarge, got Ok(bytes_written={})",
result.ok().map(|r| r.bytes_written).unwrap_or(0)
);
// Partial file must be removed on rejection.
assert!(
!path.exists(),
"rejected chunk file should be removed, but {} still exists",
path.display()
);
}
#[tokio::test]
async fn stream_body_to_path_accepts_under_cap() {
let temp_dir = tempfile::tempdir().expect("tempdir");
let path = temp_dir.path().join("chunk");
let body = Body::from(Bytes::from(vec![1u8; 1024 * 1024])); // 1 MiB
let result = stream_body_to_path(body, &path, 4 * 1024 * 1024, None).await;
let outcome = result.expect("should succeed");
assert_eq!(outcome.bytes_written, 1024 * 1024);
assert!(outcome.checksum_hex.is_none(), "no alg requested → no hash");
assert!(path.exists());
}
#[tokio::test]
async fn stream_body_to_path_caps_at_exact_boundary() {
// Edge case: body exactly equal to cap should succeed; cap+1 must fail.
let temp_dir = tempfile::tempdir().expect("tempdir");
let path = temp_dir.path().join("chunk");
let body = Body::from(Bytes::from(vec![1u8; 100]));
let outcome = stream_body_to_path(body, &path, 100, None)
.await
.expect("100 bytes at 100-byte cap should succeed");
assert_eq!(outcome.bytes_written, 100);
let path2 = temp_dir.path().join("chunk2");
let body = Body::from(Bytes::from(vec![1u8; 101]));
assert!(
stream_body_to_path(body, &path2, 100, None).await.is_err(),
"101 bytes at 100-byte cap must reject"
);
assert!(!path2.exists());
}
}