Merge pull request #438 from EdouardVanbelle/feat/cap-chunk-size-and-use-stream
This commit is contained in:
@@ -13,11 +13,11 @@ use axum::{
|
||||
http::{HeaderMap, StatusCode, header},
|
||||
response::{IntoResponse, Response},
|
||||
};
|
||||
use bytes::Bytes;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::sync::Arc;
|
||||
use utoipa::ToSchema;
|
||||
|
||||
use crate::application::ports::chunked_upload_ports::ChecksumAlg;
|
||||
use crate::application::ports::chunked_upload_ports::ChunkedUploadPort;
|
||||
use crate::application::ports::chunked_upload_ports::DEFAULT_CHUNK_SIZE;
|
||||
use crate::application::ports::file_ports::FileUploadUseCase;
|
||||
@@ -27,6 +27,7 @@ use crate::common::di::AppState;
|
||||
use crate::domain::services::authorization::Permission;
|
||||
use crate::interfaces::errors::AppError;
|
||||
use crate::interfaces::middleware::auth::AuthUser;
|
||||
use crate::interfaces::upload_spool::stream_body_to_path;
|
||||
|
||||
/// Request body for creating an upload session
|
||||
#[derive(Debug, Deserialize, ToSchema)]
|
||||
@@ -38,11 +39,17 @@ pub struct CreateUploadRequest {
|
||||
pub chunk_size: Option<usize>,
|
||||
}
|
||||
|
||||
/// Query params for chunk upload
|
||||
/// Query params for chunk upload.
|
||||
///
|
||||
/// `checksumalg` is parsed via [`ChecksumAlg::parse`] and defaults to
|
||||
/// `Md5` when absent — matching the legacy `Content-MD5` contract that
|
||||
/// older clients rely on. Unknown algorithm names produce a 400 with the
|
||||
/// offending value echoed back.
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct ChunkUploadParams {
|
||||
pub chunk_index: usize,
|
||||
pub checksum: Option<String>,
|
||||
pub checksumalg: Option<String>,
|
||||
}
|
||||
|
||||
/// Final response after completing upload
|
||||
@@ -54,6 +61,41 @@ pub struct CompleteUploadResponse {
|
||||
pub path: String,
|
||||
}
|
||||
|
||||
/// Optional body for `POST /api/uploads/{id}/complete`.
|
||||
///
|
||||
/// When the client supplies `checksum`, the server compares it against
|
||||
/// the assembled file's hash BEFORE promoting the blob to storage —
|
||||
/// failure aborts the upload atomically (no orphaned blob, no DB row).
|
||||
/// This is the end-to-end integrity check: per-chunk MD5 proves each
|
||||
/// chunk arrived intact, but only the final hash catches assembly /
|
||||
/// promotion bugs and mis-ordered chunks.
|
||||
///
|
||||
/// **`blake3` is highly recommended** — it's the algorithm the server
|
||||
/// already runs over the assembled file during hash-on-write
|
||||
/// assembly, so verification is a string comparison with zero extra
|
||||
/// I/O and zero extra CPU. It's also the same algorithm the server
|
||||
/// uses for blob-storage addressing, so the value the client sends
|
||||
/// equals the `content_hash` they'd later read back from
|
||||
/// `GET /api/files/{id}`. `md5` and `sha256` are accepted for
|
||||
/// compatibility with legacy client tooling but each triggers a
|
||||
/// second hash pass over the assembled file (~30–100 ms depending
|
||||
/// on size).
|
||||
///
|
||||
/// `Default` keeps the existing wire shape: clients that POST with no
|
||||
/// body get today's behavior (no verification, server just returns
|
||||
/// what it computed).
|
||||
#[derive(Debug, Default, Deserialize, ToSchema)]
|
||||
pub struct CompleteUploadRequest {
|
||||
/// Lowercase hex digest the client expects the assembled file to
|
||||
/// hash to. Compared case-insensitively. Omit to skip verification.
|
||||
pub checksum: Option<String>,
|
||||
/// Algorithm name. `blake3` is the recommended choice (default —
|
||||
/// matches the server's hash-on-write algorithm, zero extra cost).
|
||||
/// `md5`, `sha256` / `sha-256` are accepted but trigger an extra
|
||||
/// hash pass. Unknown values return 400.
|
||||
pub checksumalg: Option<String>,
|
||||
}
|
||||
|
||||
/// Chunked Upload Handler
|
||||
///
|
||||
/// The handler struct exists as a named grouping. All route functions are free
|
||||
@@ -120,6 +162,33 @@ impl ChunkedUploadHandler {
|
||||
.into_response();
|
||||
}
|
||||
|
||||
// ── Whole-file cap ──────────────────────────────────────────
|
||||
// Reject upfront, before any chunk is uploaded — wasting
|
||||
// bandwidth + server disk on an upload that's going to be
|
||||
// rejected at /complete is the worst-of-both-worlds outcome.
|
||||
// `max_upload_size` is the same ceiling that bounds direct
|
||||
// PUTs (per-byte during streaming there; declared per-session
|
||||
// here). When quotas are disabled, this is the only whole-file
|
||||
// limit for chunked uploads — without it a hostile client
|
||||
// could declare `total_size: 1 TB` and accumulate chunks
|
||||
// until disk fills.
|
||||
let max_upload = state.core.config.storage.max_upload_size as u64;
|
||||
if request.total_size > max_upload {
|
||||
tracing::warn!(
|
||||
"⛔ CHUNKED UPLOAD REJECTED (total_size cap): user={}, file={}, declared={}, max={}",
|
||||
auth_user.username,
|
||||
request.filename,
|
||||
request.total_size,
|
||||
max_upload
|
||||
);
|
||||
return AppError::payload_too_large(format!(
|
||||
"Declared total_size {} exceeds the server's `max_upload_size` cap ({} bytes). \
|
||||
Raise OXICLOUD_MAX_UPLOAD_SIZE on the server if larger uploads are expected.",
|
||||
request.total_size, max_upload
|
||||
))
|
||||
.into_response();
|
||||
}
|
||||
|
||||
// ── Permission pre-check: caller must have Create on the target
|
||||
// folder BEFORE we allocate a session and accept chunks. The
|
||||
// upload service re-checks at finalize time, but failing here
|
||||
@@ -200,58 +269,12 @@ impl ChunkedUploadHandler {
|
||||
}
|
||||
}
|
||||
|
||||
/// PATCH /api/uploads/:upload_id - Upload a chunk
|
||||
///
|
||||
/// Query params:
|
||||
/// - chunk_index: The index of the chunk (0-based)
|
||||
/// - checksum: Optional MD5 checksum for verification
|
||||
///
|
||||
/// Body: Raw bytes of the chunk
|
||||
pub(super) async fn upload_chunk_impl(
|
||||
State(state): State<Arc<AppState>>,
|
||||
auth_user: AuthUser,
|
||||
Path(upload_id): Path<String>,
|
||||
Query(params): Query<ChunkUploadParams>,
|
||||
headers: HeaderMap,
|
||||
body: Bytes,
|
||||
) -> impl IntoResponse {
|
||||
let chunked_service = &state.core.chunked_upload_service;
|
||||
|
||||
// Extract checksum from header or query param
|
||||
let checksum = params.checksum.or_else(|| {
|
||||
headers
|
||||
.get("Content-MD5")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.map(|s| s.to_string())
|
||||
});
|
||||
|
||||
match chunked_service
|
||||
.upload_chunk(&upload_id, auth_user.id, params.chunk_index, body, checksum)
|
||||
.await
|
||||
{
|
||||
Ok(response) => {
|
||||
let mut resp = Response::builder()
|
||||
.status(StatusCode::OK)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("Upload-Offset", response.bytes_received.to_string())
|
||||
.header(
|
||||
"Upload-Progress",
|
||||
format!("{:.2}", response.progress * 100.0),
|
||||
);
|
||||
|
||||
if response.is_complete {
|
||||
resp = resp.header("Upload-Complete", "true");
|
||||
}
|
||||
|
||||
resp.body(axum::body::Body::from(
|
||||
serde_json::to_string(&response).unwrap(),
|
||||
))
|
||||
.unwrap()
|
||||
.into_response()
|
||||
}
|
||||
Err(e) => AppError::from(e).into_response(),
|
||||
}
|
||||
}
|
||||
// PATCH /api/uploads/:upload_id — moved entirely to the free
|
||||
// function `upload_chunk` below so the body can be streamed
|
||||
// (axum::body::Body) instead of materialised as `Bytes` here.
|
||||
// The port-level `ChunkedUploadPort::upload_chunk` (Bytes-based)
|
||||
// remains for tests and any future caller that genuinely has the
|
||||
// bytes already in memory.
|
||||
|
||||
/// HEAD /api/uploads/:upload_id - Get upload status
|
||||
///
|
||||
@@ -284,19 +307,97 @@ impl ChunkedUploadHandler {
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute the requested checksum of the assembled file.
|
||||
///
|
||||
/// For `Blake3` the server already has the hash from hash-on-write
|
||||
/// assembly — we just return it (zero I/O, zero CPU). For `Md5` and
|
||||
/// `Sha256` we re-read the assembled file on the blocking pool and
|
||||
/// hash it; the cost (~30–100 ms for typical files) is the trade-off
|
||||
/// for accepting non-default algorithms.
|
||||
async fn compute_assembled_hash(
|
||||
assembled_path: &std::path::Path,
|
||||
alg: ChecksumAlg,
|
||||
blake3_already_computed: &str,
|
||||
) -> Result<String, std::io::Error> {
|
||||
match alg {
|
||||
ChecksumAlg::Blake3 => Ok(blake3_already_computed.to_string()),
|
||||
ChecksumAlg::Md5 | ChecksumAlg::Sha256 => {
|
||||
let path = assembled_path.to_path_buf();
|
||||
tokio::task::spawn_blocking(move || -> Result<String, std::io::Error> {
|
||||
use std::io::Read;
|
||||
let mut file = std::fs::File::open(&path)?;
|
||||
let mut buf = vec![0u8; 524_288];
|
||||
match alg {
|
||||
ChecksumAlg::Md5 => {
|
||||
use md5::Digest as _;
|
||||
let mut h = md5::Md5::new();
|
||||
loop {
|
||||
let n = file.read(&mut buf)?;
|
||||
if n == 0 {
|
||||
break;
|
||||
}
|
||||
h.update(&buf[..n]);
|
||||
}
|
||||
Ok(h.finalize().iter().map(|b| format!("{b:02x}")).collect())
|
||||
}
|
||||
ChecksumAlg::Sha256 => {
|
||||
use sha2::Digest as _;
|
||||
let mut h = sha2::Sha256::new();
|
||||
loop {
|
||||
let n = file.read(&mut buf)?;
|
||||
if n == 0 {
|
||||
break;
|
||||
}
|
||||
h.update(&buf[..n]);
|
||||
}
|
||||
Ok(h.finalize().iter().map(|b| format!("{b:02x}")).collect())
|
||||
}
|
||||
// Blake3 handled above — this branch is unreachable but
|
||||
// keeps the match exhaustive without an else-clause.
|
||||
ChecksumAlg::Blake3 => unreachable!(),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.map_err(|e| std::io::Error::other(format!("hash task join failed: {e}")))?
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// POST /api/uploads/:upload_id/complete - Finalize upload
|
||||
///
|
||||
/// Assembles all chunks into the final file and creates the file record
|
||||
// TODO: how is implemented security (owneship, permission ?)
|
||||
/// Assembles all chunks into the final file and creates the file record.
|
||||
/// When `body.checksum` is supplied, the assembled file's hash is
|
||||
/// verified before the blob is promoted to storage — mismatch
|
||||
/// returns 400 and the assembled temp is removed (the session
|
||||
/// itself is kept so the client can re-issue complete after
|
||||
/// diagnosing).
|
||||
pub(super) async fn complete_upload_impl(
|
||||
State(state): State<Arc<AppState>>,
|
||||
auth_user: AuthUser,
|
||||
Path(upload_id): Path<String>,
|
||||
body: CompleteUploadRequest,
|
||||
) -> impl IntoResponse {
|
||||
let chunked_service = &state.core.chunked_upload_service;
|
||||
let upload_service = &state.applications.file_upload_service;
|
||||
|
||||
// Assemble chunks (hash-on-write: SHA-256 computed during assembly)
|
||||
// ── Parse the optional algorithm BEFORE assembly so a bad
|
||||
// `checksumalg` doesn't waste the (potentially expensive)
|
||||
// hash work on a request we'll reject anyway.
|
||||
let alg = match body.checksumalg.as_deref() {
|
||||
Some(name) => match ChecksumAlg::parse(name) {
|
||||
Some(a) => Some(a),
|
||||
None => {
|
||||
return AppError::bad_request(format!(
|
||||
"Unsupported checksumalg: {name} (supported: md5, sha256, blake3)"
|
||||
))
|
||||
.into_response();
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
let expected_checksum = body.checksum.as_deref();
|
||||
|
||||
// Assemble chunks (hash-on-write: BLAKE3 computed during assembly)
|
||||
let (assembled_path, filename, folder_id, content_type, total_size, hash) =
|
||||
match chunked_service
|
||||
.complete_upload(&upload_id, auth_user.id)
|
||||
@@ -308,6 +409,46 @@ impl ChunkedUploadHandler {
|
||||
}
|
||||
};
|
||||
|
||||
// ── End-to-end integrity verification ───────────────────────
|
||||
// Only fires when the client supplied an `expected` checksum.
|
||||
// For BLAKE3 (the documented preferred choice) this is a string
|
||||
// comparison against the hash assembly already produced. For
|
||||
// MD5/SHA-256 we re-hash the assembled file on the blocking pool.
|
||||
if let Some(expected) = expected_checksum {
|
||||
let alg = alg.unwrap_or(ChecksumAlg::Blake3);
|
||||
let computed = match Self::compute_assembled_hash(&assembled_path, alg, &hash).await {
|
||||
Ok(c) => c,
|
||||
Err(e) => {
|
||||
let _ = tokio::fs::remove_file(&assembled_path).await;
|
||||
return AppError::internal_error(format!(
|
||||
"Failed to compute assembled checksum: {e}"
|
||||
))
|
||||
.into_response();
|
||||
}
|
||||
};
|
||||
if !computed.eq_ignore_ascii_case(expected) {
|
||||
let _ = tokio::fs::remove_file(&assembled_path).await;
|
||||
tracing::warn!(
|
||||
target: "audit",
|
||||
event = "chunked_upload.checksum_mismatch",
|
||||
reason = "final_checksum_mismatch",
|
||||
upload_id = %upload_id,
|
||||
user_id = %auth_user.id,
|
||||
alg = alg.as_str(),
|
||||
expected = %expected,
|
||||
actual = %computed,
|
||||
"👮🏻♂️ Chunked upload complete: client checksum mismatch — blob not promoted"
|
||||
);
|
||||
return AppError::bad_request(format!(
|
||||
"Checksum mismatch ({}): expected {}, got {}",
|
||||
alg.as_str(),
|
||||
expected,
|
||||
computed
|
||||
))
|
||||
.into_response();
|
||||
}
|
||||
}
|
||||
|
||||
// ── MIME detection (magic bytes + extension fallback) ─────
|
||||
let content_type = crate::common::mime_detect::refine_content_type_from_file(
|
||||
&assembled_path,
|
||||
@@ -420,29 +561,142 @@ pub async fn create_upload(
|
||||
params(
|
||||
("upload_id" = String, Path, description = "Upload session ID"),
|
||||
("chunk_index" = usize, Query, description = "Zero-based chunk index"),
|
||||
("checksum" = Option<String>, Query, description = "Optional MD5 checksum for integrity verification"),
|
||||
(
|
||||
"checksum" = Option<String>,
|
||||
Query,
|
||||
description = "Optional hex-encoded checksum for integrity verification. \
|
||||
Computed incrementally during the streaming write. \
|
||||
Algorithm is selected by `checksumalg` (default `md5`). \
|
||||
Also accepted via the legacy `Content-MD5` request header."
|
||||
),
|
||||
(
|
||||
"checksumalg" = Option<String>,
|
||||
Query,
|
||||
description = "Algorithm used by `checksum`. One of: `md5` (default, legacy), `sha256` / `sha-256`, `blake3`. \
|
||||
Unknown values return 400."
|
||||
),
|
||||
),
|
||||
request_body(content_type = "application/octet-stream", description = "Raw chunk bytes"),
|
||||
responses(
|
||||
(status = 200, description = "Chunk received", body = crate::application::ports::chunked_upload_ports::ChunkUploadResponseDto),
|
||||
(status = 400, description = "Invalid chunk or checksum mismatch"),
|
||||
(status = 400, description = "Invalid chunk, size mismatch, checksum mismatch, or unknown `checksumalg`"),
|
||||
(status = 404, description = "Upload session not found"),
|
||||
(status = 413, description = "Chunk exceeds `storage.chunk_max_bytes` cap"),
|
||||
),
|
||||
tag = "uploads",
|
||||
security(("bearerAuth" = []))
|
||||
)]
|
||||
pub async fn upload_chunk(
|
||||
state: State<Arc<AppState>>,
|
||||
State(state): State<Arc<AppState>>,
|
||||
auth_user: AuthUser,
|
||||
path: Path<String>,
|
||||
query: Query<ChunkUploadParams>,
|
||||
Path(upload_id): Path<String>,
|
||||
Query(params): Query<ChunkUploadParams>,
|
||||
headers: HeaderMap,
|
||||
request: Request,
|
||||
) -> impl IntoResponse {
|
||||
let body = axum::body::to_bytes(request.into_body(), usize::MAX)
|
||||
let chunked_service = &state.core.chunked_upload_service;
|
||||
let max_chunk = state.core.config.storage.chunk_max_bytes;
|
||||
|
||||
// ── Resolve the client's checksum + algorithm ────────────────────
|
||||
// Wire shape: `?checksum=<hex>&checksumalg=<name>` (or `Content-MD5`
|
||||
// header for older clients). When `checksumalg` is omitted we
|
||||
// default to MD5, matching the legacy contract — switching the
|
||||
// default would silently break any client still relying on
|
||||
// `Content-MD5` semantics.
|
||||
let expected_checksum = params.checksum.clone().or_else(|| {
|
||||
headers
|
||||
.get("Content-MD5")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.map(|s| s.to_string())
|
||||
});
|
||||
let alg = match params.checksumalg.as_deref() {
|
||||
Some(name) => match ChecksumAlg::parse(name) {
|
||||
Some(a) => a,
|
||||
None => {
|
||||
return AppError::bad_request(format!(
|
||||
"Unsupported checksumalg: {name} (supported: md5, sha256, blake3)"
|
||||
))
|
||||
.into_response();
|
||||
}
|
||||
},
|
||||
None => ChecksumAlg::Md5,
|
||||
};
|
||||
// Only compute the hash when the client supplied an `expected_checksum`
|
||||
// to verify against — saves ~30 ms per chunk for clients that don't.
|
||||
let alg_to_compute = expected_checksum.as_ref().map(|_| alg);
|
||||
|
||||
// ── Phase 1: prepare ─────────────────────────────────────────────
|
||||
// Validates session ownership + chunk index, returns the on-disk
|
||||
// path and the chunk's declared size. The handler streams the body
|
||||
// to that path; service finalises bookkeeping after the write.
|
||||
let (chunk_path, _expected_size) = match chunked_service
|
||||
.prepare_chunk(&upload_id, auth_user.id, params.chunk_index)
|
||||
.await
|
||||
.unwrap_or_default();
|
||||
ChunkedUploadHandler::upload_chunk_impl(state, auth_user, path, query, headers, body).await
|
||||
{
|
||||
Ok(p) => p,
|
||||
Err(e) => return AppError::from(e).into_response(),
|
||||
};
|
||||
|
||||
// ── Phase 2: stream the body straight to disk ────────────────────
|
||||
// Peak heap ~one HTTP frame (~64 KB) regardless of chunk size or
|
||||
// `chunk_max_bytes`. Optional incremental hashing happens here so
|
||||
// verification doesn't require reading the chunk file back.
|
||||
let streamed = match stream_body_to_path(
|
||||
request.into_body(),
|
||||
&chunk_path,
|
||||
max_chunk,
|
||||
alg_to_compute,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(s) => s,
|
||||
Err(e) => {
|
||||
tracing::warn!(
|
||||
error = ?e,
|
||||
upload_id = %upload_id,
|
||||
chunk_index = params.chunk_index,
|
||||
max_chunk,
|
||||
"Chunked upload PATCH rejected — streaming write failed (cap, transport, or IO)"
|
||||
);
|
||||
return e.into_response();
|
||||
}
|
||||
};
|
||||
|
||||
// ── Phase 3: commit ──────────────────────────────────────────────
|
||||
// Size + checksum verification + session state update. Same RAM-only
|
||||
// DashMap shard ownership pattern as the legacy `upload_chunk_inner`
|
||||
// (held only for ~µs; bitmask persist done after release).
|
||||
let response = match chunked_service
|
||||
.commit_chunk(
|
||||
&upload_id,
|
||||
auth_user.id,
|
||||
params.chunk_index,
|
||||
streamed.bytes_written,
|
||||
streamed.checksum_hex,
|
||||
expected_checksum,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(r) => r,
|
||||
Err(e) => return AppError::from(e).into_response(),
|
||||
};
|
||||
|
||||
let mut resp = Response::builder()
|
||||
.status(StatusCode::OK)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("Upload-Offset", response.bytes_received.to_string())
|
||||
.header(
|
||||
"Upload-Progress",
|
||||
format!("{:.2}", response.progress * 100.0),
|
||||
);
|
||||
if response.is_complete {
|
||||
resp = resp.header("Upload-Complete", "true");
|
||||
}
|
||||
resp.body(axum::body::Body::from(
|
||||
serde_json::to_string(&response).unwrap(),
|
||||
))
|
||||
.unwrap()
|
||||
.into_response()
|
||||
}
|
||||
|
||||
#[utoipa::path(
|
||||
@@ -472,10 +726,23 @@ pub async fn get_upload_status(
|
||||
params(
|
||||
("upload_id" = String, Path, description = "Upload session ID"),
|
||||
),
|
||||
request_body(
|
||||
content = CompleteUploadRequest,
|
||||
content_type = "application/json",
|
||||
description = "Optional. End-to-end integrity verification of the assembled file. \
|
||||
**`blake3` is highly recommended** as the `checksumalg` value — the server already \
|
||||
computes BLAKE3 over the assembled file during hash-on-write assembly, so \
|
||||
verification is a string comparison with zero extra CPU/IO. \
|
||||
Picking `md5` or `sha256` is supported for legacy client tooling but triggers a \
|
||||
second full hash pass over the assembled file. \
|
||||
Clients that POST with no body (or with an empty JSON object) get today's \
|
||||
behavior: no verification, server returns the BLAKE3 it computed."
|
||||
),
|
||||
responses(
|
||||
(status = 201, description = "File assembled and created", body = CompleteUploadResponse),
|
||||
(status = 400, description = "Unknown `checksumalg` or final-checksum mismatch"),
|
||||
(status = 404, description = "Upload session not found"),
|
||||
(status = 500, description = "Assembly or file creation failed"),
|
||||
(status = 500, description = "Assembly, hashing, or file creation failed"),
|
||||
),
|
||||
tag = "uploads",
|
||||
security(("bearerAuth" = []))
|
||||
@@ -484,8 +751,13 @@ pub async fn complete_upload(
|
||||
state: State<Arc<AppState>>,
|
||||
auth_user: AuthUser,
|
||||
path: Path<String>,
|
||||
// Empty body → `None` → default `CompleteUploadRequest`, preserving the
|
||||
// pre-checksum wire shape. Clients that DO send a body get strict
|
||||
// parsing (a malformed JSON returns 400 via the Json extractor).
|
||||
body: Option<Json<CompleteUploadRequest>>,
|
||||
) -> impl IntoResponse {
|
||||
ChunkedUploadHandler::complete_upload_impl(state, auth_user, path).await
|
||||
let req = body.map(|Json(r)| r).unwrap_or_default();
|
||||
ChunkedUploadHandler::complete_upload_impl(state, auth_user, path, req).await
|
||||
}
|
||||
|
||||
#[utoipa::path(
|
||||
|
||||
@@ -940,8 +940,10 @@ async fn handle_put(
|
||||
// another user — acceptable risk since PathResolver should always
|
||||
// be enabled in production)
|
||||
|
||||
// Hard upload size limit from config
|
||||
let max_upload = state.core.config.storage.max_upload_size;
|
||||
// Direct PUT cap — see `nextcloud/webdav_handler::handle_put` for
|
||||
// the reasoning. Files above `direct_put_max_bytes` must go through
|
||||
// the chunked-upload protocol (`/api/uploads/…`) which is resumable.
|
||||
let max_upload = state.core.config.storage.direct_put_max_bytes;
|
||||
|
||||
// Extract content type before consuming the request
|
||||
let content_type = req
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
use axum::{
|
||||
body::{self, Body},
|
||||
body::Body,
|
||||
http::{Request, StatusCode, header},
|
||||
response::Response,
|
||||
};
|
||||
@@ -10,6 +10,7 @@ use crate::common::di::AppState;
|
||||
use crate::common::mime_detect::{filename_from_path, refine_content_type_from_file};
|
||||
use crate::interfaces::errors::AppError;
|
||||
use crate::interfaces::middleware::auth::{AuthUser, CurrentUser};
|
||||
use crate::interfaces::upload_spool::stream_body_to_path;
|
||||
|
||||
/// Dispatch Nextcloud chunked upload WebDAV requests.
|
||||
///
|
||||
@@ -164,6 +165,14 @@ async fn handle_mkcol(
|
||||
}
|
||||
|
||||
/// PUT — store a chunk.
|
||||
///
|
||||
/// Streams the request body straight to the chunk file with peak heap of
|
||||
/// ~one HTTP frame, regardless of chunk size or the configured cap. The
|
||||
/// `storage.chunk_max_bytes` config (env `OXICLOUD_CHUNK_MAX_BYTES`,
|
||||
/// default 100 MB) bounds a single PUT — separate from `max_upload_size`
|
||||
/// which governs whole-file uploads. Without this separation, a client
|
||||
/// could submit a chunk up to the whole-file cap (10 GB default) and
|
||||
/// monopolise server memory.
|
||||
async fn handle_put_chunk(
|
||||
state: Arc<AppState>,
|
||||
req: Request<Body>,
|
||||
@@ -181,15 +190,17 @@ async fn handle_put_chunk(
|
||||
return Err(AppError::bad_request("Missing chunk name"));
|
||||
}
|
||||
|
||||
let max_upload = state.core.config.storage.max_upload_size;
|
||||
let body_bytes = body::to_bytes(req.into_body(), max_upload)
|
||||
.await
|
||||
.map_err(|e| AppError::bad_request(format!("Failed to read chunk body: {}", e)))?;
|
||||
let chunk_path = nc
|
||||
.chunked_uploads
|
||||
.safe_chunk_path(&user.username, upload_id, chunk_name)
|
||||
.map_err(|e| AppError::bad_request(format!("Invalid chunk path: {}", e)))?;
|
||||
|
||||
nc.chunked_uploads
|
||||
.store_chunk(&user.username, upload_id, chunk_name, &body_bytes)
|
||||
.await
|
||||
.map_err(|e| AppError::internal_error(format!("Failed to store chunk: {}", e)))?;
|
||||
let max_chunk = state.core.config.storage.chunk_max_bytes;
|
||||
// No client-side integrity contract on the NC chunked surface — the
|
||||
// NC desktop client validates the assembled-file ETag against the
|
||||
// server-side `oc:checksums` after MOVE. So we skip per-chunk
|
||||
// hashing here (peak heap stays at ~one HTTP frame).
|
||||
stream_body_to_path(req.into_body(), &chunk_path, max_chunk, None).await?;
|
||||
|
||||
Ok(Response::builder()
|
||||
.status(StatusCode::CREATED)
|
||||
@@ -228,8 +239,12 @@ async fn handle_assemble(
|
||||
let dest_subpath = extract_files_subpath(&destination, &user.username)
|
||||
.ok_or_else(|| AppError::bad_request("Invalid Destination URL"))?;
|
||||
|
||||
// Assemble chunks into a temp file (no full-file buffering in RAM).
|
||||
let (temp_path, size) = nc
|
||||
// Assemble chunks into a temp file with hash-on-write (BLAKE3 computed
|
||||
// during the same read/write loop that copies chunks into the
|
||||
// assembled file). The hash is passed downstream as `pre_computed_hash`
|
||||
// so the dedup layer never re-reads the assembled file just to compute
|
||||
// it — saves one full file-sized read pass per upload.
|
||||
let (temp_path, size, blake3_hash) = nc
|
||||
.chunked_uploads
|
||||
.assemble(&user.username, upload_id)
|
||||
.await
|
||||
@@ -238,6 +253,7 @@ async fn handle_assemble(
|
||||
// Write assembled file to storage via the upload service.
|
||||
let upload_service = &state.applications.file_upload_service;
|
||||
let file_service = &state.applications.file_retrieval_service;
|
||||
let folder_service = &state.applications.folder_service;
|
||||
|
||||
let internal_path = format!(
|
||||
"My Folder - {}/{}",
|
||||
@@ -260,7 +276,7 @@ async fn handle_assemble(
|
||||
&temp_path,
|
||||
size,
|
||||
&content_type,
|
||||
None,
|
||||
Some(blake3_hash.clone()),
|
||||
oc_mtime,
|
||||
)
|
||||
.await
|
||||
@@ -268,11 +284,12 @@ async fn handle_assemble(
|
||||
|
||||
Some(dto.etag)
|
||||
} else {
|
||||
// For new files we still need to read the temp file since create_file takes &[u8].
|
||||
let assembled = tokio::fs::read(&temp_path).await.map_err(|e| {
|
||||
AppError::internal_error(format!("Failed to read assembled file: {}", e))
|
||||
})?;
|
||||
|
||||
// New-file branch: resolve the parent folder by path and pass the
|
||||
// assembled file's path directly to `upload_file_from_path` so the
|
||||
// bytes never get read back into RAM. Previously this branch did
|
||||
// `tokio::fs::read(&temp_path)` — an extra full file-sized read
|
||||
// pass AND a peak-RAM allocation equal to the upload size, which
|
||||
// defeated the streaming model on large NC uploads.
|
||||
let (parent_sub, filename) = match dest_subpath.rsplit_once('/') {
|
||||
Some((p, n)) => (p, n),
|
||||
None => ("", dest_subpath.as_str()),
|
||||
@@ -284,8 +301,20 @@ async fn handle_assemble(
|
||||
);
|
||||
let parent_internal = parent_internal.trim_end_matches('/');
|
||||
|
||||
use crate::application::ports::folder_ports::FolderUseCase;
|
||||
let parent_folder = folder_service
|
||||
.get_folder_by_path(parent_internal)
|
||||
.await
|
||||
.map_err(|e| AppError::internal_error(format!("Parent folder lookup failed: {}", e)))?;
|
||||
|
||||
let dto = upload_service
|
||||
.create_file(parent_internal, filename, &assembled, &content_type)
|
||||
.upload_file_from_path(
|
||||
filename.to_string(),
|
||||
Some(parent_folder.id),
|
||||
content_type.to_string(),
|
||||
&temp_path,
|
||||
Some(blake3_hash),
|
||||
)
|
||||
.await
|
||||
.map_err(|e| AppError::internal_error(format!("Failed to create file: {}", e)))?;
|
||||
|
||||
|
||||
@@ -596,7 +596,14 @@ async fn handle_put(
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.and_then(|v| v.parse::<i64>().ok());
|
||||
|
||||
let max_upload = state.core.config.storage.max_upload_size;
|
||||
// ── Direct PUT cap ───────────────────────────────────────────────
|
||||
// We use `direct_put_max_bytes` (default 1 GiB), not `max_upload_size`
|
||||
// (default 10 GB). Larger files must come through the chunked upload
|
||||
// protocol (`/dav/uploads/...`) which is resumable on failure and
|
||||
// bounded per-request by `chunk_max_bytes`. Trying to stream a
|
||||
// multi-GB body through a single PUT is a footgun: a connection drop
|
||||
// at 95 % loses everything.
|
||||
let max_upload = state.core.config.storage.direct_put_max_bytes;
|
||||
|
||||
// Stream the body to a temp file + incremental hash — never buffer the
|
||||
// full upload in RAM. The old `body::to_bytes` path loaded the entire
|
||||
|
||||
@@ -6,14 +6,21 @@
|
||||
//! file (off tmpfs when [`StorageConfig::upload_temp_dir`] is configured) and
|
||||
//! BLAKE3-hashed on the fly so the dedup layer can short-circuit on a hit.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use axum::body::Body;
|
||||
use http_body_util::BodyStream;
|
||||
// The `Digest` trait (re-exported by both `md5` and `sha2` from the
|
||||
// `digest` crate) gives `Md5` and `Sha256` their `new` / `update` /
|
||||
// `finalize` methods. Importing once via `sha2` covers both —
|
||||
// otherwise every call site would need fully-qualified
|
||||
// `<md5::Md5 as md5::Digest>::…` syntax.
|
||||
use sha2::Digest as _;
|
||||
use tempfile::NamedTempFile;
|
||||
use tokio::io::AsyncWriteExt;
|
||||
use tokio_stream::StreamExt;
|
||||
|
||||
use crate::application::ports::chunked_upload_ports::ChecksumAlg;
|
||||
use crate::common::temp::new_spool_temp_file;
|
||||
use crate::interfaces::errors::AppError;
|
||||
|
||||
@@ -63,7 +70,10 @@ pub async fn spool_body_to_temp(
|
||||
drop(file);
|
||||
let _ = tokio::fs::remove_file(&temp_path).await;
|
||||
return Err(AppError::payload_too_large(format!(
|
||||
"Upload exceeds maximum size of {max_upload} bytes"
|
||||
"Upload body exceeds the direct-PUT cap ({max_upload} bytes). \
|
||||
Use the chunked-upload protocol (REST: `/api/uploads/...`, \
|
||||
NextCloud: `/remote.php/dav/uploads/...`) for files larger than this. \
|
||||
Chunked uploads are resumable on transient failure."
|
||||
)));
|
||||
}
|
||||
hasher.update(chunk);
|
||||
@@ -84,3 +94,196 @@ pub async fn spool_body_to_temp(
|
||||
size: total_bytes as u64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Result of a streamed write to a caller-supplied path.
|
||||
pub struct StreamedToPath {
|
||||
/// Total bytes written.
|
||||
pub bytes_written: u64,
|
||||
/// Lowercase hex digest, populated only when `checksum_alg=Some(_)`
|
||||
/// was passed. The algorithm is identified by [`StreamedToPath::alg`].
|
||||
pub checksum_hex: Option<String>,
|
||||
/// Algorithm used to compute `checksum_hex`. Echoed back so the
|
||||
/// caller can include it in audit logs or response headers.
|
||||
pub alg: Option<ChecksumAlg>,
|
||||
}
|
||||
|
||||
/// Stream an HTTP request body directly to a known destination file,
|
||||
/// enforcing `max_bytes` as a hard size limit.
|
||||
///
|
||||
/// Used by the chunked-upload PUT handlers — each chunk has a
|
||||
/// deterministic on-disk path (`NextcloudChunkedUploadService::safe_chunk_path`
|
||||
/// for the NC surface, `ChunkedUploadService::prepare_chunk` for the
|
||||
/// REST surface), so there's no spool/move dance. Peak heap is ~one
|
||||
/// HTTP frame regardless of chunk size or `max_bytes`.
|
||||
///
|
||||
/// `checksum_alg` is the optional client-requested integrity check
|
||||
/// (default `md5` per the legacy `Content-MD5` contract; `blake3`
|
||||
/// available for forward-compat). When `Some`, the hash is computed
|
||||
/// incrementally during streaming — no extra disk read for verification.
|
||||
///
|
||||
/// On size overflow the partial file is removed before the function
|
||||
/// returns, so a client retry against the same chunk name starts from
|
||||
/// a clean slate. On any other I/O error the partial file is also
|
||||
/// removed and the error surfaces — callers can assume the path is
|
||||
/// either fully written or absent.
|
||||
pub async fn stream_body_to_path(
|
||||
body: Body,
|
||||
path: &Path,
|
||||
max_bytes: usize,
|
||||
checksum_alg: Option<ChecksumAlg>,
|
||||
) -> Result<StreamedToPath, AppError> {
|
||||
let mut file = tokio::fs::File::create(path)
|
||||
.await
|
||||
.map_err(|e| AppError::internal_error(format!("Failed to open chunk file: {e}")))?;
|
||||
|
||||
let mut total_bytes: usize = 0;
|
||||
let mut stream = BodyStream::new(body);
|
||||
let mut hasher = checksum_alg.map(IncrementalHasher::new);
|
||||
|
||||
while let Some(frame_result) = stream.next().await {
|
||||
let frame = match frame_result {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
drop(file);
|
||||
let _ = tokio::fs::remove_file(path).await;
|
||||
return Err(AppError::bad_request(format!(
|
||||
"Failed to read request body: {e}"
|
||||
)));
|
||||
}
|
||||
};
|
||||
if let Some(chunk) = frame.data_ref() {
|
||||
total_bytes += chunk.len();
|
||||
if total_bytes > max_bytes {
|
||||
drop(file);
|
||||
let _ = tokio::fs::remove_file(path).await;
|
||||
return Err(AppError::payload_too_large(format!(
|
||||
"Chunk exceeds maximum size of {max_bytes} bytes"
|
||||
)));
|
||||
}
|
||||
if let Some(h) = hasher.as_mut() {
|
||||
h.update(chunk);
|
||||
}
|
||||
if let Err(e) = file.write_all(chunk).await {
|
||||
drop(file);
|
||||
let _ = tokio::fs::remove_file(path).await;
|
||||
return Err(AppError::internal_error(format!(
|
||||
"Failed to write chunk: {e}"
|
||||
)));
|
||||
}
|
||||
}
|
||||
}
|
||||
file.flush()
|
||||
.await
|
||||
.map_err(|e| AppError::internal_error(format!("Failed to flush chunk file: {e}")))?;
|
||||
drop(file);
|
||||
|
||||
Ok(StreamedToPath {
|
||||
bytes_written: total_bytes as u64,
|
||||
checksum_hex: hasher.map(IncrementalHasher::finalize_hex),
|
||||
alg: checksum_alg,
|
||||
})
|
||||
}
|
||||
|
||||
/// Algorithm-agnostic incremental hasher used by [`stream_body_to_path`].
|
||||
/// Per-frame `update` is sub-millisecond for all three algorithms at the
|
||||
/// 64 KB frame sizes axum's body stream produces, so we don't need
|
||||
/// `spawn_blocking` (which the old buffered path used because it hashed
|
||||
/// the full multi-MB chunk in one shot).
|
||||
enum IncrementalHasher {
|
||||
Md5(md5::Md5),
|
||||
Sha256(sha2::Sha256),
|
||||
// Boxing — blake3::Hasher is ~1.7 KB on the stack while md5::Md5
|
||||
// (~100 bytes) and sha2::Sha256 (~100 bytes) are tiny; boxing the
|
||||
// outlier keeps the enum size proportional to the common case
|
||||
// rather than the worst case.
|
||||
Blake3(Box<blake3::Hasher>),
|
||||
}
|
||||
|
||||
impl IncrementalHasher {
|
||||
fn new(alg: ChecksumAlg) -> Self {
|
||||
match alg {
|
||||
ChecksumAlg::Md5 => Self::Md5(md5::Md5::new()),
|
||||
ChecksumAlg::Sha256 => Self::Sha256(sha2::Sha256::new()),
|
||||
ChecksumAlg::Blake3 => Self::Blake3(Box::new(blake3::Hasher::new())),
|
||||
}
|
||||
}
|
||||
|
||||
fn update(&mut self, bytes: &[u8]) {
|
||||
match self {
|
||||
Self::Md5(h) => h.update(bytes),
|
||||
Self::Sha256(h) => h.update(bytes),
|
||||
Self::Blake3(h) => {
|
||||
h.update(bytes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn finalize_hex(self) -> String {
|
||||
match self {
|
||||
Self::Md5(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
|
||||
Self::Sha256(h) => h.finalize().iter().map(|b| format!("{b:02x}")).collect(),
|
||||
Self::Blake3(h) => h.finalize().to_hex().to_string(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use bytes::Bytes;
|
||||
|
||||
#[tokio::test]
|
||||
async fn stream_body_to_path_caps_oversized() {
|
||||
let temp_dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = temp_dir.path().join("chunk");
|
||||
|
||||
// 5 MiB body, 4 MiB cap → must reject.
|
||||
let body = Body::from(Bytes::from(vec![0u8; 5 * 1024 * 1024]));
|
||||
let result = stream_body_to_path(body, &path, 4 * 1024 * 1024, None).await;
|
||||
assert!(
|
||||
result.is_err(),
|
||||
"expected PayloadTooLarge, got Ok(bytes_written={})",
|
||||
result.ok().map(|r| r.bytes_written).unwrap_or(0)
|
||||
);
|
||||
// Partial file must be removed on rejection.
|
||||
assert!(
|
||||
!path.exists(),
|
||||
"rejected chunk file should be removed, but {} still exists",
|
||||
path.display()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn stream_body_to_path_accepts_under_cap() {
|
||||
let temp_dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = temp_dir.path().join("chunk");
|
||||
|
||||
let body = Body::from(Bytes::from(vec![1u8; 1024 * 1024])); // 1 MiB
|
||||
let result = stream_body_to_path(body, &path, 4 * 1024 * 1024, None).await;
|
||||
let outcome = result.expect("should succeed");
|
||||
assert_eq!(outcome.bytes_written, 1024 * 1024);
|
||||
assert!(outcome.checksum_hex.is_none(), "no alg requested → no hash");
|
||||
assert!(path.exists());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn stream_body_to_path_caps_at_exact_boundary() {
|
||||
// Edge case: body exactly equal to cap should succeed; cap+1 must fail.
|
||||
let temp_dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = temp_dir.path().join("chunk");
|
||||
|
||||
let body = Body::from(Bytes::from(vec![1u8; 100]));
|
||||
let outcome = stream_body_to_path(body, &path, 100, None)
|
||||
.await
|
||||
.expect("100 bytes at 100-byte cap should succeed");
|
||||
assert_eq!(outcome.bytes_written, 100);
|
||||
|
||||
let path2 = temp_dir.path().join("chunk2");
|
||||
let body = Body::from(Bytes::from(vec![1u8; 101]));
|
||||
assert!(
|
||||
stream_body_to_path(body, &path2, 100, None).await.is_err(),
|
||||
"101 bytes at 100-byte cap must reject"
|
||||
);
|
||||
assert!(!path2.exists());
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user