diff --git a/.github/workflows/helm-install-test.yml b/.github/workflows/helm-install-test.yml index d0571d6..cadefc9 100644 --- a/.github/workflows/helm-install-test.yml +++ b/.github/workflows/helm-install-test.yml @@ -49,7 +49,7 @@ jobs: spec: containers: - name: minio - image: minio/minio:latest + image: pgsty/minio:latest args: ["server", "/data"] env: - name: MINIO_ROOT_USER diff --git a/.gitignore b/.gitignore index 6a60685..7fceca6 100644 --- a/.gitignore +++ b/.gitignore @@ -53,3 +53,6 @@ chart/charts/*.tgz # Database backup testing db-backup-tests/ + +# Local-only cluster e2e tests +e2e/ diff --git a/Makefile b/Makefile index 4f85617..edebd16 100644 --- a/Makefile +++ b/Makefile @@ -26,10 +26,13 @@ verify-passthrough: # Integration shards for parallel CI (make test-integration-shard SHARD=memory_usage) INTEGRATION_memory_usage_TESTS = tests/integration/test_memory_usage.py +INTEGRATION_memory_usage_PYTEST_OPTS = -n0 +INTEGRATION_memory_usage_COMPOSE_OPTS = -f tests/docker-compose.oom.yml INTEGRATION_memory_leak_TESTS = tests/integration/test_memory_leak.py INTEGRATION_memory_copy_TESTS = tests/integration/test_copy_memory_governor.py tests/integration/test_copy_per_part_metrics.py tests/integration/test_upload_part_copy_passthrough_e2e.py INTEGRATION_memory_copy_PYTEST_OPTS = -n0 INTEGRATION_core_TESTS = \ + tests/integration/test_generation_roundtrip.py \ tests/integration/test_integration.py \ tests/integration/test_handlers.py \ tests/integration/test_concurrent_operations.py \ @@ -72,15 +75,15 @@ test-integration-shard: ifndef SHARD $(error SHARD is required, e.g. make test-integration-shard SHARD=memory) endif - @docker compose -f tests/docker-compose.yml down 2>/dev/null || true - @docker compose -f tests/docker-compose.yml up -d + @docker compose -f tests/docker-compose.yml $(INTEGRATION_$(SHARD)_COMPOSE_OPTS) down 2>/dev/null || true + @docker compose -f tests/docker-compose.yml $(INTEGRATION_$(SHARD)_COMPOSE_OPTS) up -d @sleep 3 @AWS_ACCESS_KEY_ID=minioadmin AWS_SECRET_ACCESS_KEY=minioadmin \ uv run pytest -m "e2e" -v \ $(if $(INTEGRATION_$(SHARD)_PYTEST_OPTS),$(INTEGRATION_$(SHARD)_PYTEST_OPTS),-n auto --dist loadgroup) \ $(INTEGRATION_$(SHARD)_TESTS); \ EXIT_CODE=$$?; \ - docker compose -f tests/docker-compose.yml down; \ + docker compose -f tests/docker-compose.yml $(INTEGRATION_$(SHARD)_COMPOSE_OPTS) down; \ exit $$EXIT_CODE # Run all tests with containers (unit + integration) diff --git a/README.md b/README.md index 4b9361b..6849f64 100644 --- a/README.md +++ b/README.md @@ -21,6 +21,10 @@ ## Overview +For the generation-format upgrade, migration requirements and measured performance, see +[Generation-bound writes and streaming changes](docs/GENERATION_FORMAT.md). + + S3's server-side encryption is great, but your cloud provider holds the keys. S3Proxy sits between your app and S3, encrypting everything **before** it leaves your infrastructure. ``` diff --git a/docs/CODE_REVIEW.md b/docs/CODE_REVIEW.md new file mode 100644 index 0000000..8dc7f22 --- /dev/null +++ b/docs/CODE_REVIEW.md @@ -0,0 +1,223 @@ +# S3Proxy Python: correctness, performance, and maintainability review + +Reviewed on September 5, 2026. Repository: `s3proxy-python`. Reviewed commit: `27b77cf92b5956c9ae671d581403294205dfed77`. + +The working tree was clean when reviewed. No application code was changed. This English report translates the earlier review and expands the proposed fixes and validation criteria. + +The largest immediate performance opportunities are removing forced garbage collection from request cleanup, reusing S3 clients and connection pools, and reducing backend round trips. Several correctness issues should be addressed first: multipart operations can report false success, overwrite data using conflicting part numbers, reuse cryptographic nonces, and leave objects unreadable after an overwrite. + +## Scope and evidence + +The review focuses on the Python backend: GET, HEAD, PUT, multipart operations, metadata persistence, request signatures, and memory management. It is not a complete security audit of the dashboard, Helm chart, or deployment environment. + +Reproductions used the repository's mock S3 client, local in-memory upload state, and synthetic credentials. No external storage service was contacted. Findings distinguish reproduced behavior from consequences inferred from code. Performance measurements are local microbenchmarks, not production throughput measurements. + +Source references below are relative to the repository root and refer to the reviewed commit. P1 means high priority because of security, data integrity, or false-success behavior; P2 means a correctness or resource-management issue that should follow. + +## Priority findings and proposed solutions + +### 1. P1 — AES-GCM nonce reuse when a multipart part is replaced + +**Source:** `s3proxy/crypto.py:367`, `s3proxy/handlers/multipart/upload_part.py:535`, `s3proxy/state/manager.py:266`. + +The nonce is derived only from the upload ID, internal part number, and, for framed encryption, frame index. Re-uploading a client part reuses its internal part numbers. If the content changes, the same DEK and nonce encrypt different plaintexts. + +**Reproduced:** upload `AAAA`, then `BBBB`, as part 1 of the same upload. Both encryptions use the same nonce, and XOR of the ciphertext payloads equals XOR of the plaintexts. + +Replacing an existing part is valid [S3 UploadPart behavior](https://docs.aws.amazon.com/AmazonS3/latest/API/API_UploadPart.html). Reusing a nonce under the same key violates [AES-GCM's security requirements](https://cryptography.io/en/latest/hazmat/primitives/aead/#cryptography.hazmat.primitives.ciphers.aead.AESGCM). + +**Proposed solution:** + +- Give every new encryption a fresh random nonce with an appropriate collision budget, or derive it from a unique encryption-attempt identity in addition to the existing fields. +- Distinguish a new encryption attempt from a network retry. A network retry can safely resend the exact same previously encrypted bytes. +- The nonce is already embedded in the ciphertext. Check legacy readers and any deterministic nonce validation before changing generation behavior. + +**Acceptance criteria:** replacing a part with different bytes never reuses a nonce/key pair; transport retries resend identical ciphertext; existing stored objects still decrypt. + +### 2. P1 — CompleteMultipartUpload can report success for the wrong upload + +**Source:** `s3proxy/handlers/multipart/lifecycle.py:169` and `:274`. + +`_try_idempotent_complete_response()` checks that the existing object's size matches its existing metadata sidecar. It does not establish that either belongs to the requested `upload_id`. This check happens before reading the requested upload state or validating the client's part list. + +**Reproduced:** after creating a multipart object, submit Complete with `uploadId=never-created` and invalid XML. The handler returns HTTP 200. Consequently, a new upload targeting an existing key can appear complete while the previous object remains in place. + +**Proposed solution:** persist a completion record tied to the exact upload ID and committed object generation, including the resulting client ETag. Accept an idempotent retry only when that record proves the requested upload completed. Do not infer completion from object size. + +**Acceptance criteria:** an unknown upload ID does not succeed because an older object exists; a retry of the actual completed upload returns its recorded result; a new upload to the same key follows the normal completion path. + +### 3. P1 — Replacing a multipart object with a small PUT leaves stale metadata + +**Source:** `s3proxy/handlers/objects/put.py:174`, `s3proxy/handlers/objects/get.py:67`. + +A buffered PUT replaces the object but leaves its old multipart sidecar. GET and HEAD prioritize that sidecar over the new object's encryption metadata. + +**Reproduced:** write `old-content` through streaming PUT, then overwrite it with `new` through a small signed PUT. HEAD still reports 11 bytes instead of 3, and GET fails using the old multipart metadata against the new ciphertext. + +**Proposed solution:** identify the storage format and generation from the current object's metadata. Only load a sidecar belonging to that generation. Clean up obsolete sidecars separately. Simply deleting a shared sidecar after every PUT is insufficient because concurrent writers can delete each other's metadata. + +**Acceptance criteria:** multipart-to-buffered and buffered-to-multipart overwrites return the new bytes and size; concurrent overwrites never combine one generation's ciphertext with another's metadata. Coordinate this work with finding 7. + +### 4. P1 — Small signed PUT requests accept a modified request body + +**Source:** `s3proxy/request_handler.py:278`, `s3proxy/client/verifier.py:376`, `s3proxy/handlers/objects/put.py:133`. + +Signature verification uses the supplied `x-amz-content-sha256` value, but the buffered PUT path never compares it with the actual body's hash. The larger streaming path performs a separate check. + +**Reproduced:** generate a valid SigV4 signature for `original`, preserve the signature and hash header, and replace the body with `changed!`. Header verification succeeds and the PUT handler returns HTTP 200 for the modified data. + +**Proposed solution:** centralize payload validation and invoke it for every signed write path. For buffered PUT, compute the actual SHA-256 and reject a mismatch before writing to S3. Preserve the intentional semantics of `UNSIGNED-PAYLOAD`; validate streaming signature formats through their dedicated verification path. + +**Acceptance criteria:** a modified signed body is rejected without replacing an existing object. Test just below, at, and above the buffering threshold. This finding does not imply arbitrary signature forgery; it shows that a valid signature does not bind the buffered body as intended. + +### 5. P1 — Multipart parts are published before their hash or signature is accepted + +**Source:** `s3proxy/handlers/multipart/upload_part.py:204`, `:471`, and `:603`. + +Both the backend upload and `add_part()` occur before late signature validation. A failed check returns an error without restoring the modified part or state. + +**Reproduced:** UploadPart rejects a wrong SHA-256, but the rejected body's MD5 is already stored in the upload's part state. The backend write can also replace a previously valid part. + +**Proposed solution:** make validation the boundary before publication. For buffered parts, validate before uploading. For streamed parts, use isolated staging or bounded disk spooling so unverified bytes cannot overwrite an accepted part. Publish the client-part mapping only after verification succeeds, and clean up failed attempts. + +Moving `add_part()` alone is insufficient: the backend part may already have been overwritten. Staging must also respect S3's part numbering, ordering, and size constraints. + +**Acceptance criteria:** a rejected replacement leaves both the previous accepted part state and its backend bytes unchanged. Inject hash failure, signature failure, cancellation, and disconnect during replacement. + +### 6. P1 — Switching part-number allocation strategies creates collisions + +**Source:** `s3proxy/state/manager.py:251`, `s3proxy/crypto.py:173`. + +The upload starts with dense numbering: client part 2 maps to internal part 2. When a client part requires multiple internal parts, the upload switches to sparse numbering without relocating or protecting previous allocations. + +**Reproduced:** allocate one internal part for client part 2: internal number 2. Then allocate two internals for client part 1: internal numbers 1–2. Both allocations include internal number 2. Concurrency is not required; out-of-order parts suffice. + +**Proposed solution:** maintain an explicit, atomically updated mapping from client part and attempt to backend allocations. Do not change the meaning of existing allocations when workload shape changes. Design final assembly to preserve client-part order and S3's maximum part count; a monotonic allocator by itself does not solve final ordering. + +**Acceptance criteria:** property-based or randomized tests cover out-of-order uploads, changing part sizes, replacements, and simultaneous allocations. No live allocations overlap, and completed plaintext is ordered correctly. + +### 7. P1 — Ciphertext and required metadata are published separately + +**Source:** `s3proxy/handlers/objects/put.py:310`, `s3proxy/state/metadata.py:265`. + +Streaming PUT completes the object before saving its sidecar. This path does not first persist the new DEK in durable upload state. If the sidecar write fails or the process dies between these operations, ciphertext is already visible and the information needed to decrypt it can be lost. This failure window is identified from code; a process-crash scenario was not executed. + +Separately, `load_multipart_metadata()` interprets all exceptions as missing metadata, including service failures, authorization failures, and corrupt compressed data. + +**Reproduced:** two simulated HTTP 503 backend errors result in `None`. Streaming PUT does not mark the main object with encryption metadata, so a subsequent GET can select unencrypted passthrough when the sidecar cannot be loaded. Other multipart formats can select an incorrect decryption path instead. + +**Proposed solution:** + +- Generate a stable object-generation identity before upload. Store the format, generation, and necessary wrapped-key information durably before publishing ciphertext. +- Bind immutable sidecars to that generation and make readers resolve only the referenced generation. +- Design a recoverable commit sequence, including crash recovery and obsolete-generation cleanup. Reordering two independent writes alone does not make them atomic. +- Return `None` only for confirmed metadata absence. Propagate service, permission, and decoding errors instead of treating them as plaintext-object detection. + +**Acceptance criteria:** inject failure at each commit step, then restart. Every visible encrypted generation must remain decryptable or produce an explicit recoverable error; never silently serve ciphertext as plaintext. A transient metadata error must not trigger format fallback. + +### 8. P1 — aws-chunked decoding accepts truncated and unverified input + +**Source:** `s3proxy/streaming/chunked.py:88`, `s3proxy/handlers/objects/put.py:90`. + +The decoder does not require a terminal zero-size chunk. It does not validate `chunk-signature`; the upload path also disables ordinary payload hash checking for `STREAMING-*`. The decoder skips the two trailing bytes after chunk data without verifying that they are CRLF. + +**Reproduced:** a complete `abc` chunk with an invalid chunk signature, followed by an incomplete chunk, produces `abc` without a decoder error. + +**Proposed solution:** implement an explicit parser state machine with header, payload, CRLF, terminal chunk, and supported trailer states. Require a valid end state at EOF. Validate the signature chain or checksum/trailer requirements for each supported encoding. Explicitly reject signed streaming variants that are not correctly verified. + +**Acceptance criteria:** test truncation at every framing boundary, invalid CRLF, invalid signatures, missing terminal chunks, and supported trailer variants. Failed validation must not publish an object or accepted part. + +### 9. P2 — ETags differ across PUT, HEAD, GET, and LIST + +**Source:** `s3proxy/handlers/objects/misc.py:72`, `s3proxy/handlers/objects/get.py:51`, `s3proxy/state/attr_cache.py:21`. + +For multipart objects, HEAD and LIST use MD5 of the plaintext size. GET uses the backend ETag. Streaming PUT returns MD5 of the plaintext content. HEAD evaluates conditional headers before replacing its effective ETag with the synthetic response ETag. + +**Reproduced:** HEAD and GET return different ETags for the same multipart object. Different contents of equal length necessarily share the synthetic HEAD/LIST ETag. + +**Proposed solution:** persist one client-facing ETag in generation-bound metadata and use it consistently for responses and conditional checks. It must distinguish content or object generations rather than only lengths. Define compatible behavior for existing objects whose metadata lacks the field. + +**Acceptance criteria:** PUT/Complete, HEAD, GET, and LIST agree. Conditional requests using the returned ETag behave consistently, including after a same-size content replacement. + +### 10. P2 — Nested memory reservations can block their own request + +**Source:** `s3proxy/handlers/objects/get.py:154`, `s3proxy/concurrency.py:105`. + +GET reserves 8 MiB at admission and can subsequently request additional memory. If that extra request is clamped to the entire budget, admission requires no memory to be reserved — while the same GET still holds its baseline reservation. + +**Reproduced at limiter level:** a 64 MiB budget, an 8 MiB baseline reservation, and an additional reservation calculated for a 40 MiB object result in SlowDown with the timeout set to zero. Production can wait through the full backpressure timeout. Large single-envelope objects are a legacy/compatibility case; new buffered PUT objects stay below the streaming threshold. Multiple smaller GETs can also hold baseline reservations while waiting for each other to release memory. + +**Proposed solution:** reserve the complete working set atomically once the object format and size are known, or implement a reservation transition that cannot leave multiple waiters holding mutually blocking partial allocations. Do not clamp a whole-buffer requirement while pretending actual memory use fits the budget. Use bounded streaming or spooling where feasible, preserving authenticated-decryption semantics. + +**Acceptance criteria:** one large compatible GET and several simultaneous smaller GETs either progress within a bounded budget or fail promptly and predictably. Cancellation releases reservations exactly once. + +## Performance improvements + +| Priority | Change | Evidence and proposed implementation | +|---|---|---| +| 1 | Remove forced full GC from every reservation release | `s3proxy/concurrency.py:199` runs `gc.collect(0)`, `(1)`, and `(2)` synchronously on the event loop. A 25-iteration local measurement had a median of approximately **30.6 ms**, versus **0.03 ms** with those calls mocked out. Retain normal automatic GC and evaluate whether any exceptional reclamation policy is needed under sustained load. Linux also calls `malloc_trim`, which was not measured on macOS. | +| 2 | Reuse S3 clients and connection pools | `s3proxy/client/s3.py:65` creates a fresh SDK client per context and closes it afterward. A shared Session reuses model loading, but does not preserve these clients' connection pools across their lifetimes. Own a bounded client registry in application lifespan, isolated by credentials and endpoint/configuration. Close clients at shutdown or safe eviction, after active streams finish. | +| 3 | Eliminate redundant metadata requests | A small GET without a sidecar performs HEAD, two sidecar probes, and data GET: **four backend requests**. Multipart performs HEAD, sidecar GET, and a second HEAD before data retrieval. Pass the first HEAD result through the read path; skip sidecar probes only when an unambiguous current-format marker permits it. Preserve safe legacy detection. | +| 4 | Fetch contiguous frames in fewer Range GETs | `s3proxy/handlers/objects/get.py:479` performs one GET per frame, normally 8 MiB. A 1 GiB object with full frames needs approximately 128 data GETs. Fetch a larger contiguous ciphertext range while reading, authenticating, and emitting one frame at a time. Preserve bounded buffering, backpressure, and frame-aligned recovery after network errors. | +| 5 | Reduce duplicate listing metadata work | `s3proxy/handlers/buckets.py:211` already uses bounded parallelism and an attribute cache. A cold listing of 1,000 multipart objects still requires roughly 1,000 HEADs and 1,000 sidecar GETs in addition to LIST. Coalesce concurrent lookups for the same generation and evaluate a generation-keyed metadata cache or index. Increasing concurrency alone increases backend pressure and memory consumption. | + +The GC measurement is a microbenchmark of reservation release, not a claim of a similar improvement in end-to-end throughput. Measure changes individually against a test backend using p50/p95/p99 latency, time to first byte, MiB/s, backend requests per operation, event-loop lag, peak RSS, and error rate. + +The 8 MiB GET reservation also understates the working set during prefetch: the current plaintext frame can remain alive while the next frame is downloaded and decrypted. Model the whole lifecycle rather than reserving only one frame's nominal size. Do not raise general concurrency before validating that model. + +## Reducing complexity + +### A shared upload pipeline + +Use common stages for PutObject and UploadPart: + +`read/decode → hash → encrypt frames → stage → verify → publish` + +The operations can share streaming, hashing, encryption, and cleanup while retaining operation-specific publication rules. This directly addresses inconsistent hash checking, buffering assumptions, and failure cleanup. For small buffered payloads, verification can occur before encryption and staging. + +### A common object descriptor + +Resolve an object into one typed descriptor containing format version, generation identity, plaintext size, client-facing ETag, encryption metadata, and frame/part index. GET, HEAD, LIST, and COPY should consume the same interpretation. This removes duplicated metadata resolution and ETag logic. + +### Explicit multipart state transitions + +Separate receiving, verified, published, and completed states. Model client part numbers separately from backend part allocations, and make attempt identity explicit. State transitions should specify what is durable and what a retry may safely repeat. + +### Centralized resource ownership + +Make request/stream lifetime own the memory reservation, backend response body, client reference, and final metrics. Prefer small explicit services over additional cross-dependent mixins. Cleanup should follow resource lifetime, including disconnects and exceptions after response headers have been sent. + +### Narrow exception handling + +Replace broad `except Exception: return None/pass` where the intended condition is “not found.” Keep authorization errors, unavailable backends, corrupt metadata, and actual absence distinct. + +Retain the current frame format during initial optimization. `FRAME_PLAINTEXT_SIZE` is part of existing read compatibility; changing it without a versioned format can make stored objects unreadable. + +## Additional observations from static inspection + +These observations were not included in the executed fault reproductions: + +| Observation | Proposed follow-up | +|---|---| +| `s3proxy/handlers/objects/put.py:57` checks If-None-Match through HEAD followed by PUT. Two writers can both pass, and non-NotFound HEAD exceptions are swallowed. | Use supported atomic backend preconditions at publication. Define consistent precondition behavior across buffered and multipart writes; test concurrent writers. | +| `s3proxy/handlers/objects/put.py:174` writes internal metadata but does not forward user `x-amz-meta-*` fields, whereas multipart initialization does. | Centralize user-metadata extraction and reserved-key handling, then test parity across upload paths. | +| `s3proxy/handlers/objects/get.py:38` exits the client context before an unencrypted StreamingResponse is consumed. The mock client does not close connections. | Verify a large, slow unencrypted download against real aiohttp/aiobotocore. Keep the backend client alive until stream cleanup. | +| `s3proxy/request_handler.py:161` acquires memory outside the main `try/finally`; rejection can leave the in-flight metric incremented. Streaming requests are recorded as complete before their bodies finish. | Include admission failures in cleanup and finalize streaming duration/status metrics at stream completion, with explicit accounting for post-header failures. | + +## Suggested implementation sequence + +1. Add regression coverage for findings 1–8. Correct nonce generation, upload identity, payload validation, and allocation collisions before expanding concurrency. +2. Design generation-bound metadata and recoverable publication together. This addresses stale sidecars, lost decryption information, incorrect format fallback, and completion identity without independent patches that conflict. +3. Remove forced per-request GC and redundant HEAD calls. Benchmark latency and RSS separately after each change. +4. Introduce lifetime-correct client pooling and contiguous frame reads. Test real backend connections, retries, disconnects, and memory limits. +5. Consolidate the shared pipeline, object descriptor, and memory ownership in small changes protected by compatibility tests. + +## Validation performed + +- Existing unit suite: **642 passed, 2 warnings in 293.07 seconds**. +- Command: `PYTHONDONTWRITEBYTECODE=1 .venv/bin/python -m pytest tests/unit -q -p no:cacheprovider --disable-warnings`. +- Lint: `.venv/bin/python -m ruff check s3proxy --no-cache` passed. +- The contents of the two warnings were not reviewed in that run. +- No integration tests against real S3/Redis and no production load tests were run. +- The separate reproductions expose cases missing from the passing unit suite. They are diagnostic reproductions, not yet committed regression tests. + diff --git a/docs/GENERATION_FORMAT.md b/docs/GENERATION_FORMAT.md new file mode 100644 index 0000000..f8c3805 --- /dev/null +++ b/docs/GENERATION_FORMAT.md @@ -0,0 +1,63 @@ +# Generation-bound writes and streaming changes + +This change implements the correctness findings and performance work from [the code review](CODE_REVIEW.md). It changes the format of newly written objects. Read this document before upgrading a running cluster. + +## Publication and retries + +New buffered PUTs carry `s3proxy-format=single-v3`. Streaming PUTs, multipart uploads and streaming copies carry `s3proxy-format=multipart-v3` and an immutable generation pointer. Required manifests are stored at `.s3proxy-internal/generations/.meta` **before** publishing ciphertext. The generation is derived from the initial wrapped random upload key; it remains the same if the first copy selects a source key before any writer starts. Readers resolve the pointer from the current HEAD response. A missing, corrupt or unavailable required manifest is an error, never a signal to return ciphertext as plaintext. + +Each UploadPart attempt writes a private staging object at `.s3proxy-internal/attempts//`. Hash/signature validation and completion of that staging object precede publication in upload state. A rejected replacement cannot modify an accepted attempt. When a state write has an uncertain outcome, its completed staging object is retained because the state write may already have succeeded. + +Complete reads the accepted client-part mapping, validates the client's ordered ETag list, and assembles the selected staging objects in client order with server-side copies. State remains available until completion succeeds. The manifest records the originating bucket, key, upload ID and client ETag, so size alone cannot prove that a retry succeeded. Redis completion locks renew their leases; loss of a lease interrupts the operation. In-memory lock entries disappear after the last waiter finishes. + +The first writer atomically freezes the upload's DEK. A whole-object UploadPartCopy may select its source DEK at that point and snapshot the ciphertext using a source ETag precondition. Subsequent copies using a different key, partial ranges and ordinary uploads encrypt with the already frozen key. No active writer can change that key. New encryption uses fresh random nonces; transport retries reuse the already sealed bytes. Full CopyObject retains native copying and publishes the corresponding manifest first. + +## Compatibility and deployment + +- Existing single-seal and framed objects remain readable. The existing 8 MiB encryption frame boundary is unchanged. Legacy multipart ETags retain their historical fallback where no stored client ETag exists. +- Upgrade all readers before allowing v3 writes. Old releases do not understand the new generation pointers. Do not run old and new writers against the same keys. Use a maintenance window to drain existing uploads and switch the fleet together. +- Drain or restart legacy in-flight multipart uploads. UploadPart and UploadPartCopy reject the old active layout. Do not rely on completing legacy active state after the upgrade. +- Use persistent Redis for uploads that must survive process restarts or move between pods. Configure its TTL above the maximum permitted upload duration. Loss of v3 upload state fails closed: accepted attempts are not reconstructed by guessing from backend parts. Committed objects do not depend on Redis. +- Keep the manifest namespace available to the same backend credentials that operate on objects. The existing internal-prefix filter hides it from proxy listings. Backend permissions now also need multipart creation/copy, listing and deletion for the attempt prefix. +- `If-None-Match: *` is passed to the backend publication operation, including CompleteMultipartUpload for streaming PUT. Verify support on the target S3-compatible service. Failed preconditions return 412. +- Supported streaming modes are `STREAMING-AWS4-HMAC-SHA256-PAYLOAD` and `STREAMING-UNSIGNED-PAYLOAD-TRAILER`. For the signed mode, the entire signature chain, terminal chunk and decoded length are checked. For the unsigned trailer mode (the default for current AWS SDKs over HTTPS), the declared `x-amz-checksum-crc32`, `-sha1` or `-sha256` trailer is verified against the decoded body before anything is published. CRC32C/CRC64NVME trailers and `STREAMING-AWS4-HMAC-SHA256-PAYLOAD-TRAILER` are explicitly rejected. +- Control request bodies are bounded at 8 MiB. Client parts are bounded at 5 GiB; assembled uploads remain subject to backend object-size and 10,000-part limits. Manifest JSON is bounded at 10 MiB and rejected before publication when it exceeds the reader limit. + +## Storage cleanup + +Successful completion and abort attempt to delete all completed staging objects for that upload generation. Replaced attempts remain immutable until terminal cleanup to avoid racing a concurrent completion snapshot. Configure a backend lifecycle rule for `.s3proxy-internal/attempts/` to expire completed orphan attempts and abort incomplete staging MPUs after a period **longer than the maximum supported upload duration and retry window**. For example, seven days is appropriate only if the deployment prohibits uploads lasting that long. This PR does not install or modify bucket lifecycle policies. + +Crashes, cancelled writes and failed cleanup can leave attempts or unpublished manifests. Generation manifests are intentionally retained: native copies and backend object versions may reference them. Do not apply an age-only deletion policy to `.s3proxy-internal/generations/`. Reclaiming those manifests requires checking all retained object versions, copy references and active uploads. Automatic generation garbage collection is outside this change; retaining metadata is the safe default. DeleteBucket through the proxy removes the internal namespace only when no client-visible object and no in-progress multipart upload remains, then retries the backend delete. + +## Read path, resources and performance + +GET, HEAD, LIST and CopyObject share an object descriptor for plaintext size, ETag and manifest interpretation. GET reuses the initial HEAD. Consecutive frames share backend range requests of up to 64 MiB while buffering and authenticating one frame at a time. Retry resumes at the first unpublished frame. Output is emitted in bounded chunks. Legacy large single seals authenticate to a temporary spool before any plaintext is exposed. Spooled bytes are AES-CTR encrypted under a random per-read key held only in memory, so plaintext never reaches local disk; configure sufficient local temporary disk for concurrent legacy reads. + +The frame reader allocates exact-size buffers instead of repeatedly growing bytearrays. This was necessary to prevent allocator fragmentation observed during repeated real HTTP reads. GC and allocator trimming no longer run synchronously on each memory release. GET reserves 32 MiB for the working set, and copy operations own their reservation instead of nesting it inside another reservation. + +An application-owned pool retains up to 32 credential-isolated S3 clients, evicts idle entries, and waits for active leases on shutdown. Response ownership keeps bodies, clients and reservations alive until streaming finishes, including failure before the body starts. Metrics finish with the stream; dashboard metric failures cannot prevent reservation release. Concurrent listing lookups for the same object/ETag coalesce and recheck the existing bounded attribute cache. Cold listings still need metadata reads for previously unseen objects; this is not a bucket-wide metadata index. + +Staging adds temporary storage and server-side copy work to ordinary multipart writes. Full compatible copies avoid re-encryption; partial copies and copies after an incompatible key selection use bounded re-encryption. The change prioritizes verified publication over overwriting unverified backend parts. It does not promise that every write workload gets faster. + +## Measured results + +A local macOS/Python 3.14.7 test compared `d45732d` with this implementation against isolated Docker MinIO. Each process wrote a 32 MiB object, performed one warm-up GET, then ten sequential GETs. A fresh process was used for each version. Requests used the same client configuration and verified every returned byte. + +| Metric | Before | After | +| --- | ---: | ---: | +| Median GET latency | 317.12 ms | 289.26 ms | +| Maximum observed latency | 332.85 ms | 331.64 ms | +| Median transfer rate | 100.91 MiB/s | 110.63 MiB/s | +| Highest sampled process RSS | 179.23 MiB | 175.06 MiB | + +This is about 8.8% lower median latency in this small local sample, not a production forecast. [Raw results](read-benchmark.json) are included. A separate 50-request run stayed below 176 MiB sampled RSS. Sampling occurred after requests and can miss transient peaks; the governor budget is not a hard process RSS limit. No p99 claim or AWS/Ceph throughput claim is made. + +Regression tests separately demonstrate eight consecutive frames using one backend GET, frame-aligned recovery without duplicate plaintext, AWS-published streaming-signature vectors, corrupt legacy seals, cancelled publication, failed completion retry, immutable key selection, coalesced listings and resource cleanup on failed response headers. + +## Validation and remaining rollout checks + +Local validation: the CI unit selection passed 779 tests, including 110 mock integration tests (two upstream deprecation warnings). The separate full unit-directory run passed 670 tests, including its slow case. Real-backend validation passed 27 compatibility tests, 11 copy/concurrency tests and nine native-copy tests covering 1,280 MiB objects and concurrent copies. Ruff lint and formatting checks passed. CI repeats the Linux integration shards; the memory-usage shard runs serially, removes all data from each unique test bucket and uses disk-backed MinIO. Its previous 4 GiB tmpfs filled during the multi-gigabyte staging workload; proxy memory limits and memory assertions are unchanged. + +Run `uv run pytest tests/unit -q`, `uv run ruff check .` and `uv run ruff format --check .`. `tests/integration/test_generation_roundtrip.py` exercises real HTTP and MinIO, including hash tampering below/at/above the buffering threshold, generation overwrite, conditional writes, ListParts, native copying and UploadPartCopy. Set `S3PROXY_TEST_REDIS_URL` to a dedicated test Redis to repeat the scenario with durable state. `S3PROXY_TEST_BACKEND` selects the backend for the shared integration fixtures. + +Before production rollout, run the target backend's compatibility suite and workload benchmarks with its real latency, upload sizes, concurrency, lifecycle policy and pod limits. Existing benchmark/compatibility tests for legacy copy internals remain explicitly separate from the public v3 route. No production deployment or bucket policy changes are part of this PR. diff --git a/docs/read-benchmark.json b/docs/read-benchmark.json new file mode 100644 index 0000000..2fedd01 --- /dev/null +++ b/docs/read-benchmark.json @@ -0,0 +1,18 @@ +{ + "baseline": { + "samples": 10, + "bytes": 33554432, + "median_ms": 317.12, + "max_ms": 332.85, + "median_MiB_s": 100.91, + "sampled_peak_RSS_MiB": 179.23 + }, + "changed": { + "samples": 10, + "bytes": 33554432, + "median_ms": 289.26, + "max_ms": 331.64, + "median_MiB_s": 110.63, + "sampled_peak_RSS_MiB": 175.06 + } +} \ No newline at end of file diff --git a/e2e/clickhouse/templates/backup-config.yaml b/e2e/clickhouse/templates/backup-config.yaml deleted file mode 100644 index d16f22a..0000000 --- a/e2e/clickhouse/templates/backup-config.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: v1 -kind: ConfigMap -metadata: - name: clickhouse-backup-config -data: - config.yml: | - general: - remote_storage: s3 - disable_progress_bar: true - backups_to_keep_local: 0 - backups_to_keep_remote: 3 - clickhouse: - host: localhost - port: 9000 - username: default - password: "" - s3: - endpoint: http://s3proxy-python-frontproxy.s3proxy:80 - bucket: clickhouse-backups - path: backups - access_key: minioadmin - secret_key: minioadmin - force_path_style: true - disable_ssl: true diff --git a/e2e/clickhouse/templates/clickhouse-installation-restore.yaml b/e2e/clickhouse/templates/clickhouse-installation-restore.yaml deleted file mode 100644 index c9060f9..0000000 --- a/e2e/clickhouse/templates/clickhouse-installation-restore.yaml +++ /dev/null @@ -1,55 +0,0 @@ -apiVersion: "clickhouse.altinity.com/v1" -kind: "ClickHouseInstallation" -metadata: - name: ${RESTORE_CLUSTER_NAME} -spec: - configuration: - clusters: - - name: "cluster" - layout: - shardsCount: 1 - replicasCount: 3 - defaults: - templates: - podTemplate: clickhouse-pod - dataVolumeClaimTemplate: data-volume - templates: - podTemplates: - - name: clickhouse-pod - spec: - containers: - - name: clickhouse - image: clickhouse/clickhouse-server:25.8 - resources: - requests: - memory: "4Gi" - cpu: "500m" - limits: - memory: "4Gi" - cpu: "2" - - name: clickhouse-backup - image: altinity/clickhouse-backup:2.6.15 - imagePullPolicy: IfNotPresent - command: - - /bin/bash - - -c - - | - echo "clickhouse-backup sidecar ready" - tail -f /dev/null - volumeMounts: - - name: clickhouse-backup-config - mountPath: /etc/clickhouse-backup - - name: data-volume - mountPath: /var/lib/clickhouse - volumes: - - name: clickhouse-backup-config - configMap: - name: clickhouse-backup-config - volumeClaimTemplates: - - name: data-volume - spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 20Gi diff --git a/e2e/clickhouse/templates/clickhouse-installation.yaml b/e2e/clickhouse/templates/clickhouse-installation.yaml deleted file mode 100644 index c8fcc5b..0000000 --- a/e2e/clickhouse/templates/clickhouse-installation.yaml +++ /dev/null @@ -1,55 +0,0 @@ -apiVersion: "clickhouse.altinity.com/v1" -kind: "ClickHouseInstallation" -metadata: - name: ${CLUSTER_NAME} -spec: - configuration: - clusters: - - name: "cluster" - layout: - shardsCount: 1 - replicasCount: 3 - defaults: - templates: - podTemplate: clickhouse-pod - dataVolumeClaimTemplate: data-volume - templates: - podTemplates: - - name: clickhouse-pod - spec: - containers: - - name: clickhouse - image: clickhouse/clickhouse-server:25.8 - resources: - requests: - memory: "4Gi" - cpu: "500m" - limits: - memory: "4Gi" - cpu: "2" - - name: clickhouse-backup - image: altinity/clickhouse-backup:2.6.15 - imagePullPolicy: IfNotPresent - command: - - /bin/bash - - -c - - | - echo "clickhouse-backup sidecar ready" - tail -f /dev/null - volumeMounts: - - name: clickhouse-backup-config - mountPath: /etc/clickhouse-backup - - name: data-volume - mountPath: /var/lib/clickhouse - volumes: - - name: clickhouse-backup-config - configMap: - name: clickhouse-backup-config - volumeClaimTemplates: - - name: data-volume - spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 20Gi diff --git a/e2e/clickhouse/test.sh b/e2e/clickhouse/test.sh deleted file mode 100755 index 4f65efd..0000000 --- a/e2e/clickhouse/test.sh +++ /dev/null @@ -1,326 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -ROOT_DIR="$(dirname "$SCRIPT_DIR")" -cd "$SCRIPT_DIR" - -# Source shared encryption verification -source "${SCRIPT_DIR}/../scripts/verify-encryption-k8s.sh" - -# Use isolated kubeconfig if not already set (running outside container) -if [ -z "${KUBECONFIG:-}" ]; then - export KUBECONFIG="${ROOT_DIR}/kubeconfig" - if [ ! -f "$KUBECONFIG" ]; then - echo "ERROR: Kubeconfig not found at $KUBECONFIG" - echo "Run ./cluster.sh up first" - exit 1 - fi -fi - -NAMESPACE="clickhouse-test" -export CLUSTER_NAME="chi-backup-test" -DATA_SIZE_ROWS=50000000 # 50M rows - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -NC='\033[0m' # No Color - -log_info() { echo -e "${GREEN}[INFO]${NC} $1"; } -log_error() { echo -e "${RED}[ERROR]${NC} $1"; } - -cleanup() { - log_info "Cleaning up..." - kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=false || true -} - -trap cleanup EXIT - -# ============================================================================ -# STEP 1: Create namespace and clickhouse-backup config -# ============================================================================ -log_info "=== Step 1: Creating namespace and clickhouse-backup config ===" - -kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - - -# Clean up any leftover data in the S3 bucket from previous runs (background) -log_info "Cleaning S3 bucket from previous test runs (background)..." -( - kubectl run bucket-cleanup --namespace "$NAMESPACE" \ - --image=mc:latest \ - --image-pull-policy=Never \ - --restart=Never \ - --command -- /bin/sh -c " - mc alias set minio http://minio.minio.svc.cluster.local:9000 minioadmin minioadmin >/dev/null 2>&1 - mc rm --recursive --force minio/clickhouse-backups/ 2>/dev/null || true - echo 'Bucket cleaned' - " 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=jsonpath='{.status.phase}'=Succeeded pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl delete pod -n "$NAMESPACE" bucket-cleanup --ignore-not-found >/dev/null 2>&1 || true -) & -BUCKET_CLEANUP_PID=$! - -# Create clickhouse-backup config -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/backup-config.yaml" - -# ============================================================================ -# STEP 2: Deploy BOTH ClickHouse clusters (source + restore target) in parallel -# ============================================================================ -log_info "=== Step 2: Deploying BOTH ClickHouse clusters (source + restore) ===" - -export RESTORE_CLUSTER_NAME="${CLUSTER_NAME}-restore" - -# Deploy source cluster -envsubst < "${SCRIPT_DIR}/templates/clickhouse-installation.yaml" | kubectl apply -n "$NAMESPACE" -f - - -# Deploy restore target cluster (same spec, different name) -envsubst < "${SCRIPT_DIR}/templates/clickhouse-installation-restore.yaml" | kubectl apply -n "$NAMESPACE" -f - - -log_info "Waiting for BOTH ClickHouse clusters to be ready (parallel)..." - -# Wait for source cluster pods -( - until kubectl get pods -n "$NAMESPACE" -l "clickhouse.altinity.com/chi=${CLUSTER_NAME}" --no-headers 2>/dev/null | grep -q .; do - sleep 5 - done - kubectl wait --namespace "$NAMESPACE" --for=condition=ready pod \ - --selector="clickhouse.altinity.com/chi=${CLUSTER_NAME}" --timeout=600s - echo "✓ Source cluster ready" -) & -SOURCE_WAIT_PID=$! - -# Wait for restore cluster pods -( - until kubectl get pods -n "$NAMESPACE" -l "clickhouse.altinity.com/chi=${RESTORE_CLUSTER_NAME}" --no-headers 2>/dev/null | grep -q .; do - sleep 5 - done - kubectl wait --namespace "$NAMESPACE" --for=condition=ready pod \ - --selector="clickhouse.altinity.com/chi=${RESTORE_CLUSTER_NAME}" --timeout=600s - echo "✓ Restore cluster ready" -) & -RESTORE_WAIT_PID=$! - -# Wait for bucket cleanup (must complete before backup) -wait $BUCKET_CLEANUP_PID || true - -wait $SOURCE_WAIT_PID || { log_error "Source cluster failed to start"; exit 1; } -wait $RESTORE_WAIT_PID || { log_error "Restore cluster failed to start"; exit 1; } - -log_info "Both ClickHouse clusters are ready" - -# Get the first pod name -CH_POD=$(kubectl get pods -n "$NAMESPACE" -l "clickhouse.altinity.com/chi=${CLUSTER_NAME}" -o jsonpath='{.items[0].metadata.name}') -log_info "Using ClickHouse pod: $CH_POD" - -# Verify clickhouse-backup is working -log_info "Verifying clickhouse-backup installation..." -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse-backup -- clickhouse-backup --version - -# ============================================================================ -# STEP 3: Generate test data using ClickHouse built-in functions -# ============================================================================ -log_info "=== Step 3: Generating ${DATA_SIZE_ROWS} rows of test data ===" - -# Create database -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query "CREATE DATABASE IF NOT EXISTS test_db;" - -# Create events table with various data types -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -CREATE TABLE IF NOT EXISTS test_db.events ( - id UInt64, - event_time DateTime64(3) DEFAULT now64(3), - user_id UInt32, - event_type LowCardinality(String), - page_url String, - referrer String, - ip_address IPv4, - user_agent String, - country_code LowCardinality(FixedString(2)), - amount Decimal64(2), - metadata String -) ENGINE = MergeTree() -PARTITION BY toYYYYMM(event_time) -ORDER BY (event_time, user_id, id); -" - -log_info "Inserting ${DATA_SIZE_ROWS} rows using generateRandom()..." -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -INSERT INTO test_db.events (id, event_time, user_id, event_type, page_url, referrer, ip_address, user_agent, country_code, amount, metadata) -SELECT - number as id, - now64(3) - toIntervalSecond(rand() % 31536000) as event_time, - rand() % 1000000 as user_id, - arrayElement(['click', 'view', 'purchase', 'signup', 'logout'], (rand() % 5) + 1) as event_type, - concat('https://example.com/page/', toString(rand() % 10000)) as page_url, - concat('https://referrer.com/', toString(rand() % 1000)) as referrer, - toIPv4(rand()) as ip_address, - concat('Mozilla/5.0 (', arrayElement(['Windows', 'Mac', 'Linux', 'iOS', 'Android'], (rand() % 5) + 1), ')') as user_agent, - arrayElement(['US', 'GB', 'DE', 'FR', 'JP', 'CN', 'BR', 'IN', 'CA', 'AU'], (rand() % 10) + 1) as country_code, - round(rand() % 100000 / 100, 2) as amount, - concat('{\"session_id\":\"', toString(generateUUIDv4()), '\",\"version\":', toString(rand() % 10), '}') as metadata -FROM numbers(${DATA_SIZE_ROWS}); -" - -# Create another table for aggregated data -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -CREATE TABLE IF NOT EXISTS test_db.daily_stats ( - date Date, - country_code LowCardinality(FixedString(2)), - event_type LowCardinality(String), - total_events UInt64, - unique_users UInt64, - total_amount Decimal128(2) -) ENGINE = SummingMergeTree() -ORDER BY (date, country_code, event_type); -" - -# Insert aggregated data -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -INSERT INTO test_db.daily_stats -SELECT - toDate(event_time) as date, - country_code, - event_type, - count() as total_events, - uniqExact(user_id) as unique_users, - sum(amount) as total_amount -FROM test_db.events -GROUP BY date, country_code, event_type; -" - -log_info "Data generation complete" - -# Get table stats -log_info "Table statistics:" -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -SELECT - database, - table, - formatReadableQuantity(sum(rows)) as rows, - formatReadableSize(sum(bytes_on_disk)) as size -FROM system.parts -WHERE active AND database = 'test_db' -GROUP BY database, table -FORMAT Pretty; -" - -# Get checksum for validation -CHECKSUM=$(kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse -- clickhouse-client --query " -SELECT cityHash64(groupArray(id)) FROM (SELECT id FROM test_db.events ORDER BY id LIMIT 10000); -") -log_info "Data checksum (first 10k rows): $CHECKSUM" - -# ============================================================================ -# STEP 4: Create backup using clickhouse-backup -# ============================================================================ -log_info "=== Step 4: Creating backup to S3 using clickhouse-backup ===" - -BACKUP_NAME="backup_$(date +%Y%m%d_%H%M%S)" - -# Create local backup first, then upload to S3 -log_info "Creating local backup..." -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse-backup -- clickhouse-backup create "$BACKUP_NAME" - -log_info "Uploading backup to S3..." -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse-backup -- clickhouse-backup upload "$BACKUP_NAME" - -log_info "Backup ${BACKUP_NAME} created and uploaded" - -# List remote backups -log_info "Remote backups:" -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse-backup -- clickhouse-backup list remote - -# Delete local backup to save space -kubectl exec -n "$NAMESPACE" "$CH_POD" -c clickhouse-backup -- clickhouse-backup delete local "$BACKUP_NAME" - -# ============================================================================ -# STEP 5 + 6: Verify encryption + Restore to pre-created cluster (parallel) -# ============================================================================ -log_info "=== Step 5 + 6: Restore to pre-created cluster (parallel with encryption verification) ===" - -# Get the restore cluster pod (already running!) -RESTORE_POD=$(kubectl get pods -n "$NAMESPACE" -l "clickhouse.altinity.com/chi=${RESTORE_CLUSTER_NAME}" -o jsonpath='{.items[0].metadata.name}') -log_info "Using restore cluster pod: $RESTORE_POD (already running)" - -# Start encryption verification in background (runs throughout restore) -verify_encryption "clickhouse-backups" "backups/" "$NAMESPACE" & -VERIFY_PID=$! - -# Restore to the PRE-CREATED restore cluster (no waiting for cluster to start!) -log_info "Downloading backup to restore cluster..." -kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse-backup -- clickhouse-backup download "$BACKUP_NAME" - -log_info "Restoring backup to restore cluster..." -kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse-backup -- clickhouse-backup restore "$BACKUP_NAME" -log_info "✓ Restore complete" - -# Clean up local backup on restore cluster -kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse-backup -- clickhouse-backup delete local "$BACKUP_NAME" - -# Now wait for encryption verification -wait $VERIFY_PID || { log_error "Encryption verification failed"; exit 1; } -log_info "✓ Encryption verified" - -# ============================================================================ -# STEP 7: Validate restored data on RESTORE cluster -# ============================================================================ -log_info "=== Step 7: Validating restored data on restore cluster ===" - -# Get table stats after restore -log_info "Restored table statistics:" -kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse -- clickhouse-client --query " -SELECT - database, - table, - formatReadableQuantity(sum(rows)) as rows, - formatReadableSize(sum(bytes_on_disk)) as size -FROM system.parts -WHERE active AND database = 'test_db' -GROUP BY database, table -FORMAT Pretty; -" - -# Get row count from RESTORE cluster -RESTORED_COUNT=$(kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse -- clickhouse-client --query " -SELECT count() FROM test_db.events; -") -log_info "Restored row count: $RESTORED_COUNT" - -# Get checksum for validation from RESTORE cluster -RESTORED_CHECKSUM=$(kubectl exec -n "$NAMESPACE" "$RESTORE_POD" -c clickhouse -- clickhouse-client --query " -SELECT cityHash64(groupArray(id)) FROM (SELECT id FROM test_db.events ORDER BY id LIMIT 10000); -") -log_info "Restored data checksum: $RESTORED_CHECKSUM" - -# Validate -if [ "$CHECKSUM" = "$RESTORED_CHECKSUM" ]; then - log_info "=== VALIDATION PASSED: Checksums match! ===" - log_info "Original: $CHECKSUM" - log_info "Restored: $RESTORED_CHECKSUM" -else - log_error "=== VALIDATION FAILED: Checksums do not match! ===" - log_error "Original: $CHECKSUM" - log_error "Restored: $RESTORED_CHECKSUM" - exit 1 -fi - -if [ "$RESTORED_COUNT" = "$DATA_SIZE_ROWS" ]; then - log_info "=== ROW COUNT PASSED: $RESTORED_COUNT rows ===" -else - log_error "=== ROW COUNT MISMATCH: Expected $DATA_SIZE_ROWS, got $RESTORED_COUNT ===" - exit 1 -fi - -# ============================================================================ -# STEP 8: Cleanup -# ============================================================================ -log_info "=== Step 8: Cleanup ===" -log_info "Test completed successfully!" -log_info "" -log_info "The namespace $NAMESPACE will be deleted on script exit." -log_info "To keep the cluster for inspection, press Ctrl+C within 10 seconds..." -sleep 10 - -log_info "=== ClickHouse Backup/Restore Test PASSED ===" diff --git a/e2e/cluster.sh b/e2e/cluster.sh deleted file mode 100755 index 5ee91e1..0000000 --- a/e2e/cluster.sh +++ /dev/null @@ -1,198 +0,0 @@ -#!/bin/bash -set -e - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -cd "$SCRIPT_DIR" - -COMPOSE_FILE="docker-compose.yml" - -case "${1:-help}" in - up) - echo "=== Starting database test cluster ===" - docker compose -f $COMPOSE_FILE up --build -d - - echo "Waiting for cluster to be ready..." - ( docker compose -f $COMPOSE_FILE logs -f & ) | while read -r line; do - echo "$line" - if echo "$line" | grep -q "Cluster is ready"; then - break - fi - done - - echo "" - echo "==========================================" - echo "Cluster is running in background." - echo "" - echo "Run tests:" - echo " ./cluster.sh postgres" - echo " ./cluster.sh elasticsearch" - echo " ./cluster.sh scylla" - echo "" - echo "Or open a shell:" - echo " ./cluster.sh shell" - echo "" - echo "Cleanup when done:" - echo " ./cluster.sh down" - echo "==========================================" - ;; - - postgres) - echo "=== Running PostgreSQL (CloudNativePG) test ===" - docker compose -f $COMPOSE_FILE exec db-test ./postgres/test.sh - ;; - - elasticsearch|es) - echo "=== Running Elasticsearch (ECK) test ===" - docker compose -f $COMPOSE_FILE exec db-test ./elasticsearch/test.sh - ;; - - scylla) - echo "=== Running ScyllaDB test ===" - docker compose -f $COMPOSE_FILE exec db-test ./scylla/test.sh - ;; - - clickhouse|ch) - echo "=== Running ClickHouse test ===" - docker compose -f $COMPOSE_FILE exec db-test ./clickhouse/test.sh - ;; - - s3-compat|s3) - echo "=== Running S3 Compatibility (Ceph s3-tests) ===" - docker compose -f $COMPOSE_FILE exec db-test ./s3-compatibility/test.sh - ;; - - all) - echo "=== Running all database tests ===" - docker compose -f $COMPOSE_FILE exec db-test ./postgres/test.sh - docker compose -f $COMPOSE_FILE exec db-test ./elasticsearch/test.sh - docker compose -f $COMPOSE_FILE exec db-test ./scylla/test.sh - docker compose -f $COMPOSE_FILE exec db-test ./clickhouse/test.sh - echo "=== All tests completed ===" - ;; - - load-test) - echo "=== Running S3 load test (3 concurrent 10MB uploads) ===" - docker compose -f $COMPOSE_FILE exec db-test sh -c ' - PODS=$(kubectl get pods -n s3proxy -l app=s3proxy-python -o jsonpath="{.items[*].metadata.name}") - POD_COUNT=$(echo $PODS | wc -w) - echo "Found $POD_COUNT s3proxy pods: $PODS" - - mkdir -p /tmp/lb-test - for pod in $PODS; do - kubectl logs $pod -n s3proxy 2>/dev/null | wc -l > /tmp/lb-test/$pod.start - done - - echo "=== Creating test pod with AWS CLI ===" - kubectl run s3-load-test -n s3proxy --rm -i --restart=Never \ - --image=amazon/aws-cli:latest \ - --env="AWS_ACCESS_KEY_ID=minioadmin" \ - --env="AWS_SECRET_ACCESS_KEY=minioadmin" \ - --env="AWS_DEFAULT_REGION=us-east-1" \ - --command -- /bin/sh -c " - aws --endpoint-url http://s3proxy-python-frontproxy.s3proxy s3 mb s3://load-test-bucket 2>/dev/null || true - - echo \"Generating 10MB test files...\" - mkdir -p /tmp/testfiles - for i in 1 2 3; do - dd if=/dev/urandom of=/tmp/testfiles/file-\$i.bin bs=1M count=10 2>/dev/null & - done - wait - ls -lh /tmp/testfiles/ - - echo \"=== Starting concurrent uploads ===\" - START=\$(date +%s) - for i in 1 2 3; do - aws --endpoint-url http://s3proxy-python-frontproxy.s3proxy s3 cp /tmp/testfiles/file-\$i.bin s3://load-test-bucket/file-\$i.bin & - done - wait - END=\$(date +%s) - echo \"Upload complete in \$((END - START))s\" - - echo \"=== Verifying uploads ===\" - aws --endpoint-url http://s3proxy-python-frontproxy.s3proxy s3 ls s3://load-test-bucket/ - - echo \"=== Downloading and verifying ===\" - mkdir -p /tmp/downloads - for i in 1 2 3; do - aws --endpoint-url http://s3proxy-python-frontproxy.s3proxy s3 cp s3://load-test-bucket/file-\$i.bin /tmp/downloads/file-\$i.bin & - done - wait - - md5sum /tmp/testfiles/*.bin > /tmp/orig.md5 - md5sum /tmp/downloads/*.bin > /tmp/down.md5 - ORIG_SUMS=\$(cat /tmp/orig.md5 | while read sum name; do echo \$sum; done | sort) - DOWN_SUMS=\$(cat /tmp/down.md5 | while read sum name; do echo \$sum; done | sort) - - if [ \"\$ORIG_SUMS\" = \"\$DOWN_SUMS\" ]; then - echo \"✓ Checksums match - round-trip successful\" - else - echo \"✗ Checksum mismatch!\" - exit 1 - fi - " - - echo "" - echo "=== Checking load balancing ===" - sleep 2 - PODS_HIT=0 - for pod in $PODS; do - START_LINE=$(cat /tmp/lb-test/$pod.start 2>/dev/null || echo "0") - REQUEST_COUNT=$(kubectl logs $pod -n s3proxy 2>/dev/null | tail -n +$((START_LINE + 1)) | grep -c -E "GET|POST|PUT|HEAD" || echo "0") - if [ "$REQUEST_COUNT" -gt 0 ]; then - PODS_HIT=$((PODS_HIT + 1)) - echo "✓ Pod $pod: received $REQUEST_COUNT requests" - else - echo " Pod $pod: received 0 requests" - fi - done - rm -rf /tmp/lb-test - - if [ "$PODS_HIT" -ge 2 ]; then - echo "✓ Load balancing verified - traffic distributed across $PODS_HIT pods" - else - echo "⚠ Traffic went to only $PODS_HIT pod(s)" - fi - ' - ;; - - down) - echo "=== Cleaning up ===" - docker compose -f $COMPOSE_FILE down 2>/dev/null || true - # Remove all e2e volumes EXCEPT registry cache - docker volume ls -q --filter name=e2e_ | grep -v e2e_registry-data | xargs -r docker volume rm 2>/dev/null || true - # Remove anonymous volumes (64-char hex names from Kind/Docker) - docker volume ls -q | grep -E '^[a-f0-9]{64}$' | xargs -r docker volume rm 2>/dev/null || true - # Delete Kind cluster containers directly - docker rm -f db-backup-test-control-plane db-backup-test-worker db-backup-test-worker2 db-backup-test-worker3 2>/dev/null || true - # Clean up Kind network - docker network rm kind 2>/dev/null || true - echo "Cleanup complete (registry cache preserved)" - ;; - - *) - echo "Usage: $0 " - echo "" - echo "Commands:" - echo " up - Start Kind cluster + s3proxy + MinIO" - echo " down - Stop and cleanup everything" - echo " status - Show cluster status" - echo " logs - Show cluster logs" - echo " shell - Open shell in test container" - echo "" - echo "Tests:" - echo " load-test - Run S3 load test (upload/download verification)" - echo " s3-compat - Run S3 compatibility tests (Ceph s3-tests)" - echo " postgres - Run PostgreSQL (CloudNativePG) backup test" - echo " elasticsearch - Run Elasticsearch (ECK) backup test" - echo " scylla - Run ScyllaDB backup test" - echo " clickhouse - Run ClickHouse backup test" - echo " all - Run all database backup tests" - echo "" - echo "Example:" - echo " ./cluster.sh up" - echo " ./cluster.sh load-test" - echo " ./cluster.sh down" - echo "" - exit 1 - ;; -esac diff --git a/e2e/common/s3-credentials.yaml b/e2e/common/s3-credentials.yaml deleted file mode 100644 index c83c867..0000000 --- a/e2e/common/s3-credentials.yaml +++ /dev/null @@ -1,11 +0,0 @@ -# S3 credentials template for MinIO -# Each database namespace needs its own copy of this secret ---- -apiVersion: v1 -kind: Secret -metadata: - name: s3-credentials -type: Opaque -stringData: - ACCESS_KEY_ID: "minioadmin" - ACCESS_SECRET_KEY: "minioadmin" diff --git a/e2e/docker-compose.yml b/e2e/docker-compose.yml deleted file mode 100644 index 229c9f1..0000000 --- a/e2e/docker-compose.yml +++ /dev/null @@ -1,519 +0,0 @@ -services: - # Local registry as pull-through cache for Docker Hub - avoids rate limits - registry: - image: registry:3 - restart: always - ports: - - "5001:5000" - volumes: - - registry-data:/var/lib/registry - environment: - REGISTRY_PROXY_REMOTEURL: https://registry-1.docker.io - REGISTRY_LOG_LEVEL: info - OTEL_TRACES_EXPORTER: none - OTEL_SDK_DISABLED: "true" - - db-test: - image: docker:29-cli - privileged: true - depends_on: - - registry - # Publish the s3proxy dashboard to the host. After the cluster is up, the - # entrypoint runs `kubectl port-forward` (on 0.0.0.0) onto container :14433, - # so the dashboard is reachable at http://localhost:14433/dashboard (admin/admin). - ports: - - "14433:14433" - volumes: - - /var/run/docker.sock:/var/run/docker.sock - - ..:/repo - - db-test-data:/root/.kube - working_dir: /repo/e2e - environment: - - KUBECONFIG=/root/.kube/config - - REGISTRY_HOST=registry:5000 - entrypoint: ["/bin/sh", "-c"] - command: - - | - set -e - echo "=== Installing tools ===" - apk add --no-cache curl wget bash openssl jq python3 py3-pip gettext - - # Detect architecture - RAW_ARCH=$$(cat /etc/apk/arch 2>/dev/null || uname -m) - echo "Raw architecture: $$RAW_ARCH" - if [ "$$RAW_ARCH" = "x86_64" ]; then - ARCH="amd64" - else - ARCH="arm64" - fi - echo "Using architecture: $$ARCH" - - echo "=== Installing kind, kubectl, helm (parallel) ===" - wget -qO /usr/local/bin/kind "https://kind.sigs.k8s.io/dl/v0.31.0/kind-linux-$${ARCH}" & - wget -qO /usr/local/bin/kubectl "https://dl.k8s.io/release/v1.34.3/bin/linux/$${ARCH}/kubectl" & - wget -qO- "https://get.helm.sh/helm-v3.20.2-linux-$${ARCH}.tar.gz" | tar xz -C /tmp & - wait - chmod +x /usr/local/bin/kind /usr/local/bin/kubectl - mv /tmp/linux-$${ARCH}/helm /usr/local/bin/helm - chmod +x /usr/local/bin/helm - - echo "=== Starting parallel prep (docker build + helm + kind cluster) ===" - - # 1. Docker builds (background, parallel) - docker build -t s3proxy:latest -f /repo/s3proxy/Dockerfile /repo & - BUILD_PID=$$! - - # Dashboard image (Svelte static build served by nginx) - docker build -t s3proxy-dashboard:latest /repo/dashboard-frontend & - DASHBOARD_BUILD_PID=$$! - - # Build esrally image with dependencies pre-installed (parallel) - # Pinned to 3.12 because yappi (esrally dep) has no 3.14 wheel yet - ( - printf '%s\n' \ - 'FROM python:3.12-slim' \ - 'RUN apt-get update && apt-get install -y --no-install-recommends gcc python3-dev git curl && rm -rf /var/lib/apt/lists/*' \ - 'RUN pip install --no-cache-dir esrally' \ - | docker build -t esrally:latest -f - /tmp - echo "✓ esrally image built" - ) & - ESRALLY_BUILD_PID=$$! - - # Build mc (MinIO client) image for encryption verification (parallel) - ( - printf '%s\n' \ - 'FROM alpine:latest' \ - 'RUN apk add --no-cache ca-certificates wget && ARCH=$$(uname -m) && [ "$$ARCH" = "aarch64" ] || [ "$$ARCH" = "arm64" ] && MC_ARCH=arm64 || MC_ARCH=amd64 && wget -q https://dl.min.io/client/mc/release/linux-$${MC_ARCH}/mc -O /usr/local/bin/mc && chmod +x /usr/local/bin/mc && apk del wget' \ - | docker build -t mc:latest -f - /tmp - echo "✓ mc image built" - ) & - MC_BUILD_PID=$$! - - # 2. Helm repo setup + dependency build (background) - ( - helm repo add dandydeveloper https://dandydeveloper.github.io/charts - helm repo update - helm dependency build /repo/chart - echo "✓ Helm repos and dependencies ready" - ) & - HELM_PREP_PID=$$! - - # 3. Kind cluster creation (foreground - other tasks run in parallel) - echo "=== Creating kind cluster ===" - # Cleanup any leftover kind containers and network - CONTAINERS=$$(docker ps -aq --filter "label=io.x-k8s.kind.cluster=db-backup-test" 2>/dev/null || true) - if [ -n "$$CONTAINERS" ]; then - docker rm -f $$CONTAINERS 2>/dev/null || true - fi - docker network rm kind 2>/dev/null || true - kind delete cluster --name db-backup-test 2>/dev/null || true - - # Get the real host path that maps to /repo (for Kind extraMounts) - REPO_HOST_PATH=$$(docker inspect $$(hostname) 2>/dev/null | jq -r '.[0].Mounts[] | select(.Destination=="/repo") | .Source' || echo "") - echo "Detected repo host path: $$REPO_HOST_PATH" - - # Create kind config with 3 worker nodes for database replicas - HOST_PATH="$${REPO_HOST_PATH:-/tmp}" - - printf '%s\n' \ - "kind: Cluster" \ - "apiVersion: kind.x-k8s.io/v1alpha4" \ - "name: db-backup-test" \ - "networking:" \ - " apiServerAddress: \"0.0.0.0\"" \ - "containerdConfigPatches:" \ - "- |-" \ - " [plugins.\"io.containerd.grpc.v1.cri\".registry]" \ - " config_path = \"/etc/containerd/certs.d\"" \ - "nodes:" \ - " - role: control-plane" \ - " image: kindest/node:v1.34.3" \ - " extraMounts:" \ - " - hostPath: $$HOST_PATH" \ - " containerPath: /repo" \ - " - role: worker" \ - " image: kindest/node:v1.34.3" \ - " extraMounts:" \ - " - hostPath: $$HOST_PATH" \ - " containerPath: /repo" \ - " - role: worker" \ - " image: kindest/node:v1.34.3" \ - " extraMounts:" \ - " - hostPath: $$HOST_PATH" \ - " containerPath: /repo" \ - " - role: worker" \ - " image: kindest/node:v1.34.3" \ - " extraMounts:" \ - " - hostPath: $$HOST_PATH" \ - " containerPath: /repo" \ - > /tmp/kind-config.yaml - - kind create cluster --name db-backup-test --wait 120s --config /tmp/kind-config.yaml - - # Configure registry mirror on all Kind nodes (inject hosts.toml after cluster creation) - echo "=== Configuring registry mirror on Kind nodes ===" - # Get registry IP on kind network (after connecting it below) - REGISTRY_CONTAINER=$$(docker ps -qf "name=registry" | head -1) - docker network connect kind $$REGISTRY_CONTAINER 2>/dev/null || true - REGISTRY_IP=$$(docker inspect -f '{{range $$k, $$v := .NetworkSettings.Networks}}{{if eq $$k "kind"}}{{$$v.IPAddress}}{{end}}{{end}}' $$REGISTRY_CONTAINER) - echo "Registry IP on kind network: $$REGISTRY_IP" - - # Inject hosts.toml into each Kind node - for NODE in $$(kind get nodes --name db-backup-test); do - echo "Configuring registry mirror on $$NODE..." - docker exec $$NODE mkdir -p /etc/containerd/certs.d/docker.io - printf '%s\n' \ - 'server = "https://registry-1.docker.io"' \ - '' \ - "[host.\"http://$$REGISTRY_IP:5000\"]" \ - ' capabilities = ["pull", "resolve"]' \ - ' skip_verify = true' \ - | docker exec -i $$NODE tee /etc/containerd/certs.d/docker.io/hosts.toml > /dev/null - # Restart containerd to pick up the new config - docker exec $$NODE systemctl restart containerd - echo "✓ $$NODE configured" - done - echo "✓ Registry mirror configured on all nodes" - - # Verify registry mirror setup - echo "=== Verifying registry mirror setup ===" - CONTROL_PLANE=$$(kind get nodes --name db-backup-test | head -1) - echo "Checking hosts.toml on $$CONTROL_PLANE..." - docker exec $$CONTROL_PLANE cat /etc/containerd/certs.d/docker.io/hosts.toml - echo "" - echo "Testing registry connectivity from Kind node..." - docker exec $$CONTROL_PLANE curl -sf "http://$$REGISTRY_IP:5000/v2/" && echo "✓ Registry reachable from Kind nodes" - - echo "=== Configuring kubectl ===" - mkdir -p /root/.kube - - # Connect kind container to compose network - COMPOSE_NETWORK=$$(docker network ls --filter name=e2e_default -q | head -1) - if [ -z "$$COMPOSE_NETWORK" ]; then - COMPOSE_NETWORK=$$(docker network ls --filter name=e2e -q | head -1) - fi - echo "Connecting to compose network: $$COMPOSE_NETWORK" - docker network connect $$COMPOSE_NETWORK db-backup-test-control-plane 2>/dev/null || true - - - # Get kubeconfig - kind get kubeconfig --name db-backup-test > /root/.kube/config - sed -i 's|server:.*|server: https://db-backup-test-control-plane:6443|g' /root/.kube/config - - echo "Waiting for API server..." - for i in 1 2 3 4 5 6 7 8 9 10; do - kubectl cluster-info && break - echo "Retry $$i..." - sleep 5 - done - kubectl get nodes - - # === ALL PARALLEL: Operators + Infrastructure start together === - echo "=== Starting ALL parallel tasks (operators + infrastructure) ===" - - # 1. ECK (Elasticsearch) operator - ( - set -e - # retry wrapper: apply a remote manifest, failing hard only after N attempts - eck_apply() { - for i in 1 2 3 4 5; do - kubectl apply $$2 -f "$$1" && return 0 - echo "apply $$1 failed, retry $$i..."; sleep 10 - done - echo "ERROR: apply $$1 failed after 5 attempts"; return 1 - } - # wait for a CRD to exist, then to be Established (avoids the - # "no matching resources" race when wait runs before apply propagates) - eck_wait_crd() { - until kubectl get "crd/$$1" >/dev/null 2>&1; do sleep 2; done - kubectl wait --for=condition=Established "crd/$$1" --timeout=180s - } - - echo "Installing ECK CRDs..." - eck_apply https://download.elastic.co/downloads/eck/3.3.2/crds.yaml - echo "Waiting for ECK CRDs to be established..." - eck_wait_crd elasticsearches.elasticsearch.k8s.elastic.co - eck_wait_crd kibanas.kibana.k8s.elastic.co - - echo "Installing ECK operator..." - eck_apply https://download.elastic.co/downloads/eck/3.3.2/operator.yaml - # statefulset may not be registered the instant apply returns; wait for it - until kubectl get --namespace elastic-system statefulset/elastic-operator >/dev/null 2>&1; do sleep 2; done - for i in 1 2 3; do - kubectl rollout status --namespace elastic-system statefulset/elastic-operator --timeout=300s && break - echo "elastic-operator rollout not ready, retry $$i..."; sleep 10 - done - echo "✓ ECK operator ready" - ) & - ECK_PID=$$! - - # 2. CloudNativePG (PostgreSQL) operator - ( - helm repo add cnpg https://cloudnative-pg.github.io/charts 2>/dev/null || true - helm repo update cnpg 2>/dev/null || true - helm upgrade --install cnpg cnpg/cloudnative-pg --namespace cnpg-system --create-namespace --wait --timeout 300s - echo "✓ CloudNativePG operator ready" - ) & - CNPG_PID=$$! - - # 3. ClickHouse operator - ( - set -e - for i in 1 2 3 4 5; do - kubectl apply -f https://raw.githubusercontent.com/Altinity/clickhouse-operator/master/deploy/operator/clickhouse-operator-install-bundle.yaml && break - echo "ClickHouse operator apply failed, retry $$i..."; sleep 10 - [ "$$i" = 5 ] && { echo "ERROR: ClickHouse operator apply failed after 5 attempts"; exit 1; } - done - # pod may not be scheduled the instant apply returns; wait for it to exist first - until kubectl get pods --namespace kube-system --selector=app=clickhouse-operator 2>/dev/null | grep -q .; do sleep 2; done - kubectl wait --namespace kube-system --for=condition=ready pod --selector=app=clickhouse-operator --timeout=300s - echo "✓ ClickHouse operator ready" - ) & - CLICKHOUSE_PID=$$! - - # 4. cert-manager + Scylla Operator + Scylla Manager (sequential chain in single background job) - ( - set -e - # retry wrapper for remote manifests (network-flaky downloads) - remote_apply() { - for i in 1 2 3 4 5; do - kubectl apply $$2 -f "$$1" && return 0 - echo "apply $$1 failed, retry $$i..."; sleep 10 - done - echo "ERROR: apply $$1 failed after 5 attempts"; return 1 - } - # cert-manager - remote_apply https://github.com/cert-manager/cert-manager/releases/download/v1.19.0/cert-manager.yaml - # wait for the 3 cert-manager deployments to exist before waiting on pods - # (--all matches zero pods and returns immediately if none scheduled yet) - for d in cert-manager cert-manager-cainjector cert-manager-webhook; do - until kubectl get --namespace cert-manager "deployment/$$d" >/dev/null 2>&1; do sleep 2; done - kubectl rollout status --namespace cert-manager "deployment/$$d" --timeout=300s - done - echo "✓ cert-manager ready" - - # Scylla Operator (requires cert-manager) - remote_apply https://raw.githubusercontent.com/scylladb/scylla-operator/v1.20.2/deploy/operator.yaml --server-side - until kubectl get pods -l app.kubernetes.io/name=scylla-operator -n scylla-operator 2>/dev/null | grep -q .; do sleep 2; done - kubectl wait --for=condition=ready pod -l app.kubernetes.io/name=scylla-operator -n scylla-operator --timeout=300s - kubectl wait --for=condition=ready pod -l app.kubernetes.io/name=webhook-server -n scylla-operator --timeout=300s - # Wait for webhook to be fully ready - sleep 15 - echo "✓ Scylla Operator ready" - - # Scylla Manager (requires Scylla Operator) - # Label worker nodes for ScyllaDB scheduling - for node in $$(kubectl get nodes -o name | grep -v control-plane); do - kubectl label "$$node" scylla.scylladb.com/node-type=scylla --overwrite 2>/dev/null || true - done - - # Create storage class needed by manager-prod.yaml's ScyllaCluster - printf '%s\n' \ - 'apiVersion: storage.k8s.io/v1' \ - 'kind: StorageClass' \ - 'metadata:' \ - ' name: scylladb-local-xfs' \ - 'provisioner: rancher.io/local-path' \ - 'volumeBindingMode: WaitForFirstConsumer' \ - 'reclaimPolicy: Delete' \ - | kubectl apply -f - - - # manager-prod.yaml creates a ScyllaCluster backend for the manager - kubectl apply --server-side -f https://raw.githubusercontent.com/scylladb/scylla-operator/v1.20.2/deploy/manager-prod.yaml - - # Wait for backend DB pods to appear - echo "Waiting for Scylla Manager backend DB pods..." - until kubectl get pods -n scylla-manager -l scylla/cluster=scylla-manager-cluster 2>/dev/null | grep -q "."; do - sleep 2 - done - - # Wait for backend DB to be ready - kubectl wait --for=condition=ready pod -l scylla/cluster=scylla-manager-cluster -n scylla-manager --timeout=600s - - # Allow backend DB to initialize schema - sleep 20 - - # Wait for manager deployment - kubectl rollout status deployment/scylla-manager -n scylla-manager --timeout=300s - echo "✓ Scylla Manager ready" - ) & - SCYLLA_PID=$$! - - # 7. MinIO (no helm prep needed) - echo " - MinIO deployment starting..." - kubectl create namespace minio 2>/dev/null || true - cat </dev/null || true - ENCRYPT_KEY=$$(openssl rand -base64 32) - - # 5. Wait for MinIO pod only (not buckets - those can be created in parallel) - echo "=== Waiting for MinIO pod ===" - kubectl wait --for=condition=ready pod -l app=minio -n minio --timeout=120s - echo "✓ MinIO ready" - - # 6. Start bucket creation in background (doesn't block s3proxy) - echo " - Creating buckets (background)..." - kubectl run minio-setup --namespace minio --rm -i \ - --image=minio/mc:latest \ - --restart=Never \ - --command -- /bin/sh -c " - mc alias set local http://minio:9000 minioadmin minioadmin && \ - mc mb local/postgres-backups --ignore-existing && \ - mc mb local/elasticsearch-backups --ignore-existing && \ - mc mb local/scylla-backups --ignore-existing && \ - mc mb local/clickhouse-backups --ignore-existing && \ - echo 'Buckets created successfully' - " & - BUCKETS_PID=$$! - - # 7. Start s3proxy install (background) - MinIO is ready - echo "=== Installing s3proxy (background) ===" - helm upgrade --install s3proxy /repo/chart \ - -n s3proxy --wait --timeout 600s \ - --set image.repository=s3proxy \ - --set image.pullPolicy=IfNotPresent \ - --set s3.host="http://minio.minio.svc.cluster.local:9000" \ - --set secrets.credentials[0].accessKey="minioadmin" \ - --set secrets.credentials[0].secretKey="minioadmin" \ - --set secrets.credentials[0].kek="$$ENCRYPT_KEY" \ - --set dashboard.secret="$$ENCRYPT_KEY" \ - --set logLevel="DEBUG" \ - --set performance.memoryLimitMb=64 \ - --set frontproxy.enabled=true \ - --set dashboard.enabled=true \ - --set dashboard.frontend.image.repository=s3proxy-dashboard \ - --set dashboard.frontend.image.tag=latest \ - --set dashboard.frontend.image.pullPolicy=IfNotPresent \ - --set redis.enabled=true \ - --set redis.auth.enabled=true \ - --set redis.auth.password=testredispassword & - S3PROXY_PID=$$! - - # 8. Wait for ALL parallel tasks - echo "=== Waiting for all parallel tasks ===" - - # Infrastructure - wait $$BUCKETS_PID || { echo "Bucket creation failed"; exit 1; }; echo "✓ Buckets" - wait $$S3PROXY_PID || { echo "S3proxy failed"; exit 1; }; echo "✓ S3Proxy" - - # Operators - wait $$ECK_PID || { echo "ECK failed"; exit 1; }; echo "✓ ECK" - wait $$CNPG_PID || { echo "CNPG failed"; exit 1; }; echo "✓ CNPG" - wait $$CLICKHOUSE_PID || { echo "ClickHouse failed"; exit 1; }; echo "✓ ClickHouse" - wait $$SCYLLA_PID || { echo "Scylla failed"; exit 1; }; echo "✓ Scylla" - - echo "✓ All parallel tasks complete" - - # Show s3proxy status - kubectl get pods -n s3proxy - kubectl get svc -n s3proxy - - # Expose the s3proxy dashboard to the host. The dashboard is served - # by the dashboard Deployment's nginx, which also reverse-proxies the API + - # SSE to the proxy service — so a single forward to the dashboard service - # makes the whole dashboard reachable at http://localhost:14433/dashboard. - ( while true; do - kubectl port-forward -n s3proxy --address 0.0.0.0 svc/s3proxy-python-dashboard 14433:80 >/dev/null 2>&1 || true - sleep 2 - done ) & - - echo "" - echo "S3 Proxy endpoint for databases: http://s3proxy-python-frontproxy.s3proxy.svc.cluster.local:80" - echo "Direct MinIO (unencrypted): http://minio.minio.svc.cluster.local:9000" - echo "Dashboard (host browser): http://localhost:14433/dashboard (login: admin / admin)" - - echo "" - echo "==========================================" - echo "Cluster is ready" - echo "==========================================" - echo "" - echo "Run database tests with:" - echo " ./cluster.sh postgres" - echo " ./cluster.sh elasticsearch" - echo " ./cluster.sh scylla" - echo "" - echo "Or open a shell:" - echo " ./cluster.sh shell" - echo "" - # Keep container alive - tail -f /dev/null - -volumes: - db-test-data: - registry-data: diff --git a/e2e/elasticsearch/templates/elasticsearch-cluster-restore.yaml b/e2e/elasticsearch/templates/elasticsearch-cluster-restore.yaml deleted file mode 100644 index 88c069c..0000000 --- a/e2e/elasticsearch/templates/elasticsearch-cluster-restore.yaml +++ /dev/null @@ -1,39 +0,0 @@ -apiVersion: elasticsearch.k8s.elastic.co/v1 -kind: Elasticsearch -metadata: - name: ${CLUSTER_NAME}-restored -spec: - version: 9.2.4 - nodeSets: - - name: default - count: 3 - config: - node.store.allow_mmap: false - s3.client.default.endpoint: "s3proxy-python-frontproxy.s3proxy:80" - s3.client.default.protocol: "http" - s3.client.default.path_style_access: true - podTemplate: - spec: - containers: - - name: elasticsearch - env: - - name: ES_JAVA_OPTS - value: "-Xms1g -Xmx1g" - resources: - requests: - memory: 2Gi - cpu: 500m - limits: - memory: 3Gi - cpu: 2 - volumeClaimTemplates: - - metadata: - name: elasticsearch-data - spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 20Gi - secureSettings: - - secretName: s3-credentials diff --git a/e2e/elasticsearch/templates/elasticsearch-cluster.yaml b/e2e/elasticsearch/templates/elasticsearch-cluster.yaml deleted file mode 100644 index 9480fbb..0000000 --- a/e2e/elasticsearch/templates/elasticsearch-cluster.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: elasticsearch.k8s.elastic.co/v1 -kind: Elasticsearch -metadata: - name: ${CLUSTER_NAME} -spec: - version: 9.2.4 - nodeSets: - - name: default - count: 3 - config: - node.store.allow_mmap: false - # S3 repository settings - s3.client.default.endpoint: "s3proxy-python-frontproxy.s3proxy:80" - s3.client.default.protocol: "http" - s3.client.default.path_style_access: true - podTemplate: - spec: - containers: - - name: elasticsearch - env: - - name: ES_JAVA_OPTS - value: "-Xms1g -Xmx1g" - resources: - requests: - memory: 2Gi - cpu: 500m - limits: - memory: 3Gi - cpu: 2 - volumeClaimTemplates: - - metadata: - name: elasticsearch-data - spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 20Gi - secureSettings: - - secretName: s3-credentials diff --git a/e2e/elasticsearch/templates/esrally-job.yaml b/e2e/elasticsearch/templates/esrally-job.yaml deleted file mode 100644 index f4d0be5..0000000 --- a/e2e/elasticsearch/templates/esrally-job.yaml +++ /dev/null @@ -1,52 +0,0 @@ -apiVersion: batch/v1 -kind: Job -metadata: - name: geonames-loader -spec: - backoffLimit: 2 - template: - spec: - restartPolicy: Never - containers: - - name: esrally - image: esrally:latest - imagePullPolicy: Never - env: - - name: ES_PASSWORD - valueFrom: - secretKeyRef: - name: ${CLUSTER_NAME}-es-elastic-user - key: elastic - command: - - /bin/bash - - -c - - | - set -e - ES_HOST="${CLUSTER_NAME}-es-http" - ES_URL="https://${ES_HOST}:9200" - - echo "=== Waiting for Elasticsearch to be ready ===" - until curl -sk -u "elastic:${ES_PASSWORD}" "${ES_URL}/_cluster/health" | grep -q '"status":"green"\|"status":"yellow"'; do - echo "Waiting for ES cluster..." - sleep 5 - done - echo "ES cluster is ready!" - - echo "=== Running esrally geonames track ===" - esrally race \ - --track=geonames \ - --target-hosts="${ES_HOST}:9200" \ - --pipeline=benchmark-only \ - --client-options="use_ssl:true,verify_certs:false,basic_auth_user:'elastic',basic_auth_password:'${ES_PASSWORD}'" \ - --include-tasks="delete-index,create-index,check-cluster-health,index-append,refresh-after-index,force-merge" \ - --on-error=abort \ - --kill-running-processes - - echo "esrally completed!" - resources: - requests: - memory: 4Gi - cpu: 1 - limits: - memory: 8Gi - cpu: 4 diff --git a/e2e/elasticsearch/templates/s3-credentials.yaml b/e2e/elasticsearch/templates/s3-credentials.yaml deleted file mode 100644 index f173e85..0000000 --- a/e2e/elasticsearch/templates/s3-credentials.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: v1 -kind: Secret -metadata: - name: s3-credentials -type: Opaque -stringData: - s3.client.default.access_key: "minioadmin" - s3.client.default.secret_key: "minioadmin" diff --git a/e2e/elasticsearch/test.sh b/e2e/elasticsearch/test.sh deleted file mode 100755 index 0c17a64..0000000 --- a/e2e/elasticsearch/test.sh +++ /dev/null @@ -1,299 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -ROOT_DIR="$(dirname "$SCRIPT_DIR")" -cd "$SCRIPT_DIR" - -# Source shared encryption verification -source "${SCRIPT_DIR}/../scripts/verify-encryption-k8s.sh" - -# Use isolated kubeconfig if not already set (running outside container) -if [ -z "${KUBECONFIG:-}" ]; then - export KUBECONFIG="${ROOT_DIR}/kubeconfig" - if [ ! -f "$KUBECONFIG" ]; then - echo "ERROR: Kubeconfig not found at $KUBECONFIG" - echo "Run ./cluster.sh up first" - exit 1 - fi -fi - -NAMESPACE="elasticsearch-test" -export CLUSTER_NAME="es-cluster" - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -NC='\033[0m' # No Color - -log_info() { echo -e "${GREEN}[INFO]${NC} $1"; } -log_error() { echo -e "${RED}[ERROR]${NC} $1"; } - -cleanup() { - log_info "Cleaning up..." - kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=false || true -} - -trap cleanup EXIT - -# ============================================================================ -# STEP 1: Create namespace and S3 credentials -# ============================================================================ -log_info "=== Step 1: Creating namespace and credentials ===" -log_info "(ECK operator already installed by cluster up)" - -kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - - -# Clean up any leftover data in the S3 bucket from previous runs (background) -log_info "Cleaning S3 bucket from previous test runs (background)..." -( - kubectl run bucket-cleanup --namespace "$NAMESPACE" \ - --image=mc:latest \ - --image-pull-policy=Never \ - --restart=Never \ - --command -- /bin/sh -c " - mc alias set minio http://minio.minio.svc.cluster.local:9000 minioadmin minioadmin >/dev/null 2>&1 - mc rm --recursive --force minio/es-snapshots/ 2>/dev/null || true - echo 'Bucket cleaned' - " 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=jsonpath='{.status.phase}'=Succeeded pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl delete pod -n "$NAMESPACE" bucket-cleanup --ignore-not-found >/dev/null 2>&1 || true -) & -BUCKET_CLEANUP_PID=$! - -# Create S3 credentials secret for Elasticsearch -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/s3-credentials.yaml" - -# ============================================================================ -# STEP 2: Deploy Elasticsearch cluster (3 nodes) -# ============================================================================ -log_info "=== Step 2: Deploying Elasticsearch cluster (3 nodes) ===" - -envsubst < "${SCRIPT_DIR}/templates/elasticsearch-cluster.yaml" | kubectl apply -n "$NAMESPACE" -f - - -# Wait for bucket cleanup (must complete before snapshot creation) -wait $BUCKET_CLEANUP_PID || true - -# ============================================================================ -# STEP 3: Start esrally job (pre-built image, waits for ES then runs) -# ============================================================================ -log_info "=== Step 3: Starting esrally job (using pre-built image) ===" - -# Wait for ECK to create the password secret (wait for existence first, then data) -log_info "Waiting for ES password secret..." -until kubectl get secret -n "$NAMESPACE" ${CLUSTER_NAME}-es-elastic-user &>/dev/null; do - echo " Waiting for secret to be created..." - sleep 2 -done -kubectl wait --namespace "$NAMESPACE" \ - --for=jsonpath='{.data.elastic}' secret/${CLUSTER_NAME}-es-elastic-user \ - --timeout=120s - -# Start esrally job NOW - it will do apt-get + pip install while ES is still starting -log_info "Starting esrally loader job..." -envsubst '$CLUSTER_NAME' < "${SCRIPT_DIR}/templates/esrally-job.yaml" | kubectl apply -n "$NAMESPACE" -f - - -# Follow esrally logs in background -kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod -l job-name=geonames-loader --timeout=120s 2>/dev/null || true -kubectl logs -n "$NAMESPACE" -f job/geonames-loader 2>/dev/null & -LOGS_PID=$! - -# ============================================================================ -# STEP 4: Wait for ES ready + register S3 repo -# ============================================================================ -log_info "=== Step 4: Waiting for ES cluster ===" - -kubectl wait --namespace "$NAMESPACE" \ - --for=jsonpath='{.status.phase}'=Ready elasticsearch/${CLUSTER_NAME} \ - --timeout=900s - -log_info "Elasticsearch cluster is ready" - -# Get ES password and register S3 repo -ES_PASSWORD=$(kubectl get secret -n "$NAMESPACE" ${CLUSTER_NAME}-es-elastic-user -o jsonpath='{.data.elastic}' | base64 -d) -ES_POD="${CLUSTER_NAME}-es-default-0" - -log_info "Registering S3 snapshot repository..." -kubectl exec -n "$NAMESPACE" "${ES_POD}" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" -X PUT "https://localhost:9200/_snapshot/s3_backup" \ - -H "Content-Type: application/json" \ - -d '{ - "type": "s3", - "settings": { - "bucket": "elasticsearch-backups", - "endpoint": "s3proxy-python-frontproxy.s3proxy:80", - "protocol": "http", - "path_style_access": true - } - }' -echo "" -log_info "S3 snapshot repository registered" - -# ============================================================================ -# STEP 5: Wait for esrally to complete -# ============================================================================ -log_info "=== Step 5: Waiting for esrally data loading ===" - -kubectl wait --namespace "$NAMESPACE" \ - --for=condition=complete job/geonames-loader \ - --timeout=3600s - -kill $LOGS_PID 2>/dev/null || true - -# Check job status -JOB_STATUS=$(kubectl get job -n "$NAMESPACE" geonames-loader -o jsonpath='{.status.succeeded}') -if [ "$JOB_STATUS" != "1" ]; then - log_error "Loader job failed!" - kubectl logs -n "$NAMESPACE" job/geonames-loader --tail=50 - exit 1 -fi - -log_info "Data loading complete" - -# Refresh index via kubectl exec (no port-forward needed) -kubectl exec -n "$NAMESPACE" "${CLUSTER_NAME}-es-default-0" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" -X POST "https://localhost:9200/geonames/_refresh" > /dev/null - -# Get cluster stats via kubectl exec -log_info "Cluster stats:" -kubectl exec -n "$NAMESPACE" "${CLUSTER_NAME}-es-default-0" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" "https://localhost:9200/_cat/indices?v" -echo "" - -TOTAL_DOCS=$(kubectl exec -n "$NAMESPACE" "${CLUSTER_NAME}-es-default-0" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" "https://localhost:9200/_cat/count?h=count" | tr -d '[:space:]') -log_info "Total documents: $TOTAL_DOCS" - -# ============================================================================ -# STEP 6: Create snapshot (backup) -# ============================================================================ -log_info "=== Step 6: Creating snapshot ===" - -SNAPSHOT_NAME="snapshot-$(date +%Y%m%d-%H%M%S)" - -kubectl exec -n "$NAMESPACE" "${CLUSTER_NAME}-es-default-0" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" -X PUT "https://localhost:9200/_snapshot/s3_backup/${SNAPSHOT_NAME}?wait_for_completion=true" \ - -H "Content-Type: application/json" \ - -d '{ - "indices": "geonames", - "ignore_unavailable": true, - "include_global_state": false - }' - -echo "" -log_info "Snapshot ${SNAPSHOT_NAME} created" - -# Get snapshot info -kubectl exec -n "$NAMESPACE" "${CLUSTER_NAME}-es-default-0" -- \ - curl -sk -u "elastic:${ES_PASSWORD}" "https://localhost:9200/_snapshot/s3_backup/${SNAPSHOT_NAME}" | jq . - -# ============================================================================ -# STEP 7: Verify encryption + Delete cluster + Create new cluster (ALL PARALLEL) -# ============================================================================ -log_info "=== Step 7: Parallel - verify encryption, delete old, create new ===" - -# 1. Start encryption verification in background -verify_encryption "elasticsearch-backups" "" "$NAMESPACE" "ALL" & -VERIFY_PID=$! - -# 2. Delete old cluster in background -( - kubectl delete elasticsearch -n "$NAMESPACE" ${CLUSTER_NAME} --wait - kubectl wait --namespace "$NAMESPACE" --for=delete pod -l elasticsearch.k8s.elastic.co/cluster-name=${CLUSTER_NAME} --timeout=120s 2>/dev/null || true - log_info "✓ Old cluster deleted" -) & -DELETE_PID=$! - -# 3. Create new cluster immediately (different name, can coexist) -log_info "Creating restored cluster (parallel with deletion)..." -envsubst < "${SCRIPT_DIR}/templates/elasticsearch-cluster-restore.yaml" | kubectl apply -n "$NAMESPACE" -f - - -# Wait for all parallel operations -wait $VERIFY_PID || { log_error "Encryption verification failed"; exit 1; } -log_info "✓ Encryption verified" - -wait $DELETE_PID || { log_error "Old cluster deletion failed"; exit 1; } - -log_info "Waiting for restored cluster to be ready..." -kubectl wait --namespace "$NAMESPACE" \ - --for=jsonpath='{.status.phase}'=Ready elasticsearch/${CLUSTER_NAME}-restored \ - --timeout=900s - -# Get new password -NEW_ES_PASSWORD=$(kubectl get secret -n "$NAMESPACE" ${CLUSTER_NAME}-restored-es-elastic-user -o jsonpath='{.data.elastic}' | base64 -d) -RESTORED_POD="${CLUSTER_NAME}-restored-es-default-0" - -# ============================================================================ -# STEP 8: Register repository and restore snapshot -# ============================================================================ -log_info "=== Step 8: Restoring from snapshot ===" - -# Register repository via kubectl exec -kubectl exec -n "$NAMESPACE" "${RESTORED_POD}" -- \ - curl -sk -u "elastic:${NEW_ES_PASSWORD}" -X PUT "https://localhost:9200/_snapshot/s3_backup" \ - -H "Content-Type: application/json" \ - -d '{ - "type": "s3", - "settings": { - "bucket": "elasticsearch-backups", - "endpoint": "s3proxy-python-frontproxy.s3proxy:80", - "protocol": "http", - "path_style_access": true - } - }' - -echo "" - -# Restore snapshot -log_info "Restoring snapshot ${SNAPSHOT_NAME}..." -kubectl exec -n "$NAMESPACE" "${RESTORED_POD}" -- \ - curl -sk -u "elastic:${NEW_ES_PASSWORD}" -X POST "https://localhost:9200/_snapshot/s3_backup/${SNAPSHOT_NAME}/_restore?wait_for_completion=true" \ - -H "Content-Type: application/json" \ - -d '{ - "indices": "geonames", - "ignore_unavailable": true, - "include_global_state": false - }' - -echo "" -log_info "Restore complete!" - -# ============================================================================ -# STEP 9: Validate restored data -# ============================================================================ -log_info "=== Step 9: Validating restored data ===" - -# Wait for index to be green -kubectl exec -n "$NAMESPACE" "${RESTORED_POD}" -- \ - curl -sk -u "elastic:${NEW_ES_PASSWORD}" \ - "https://localhost:9200/_cluster/health/geonames?wait_for_status=green&timeout=60s" > /dev/null 2>&1 || true - -log_info "Restored indices:" -kubectl exec -n "$NAMESPACE" "${RESTORED_POD}" -- \ - curl -sk -u "elastic:${NEW_ES_PASSWORD}" "https://localhost:9200/_cat/indices?v" -echo "" - -RESTORED_DOCS=$(kubectl exec -n "$NAMESPACE" "${RESTORED_POD}" -- \ - curl -sk -u "elastic:${NEW_ES_PASSWORD}" "https://localhost:9200/_cat/count?h=count" | tr -d '[:space:]') -log_info "Total documents after restore: $RESTORED_DOCS" - -# Validate counts -if [ "$TOTAL_DOCS" = "$RESTORED_DOCS" ]; then - log_info "=== VALIDATION PASSED: Document counts match! ===" - log_info "Original: $TOTAL_DOCS documents" - log_info "Restored: $RESTORED_DOCS documents" -else - log_error "=== VALIDATION FAILED: Document counts do not match! ===" - log_error "Original: $TOTAL_DOCS documents" - log_error "Restored: $RESTORED_DOCS documents" - exit 1 -fi - -# ============================================================================ -# STEP 10: Cleanup -# ============================================================================ -log_info "=== Step 10: Cleanup ===" -log_info "Test completed successfully!" - -log_info "=== Elasticsearch Backup/Restore Test PASSED ===" diff --git a/e2e/postgres/templates/backup.yaml b/e2e/postgres/templates/backup.yaml deleted file mode 100644 index c867dbc..0000000 --- a/e2e/postgres/templates/backup.yaml +++ /dev/null @@ -1,7 +0,0 @@ -apiVersion: postgresql.cnpg.io/v1 -kind: Backup -metadata: - name: ${CLUSTER_NAME}-backup-1 -spec: - cluster: - name: ${CLUSTER_NAME} diff --git a/e2e/postgres/templates/postgres-cluster-restore.yaml b/e2e/postgres/templates/postgres-cluster-restore.yaml deleted file mode 100644 index a5a8937..0000000 --- a/e2e/postgres/templates/postgres-cluster-restore.yaml +++ /dev/null @@ -1,37 +0,0 @@ -apiVersion: postgresql.cnpg.io/v1 -kind: Cluster -metadata: - name: ${CLUSTER_NAME}-restored -spec: - instances: 3 - imageName: ghcr.io/cloudnative-pg/postgresql:17.5 - - storage: - size: 20Gi - - resources: - requests: - memory: "1Gi" - cpu: "500m" - limits: - memory: "2Gi" - cpu: "2" - - bootstrap: - recovery: - source: ${CLUSTER_NAME} - - externalClusters: - - name: ${CLUSTER_NAME} - barmanObjectStore: - destinationPath: "s3://postgres-backups/" - endpointURL: "http://s3proxy-python-frontproxy.s3proxy:80" - s3Credentials: - accessKeyId: - name: s3-credentials - key: ACCESS_KEY_ID - secretAccessKey: - name: s3-credentials - key: ACCESS_SECRET_KEY - wal: - compression: gzip diff --git a/e2e/postgres/templates/postgres-cluster.yaml b/e2e/postgres/templates/postgres-cluster.yaml deleted file mode 100644 index 6f6adfe..0000000 --- a/e2e/postgres/templates/postgres-cluster.yaml +++ /dev/null @@ -1,46 +0,0 @@ -apiVersion: postgresql.cnpg.io/v1 -kind: Cluster -metadata: - name: ${CLUSTER_NAME} -spec: - instances: 3 - imageName: ghcr.io/cloudnative-pg/postgresql:17.5 - - storage: - size: 20Gi - - resources: - requests: - memory: "1Gi" - cpu: "500m" - limits: - memory: "2Gi" - cpu: "2" - - postgresql: - parameters: - max_connections: "200" - shared_buffers: "256MB" - - backup: - barmanObjectStore: - destinationPath: "s3://postgres-backups/" - endpointURL: "http://s3proxy-python-frontproxy.s3proxy:80" - s3Credentials: - accessKeyId: - name: s3-credentials - key: ACCESS_KEY_ID - secretAccessKey: - name: s3-credentials - key: ACCESS_SECRET_KEY - wal: - compression: gzip - data: - # Compression intentionally OFF: pgbench data is highly compressible, so - # gzip would shrink the backup below one 512MB chunk and we'd never get - # concurrent large uploads. Uncompressed, the dataset spans several 512MB - # chunks which (with jobs=4) reproduces the multipart-upload OOM. - jobs: 4 - additionalCommandArgs: - - "--min-chunk-size=512MB" - retentionPolicy: "7d" diff --git a/e2e/postgres/templates/s3-credentials.yaml b/e2e/postgres/templates/s3-credentials.yaml deleted file mode 100644 index b607c28..0000000 --- a/e2e/postgres/templates/s3-credentials.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: v1 -kind: Secret -metadata: - name: s3-credentials -type: Opaque -stringData: - ACCESS_KEY_ID: "minioadmin" - ACCESS_SECRET_KEY: "minioadmin" diff --git a/e2e/postgres/test.sh b/e2e/postgres/test.sh deleted file mode 100755 index 23349e2..0000000 --- a/e2e/postgres/test.sh +++ /dev/null @@ -1,368 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -ROOT_DIR="$(dirname "$SCRIPT_DIR")" -cd "$SCRIPT_DIR" - -# Source shared encryption verification -source "${SCRIPT_DIR}/../scripts/verify-encryption-k8s.sh" - -# Use isolated kubeconfig if not already set (running outside container) -if [ -z "${KUBECONFIG:-}" ]; then - export KUBECONFIG="${ROOT_DIR}/kubeconfig" - if [ ! -f "$KUBECONFIG" ]; then - echo "ERROR: Kubeconfig not found at $KUBECONFIG" - echo "Run ./cluster.sh up first" - exit 1 - fi -fi - -NAMESPACE="postgres-test" -export CLUSTER_NAME="pg-cluster" -DATA_SIZE_GB=3 -SCALE_FACTOR=$((DATA_SIZE_GB * 70)) # pgbench scale: ~15MB per scale factor - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -NC='\033[0m' # No Color - -log_info() { echo -e "${GREEN}[INFO]${NC} $1"; } -log_warn() { echo -e "${YELLOW}[WARN]${NC} $1"; } -log_error() { echo -e "${RED}[ERROR]${NC} $1"; } - -S3_ENDPOINT="http://s3proxy-python-frontproxy.s3proxy:80" -S3_WAL_CHECK_POD="s3-wal-check" - -ensure_s3_wal_checker() { - if kubectl get pod -n "$NAMESPACE" "$S3_WAL_CHECK_POD" >/dev/null 2>&1; then - kubectl wait -n "$NAMESPACE" --for=condition=Ready "pod/${S3_WAL_CHECK_POD}" --timeout=120s - return - fi - kubectl run "$S3_WAL_CHECK_POD" --restart=Never -n "$NAMESPACE" \ - --image=amazon/aws-cli:2.15.0 \ - --overrides='{"spec":{"containers":[{"name":"'"$S3_WAL_CHECK_POD"'","image":"amazon/aws-cli:2.15.0","command":["sleep","3600"],"envFrom":[{"secretRef":{"name":"s3-credentials"}}]}]}}' - kubectl wait -n "$NAMESPACE" --for=condition=Ready "pod/${S3_WAL_CHECK_POD}" --timeout=120s -} - -wait_for_end_wal_in_s3() { - local end_wal="$1" - local timeline_prefix="${end_wal:0:16}" - local wal_object="pg-cluster/wals/${timeline_prefix}/${end_wal}.gz" - - log_info "Waiting for backup end WAL in S3: ${wal_object}" - ensure_s3_wal_checker - - local deadline=$((SECONDS + 600)) - while [ "$SECONDS" -lt "$deadline" ]; do - if kubectl exec -n "$NAMESPACE" "$S3_WAL_CHECK_POD" -- sh -c " - export AWS_ACCESS_KEY_ID=\$ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY=\$ACCESS_SECRET_KEY - aws --endpoint-url ${S3_ENDPOINT} s3 ls s3://postgres-backups/${wal_object} >/dev/null 2>&1 - "; then - log_info "✓ End WAL archived: ${end_wal}.gz" - return 0 - fi - sleep 5 - done - - log_error "Timeout waiting for end WAL ${end_wal}.gz in S3" - return 1 -} - -cleanup() { - log_info "Cleaning up..." - kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=false || true -} - -trap cleanup EXIT - -# ============================================================================ -# STEP 1: Create namespace and S3 credentials -# ============================================================================ -log_info "=== Step 1: Creating namespace and credentials ===" - -kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - - -# Clean up any leftover data in the S3 bucket from previous runs (background) -# CNPG's barman-cloud-check-wal-archive fails with "Expected empty archive" if bucket has old data -log_info "Cleaning S3 bucket from previous test runs (background)..." -( - kubectl run bucket-cleanup --namespace "$NAMESPACE" \ - --image=mc:latest \ - --image-pull-policy=Never \ - --restart=Never \ - --command -- /bin/sh -c " - mc alias set minio http://minio.minio.svc.cluster.local:9000 minioadmin minioadmin >/dev/null 2>&1 - mc rm --recursive --force minio/postgres-backups/ 2>/dev/null || true - echo 'Bucket cleaned' - " 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl wait --namespace "$NAMESPACE" --for=jsonpath='{.status.phase}'=Succeeded pod/bucket-cleanup --timeout=60s 2>/dev/null || true - kubectl delete pod -n "$NAMESPACE" bucket-cleanup --ignore-not-found >/dev/null 2>&1 || true -) & -BUCKET_CLEANUP_PID=$! - -# Create S3 credentials secret -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/s3-credentials.yaml" - -# ============================================================================ -# STEP 2: Deploy PostgreSQL cluster (3 replicas) -# ============================================================================ -log_info "=== Step 2: Deploying PostgreSQL cluster (3 replicas) ===" - -envsubst < "${SCRIPT_DIR}/templates/postgres-cluster.yaml" | kubectl apply -n "$NAMESPACE" -f - - -# Wait for bucket cleanup before cluster tries to access S3 -wait $BUCKET_CLEANUP_PID || true - -log_info "Waiting for PostgreSQL cluster to be ready..." -kubectl wait --namespace "$NAMESPACE" \ - --for=condition=Ready cluster/${CLUSTER_NAME} \ - --timeout=600s - -log_info "PostgreSQL cluster is ready" - -# ============================================================================ -# STEP 3: Generate test data using pgbench -# ============================================================================ -log_info "=== Step 3: Generating ${DATA_SIZE_GB}GB of test data ===" - -# Get the primary pod -PRIMARY_POD=$(kubectl get pods -n "$NAMESPACE" -l cnpg.io/cluster=${CLUSTER_NAME},role=primary -o jsonpath='{.items[0].metadata.name}') -log_info "Primary pod: $PRIMARY_POD" - -# Initialize pgbench tables -log_info "Initializing pgbench with scale factor $SCALE_FACTOR (this may take a while)..." -kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- \ - pgbench -i -s "$SCALE_FACTOR" -U postgres app - -# Run some transactions to generate more data -log_info "Running pgbench transactions..." -kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- \ - pgbench -U postgres -c 10 -j 2 -t 10000 app - -# Create additional tables with fake data -log_info "Creating additional tables with fake data..." -kubectl exec -i -n "$NAMESPACE" "$PRIMARY_POD" -- psql -U postgres app <<'EOSQL' --- Enable pgcrypto for gen_random_bytes() -CREATE EXTENSION IF NOT EXISTS pgcrypto; - --- Create users table with fake data -CREATE TABLE IF NOT EXISTS users ( - id SERIAL PRIMARY KEY, - username VARCHAR(50), - email VARCHAR(100), - created_at TIMESTAMP DEFAULT NOW(), - data BYTEA -); - --- Create orders table -CREATE TABLE IF NOT EXISTS orders ( - id SERIAL PRIMARY KEY, - user_id INTEGER, - amount DECIMAL(10,2), - status VARCHAR(20), - created_at TIMESTAMP DEFAULT NOW(), - metadata JSONB -); - --- Create products table -CREATE TABLE IF NOT EXISTS products ( - id SERIAL PRIMARY KEY, - name VARCHAR(100), - description TEXT, - price DECIMAL(10,2), - inventory INTEGER, - attributes JSONB -); - --- Insert fake users (with some random binary data for size) -INSERT INTO users (username, email, data) -SELECT - 'user_' || i, - 'user_' || i || '@example.com', - gen_random_bytes(1000) -FROM generate_series(1, 20000) AS i; - --- Insert fake orders -INSERT INTO orders (user_id, amount, status, metadata) -SELECT - (random() * 20000)::int, - (random() * 1000)::decimal(10,2), - (ARRAY['pending', 'completed', 'shipped', 'cancelled'])[floor(random() * 4 + 1)], - jsonb_build_object('source', 'web', 'version', floor(random() * 10)) -FROM generate_series(1, 100000) AS i; - --- Insert fake products -INSERT INTO products (name, description, price, inventory, attributes) -SELECT - 'Product ' || i, - 'Description for product ' || i || '. ' || repeat('Lorem ipsum dolor sit amet. ', 10), - (random() * 500)::decimal(10,2), - (random() * 1000)::int, - jsonb_build_object('category', 'cat_' || (i % 50), 'tags', ARRAY['tag1', 'tag2']) -FROM generate_series(1, 20000) AS i; - --- Create indexes -CREATE INDEX IF NOT EXISTS idx_users_email ON users(email); -CREATE INDEX IF NOT EXISTS idx_orders_user_id ON orders(user_id); -CREATE INDEX IF NOT EXISTS idx_orders_status ON orders(status); -CREATE INDEX IF NOT EXISTS idx_products_price ON products(price); - --- Analyze tables -ANALYZE; -EOSQL - -# Get database size -DB_SIZE=$(kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- psql -U postgres -t -c "SELECT pg_size_pretty(pg_database_size('app'));") -log_info "Database size: $DB_SIZE" - -# Update statistics and get row counts -log_info "Analyzing tables and getting row counts:" -kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- psql -U postgres app -c "ANALYZE;" -kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- psql -U postgres app -c " -SELECT - schemaname || '.' || relname as table_name, - n_live_tup as estimated_rows -FROM pg_stat_user_tables -WHERE n_live_tup > 0 -ORDER BY n_live_tup DESC; -" - -# Store checksum for validation -CHECKSUM=$(kubectl exec -n "$NAMESPACE" "$PRIMARY_POD" -- psql -U postgres -t -d app -c " -SELECT md5(string_agg(md5(row::text), '')) -FROM ( - SELECT * FROM users ORDER BY id LIMIT 1000 -) row; -") -log_info "Data checksum (first 1000 users): $CHECKSUM" - -# ============================================================================ -# STEP 4: Trigger backup to S3 -# ============================================================================ -log_info "=== Step 4: Triggering backup to S3 ===" - -envsubst < "${SCRIPT_DIR}/templates/backup.yaml" | kubectl apply -n "$NAMESPACE" -f - - -log_info "Waiting for backup to complete..." -kubectl wait --namespace "$NAMESPACE" \ - --for=jsonpath='{.status.phase}'=completed backup/${CLUSTER_NAME}-backup-1 \ - --timeout=1800s - -log_info "Backup completed!" -kubectl get backup -n "$NAMESPACE" ${CLUSTER_NAME}-backup-1 -o yaml | grep -A5 "status:" - -# ============================================================================ -# STEP 4b: Assert the s3proxy pods survived (the OOM we are gating against) -# ============================================================================ -log_info "=== Checking s3proxy pods were not OOM-killed during backup ===" -oom_found=0 -for p in $(kubectl get pods -n s3proxy -l app.kubernetes.io/name=s3proxy-python,app.kubernetes.io/component=server -o name); do - rc=$(kubectl get -n s3proxy "$p" -o jsonpath='{.status.containerStatuses[0].restartCount}' 2>/dev/null || echo 0) - reason=$(kubectl get -n s3proxy "$p" -o jsonpath='{.status.containerStatuses[0].lastState.terminated.reason}' 2>/dev/null || echo "") - log_info " $p restarts=${rc:-0} lastTerminated=${reason:-none}" - if [ "${reason}" = "OOMKilled" ] || [ "${rc:-0}" -gt 0 ]; then oom_found=1; fi -done -if [ "$oom_found" -eq 1 ]; then - log_error "s3proxy was OOM-killed/restarted during backup" - exit 1 -fi -log_info "✓ s3proxy survived the backup" - -# ============================================================================ -# STEP 4c: Wait for backup end WAL to reach S3 (restore needs it) -# CNPG may mark backup completed before the archiver uploads the final segment. -# ============================================================================ -END_WAL=$(kubectl get backup -n "$NAMESPACE" "${CLUSTER_NAME}-backup-1" -o jsonpath='{.status.endWal}') -if [ -z "$END_WAL" ]; then - log_error "Backup status.endWal is empty" - exit 1 -fi -log_info "=== Step 4c: Waiting for end WAL ${END_WAL} in S3 ===" -wait_for_end_wal_in_s3 "$END_WAL" - -# ============================================================================ -# STEP 5: Verify encryption, restore, then delete source cluster -# Keep the source cluster alive until end WAL is archived and restore succeeds. -# ============================================================================ -log_info "=== Step 5: Verify encryption and restore from backup ===" - -verify_encryption "postgres-backups" "" "$NAMESPACE" ".gz|.tar|.backup|.data" & -VERIFY_PID=$! - -log_info "Creating restored cluster..." -envsubst < "${SCRIPT_DIR}/templates/postgres-cluster-restore.yaml" | kubectl apply -n "$NAMESPACE" -f - - -wait $VERIFY_PID || { log_error "Encryption verification failed"; exit 1; } -log_info "✓ Encryption verified" - -log_info "Waiting for restored cluster to be ready..." -kubectl wait --namespace "$NAMESPACE" \ - --for=condition=Ready cluster/${CLUSTER_NAME}-restored \ - --timeout=1800s - -log_info "Restored cluster is ready!" - -log_info "Deleting source cluster (no longer needed)..." -kubectl delete cluster -n "$NAMESPACE" "${CLUSTER_NAME}" --wait -kubectl wait --namespace "$NAMESPACE" \ - --for=delete pod -l "cnpg.io/cluster=${CLUSTER_NAME}" \ - --timeout=300s || true -log_info "✓ Old cluster deleted" - -# ============================================================================ -# STEP 6: Validate restored data -# ============================================================================ -log_info "=== Step 6: Validating restored data ===" - -RESTORED_PRIMARY=$(kubectl get pods -n "$NAMESPACE" -l cnpg.io/cluster=${CLUSTER_NAME}-restored,role=primary -o jsonpath='{.items[0].metadata.name}') -log_info "Restored primary pod: $RESTORED_PRIMARY" - -# Get database size -RESTORED_DB_SIZE=$(kubectl exec -n "$NAMESPACE" "$RESTORED_PRIMARY" -- psql -U postgres -t -c "SELECT pg_size_pretty(pg_database_size('app'));") -log_info "Restored database size: $RESTORED_DB_SIZE" - -# Update statistics and get row counts -log_info "Analyzing restored tables and getting row counts:" -kubectl exec -n "$NAMESPACE" "$RESTORED_PRIMARY" -- psql -U postgres app -c "ANALYZE;" -kubectl exec -n "$NAMESPACE" "$RESTORED_PRIMARY" -- psql -U postgres app -c " -SELECT - schemaname || '.' || relname as table_name, - n_live_tup as estimated_rows -FROM pg_stat_user_tables -WHERE n_live_tup > 0 -ORDER BY n_live_tup DESC; -" - -# Validate checksum -RESTORED_CHECKSUM=$(kubectl exec -n "$NAMESPACE" "$RESTORED_PRIMARY" -- psql -U postgres -t -d app -c " -SELECT md5(string_agg(md5(row::text), '')) -FROM ( - SELECT * FROM users ORDER BY id LIMIT 1000 -) row; -") -log_info "Restored data checksum: $RESTORED_CHECKSUM" - -if [ "$CHECKSUM" = "$RESTORED_CHECKSUM" ]; then - log_info "=== VALIDATION PASSED: Checksums match! ===" -else - log_error "=== VALIDATION FAILED: Checksums do not match! ===" - log_error "Original: $CHECKSUM" - log_error "Restored: $RESTORED_CHECKSUM" - exit 1 -fi - -# ============================================================================ -# STEP 7: Cleanup -# ============================================================================ -log_info "=== Step 7: Cleanup ===" -log_info "Test completed successfully!" -log_info "" -log_info "The namespace $NAMESPACE will be deleted on script exit." -log_info "To keep the restored cluster for inspection, press Ctrl+C within 10 seconds..." -sleep 10 - -log_info "=== PostgreSQL Backup/Restore Test PASSED ===" diff --git a/e2e/s3-compatibility/templates/s3-tests-config.yaml b/e2e/s3-compatibility/templates/s3-tests-config.yaml deleted file mode 100644 index 43cfe17..0000000 --- a/e2e/s3-compatibility/templates/s3-tests-config.yaml +++ /dev/null @@ -1,63 +0,0 @@ -apiVersion: v1 -kind: ConfigMap -metadata: - name: s3-tests-config -data: - s3tests.conf: | - [DEFAULT] - # Front proxy service (even load distribution across s3proxy pods) - host = s3proxy-python-frontproxy.s3proxy - port = 80 - is_secure = False - ssl_verify = False - - [fixtures] - bucket prefix = s3compat-{random}- - - [s3 main] - user_id = testuser - display_name = Test User - email = testuser@example.com - access_key = minioadmin - secret_key = minioadmin - api_name = default - - [s3 alt] - user_id = altuser - display_name = Alt User - email = altuser@example.com - access_key = minioadmin - secret_key = minioadmin - - [s3 tenant] - user_id = tenantuser - display_name = Tenant User - email = tenantuser@example.com - access_key = minioadmin - secret_key = minioadmin - tenant = test - - [iam] - access_key = minioadmin - secret_key = minioadmin - display_name = IAM User - user_id = iamuser - email = iamuser@example.com - - [iam root] - access_key = minioadmin - secret_key = minioadmin - user_id = iamrootuser - email = iamroot@example.com - - [iam alt root] - access_key = minioadmin - secret_key = minioadmin - user_id = iamaltrootuser - email = iamaltroot@example.com - - [webidentity] - token = - aud = - thumbprint = - KC_REALM = diff --git a/e2e/s3-compatibility/templates/s3-tests-runner-job.yaml b/e2e/s3-compatibility/templates/s3-tests-runner-job.yaml deleted file mode 100644 index ec6c164..0000000 --- a/e2e/s3-compatibility/templates/s3-tests-runner-job.yaml +++ /dev/null @@ -1,204 +0,0 @@ -apiVersion: batch/v1 -kind: Job -metadata: - name: s3-tests-runner -spec: - backoffLimit: 0 - ttlSecondsAfterFinished: 3600 - template: - spec: - restartPolicy: Never - containers: - - name: s3-tests - image: python:3.12-slim - volumeMounts: - - name: config - mountPath: /config - - name: results - mountPath: /results - env: - - name: S3TEST_CONF - value: /config/s3tests.conf - command: - - /bin/bash - - -c - - | - set -e - - echo "=== Installing dependencies ===" - apt-get update -qq && apt-get install -y -qq git > /dev/null - - echo "=== Cloning s3-tests ===" - git clone --depth 1 https://github.com/ceph/s3-tests.git /s3-tests - cd /s3-tests - - echo "=== Installing s3-tests ===" - pip install --quiet -r requirements.txt - pip install --quiet -e . - pip install --quiet pytest pytest-json-report pytz - - echo "=== Running S3 compatibility tests ===" - echo "" - echo "Running core S3 operations tests..." - echo "Note: Many tests will fail by design (features handled by backend S3)" - echo "" - - # Discover the test file structure - echo "Discovering test structure..." - ls -la - echo "" - find . -name "test_s3*.py" -o -name "*test*.py" 2>/dev/null | head -20 || true - echo "" - - # Find the correct test file path - TEST_FILE=$(find . -path "*/functional/test_s3.py" -o -path "*/boto3/test_s3.py" 2>/dev/null | head -1) - if [ -z "$TEST_FILE" ]; then - # Try alternative locations - TEST_FILE=$(find . -name "test_s3.py" 2>/dev/null | head -1) - fi - - if [ -z "$TEST_FILE" ]; then - echo "ERROR: Could not find test_s3.py in the repository" - echo "Repository structure:" - find . -type f -name "*.py" | head -30 - exit 1 - fi - - echo "Found test file at: $TEST_FILE" - echo "" - - # Test patterns for proxy-relevant operations: - # - Bucket operations: list, create, delete, head - # - Object operations: put, get, delete, head, metadata - # - Multipart uploads - # - Copy operations - # - Range requests - # - # Excluded tests (features delegated to backend or not supported): - # - ACL tests (acl, grant, canned_acl) - # - Policy tests (policy) - # - Versioning tests (versioning, version) - # - Anonymous access tests (anon, anonymous, public) - # - Checksum algorithms (checksum, crc32, sha256, sha1) - # - Object lock (object_lock, retention, legal_hold) - # - Website/CORS (website, cors) - # - Lifecycle (lifecycle, expiration) - # - SSE/encryption tests that conflict (sse_s3, sse_c, sse_kms) - # Exclude tests for features we don't support or delegate to backend - # - ACL/policy/versioning: delegated to backend - # - Multi-object delete key limit: backend specific limitation - # - Multipart validation errors: MinIO returns 404 instead of 400 - # - PartsCount: advanced multipart feature not supported - # - Extended headers: rgw-specific headers - # - return_data: requires GetObjectAcl which is delegated - # - put_object_current: versioning-dependent conditional PUT - # - multipart_copy_special_names: SigV4 mismatch with URL-encoded special chars - # - delimiter_percentage, delimiter_prefix_ends_with_delimiter, - # delimiter_prefix_underscore, delimiter_unreadable: MinIO NextMarker quirks - # Note: Conditional headers (if*) are only implemented for GET, not PUT/complete_multipart - EXCLUDE_PATTERN="acl or policy or versioning or version or anon or anonymous or public or \ - checksum or crc32 or sha256 or sha1 or object_lock or retention or legal_hold or \ - website or cors or lifecycle or expiration or sse_s3 or sse_c or sse_kms or \ - grant or canned or encryption or \ - ownership or torrent or unordered or attributes or \ - not_owned or nonowner or rgw or \ - key_limit or \ - upload_empty or size_too_small or missing_part or incorrect_etag or \ - get_part or extended or return_data or \ - put_object_ifmatch or put_object_ifnonmatch or put_object_if_match or put_object_if_none or \ - multipart_put_object_if or multipart_put_current or \ - put_object_current or \ - multipart_copy_special_names or \ - delimiter_percentage or delimiter_prefix_ends_with_delimiter or \ - delimiter_prefix_underscore or delimiter_unreadable" - - TEST_PATTERN="((test_bucket_list) or \ - (test_bucket_create) or \ - (test_bucket_delete) or \ - (test_bucket_head) or \ - (test_object_write) or \ - (test_object_read) or \ - (test_object_head) or \ - (test_get_obj) or \ - (test_put_obj) or \ - (test_multipart) or \ - (test_object_copy) or \ - (test_ranged) or \ - (test_multi_object_delete)) and not ($EXCLUDE_PATTERN)" - - # Run tests and capture results - echo "" - echo "Running S3 compatibility tests with pattern matching..." - echo "Pattern: tests for bucket ops, object ops, multipart, copy, range requests" - echo "(excluding ACL and policy tests which are delegated to backend)" - echo "" - - pytest "$TEST_FILE" \ - -k "$TEST_PATTERN" \ - --json-report --json-report-file=/results/report.json \ - -v --tb=short 2>&1 | tee /results/output.txt || true - - echo "" - echo "=== Test Results Summary ===" - - # Parse and display results - python3 << 'PYTHON' - import json - import sys - - try: - with open('/results/report.json') as f: - report = json.load(f) - - summary = report.get('summary', {}) - passed = summary.get('passed', 0) - failed = summary.get('failed', 0) - error = summary.get('error', 0) - skipped = summary.get('skipped', 0) - total = summary.get('total', 0) - - print(f"\nTotal: {total}") - print(f"Passed: {passed}") - print(f"Failed: {failed}") - print(f"Error: {error}") - print(f"Skipped: {skipped}") - - if failed > 0 or error > 0: - print("\n=== Failed Tests ===") - for test in report.get('tests', []): - if test.get('outcome') in ('failed', 'error'): - print(f"\n❌ {test.get('nodeid')}") - call = test.get('call', {}) - if call.get('longrepr'): - # Print just the last part of the error - lines = str(call['longrepr']).split('\n') - for line in lines[-5:]: - print(f" {line}") - - # Save summary for easy access - with open('/results/summary.txt', 'w') as f: - f.write(f"passed={passed}\n") - f.write(f"failed={failed}\n") - f.write(f"error={error}\n") - f.write(f"total={total}\n") - - except Exception as e: - print(f"Error parsing results: {e}") - sys.exit(1) - PYTHON - - echo "" - echo "=== S3 Compatibility Test Complete ===" - resources: - requests: - memory: 512Mi - cpu: 500m - limits: - memory: 1Gi - cpu: 2 - volumes: - - name: config - configMap: - name: s3-tests-config - - name: results - emptyDir: {} diff --git a/e2e/s3-compatibility/test.sh b/e2e/s3-compatibility/test.sh deleted file mode 100755 index d354977..0000000 --- a/e2e/s3-compatibility/test.sh +++ /dev/null @@ -1,111 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -ROOT_DIR="$(dirname "$SCRIPT_DIR")" -cd "$SCRIPT_DIR" - -# Use isolated kubeconfig if not already set (running outside container) -if [ -z "${KUBECONFIG:-}" ]; then - export KUBECONFIG="${ROOT_DIR}/kubeconfig" - if [ ! -f "$KUBECONFIG" ]; then - echo "ERROR: Kubeconfig not found at $KUBECONFIG" - echo "Run ./cluster.sh up first" - exit 1 - fi -fi - -NAMESPACE="s3-compat-test" - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -NC='\033[0m' # No Color - -log_info() { echo -e "${GREEN}[INFO]${NC} $1"; } -log_warn() { echo -e "${YELLOW}[WARN]${NC} $1"; } -log_error() { echo -e "${RED}[ERROR]${NC} $1"; } - -cleanup() { - log_info "Cleaning up..." - kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=false || true -} - -trap cleanup EXIT - -# ============================================================================ -# STEP 1: Create namespace -# ============================================================================ -log_info "=== Step 1: Creating namespace ===" - -kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - - -# ============================================================================ -# STEP 2: Create test configuration -# ============================================================================ -log_info "=== Step 2: Creating s3-tests configuration ===" - -# Create ConfigMap with s3-tests config -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/s3-tests-config.yaml" - -log_info "Configuration created" - -# ============================================================================ -# STEP 3: Run s3-tests -# ============================================================================ -log_info "=== Step 3: Running Ceph s3-tests ===" - -# Create Job to run tests -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/s3-tests-runner-job.yaml" - -log_info "Test job created, waiting for completion..." - -# Wait for pod to start -sleep 10 - -# Follow logs -kubectl logs -n "$NAMESPACE" -f job/s3-tests-runner 2>/dev/null || true - -# Wait for job completion -kubectl wait --namespace "$NAMESPACE" \ - --for=condition=complete job/s3-tests-runner \ - --timeout=1800s || true - -# Check job status -JOB_STATUS=$(kubectl get job -n "$NAMESPACE" s3-tests-runner -o jsonpath='{.status.succeeded}' 2>/dev/null || echo "0") - -if [ "$JOB_STATUS" == "1" ]; then - log_info "=== S3 Compatibility Tests Completed ===" -else - log_warn "Tests completed with some failures (expected for proxy architecture)" -fi - -# ============================================================================ -# STEP 4: Summary -# ============================================================================ -log_info "=== Step 4: Summary ===" - -echo "" -echo "The Ceph s3-tests validate S3 API compatibility." -echo "" -echo "Excluded tests (delegated to backend or not supported):" -echo " - ACL/Grant tests (delegated to backend)" -echo " - Policy tests (delegated to backend)" -echo " - Versioning tests (delegated to backend)" -echo " - Lifecycle tests (delegated to backend)" -echo " - CORS/Website tests (delegated to backend)" -echo " - Anonymous/public access tests (not supported)" -echo " - Checksum algorithm tests (CRC32, SHA256 - not supported)" -echo " - Object lock/retention tests (delegated to backend)" -echo " - SSE tests (proxy handles encryption differently)" -echo "" -echo "Tests that should pass:" -echo " - Basic CRUD operations" -echo " - Multipart uploads" -echo " - Copy operations" -echo " - List operations" -echo " - Range requests" -echo "" - -log_info "=== S3 Compatibility Test Complete ===" diff --git a/e2e/scripts/verify-encryption-k8s.sh b/e2e/scripts/verify-encryption-k8s.sh deleted file mode 100755 index a4fc31f..0000000 --- a/e2e/scripts/verify-encryption-k8s.sh +++ /dev/null @@ -1,181 +0,0 @@ -#!/usr/bin/env bash -# Kubernetes wrapper for encryption verification -# Source this script and call: verify_encryption - -# Assert the dashboard reports a MULTIPART object as encrypted (issue #47 -# #6). Multipart objects keep their wrapped DEK in a sidecar, not an on-object -# tag, so the dashboard API must consult the sidecar. The byte-level verify_encryption -# above can't catch a mislabel in the dashboard — this closes that gap. -# -# verify_dashboard_encryption -# Picks a real multipart backup object (ETag ending in "-") and asserts the -# dashboard object-detail API returns "encrypted": true for it. -verify_dashboard_encryption() { - local BUCKET="$1" - local NAMESPACE="${2:-default}" - - echo "=== Dashboard Encryption-Status Check (multipart) ===" - - # Find a multipart object via MinIO: a *-big-Data.db SSTable, which Scylla - # uploads as a multipart object (ETag ends in -). Skip sidecars. - local KEY - KEY=$(kubectl run dashboard-enc-find --namespace minio --rm -i --restart=Never \ - --image=mc:latest --image-pull-policy=Never --command -- /bin/sh -c " - mc alias set m http://minio.minio.svc.cluster.local:9000 minioadmin minioadmin >/dev/null 2>&1 - mc ls -r m/$BUCKET 2>/dev/null | awk '{print \$NF}' \ - | grep -v '^\.s3proxy-internal/' | grep -- '-big-Data.db' | head -1 - " 2>/dev/null | tr -d '\r' | tail -1) - - if [ -z "$KEY" ]; then - echo "✗ No multipart object found to check" - return 1 - fi - echo "Multipart object: $KEY" - - # Query the dashboard object-detail API from inside an s3proxy pod (which has - # Python + reaches its own dashboard API on localhost). The {key:path} route - # takes the slashed key verbatim. - local POD - POD=$(kubectl get pod -n "$NAMESPACE" -l app.kubernetes.io/name=s3proxy-python \ - -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) - local RESP - RESP=$(kubectl exec -n "$NAMESPACE" "$POD" -- python -c " -import base64, json, urllib.request -key = '''$KEY''' -url = 'http://localhost:4433/dashboard/api/objects/$BUCKET/' + key -req = urllib.request.Request(url) -req.add_header('Authorization', 'Basic ' + base64.b64encode(b'admin:admin').decode()) -print(urllib.request.urlopen(req, timeout=15).read().decode()) -" 2>&1) - echo "Dashboard response: $RESP" - - if echo "$RESP" | grep -q '"encrypted": *true'; then - echo "✓ Dashboard API reports multipart object encrypted" - return 0 - fi - echo "✗ Dashboard API reported multipart object as NOT encrypted" - return 1 -} - -verify_encryption() { - local BUCKET="$1" - local PATH_PREFIX="${2:-}" - local NAMESPACE="${3:-default}" - - kubectl run encryption-check --namespace "$NAMESPACE" \ - --image=mc:latest \ - --image-pull-policy=Never \ - --restart=Never \ - --command -- /bin/sh -c " - set -e - echo '[1/3] Connecting to MinIO...' - timeout 30 mc alias set minio http://minio.minio.svc.cluster.local:9000 minioadmin minioadmin >/dev/null 2>&1 - echo ' ✓ Connected to MinIO' - - echo '[2/3] Listing backup files...' - echo '' - echo '=== Encryption Verification ===' - echo 'Bucket: $BUCKET' - echo 'Path: ${PATH_PREFIX:-}' - echo '' - - # List ALL files, excluding .s3proxy-internal/ metadata - FILES=\$(timeout 60 mc ls -r minio/$BUCKET/${PATH_PREFIX} 2>/dev/null | awk '{print \$NF}' | grep -v '^\.s3proxy-internal/' || true) - COUNT=\$(echo \"\$FILES\" | grep -c . || echo 0) - - echo \" ✓ Found \$COUNT files to verify\" - [ \"\$COUNT\" -eq 0 ] && { echo '✗ No files found!'; exit 1; } - - echo '[3/3] Verifying encryption (magic byte + entropy check)...' - CHECKED=0 PASSED=0 FAILED=0 SKIPPED=0 - FAILED_FILES='' SKIPPED_FILES='' - - for F in \$FILES; do - [ -z \"\$F\" ] && continue - - FAIL_REASON='' - - # Entropy check - encrypted data should have high entropy - # Stream only first 4KB instead of downloading entire file - if ! timeout 30 mc cat \"minio/$BUCKET/${PATH_PREFIX}\$F\" 2>/dev/null | head -c 4096 > /tmp/f; then - SKIPPED=\$((SKIPPED + 1)) - SKIPPED_FILES=\"\${SKIPPED_FILES} - \$F (download failed)\n\" - continue - fi - - SIZE=\$(stat -c%s /tmp/f 2>/dev/null || stat -f%z /tmp/f) - if [ \"\$SIZE\" -lt 100 ]; then - SKIPPED=\$((SKIPPED + 1)) - SKIPPED_FILES=\"\${SKIPPED_FILES} - \$F (too small: \${SIZE} bytes)\n\" - rm -f /tmp/f - continue - fi - - CHECKED=\$((CHECKED + 1)) - - # Entropy check - encrypted data should have high entropy (>6.0 bits/byte) - ENT=\$(cat /tmp/f | od -A n -t u1 | tr ' ' '\n' | grep -v '^\$' | sort | uniq -c | awk ' - BEGIN{t=0;e=0}{c[\$2]=\$1;t+=\$1}END{for(b in c){p=c[b]/t;if(p>0)e-=p*log(p)/log(2)}printf\"%.2f\",e}') - rm -f /tmp/f - - if awk \"BEGIN{exit!(\$ENT<6.0)}\"; then - [ -n \"\$FAIL_REASON\" ] && FAIL_REASON=\"\$FAIL_REASON + \" - FAIL_REASON=\"\${FAIL_REASON}low entropy: \$ENT\" - fi - - if [ -n \"\$FAIL_REASON\" ]; then - FAILED=\$((FAILED + 1)) - FAILED_FILES=\"\${FAILED_FILES} ✗ \$F (\$FAIL_REASON)\n\" - else - PASSED=\$((PASSED + 1)) - fi - - [ \$((CHECKED % 10)) -eq 0 ] && echo \" Progress: \$CHECKED/\$COUNT files checked (Encrypted: \$PASSED, Unencrypted: \$FAILED, Skipped: \$SKIPPED)\" - done - - echo '' - echo '=== SUMMARY ===' - echo \"Total found: \$COUNT\" - echo \"Checked: \$CHECKED\" - echo \"Encrypted: \$PASSED\" - echo \"Unencrypted: \$FAILED\" - echo \"Skipped: \$SKIPPED\" - echo '' - - if [ \"\$SKIPPED\" -gt 0 ]; then - echo ''; echo 'SKIPPED FILES:'; echo -e \"\$SKIPPED_FILES\" - fi - - if [ \"\$FAILED\" -gt 0 ]; then - echo ''; echo 'UNENCRYPTED FILES:'; echo -e \"\$FAILED_FILES\" - echo '✗ ENCRYPTION VERIFICATION FAILED!' - exit 1 - fi - - echo '✓ ALL FILES ENCRYPTED' - " - - kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod/encryption-check --timeout=60s || true - # Wait for pod to complete (Succeeded or Failed), not just be Ready - # Poll for completion instead of sequential waits to avoid long timeouts on failure - for i in $(seq 1 60); do - PHASE=$(kubectl get pod -n "$NAMESPACE" encryption-check -o jsonpath='{.status.phase}' 2>/dev/null || echo "") - if [ "$PHASE" = "Succeeded" ] || [ "$PHASE" = "Failed" ]; then - break - fi - sleep 5 - done - kubectl logs -n "$NAMESPACE" encryption-check || true - local EXIT_CODE=$(kubectl get pod -n "$NAMESPACE" encryption-check -o jsonpath='{.status.containerStatuses[0].state.terminated.exitCode}' 2>/dev/null || echo "") - kubectl delete pod -n "$NAMESPACE" encryption-check --ignore-not-found >/dev/null 2>&1 - - if [ -z "$EXIT_CODE" ]; then - echo "ERROR: Could not determine pod exit code (pod may not have terminated)" - return 1 - fi - if [ "$EXIT_CODE" != "0" ]; then - echo "Encryption verification failed with exit code: $EXIT_CODE" - return 1 - fi - return 0 -} diff --git a/e2e/scylla/templates/agent-config-secret.yaml b/e2e/scylla/templates/agent-config-secret.yaml deleted file mode 100644 index 01dd138..0000000 --- a/e2e/scylla/templates/agent-config-secret.yaml +++ /dev/null @@ -1,12 +0,0 @@ -apiVersion: v1 -kind: Secret -metadata: - name: scylla-agent-config -type: Opaque -stringData: - scylla-manager-agent.yaml: | - s3: - access_key_id: minioadmin - secret_access_key: minioadmin - provider: Minio - endpoint: http://s3proxy-python-frontproxy.s3proxy:80 diff --git a/e2e/scylla/templates/scylla-cluster-restore.yaml b/e2e/scylla/templates/scylla-cluster-restore.yaml deleted file mode 100644 index 6abdcf8..0000000 --- a/e2e/scylla/templates/scylla-cluster-restore.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: scylla.scylladb.com/v1 -kind: ScyllaCluster -metadata: - name: ${CLUSTER_NAME}-new -spec: - version: 2025.3.5 - agentVersion: 3.7.0 - developerMode: true - datacenter: - name: dc1 - racks: - - name: rack1 - members: 3 - storage: - capacity: 10Gi - resources: - requests: - cpu: 500m - memory: 2Gi - limits: - cpu: 2 - memory: 4Gi - volumes: - - name: agent-config - secret: - secretName: scylla-agent-config - agentVolumeMounts: - - name: agent-config - mountPath: /etc/scylla-manager-agent/scylla-manager-agent.yaml - subPath: scylla-manager-agent.yaml diff --git a/e2e/scylla/templates/scylla-cluster.yaml b/e2e/scylla/templates/scylla-cluster.yaml deleted file mode 100644 index 407f3b3..0000000 --- a/e2e/scylla/templates/scylla-cluster.yaml +++ /dev/null @@ -1,32 +0,0 @@ -apiVersion: scylla.scylladb.com/v1 -kind: ScyllaCluster -metadata: - name: ${CLUSTER_NAME} -spec: - version: 2025.3.5 - agentVersion: 3.7.0 - developerMode: true - sysctls: - - fs.aio-max-nr=1048576 - datacenter: - name: dc1 - racks: - - name: rack1 - members: 3 - storage: - capacity: 10Gi - resources: - requests: - cpu: 500m - memory: 2Gi - limits: - cpu: 2 - memory: 4Gi - volumes: - - name: agent-config - secret: - secretName: scylla-agent-config - agentVolumeMounts: - - name: agent-config - mountPath: /etc/scylla-manager-agent/scylla-manager-agent.yaml - subPath: scylla-manager-agent.yaml diff --git a/e2e/scylla/templates/storage-class.yaml b/e2e/scylla/templates/storage-class.yaml deleted file mode 100644 index c0fdba9..0000000 --- a/e2e/scylla/templates/storage-class.yaml +++ /dev/null @@ -1,7 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: scylladb-local-xfs -provisioner: rancher.io/local-path -volumeBindingMode: WaitForFirstConsumer -reclaimPolicy: Delete diff --git a/e2e/scylla/test.sh b/e2e/scylla/test.sh deleted file mode 100755 index 456e012..0000000 --- a/e2e/scylla/test.sh +++ /dev/null @@ -1,362 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" -ROOT_DIR="$(dirname "$SCRIPT_DIR")" -cd "$SCRIPT_DIR" - -source "${SCRIPT_DIR}/../scripts/verify-encryption-k8s.sh" - -if [ -z "${KUBECONFIG:-}" ]; then - export KUBECONFIG="${ROOT_DIR}/kubeconfig" - if [ ! -f "$KUBECONFIG" ]; then - echo "ERROR: Kubeconfig not found. Run ./cluster.sh up first" - exit 1 - fi -fi - -NAMESPACE="scylla-test" -export CLUSTER_NAME="scylla-cluster" - -log_info() { echo -e "\033[0;32m[INFO]\033[0m $1"; } -log_warn() { echo -e "\033[1;33m[WARN]\033[0m $1"; } -log_error() { echo -e "\033[0;31m[ERROR]\033[0m $1"; } - -cleanup() { - log_info "Cleaning up..." - kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=false || true -} -trap cleanup EXIT - -# Delete existing namespace -if kubectl get namespace "$NAMESPACE" &>/dev/null; then - # Clear finalizers from scylla resources first - for r in $(kubectl api-resources --namespaced -o name 2>/dev/null | grep scylla); do - kubectl get "$r" -n "$NAMESPACE" -o name 2>/dev/null | xargs -I{} kubectl patch {} -n "$NAMESPACE" -p '{"metadata":{"finalizers":null}}' --type=merge 2>/dev/null || true - done - kubectl delete namespace "$NAMESPACE" --timeout=10s 2>/dev/null || \ - kubectl get namespace "$NAMESPACE" -o json | jq '.spec.finalizers=[]' | kubectl replace --raw "/api/v1/namespaces/$NAMESPACE/finalize" -f - 2>/dev/null || true - sleep 2 -fi - -# Label nodes for ScyllaDB -for node in $(kubectl get nodes -o name | grep -v control-plane); do - kubectl label "$node" scylla.scylladb.com/node-type=scylla --overwrite 2>/dev/null || true -done - -# Create storage class for Scylla (uses default provisioner) -kubectl apply -f "${SCRIPT_DIR}/templates/storage-class.yaml" - -kubectl create namespace "$NAMESPACE" - -# S3 credentials for Scylla agent -kubectl apply -n "$NAMESPACE" -f "${SCRIPT_DIR}/templates/agent-config-secret.yaml" - -# ============================================================================ -# STEP 1: Create cluster -# ============================================================================ -log_info "=== Step 1: Creating ScyllaDB cluster ===" - -envsubst < "${SCRIPT_DIR}/templates/scylla-cluster.yaml" | kubectl apply -n "$NAMESPACE" -f - - -log_info "Waiting for pods to be created..." -until kubectl get pods -n "$NAMESPACE" -l app.kubernetes.io/name=scylla --no-headers 2>/dev/null | grep -q .; do - sleep 5 -done - -log_info "Waiting for all 3 pods to be ready..." -until [ "$(kubectl get pods -n "$NAMESPACE" -l app.kubernetes.io/name=scylla --no-headers 2>/dev/null | grep -c Running)" -eq 3 ]; do - sleep 5 -done -kubectl wait --namespace "$NAMESPACE" --for=condition=ready pod -l app.kubernetes.io/name=scylla --timeout=600s - -SCYLLA_POD=$(kubectl get pods -n "$NAMESPACE" -l app.kubernetes.io/name=scylla -o jsonpath='{.items[0].metadata.name}') -log_info "Cluster ready: $SCYLLA_POD" - -# ============================================================================ -# STEP 2: Generate ~2GB data using scylla-bench -# ============================================================================ -log_info "=== Step 2: Generating ~2GB test data ===" - -# Clean up any leftover bench pod -log_info "=== Step 2: Generating ~2GB test data ===" - -# 1. Give the network and schema a moment to settle -log_info "Waiting for service endpoints to be ready..." -sleep 20 - -# 2. Run bench in the background (no -i, no --rm) so we can control the wait -kubectl delete pod scylla-bench -n "$NAMESPACE" --ignore-not-found - -kubectl run scylla-bench --namespace "$NAMESPACE" \ - --image=scylladb/scylla-bench:0.3.6 \ - --restart=Never \ - -- \ - -workload sequential -mode write \ - -nodes "${CLUSTER_NAME}-client" \ - -partition-count 20000 -clustering-row-count 100 -clustering-row-size 1024 \ - -replication-factor 3 -consistency-level quorum - -# 3. Stream logs while waiting for completion -log_info "Waiting for benchmark pod to start..." -kubectl wait --namespace "$NAMESPACE" --for=condition=Ready pod/scylla-bench --timeout=120s || true -log_info "Benchmarking in progress (streaming logs)..." -kubectl logs -f -n "$NAMESPACE" scylla-bench & -LOGS_PID=$! - -# Wait for pod to succeed (plain pods don't have 'complete' condition, use jsonpath) -if ! kubectl wait --namespace "$NAMESPACE" --for=jsonpath='{.status.phase}'=Succeeded pod/scylla-bench --timeout=900s; then - kill $LOGS_PID 2>/dev/null || true - log_error "Data generation failed or timed out." - kubectl logs -n "$NAMESPACE" scylla-bench | tail -n 20 - exit 1 -fi -kill $LOGS_PID 2>/dev/null || true - -# 4. Cleanup the pod manually after success -kubectl delete pod scylla-bench -n "$NAMESPACE" - -# 5. Verify data -log_info "Verifying row count..." -ROW_COUNT=$(kubectl exec -n "$NAMESPACE" "$SCYLLA_POD" -- cqlsh -e "SELECT COUNT(*) FROM scylla_bench.test;" | grep -oE '[0-9]+' | head -1 | tr -d '\n\r ' || echo "0") -log_info "Generated approximately $ROW_COUNT rows" - -# ============================================================================ -# STEP 3: Backup -# ============================================================================ -log_info "=== Step 3: Creating backup ===" - -MANAGER_POD=$(kubectl get pods -n scylla-manager -l app.kubernetes.io/name=scylla-manager -o jsonpath='{.items[0].metadata.name}') - -# Wait for cluster to be registered with Scylla Manager -log_info "Waiting for cluster registration with Scylla Manager..." -until kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool cluster list 2>/dev/null | grep -q "${NAMESPACE}/${CLUSTER_NAME}"; do - sleep 5 -done -log_info "Cluster registered" - -# Run backup (returns task ID immediately) -log_info "Running backup..." -BACKUP_OUTPUT=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool backup \ - -c "${NAMESPACE}/${CLUSTER_NAME}" \ - -L "s3:scylla-backups" 2>&1) || true -echo "$BACKUP_OUTPUT" - -# Extract task ID from output (format: backup/uuid) -TASK_ID=$(echo "$BACKUP_OUTPUT" | grep -oE 'backup/[a-f0-9-]+' | head -1 || true) -if [ -n "$TASK_ID" ]; then - log_info "Backup task started: $TASK_ID" - log_info "Waiting for backup to complete..." - - # Wait for backup task to complete (poll progress) - WAIT_COUNT=0 - while [ $WAIT_COUNT -lt 120 ]; do - PROGRESS=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool progress \ - -c "${NAMESPACE}/${CLUSTER_NAME}" "$TASK_ID" 2>&1) || true - - # Check if backup is done (only check overall Status, not individual host %) - if echo "$PROGRESS" | grep -qE "^Status:\s+DONE"; then - log_info "Backup completed" - break - fi - - # Check for errors - look for Status: ERROR/FAILED, not column headers - if echo "$PROGRESS" | grep -qE "Status:\s+(ERROR|FAILED)"; then - log_error "Backup failed" - echo "$PROGRESS" - break - fi - - # Show progress (extract from "Progress: XX%" line, not individual hosts) - PERCENT=$(echo "$PROGRESS" | grep -E "^Progress:" | grep -oE '[0-9]+%' || echo "?%") - log_info "Backup progress: $PERCENT (attempt $((WAIT_COUNT + 1))/120)" - sleep 5 - WAIT_COUNT=$((WAIT_COUNT + 1)) - done - - if [ $WAIT_COUNT -ge 120 ]; then - log_warn "Backup timeout - may still be in progress" - fi -else - log_warn "Could not extract backup task ID" -fi - -sleep 2 -log_info "Getting backup list..." -BACKUP_LIST=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool backup list \ - -c "${NAMESPACE}/${CLUSTER_NAME}" \ - -L "s3:scylla-backups" 2>&1) || true -echo "$BACKUP_LIST" - -# Extract snapshot tag (format: sm_YYYYMMDDHHMMSSUTC) -SNAPSHOT_TAG=$(echo "$BACKUP_LIST" | grep -oE 'sm_[0-9]{14}UTC' | tail -1 || true) -if [ -z "$SNAPSHOT_TAG" ]; then - log_error "Could not extract snapshot tag from backup list" - log_info "Backup list output: $BACKUP_LIST" - exit 1 -fi -log_info "Backup snapshot tag: $SNAPSHOT_TAG" - -# Verify encryption -verify_encryption "scylla-backups" "" "$NAMESPACE" || log_warn "Encryption check skipped" - -# Verify the dashboard correctly reports multipart backup objects as -# encrypted (issue #47 #6 — sidecar-aware detection). -verify_dashboard_encryption "scylla-backups" "$NAMESPACE" || log_warn "Dashboard encryption check skipped" - -# ============================================================================ -# STEP 4: Delete cluster -# ============================================================================ -log_info "=== Step 4: Deleting cluster ===" - -kubectl delete scyllacluster -n "$NAMESPACE" "$CLUSTER_NAME" --wait -kubectl wait --namespace "$NAMESPACE" --for=delete pod -l scylla/cluster=${CLUSTER_NAME} --timeout=300s || true -log_info "Cluster deleted" - -# ============================================================================ -# STEP 5: Create new cluster -# ============================================================================ -log_info "=== Step 5: Creating new cluster ===" - -envsubst < "${SCRIPT_DIR}/templates/scylla-cluster-restore.yaml" | kubectl apply -n "$NAMESPACE" -f - - -log_info "Waiting for all 3 new cluster pods to be ready..." -# Wait until we have 3 pods created -until [ "$(kubectl get pods -n "$NAMESPACE" -l scylla/cluster=${CLUSTER_NAME}-new --no-headers 2>/dev/null | wc -l)" -ge 3 ]; do - CURRENT=$(kubectl get pods -n "$NAMESPACE" -l scylla/cluster=${CLUSTER_NAME}-new --no-headers 2>/dev/null | wc -l) - log_info "Waiting for pods... ($CURRENT/3 created)" - sleep 10 -done - -# Wait for ALL pods to be ready (kubectl wait waits for all matching pods) -kubectl wait --namespace "$NAMESPACE" --for=condition=ready pod -l scylla/cluster=${CLUSTER_NAME}-new --timeout=600s - -# Verify all 3 are ready -READY_COUNT=$(kubectl get pods -n "$NAMESPACE" -l scylla/cluster=${CLUSTER_NAME}-new --no-headers 2>/dev/null | grep -c "Running" || echo 0) -log_info "New cluster ready: $READY_COUNT/3 pods running" - -RESTORED_POD=$(kubectl get pods -n "$NAMESPACE" -l scylla/cluster=${CLUSTER_NAME}-new -o jsonpath='{.items[0].metadata.name}') - -# ============================================================================ -# STEP 6: Restore -# ============================================================================ -log_info "=== Step 6: Restoring from backup ===" - -# Wait for new cluster to be registered with Scylla Manager -log_info "Waiting for new cluster registration with Scylla Manager..." -WAIT_COUNT=0 -while [ $WAIT_COUNT -lt 60 ]; do - CLUSTER_LIST=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool cluster list 2>&1) || true - if echo "$CLUSTER_LIST" | grep -q "${CLUSTER_NAME}-new"; then - log_info "New cluster registered" - break - fi - log_info "Waiting... (attempt $((WAIT_COUNT + 1))/60)" - echo "$CLUSTER_LIST" | head -5 - sleep 5 - WAIT_COUNT=$((WAIT_COUNT + 1)) -done - -if [ $WAIT_COUNT -ge 60 ]; then - log_error "Timeout waiting for new cluster registration" - log_info "Final cluster list:" - kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool cluster list - exit 1 -fi - -# Give manager a moment to fully sync with new cluster -sleep 10 - -if [ -z "$SNAPSHOT_TAG" ]; then - log_error "No snapshot tag found, cannot restore" - exit 1 -fi - -log_info "Restoring schema from snapshot $SNAPSHOT_TAG..." -SCHEMA_RESTORE_OUTPUT=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool restore \ - -c "${NAMESPACE}/${CLUSTER_NAME}-new" \ - -L "s3:scylla-backups" \ - --snapshot-tag "$SNAPSHOT_TAG" \ - --restore-schema 2>&1) || true -echo "$SCHEMA_RESTORE_OUTPUT" - -# Extract and wait for schema restore task -SCHEMA_TASK_ID=$(echo "$SCHEMA_RESTORE_OUTPUT" | grep -oE 'restore/[a-f0-9-]+' | head -1 || true) -if [ -n "$SCHEMA_TASK_ID" ]; then - log_info "Schema restore task: $SCHEMA_TASK_ID - waiting for completion..." - WAIT_COUNT=0 - while [ $WAIT_COUNT -lt 60 ]; do - PROGRESS=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool progress \ - -c "${NAMESPACE}/${CLUSTER_NAME}-new" "$SCHEMA_TASK_ID" 2>&1) || true - if echo "$PROGRESS" | grep -qE "^Status:\s+DONE"; then - log_info "Schema restore completed" - break - fi - if echo "$PROGRESS" | grep -qE "Status:\s+(ERROR|FAILED)"; then - log_error "Schema restore failed" - echo "$PROGRESS" - break - fi - PERCENT=$(echo "$PROGRESS" | grep -E "^Progress:" | grep -oE '[0-9]+%' || echo "?%") - log_info "Schema restore progress: $PERCENT (attempt $((WAIT_COUNT + 1))/60)" - sleep 5 - WAIT_COUNT=$((WAIT_COUNT + 1)) - done -else - log_warn "Could not extract schema restore task ID" -fi - -sleep 5 - -log_info "Restoring data from snapshot $SNAPSHOT_TAG..." -DATA_RESTORE_OUTPUT=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool restore \ - -c "${NAMESPACE}/${CLUSTER_NAME}-new" \ - -L "s3:scylla-backups" \ - --snapshot-tag "$SNAPSHOT_TAG" \ - --restore-tables 2>&1) || true -echo "$DATA_RESTORE_OUTPUT" - -# Extract and wait for data restore task -DATA_TASK_ID=$(echo "$DATA_RESTORE_OUTPUT" | grep -oE 'restore/[a-f0-9-]+' | head -1 || true) -if [ -n "$DATA_TASK_ID" ]; then - log_info "Data restore task: $DATA_TASK_ID - waiting for completion..." - WAIT_COUNT=0 - while [ $WAIT_COUNT -lt 120 ]; do - PROGRESS=$(kubectl exec -n scylla-manager "$MANAGER_POD" -- sctool progress \ - -c "${NAMESPACE}/${CLUSTER_NAME}-new" "$DATA_TASK_ID" 2>&1) || true - if echo "$PROGRESS" | grep -qE "^Status:\s+DONE"; then - log_info "Data restore completed" - break - fi - if echo "$PROGRESS" | grep -qE "Status:\s+(ERROR|FAILED)"; then - log_error "Data restore failed" - echo "$PROGRESS" - break - fi - PERCENT=$(echo "$PROGRESS" | grep -E "^Progress:" | grep -oE '[0-9]+%' || echo "?%") - log_info "Data restore progress: $PERCENT (attempt $((WAIT_COUNT + 1))/120)" - sleep 5 - WAIT_COUNT=$((WAIT_COUNT + 1)) - done -else - log_warn "Could not extract data restore task ID" -fi - -sleep 10 - -# ============================================================================ -# STEP 7: Verify -# ============================================================================ -log_info "=== Step 7: Verifying restore ===" - -RESTORED_COUNT=$(kubectl exec -n "$NAMESPACE" "$RESTORED_POD" -- cqlsh -e "SELECT COUNT(*) FROM scylla_bench.test LIMIT 1000000;" 2>/dev/null | grep -oE '[0-9]+' | head -1 | tr -d '\n\r ' || echo "0") -log_info "Restored rows: $RESTORED_COUNT (original: $ROW_COUNT)" - -if [ "$RESTORED_COUNT" = "$ROW_COUNT" ]; then - log_info "=== TEST PASSED ===" -else - log_warn "Row count mismatch - restore may still be in progress" -fi - -log_info "Test completed. Cleaning up in 5 seconds..." -sleep 5 diff --git a/pyproject.toml b/pyproject.toml index bf2eb2f..8006281 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,7 +21,7 @@ keywords = [ requires-python = ">=3.14" dependencies = [ "fastapi>=0.141.1", - "uvicorn[standard]>=0.52.4", + "uvicorn[standard]>=0.53.0", "aioboto3>=15.5.0", "boto3>=1.40.0", "botocore>=1.40.0", @@ -49,10 +49,10 @@ dev = [ "pytest-cov>=7.1.0", "pytest-xdist>=3.8.0", "moto[s3]>=5.2.3", - "ruff>=0.16.5", + "ruff>=0.16.8", "mypy>=2.3.1", - "boto3-stubs[s3]>=1.43.83", - "fakeredis>=2.37.1", + "boto3-stubs[s3]>=1.43.98", + "fakeredis>=2.38.0", "requests>=2.34.2", ] diff --git a/s3proxy/app.py b/s3proxy/app.py index 41701a2..9238032 100644 --- a/s3proxy/app.py +++ b/s3proxy/app.py @@ -200,14 +200,16 @@ async def lifespan(app: FastAPI) -> AsyncIterator[None]: tracemalloc_task = _maybe_start_tracemalloc() - yield - - if tracemalloc_task is not None: - tracemalloc_task.cancel() - await stats_store.aclose() # flush buffered samples before Redis closes - await close_redis() - await close_http_client() - logger.info("Shutting down") + try: + yield + finally: + if tracemalloc_task is not None: + tracemalloc_task.cancel() + await stats_store.aclose() # flush buffered samples before Redis closes + await close_redis() + await close_http_client() + await handler.client_pool.close() + logger.info("Shutting down") return lifespan diff --git a/s3proxy/client/pool.py b/s3proxy/client/pool.py new file mode 100644 index 0000000..b5b16a5 --- /dev/null +++ b/s3proxy/client/pool.py @@ -0,0 +1,51 @@ +"""Bounded credential-isolated S3 client pool owned by one application lifespan.""" + +import asyncio +from contextlib import asynccontextmanager + +from ..errors import S3Error +from .s3 import S3Client + + +class S3ClientPool: + def __init__(self, settings, max_clients=32): + self.settings = settings + self.max_clients = max_clients + self.entries = {} + self.condition = asyncio.Condition() + self.closed = False + + @asynccontextmanager + async def acquire(self, credentials): + key = (credentials.access_key, credentials.secret_key, credentials.region) + async with self.condition: + if self.closed: + raise S3Error.slow_down("S3 client pool is shutting down") + if key not in self.entries: + if len(self.entries) >= self.max_clients: + idle = next((k for k, (_, refs) in self.entries.items() if refs == 0), None) + if idle is None: + raise S3Error.slow_down("S3 client pool is busy") + client, _ = self.entries.pop(idle) + await client.__aexit__(None, None, None) + client = S3Client(self.settings, credentials) + await client.__aenter__() + self.entries[key] = [client, 0] + entry = self.entries[key] + entry[1] += 1 + try: + yield entry[0] + finally: + async with self.condition: + entry[1] -= 1 + self.condition.notify_all() + + async def close(self): + async with self.condition: + self.closed = True + await self.condition.wait_for( + lambda: all(refs == 0 for _, refs in self.entries.values()) + ) + entries, self.entries = self.entries, {} + for client, _ in entries.values(): + await client.__aexit__(None, None, None) diff --git a/s3proxy/client/s3.py b/s3proxy/client/s3.py index 3b82b7c..165fea1 100644 --- a/s3proxy/client/s3.py +++ b/s3proxy/client/s3.py @@ -41,7 +41,7 @@ class S3Client: Memory management: - Uses a shared aioboto3 Session to avoid repeated JSON model loading - - Creates fresh clients per request for proper connection cleanup + - Clients are leased from the credential-isolated application pool - Each session load costs ~30-150MB (botocore service definitions) See: https://github.com/boto/boto3/issues/1670 @@ -118,6 +118,7 @@ async def put_object( tagging: str | None = None, cache_control: str | None = None, expires: str | None = None, + if_none_match: str | None = None, ) -> dict[str, Any]: """Put object to S3.""" kwargs: dict[str, Any] = {"Bucket": bucket, "Key": key, "Body": body} @@ -128,6 +129,7 @@ async def put_object( Tagging=tagging, CacheControl=cache_control, Expires=expires, + IfNoneMatch=if_none_match, ) return await self._cached_client.put_object(**kwargs) @@ -213,6 +215,7 @@ async def complete_multipart_upload( key: str, upload_id: str, parts: list[dict[str, Any]], + if_none_match: str | None = None, ) -> dict[str, Any]: """Complete multipart upload.""" start = time.monotonic() @@ -221,6 +224,7 @@ async def complete_multipart_upload( Key=key, UploadId=upload_id, MultipartUpload={"Parts": parts}, + **({"IfNoneMatch": if_none_match} if if_none_match is not None else {}), ) duration = time.monotonic() - start logger.info( @@ -284,6 +288,9 @@ async def copy_object( content_type: str | None = None, tagging_directive: str | None = None, tagging: str | None = None, + copy_source_if_match: str | None = None, + cache_control: str | None = None, + expires=None, ) -> dict[str, Any]: """Copy object within S3.""" kwargs: dict[str, Any] = { @@ -300,6 +307,12 @@ async def copy_object( kwargs["TaggingDirective"] = tagging_directive if tagging and tagging_directive == "REPLACE": kwargs["Tagging"] = tagging + _add_optional_kwargs( + kwargs, + CopySourceIfMatch=copy_source_if_match, + CacheControl=cache_control, + Expires=expires, + ) return await self._cached_client.copy_object(**kwargs) async def delete_objects( @@ -399,6 +412,7 @@ async def upload_part_copy( part_number: int, copy_source: str, copy_source_range: str | None = None, + copy_source_if_match: str | None = None, ) -> dict[str, Any]: """Copy a part from another object.""" kwargs: dict[str, Any] = { @@ -408,5 +422,7 @@ async def upload_part_copy( "PartNumber": part_number, "CopySource": copy_source, } - _add_optional_kwargs(kwargs, CopySourceRange=copy_source_range) + _add_optional_kwargs( + kwargs, CopySourceRange=copy_source_range, CopySourceIfMatch=copy_source_if_match + ) return await self._cached_client.upload_part_copy(**kwargs) diff --git a/s3proxy/concurrency.py b/s3proxy/concurrency.py index b1a638e..c203d47 100644 --- a/s3proxy/concurrency.py +++ b/s3proxy/concurrency.py @@ -4,11 +4,7 @@ import asyncio import contextlib -import ctypes -import gc import os -import sys -from collections.abc import Callable import structlog @@ -29,26 +25,6 @@ MAX_BUFFER_SIZE = 8 * 1024 * 1024 # 8MB streaming buffer size -def _create_malloc_release() -> Callable[[], int] | None: - """Create platform-specific function to release memory back to OS. - - Only works on Linux via malloc_trim(0). Returns None on other platforms. - """ - if sys.platform != "linux": - return None - - try: - libc = ctypes.CDLL("libc.so.6") - libc.malloc_trim.argtypes = [ctypes.c_size_t] - libc.malloc_trim.restype = ctypes.c_int - return lambda: libc.malloc_trim(0) - except OSError, AttributeError: - return None - - -_malloc_release = _create_malloc_release() - - BACKPRESSURE_TIMEOUT = int(os.environ.get("S3PROXY_BACKPRESSURE_TIMEOUT", "120")) @@ -195,18 +171,6 @@ async def release(self, bytes_reserved: int) -> None: MEMORY_RESERVED_BYTES.set(self._active_bytes) self._condition.notify_all() - # Run garbage collection and release memory to OS - gc.collect(0) - gc.collect(1) - gc.collect(2) - - if _malloc_release: - with contextlib.suppress(OSError): - _malloc_release() - - # Yield to allow OS memory reclaim - await asyncio.sleep(0) - # Default instance used by module-level functions _default = ConcurrencyLimiter(limit_mb=int(os.environ.get("S3PROXY_MEMORY_LIMIT_MB", "64"))) @@ -225,9 +189,9 @@ def estimate_memory_footprint(method: str, content_length: int) -> int: if method in ("HEAD", "DELETE"): return 0 if method == "GET": - return MAX_BUFFER_SIZE + return 4 * MAX_BUFFER_SIZE if method == "POST": - return MIN_RESERVATION + return max(MIN_RESERVATION, 2 * min(max(content_length, 0), MAX_BUFFER_SIZE)) return max(MIN_RESERVATION, governor_memory_footprint(content_length)) diff --git a/s3proxy/crypto.py b/s3proxy/crypto.py index b58ca10..666f4c7 100644 --- a/s3proxy/crypto.py +++ b/s3proxy/crypto.py @@ -399,8 +399,8 @@ def framed_ciphertext_size(plaintext_size: int) -> int: def encrypt_frame( plaintext: bytes, dek: bytes, upload_id: str, part_number: int, frame_index: int ) -> bytes: - """Encrypt a single frame (nonce || ciphertext || tag) with its derived nonce.""" - return encrypt(plaintext, dek, derive_frame_nonce(upload_id, part_number, frame_index)) + """Encrypt a frame with a fresh nonce; network retries must reuse these sealed bytes.""" + return encrypt(plaintext, dek) def ciphertext_frame_byte_sizes(plaintext_size: int, ciphertext_size: int) -> list[int]: @@ -529,7 +529,7 @@ def decrypt(ciphertext: bytes, dek: bytes) -> bytes: ) nonce = ciphertext[:NONCE_SIZE] - ct_with_tag = ciphertext[NONCE_SIZE:] + ct_with_tag = memoryview(ciphertext)[NONCE_SIZE:] try: aesgcm = AESGCM(dek) diff --git a/s3proxy/errors.py b/s3proxy/errors.py index 11c3e04..b00bcf0 100644 --- a/s3proxy/errors.py +++ b/s3proxy/errors.py @@ -93,6 +93,16 @@ def malformed_xml(cls, message: str = "The XML you provided was not well-formed" def invalid_request(cls, message: str) -> S3Error: return cls(400, "InvalidRequest", message) + @classmethod + def bad_digest(cls, checksum: str) -> S3Error: + return cls( + 400, "BadDigest", f"The {checksum} you specified did not match the calculated checksum." + ) + + @classmethod + def invalid_digest(cls, checksum: str) -> S3Error: + return cls(400, "InvalidDigest", f"The {checksum} you specified was invalid.") + # 403 Forbidden variants @classmethod def access_denied(cls, message: str = "Access Denied") -> S3Error: @@ -250,6 +260,14 @@ def raise_for_client_error( msg = e.response.get("Error", {}).get("Message", str(e)) _log_upstream_failure(source="client_error", exc=e, bucket=bucket, key=key) + if code == "PreconditionFailed": + raise S3Error.precondition_failed(msg) from e + if code in ("AccessDenied", "403"): + raise S3Error.access_denied(msg) from e + if code == "InvalidRange": + raise S3Error.invalid_range(msg) from e + if code == "ConditionalRequestConflict": + raise S3Error(409, code, msg) from e if code == "NoSuchUpload": raise S3Error.no_such_upload(msg) from e if code in ("NoSuchKey", "404"): diff --git a/s3proxy/handlers/base.py b/s3proxy/handlers/base.py index 7fcf129..a406b5f 100644 --- a/s3proxy/handlers/base.py +++ b/s3proxy/handlers/base.py @@ -220,9 +220,12 @@ def __init__( self.multipart_manager = multipart_manager self.complete_upload_lock = complete_upload_lock or create_complete_upload_lock() self.keyring = settings.keyring + from ..client.pool import S3ClientPool + + self.client_pool = S3ClientPool(settings) def _client(self, creds: S3Credentials) -> S3Client: - return S3Client(self.settings, creds) + return self.client_pool.acquire(creds) def _parse_path(self, path: str) -> tuple[str, str]: if m := PATH_RE.match(path): @@ -251,6 +254,16 @@ def _internal_meta_keys(self) -> set[str]: self.settings.kidtag_name.lower(), "client-etag", "plaintext-size", + "s3proxy-format", + "s3proxy-generation", + } + + def _user_metadata(self, request: Request) -> dict[str, str]: + internal = self._internal_meta_keys() + return { + k[11:]: v + for k, v in request.headers.items() + if k.startswith("x-amz-meta-") and k[11:] not in internal } def _parse_range(self, header: str, size: int) -> tuple[int, int]: @@ -374,6 +387,29 @@ def _check_conditional_headers( return None + async def _resolve_object(self, client, bucket, key, head=None): + from ..state.metadata import load_multipart_metadata, multipart_etag + from ..state.object import ObjectDescriptor + + if head is None: + head = await client.head_object(bucket, key) + multipart = await load_multipart_metadata(client, bucket, key, head) + metadata = head.get("Metadata", {}) + return ObjectDescriptor( + head=head, + multipart=multipart, + plaintext_size=( + multipart.total_plaintext_size + if multipart + else self._get_plaintext_size(metadata, head.get("ContentLength", 0)) + ), + etag=( + multipart_etag(multipart) + if multipart + else self._get_effective_etag(metadata, head.get("ETag", "")) + ), + ) + async def _download_encrypted_single( self, client: S3Client, bucket: str, key: str, wrapped_dek_b64: str, kid: str = "" ) -> bytes: @@ -381,6 +417,27 @@ async def _download_encrypted_single( wrapped_dek = base64.b64decode(wrapped_dek_b64) return crypto.decrypt_object(ciphertext, wrapped_dek, self.keyring.key_by_id(kid)) + async def _iter_single_plaintext( + self, client, bucket, key, wrapped_dek_b64, kid="", start=0, end=None, if_match=None + ): + from contextlib import aclosing + + from ..streaming.authenticated import decrypt_to_file, file_range + + dek = crypto.unwrap_key(base64.b64decode(wrapped_dek_b64), self.keyring.key_by_id(kid)) + response = await client.get_object( + bucket, key, **({"if_match": if_match} if if_match else {}) + ) + spool, length = await decrypt_to_file(response["Body"], dek) + try: + async with aclosing( + file_range(spool, start, length - 1 if end is None else end) + ) as stream: + async for chunk in stream: + yield chunk + finally: + spool.close() + async def _iter_multipart_plaintext( self, client: S3Client, @@ -390,6 +447,8 @@ async def _iter_multipart_plaintext( dek: bytes, range_start: int | None = None, range_end: int | None = None, + *, + if_match=None, ) -> AsyncIterator[bytes]: """Yield decrypted plaintext for a multipart-encrypted object, one frame at a time. @@ -405,44 +464,17 @@ async def _iter_multipart_plaintext( ~50MB client part. Frames outside the requested plaintext range are skipped (no fetch); frames that partially overlap are trimmed before yielding. """ - sorted_parts = sorted(meta.parts, key=lambda p: p.part_number) - pt_offset = 0 - ct_offset = 0 - - for part in sorted_parts: - if part.internal_parts: - segments = [ - (ip.plaintext_size, ip.ciphertext_size) - for ip in sorted(part.internal_parts, key=lambda p: p.internal_part_number) - ] - else: - segments = [(part.plaintext_size, part.ciphertext_size)] - - for seg_pt_size, seg_ct_size in segments: - for fsize in crypto.ciphertext_frame_byte_sizes(seg_pt_size, seg_ct_size): - frame_pt_size = fsize - crypto.ENCRYPTION_OVERHEAD - frame_pt_end = pt_offset + frame_pt_size - 1 + from contextlib import aclosing - in_range = range_start is None or ( - frame_pt_end >= range_start and pt_offset <= range_end - ) + from ..streaming.frames import plaintext_frames - if in_range: - ct_end = ct_offset + fsize - 1 - ciphertext = await read_source_bytes( - client, bucket, key, f"bytes={ct_offset}-{ct_end}" - ) - plaintext = crypto.decrypt(ciphertext, dek) - - if range_start is not None: - trim_start = max(0, range_start - pt_offset) - trim_end = min(frame_pt_size, range_end - pt_offset + 1) - plaintext = plaintext[trim_start:trim_end] - - yield plaintext - - pt_offset += frame_pt_size - ct_offset += fsize + async with aclosing( + plaintext_frames( + client, bucket, key, meta, dek, range_start, range_end, if_match=if_match + ) + ) as stream: + async for chunk in stream: + yield chunk async def _download_encrypted_multipart( self, diff --git a/s3proxy/handlers/buckets.py b/s3proxy/handlers/buckets.py index c2c8436..50bb711 100644 --- a/s3proxy/handlers/buckets.py +++ b/s3proxy/handlers/buckets.py @@ -19,9 +19,7 @@ INTERNAL_PREFIX, META_SUFFIX_LEGACY, delete_multipart_metadata, - load_multipart_metadata, plaintext_attr_cache, - synthetic_multipart_etag, ) from ..xml_utils import find_element, find_elements from .base import BaseHandler @@ -223,26 +221,29 @@ async def resolve(obj: dict) -> dict: if cached is not None: size, etag = cached return self._list_entry(obj, size, etag) - async with sem: + async with plaintext_attr_cache.coalesce(bucket, obj["Key"], backend_etag), sem: + cached = plaintext_attr_cache.get(bucket, obj["Key"], backend_etag) + if cached is not None: + return self._list_entry(obj, *cached) try: head = await client.head_object(bucket, obj["Key"]) - meta = head.get("Metadata", {}) - if "plaintext-size" in meta: - size = self._get_plaintext_size(meta, obj.get("Size", 0)) - etag = self._get_effective_etag(meta, obj.get("ETag", "")) - elif mp_meta := await load_multipart_metadata(client, bucket, obj["Key"]): - # Multipart objects can't carry plaintext-size in user - # metadata (it is fixed at CreateMultipartUpload); the - # size lives in the .meta sidecar. Reporting the backend - # Size here would leak the ciphertext size and make sync - # clients re-upload every multipart object on each pass. - size = mp_meta.total_plaintext_size - etag = synthetic_multipart_etag(size) - else: - size = self._get_plaintext_size(meta, obj.get("Size", 0)) - etag = self._get_effective_etag(meta, obj.get("ETag", "")) - plaintext_attr_cache.put(bucket, obj["Key"], backend_etag, size, etag) - except Exception: + head.setdefault("ContentLength", obj.get("Size", 0)) + head.setdefault("ETag", obj.get("ETag", "")) + descriptor = await self._resolve_object(client, bucket, obj["Key"], head) + size, etag = descriptor.plaintext_size, descriptor.etag + plaintext_attr_cache.put( + bucket, + obj["Key"], + str(head.get("ETag", backend_etag)).strip('"'), + size, + etag, + ) + except ClientError as error: + from ..state.metadata import is_not_found + + if not is_not_found(error): + raise + # LIST and HEAD are separate snapshots; the object may have been deleted. size, etag = obj.get("Size", 0), backend_etag return self._list_entry(obj, size, etag) @@ -278,7 +279,14 @@ async def handle_delete_bucket(self, request: Request, creds: S3Credentials) -> bucket = self._parse_bucket(request.url.path) async with self._client(creds) as client: try: - await client.delete_bucket(bucket) + try: + await client.delete_bucket(bucket) + except ClientError as e: + if e.response["Error"]["Code"] != "BucketNotEmpty": + raise + if not await self._purge_internal_keys(client, bucket): + raise + await client.delete_bucket(bucket) return Response(status_code=204) except ClientError as e: code = e.response["Error"]["Code"] @@ -288,6 +296,27 @@ async def handle_delete_bucket(self, request: Request, creds: S3Credentials) -> raise S3Error.bucket_not_empty(bucket) from None raise S3Error.internal_error(str(e)) from e + async def _purge_internal_keys(self, client, bucket: str) -> bool: + """Remove proxy metadata only once no client object or upload can reference it.""" + if (await client.list_multipart_uploads(bucket, max_uploads=1)).get("Uploads"): + return False + keys: list[str] = [] + token = None + while True: + page = await client.list_objects_v2(bucket, continuation_token=token) + for obj in page.get("Contents", []): + if not self._is_internal_key(obj["Key"]): + return False + keys.append(obj["Key"]) + if not page.get("IsTruncated"): + break + token = page["NextContinuationToken"] + for offset in range(0, len(keys), 1000): + batch = [{"Key": key} for key in keys[offset : offset + 1000]] + if (await client.delete_objects(bucket, batch, quiet=True)).get("Errors"): + return False + return bool(keys) + async def handle_head_bucket(self, request: Request, creds: S3Credentials) -> Response: bucket = self._parse_bucket(request.url.path) async with self._client(creds) as client: diff --git a/s3proxy/handlers/multipart/copy.py b/s3proxy/handlers/multipart/copy.py index f796de6..c6a727f 100644 --- a/s3proxy/handlers/multipart/copy.py +++ b/s3proxy/handlers/multipart/copy.py @@ -89,6 +89,18 @@ class _PlaintextRangeSplit: class CopyPartMixin(BaseHandler): async def handle_upload_part_copy(self, request: Request, creds: S3Credentials) -> Response: + bucket, key = self._parse_path(request.url.path) + upload_id, _ = self._extract_multipart_params(request) + state = await self.multipart_manager.get_upload(bucket, key, upload_id) + if state is None: + raise S3Error.no_such_upload(upload_id) + if state.layout_version < 3: + raise S3Error.invalid_request( + "Legacy in-flight uploads must be restarted after upgrade" + ) + return await self._copy_part_impl(request, creds) + + async def _copy_part_impl(self, request: Request, creds: S3Credentials) -> Response: bucket, key = self._parse_path(request.url.path) async with self._client(creds) as client: upload_id, part_num = self._extract_multipart_params(request) @@ -107,7 +119,7 @@ async def handle_upload_part_copy(self, request: Request, creds: S3Credentials) try: head_resp = await client.head_object(src_bucket, src_key) - except Exception as e: + except ClientError as e: logger.error( "UPLOAD_PART_COPY_HEAD_FAILED", bucket=bucket, @@ -118,11 +130,13 @@ async def handle_upload_part_copy(self, request: Request, creds: S3Credentials) error_type=type(e).__name__, error=str(e), ) - raise S3Error.no_such_key(src_key) from e + self._raise_s3_error(e, src_bucket, src_key) src_metadata = head_resp.get("Metadata", {}) src_wrapped_dek = src_metadata.get(self.settings.dektag_name) - src_multipart_meta = await load_multipart_metadata(client, src_bucket, src_key) + src_multipart_meta = await load_multipart_metadata( + client, src_bucket, src_key, head_resp + ) total_plaintext = self._copy_plaintext_size( head_resp, None, src_wrapped_dek, src_multipart_meta @@ -134,6 +148,78 @@ async def handle_upload_part_copy(self, request: Request, creds: S3Credentials) head_resp, copy_source_range, src_wrapped_dek, src_multipart_meta ) + if state.layout_version >= 3: + from .staged import stage_ciphertext_copy, stage_part + + full_source = ( + copy_source_range is None + or copy_source_range == f"bytes=0-{total_plaintext - 1}" + ) + source_etag = ( + src_multipart_meta.client_etag + if src_multipart_meta + else src_metadata.get("client-etag", "") + ) + native = False + if full_source and source_etag and (src_multipart_meta or src_wrapped_dek): + source_dek, source_kid = self._resolve_source_dek( + src_multipart_meta, src_wrapped_dek, src_metadata, creds + ) + state = await self.multipart_manager.begin_write( + bucket, key, upload_id, source_dek, source_kid + ) + native = state.dek == source_dek + else: + state = await self.multipart_manager.begin_write(bucket, key, upload_id) + + async def stage_copy(): + if native: + async with self._client(creds) as work_client: + part = await stage_ciphertext_copy( + self, + work_client, + state, + part_num, + copy_source, + head_resp, + self._source_ciphertext_segments( + src_multipart_meta, head_resp, src_wrapped_dek, src_metadata + ), + source_etag, + ) + logger.info("UPLOAD_PART_COPY_PASSTHROUGH", bucket=bucket, key=key) + return xml_responses.upload_part_copy_result( + part.md5, format_iso8601(datetime.now(UTC)) + ).encode() + + async with ( + self._client(creds) as work_client, + concurrency.reserve_copy_memory(4 * crypto.FRAME_PLAINTEXT_SIZE), + ): + source = self._iter_copy_source( + work_client, + src_bucket, + src_key, + copy_source_range, + src_wrapped_dek, + src_multipart_meta, + head_resp, + src_metadata, + ) + part = await stage_part( + self, request, work_client, state, part_num, source, verify=False + ) + return xml_responses.upload_part_copy_result( + part.md5, format_iso8601(datetime.now(UTC)) + ).encode() + + return StreamingResponse( + self._keepalive_copy_stream( + stage_copy(), bucket=bucket, key=key, part_num=part_num + ), + media_type="application/xml", + ) + passthrough_block = self._passthrough_block_reason( copy_source_range, raw_copy_source_range, @@ -1473,13 +1559,26 @@ async def _iter_copy_source( yield chunk elif src_wrapped_dek: src_kid = src_metadata.get(self.settings.kidtag_name, "") - plaintext = await self._download_encrypted_single( - client, src_bucket, src_key, src_wrapped_dek, src_kid - ) + from contextlib import aclosing + + start, end = 0, None if copy_source_range: - start, end = self._parse_copy_source_range(copy_source_range, len(plaintext)) - plaintext = plaintext[start : end + 1] - yield plaintext + total = crypto.plaintext_size(head_resp["ContentLength"]) + start, end = self._parse_copy_source_range(copy_source_range, total) + async with aclosing( + self._iter_single_plaintext( + client, + src_bucket, + src_key, + src_wrapped_dek, + src_kid, + start, + end, + if_match=head_resp.get("ETag"), + ) + ) as stream: + async for chunk in stream: + yield chunk else: async for chunk in self._stream_raw_source_with_resume( client, src_bucket, src_key, copy_source_range diff --git a/s3proxy/handlers/multipart/lifecycle.py b/s3proxy/handlers/multipart/lifecycle.py index 7e0901e..002e094 100644 --- a/s3proxy/handlers/multipart/lifecycle.py +++ b/s3proxy/handlers/multipart/lifecycle.py @@ -29,6 +29,7 @@ save_multipart_metadata, synthetic_multipart_etag, ) +from ...state.metadata import GENERATION_KEY, multipart_etag, multipart_headers from ...xml_utils import find_elements, get_element_text from ..base import BaseHandler, is_retryable_source_error @@ -90,12 +91,11 @@ async def handle_create_multipart_upload( # Build metadata (include user's x-amz-meta-*) upload_metadata = { + **self._user_metadata(request), + **multipart_headers(wrapped_dek), self.settings.dektag_name: base64.b64encode(wrapped_dek).decode(), self.settings.kidtag_name: kid, } - for hdr, val in request.headers.items(): - if hdr.lower().startswith("x-amz-meta-"): - upload_metadata[hdr[11:]] = val resp = await client.create_multipart_upload( bucket, @@ -109,12 +109,22 @@ async def handle_create_multipart_upload( upload_id = resp["UploadId"] # Store state in Redis/memory first, then persist to S3 as backup - await self.multipart_manager.create_upload(bucket, key, upload_id, dek, kid) + await self.multipart_manager.create_upload( + bucket, + key, + upload_id, + dek, + kid, + generation=upload_metadata[GENERATION_KEY], + layout_version=3, + ) # Persist DEK to S3 as backup - retry once on failure for attempt in range(2): try: - await persist_upload_state(client, bucket, key, upload_id, wrapped_dek, kid) + await persist_upload_state( + client, bucket, key, upload_id, wrapped_dek, kid, layout_version=3 + ) break except Exception as e: if attempt == 0: @@ -170,12 +180,21 @@ async def _handle_complete_multipart_upload_locked( if idempotent is not None: return idempotent - state = await self.multipart_manager.complete_upload(bucket, key, upload_id) + state = await self.multipart_manager.get_upload(bucket, key, upload_id) if not state: state = await self._recover_upload_state( client, bucket, key, upload_id, context="for complete" ) + if state.layout_version >= 3: + from .staged import cleanup_attempts, complete_staged + + response = await complete_staged(self, request, client, state) + await self.multipart_manager.abort_upload(bucket, key, upload_id) + await delete_upload_state(client, bucket, key, upload_id) + await cleanup_attempts(self, client, state) + return response + if state.deferred_copy_tail: logger.info( "COMPLETE_MULTIPART_DEFERRED_TAIL_PENDING", @@ -243,6 +262,7 @@ async def _handle_complete_multipart_upload_locked( key, MultipartMetadata( version=2, + upload_id=upload_id, part_count=len(completed_parts), total_plaintext_size=total_plaintext, parts=completed_parts, @@ -250,6 +270,7 @@ async def _handle_complete_multipart_upload_locked( kid=kid, ), ) + await self.multipart_manager.abort_upload(bucket, key, upload_id) await delete_upload_state(client, bucket, key, upload_id) logger.info( @@ -276,7 +297,9 @@ async def _try_idempotent_complete_response( ) -> Response | None: """Return success if a peer pod already finished this upload.""" meta = await load_multipart_metadata(client, bucket, key) - if meta is None: + if meta is None or meta.upload_id != upload_id: + return None + if meta.generation and (meta.upload_bucket, meta.upload_key) != (bucket, key): return None try: @@ -297,23 +320,27 @@ async def _try_idempotent_complete_response( ) location = f"{self.settings.s3_endpoint}/{bucket}/{key}" - etag = hashlib.md5( - str(meta.total_plaintext_size).encode(), usedforsecurity=False - ).hexdigest() + etag = multipart_etag(meta) return Response( content=xml_responses.complete_multipart(location, bucket, key, etag), media_type="application/xml", ) def _parse_client_parts(self, body: bytes) -> list[dict]: - client_parts = [] - root = ET.fromstring(body.decode()) - for part in find_elements(root, "Part"): - pn_text = get_element_text(part, "PartNumber") - etag_text = get_element_text(part, "ETag") - if pn_text and etag_text: - client_parts.append({"PartNumber": int(pn_text), "ETag": etag_text}) - return client_parts + try: + root = ET.fromstring(body) + client_parts = [] + for part in find_elements(root, "Part"): + number = int(get_element_text(part, "PartNumber") or "") + etag = get_element_text(part, "ETag") + if not 1 <= number <= 10000 or not etag: + raise ValueError("Invalid part") + client_parts.append({"PartNumber": number, "ETag": etag}) + if not client_parts: + raise ValueError("Empty part list") + return client_parts + except (ET.ParseError, ValueError, TypeError) as error: + raise S3Error.malformed_xml() from error def _build_s3_parts( self, @@ -408,7 +435,7 @@ async def _complete_multipart_upload_with_retry( # already be invalidated. Confirm before treating this as failed. if error_code == "NoSuchUpload" and attempt > 1: verified = await self._verify_already_completed( - client, bucket, key, expected_ciphertext_size + client, bucket, key, expected_ciphertext_size, upload_id ) if verified is not None: logger.warning( @@ -440,7 +467,12 @@ async def _complete_multipart_upload_with_retry( raise last_exc async def _verify_already_completed( - self, client: S3Client, bucket: str, key: str, expected_ciphertext_size: int + self, + client: S3Client, + bucket: str, + key: str, + expected_ciphertext_size: int, + upload_id: str = "", ) -> dict[str, Any] | None: """Check whether a retried CompleteMultipartUpload's NoSuchUpload means the prior attempt actually succeeded (backend assembled the object, then the @@ -449,7 +481,13 @@ async def _verify_already_completed( head = await client.head_object(bucket, key) except ClientError: return None - if head.get("ContentLength") == expected_ciphertext_size: + meta = await load_multipart_metadata(client, bucket, key, head) + if ( + meta is not None + and meta.upload_id == upload_id + and (not meta.generation or (meta.upload_bucket, meta.upload_key) == (bucket, key)) + and head.get("ContentLength") == expected_ciphertext_size + ): return {"ETag": head.get("ETag", "")} return None @@ -499,10 +537,15 @@ async def handle_abort_multipart_upload( upload_id=upload_id[:20] + "...", ) + state = await self.multipart_manager.get_upload(bucket, key, upload_id) await asyncio.gather( self.multipart_manager.abort_upload(bucket, key, upload_id), self._safe_abort(client, bucket, key, upload_id), delete_upload_state(client, bucket, key, upload_id), ) + if state is not None and state.layout_version >= 3: + from .staged import cleanup_attempts + + await cleanup_attempts(self, client, state) return Response(status_code=204) diff --git a/s3proxy/handlers/multipart/list.py b/s3proxy/handlers/multipart/list.py index 8ce4654..202dd16 100644 --- a/s3proxy/handlers/multipart/list.py +++ b/s3proxy/handlers/multipart/list.py @@ -25,6 +25,37 @@ async def handle_list_parts(self, request: Request, creds: S3Credentials) -> Res part_number_marker = int(part_number_marker) if part_number_marker else None max_parts = int(query.get("max-parts", ["1000"])[0]) + if not 0 <= max_parts <= 1000 or (part_number_marker or 0) < 0: + raise S3Error.invalid_argument("Invalid ListParts pagination") + state = await self.multipart_manager.get_upload(bucket, key, upload_id) + if state is not None and state.layout_version >= 3: + accepted = sorted( + (p for p in state.parts.values() if p.part_number > (part_number_marker or 0)), + key=lambda p: p.part_number, + ) + page = accepted[:max_parts] + return Response( + content=xml_responses.list_parts( + bucket=bucket, + key=key, + upload_id=upload_id, + parts=[ + { + "PartNumber": p.part_number, + "ETag": p.md5, + "Size": p.plaintext_size, + "LastModified": state.created_at.isoformat(), + } + for p in page + ], + part_number_marker=part_number_marker, + next_part_number_marker=page[-1].part_number if page else None, + max_parts=max_parts, + is_truncated=len(accepted) > len(page), + storage_class="STANDARD", + ), + media_type="application/xml", + ) try: resp = await client.list_parts( bucket, key, upload_id, part_number_marker, max_parts diff --git a/s3proxy/handlers/multipart/staged.py b/s3proxy/handlers/multipart/staged.py new file mode 100644 index 0000000..64fb8a7 --- /dev/null +++ b/s3proxy/handlers/multipart/staged.py @@ -0,0 +1,292 @@ +"""Generation-bound multipart writes with immutable, verified part attempts. + +Each client part is encrypted into a private temporary object. Only a validated, +completed attempt is published in upload state. Final assembly uses server-side +copies in client order, so replacements cannot overwrite accepted backend bytes +and arbitrary arrival order never changes the plaintext order. +""" + +import contextlib +import hashlib +import math +import uuid +from collections.abc import AsyncIterator +from urllib.parse import quote + +from fastapi import Request, Response + +from ... import crypto, xml_responses +from ...errors import S3Error +from ...signature import verify_payload_hash +from ...state import InternalPartMetadata, MultipartMetadata, PartMetadata +from ...state.metadata import ( + INTERNAL_PREFIX, + save_multipart_metadata, +) + + +async def stage_part( + handler, + request: Request, + client, + state, + part_number: int, + source: AsyncIterator[bytes], + *, + verify: bool = True, +) -> PartMetadata: + if not 1 <= part_number <= 10000: + raise S3Error.invalid_part("PartNumber must be between 1 and 10000") + # Local import avoids mixing the legacy pipeline's implementation into ours. + from .upload_part import _PlaintextReader + + state = await handler.multipart_manager.begin_write(state.bucket, state.key, state.upload_id) + stage_key = f"{INTERNAL_PREFIX}attempts/{state.generation}/{uuid.uuid4().hex}" + created = await client.create_multipart_upload(state.bucket, stage_key) + stage_id = created["UploadId"] + md5 = hashlib.md5(usedforsecurity=False) + sha = hashlib.sha256() + reader = _PlaintextReader(source) + parts = [] + uploaded = [] + total = 0 + publishing = False + try: + while data := await reader.read(crypto.FRAME_PLAINTEXT_SIZE): + md5.update(data) + sha.update(data) + total += len(data) + if total > 5 * 1024**3: + raise S3Error.invalid_argument("Client parts cannot exceed 5 GiB") + ciphertext = crypto.encrypt(data, state.dek) + number = len(parts) + 1 + response = await handler._upload_part_with_retry( + client, + state.bucket, + stage_key, + stage_id, + number, + ciphertext, + client_part_num=part_number, + ) + parts.append( + InternalPartMetadata( + number, len(data), len(ciphertext), response["ETag"].strip('"') + ) + ) + uploaded.append({"PartNumber": number, "ETag": response["ETag"]}) + del data, ciphertext + if verify: + verify_payload_hash(request, sha.hexdigest()) + expected = request.headers.get("content-length") + chunked = "aws-chunked" in request.headers.get( + "content-encoding", "" + ) or request.headers.get("x-amz-content-sha256", "").startswith("STREAMING-") + if expected is not None and not chunked and int(expected) != total: + raise S3Error.bad_request("Content-Length does not match uploaded body") + if not parts: + ciphertext = crypto.encrypt(b"", state.dek) + response = await client.upload_part(state.bucket, stage_key, stage_id, 1, ciphertext) + parts.append(InternalPartMetadata(1, 0, len(ciphertext), response["ETag"].strip('"'))) + uploaded.append({"PartNumber": 1, "ETag": response["ETag"]}) + await client.complete_multipart_upload(state.bucket, stage_key, stage_id, uploaded) + part = PartMetadata( + part_number, + total, + sum(p.ciphertext_size for p in parts), + md5.hexdigest(), + md5.hexdigest(), + internal_parts=parts, + staging_key=stage_key, + ) + # The old attempt remains immutable. Replaced attempts are cleaned by a + # bucket lifecycle rule; deleting here could race a Complete snapshot. + publishing = True + await handler.multipart_manager.add_part(state.bucket, state.key, state.upload_id, part) + return part + except BaseException: + with contextlib.suppress(Exception): + await handler._safe_abort(client, state.bucket, stage_key, stage_id) + # A cancelled Redis write may already have published the reference. + if not publishing: + with contextlib.suppress(Exception): + await client.delete_object(state.bucket, stage_key) + raise + finally: + close = getattr(source, "aclose", None) + if close is not None: + await close() + + +def select_parts(handler, body: bytes, state) -> list[PartMetadata]: + requested = handler._parse_client_parts(body) + numbers = [p["PartNumber"] for p in requested] + if not numbers or numbers != sorted(set(numbers)): + raise S3Error.invalid_part("Parts must be unique and ordered") + parts = [] + for item in requested: + part = state.parts.get(item["PartNumber"]) + if part is None or part.md5 != item["ETag"].strip('"') or not part.staging_key: + raise S3Error.invalid_part("Part or ETag does not match an accepted upload") + parts.append(part) + if any(p.plaintext_size < crypto.MIN_PART_SIZE for p in parts[:-1]): + raise S3Error.entity_too_small("All client parts except the last must be at least 5 MiB") + return parts + + +async def complete_staged(handler, request, client, state) -> Response: + parts = select_parts(handler, await request.body(), state) + # One backend copy for almost all legal client parts. Split ciphertext just + # above S3's 5 GiB CopyPart limit into balanced ranges, never a tiny tail. + max_copy = 5 * 1024**3 + counts = [math.ceil(p.ciphertext_size / max_copy) for p in parts] + if sum(counts) > 10000: + raise S3Error.invalid_request("Encrypted upload exceeds S3's 10000 backend part limit") + copies = [] + for part, count in zip(parts, counts, strict=True): + size = math.ceil(part.ciphertext_size / count) + for start in range(0, part.ciphertext_size, size): + number = len(copies) + 1 + end = min(start + size, part.ciphertext_size) - 1 + response = await copy_with_retry( + client, + state.bucket, + state.key, + state.upload_id, + number, + f"{state.bucket}/{quote(part.staging_key, safe='/')}", + f"bytes={start}-{end}", + ) + copies.append({"PartNumber": number, "ETag": response["CopyPartResult"]["ETag"]}) + # ETag is consistent across Complete, HEAD, GET and LIST; no size-only hash. + etag = ( + hashlib.md5( + b"".join(bytes.fromhex(p.md5) for p in parts), usedforsecurity=False + ).hexdigest() + + f"-{len(parts)}" + ) + wrapped = crypto.wrap_key(state.dek, handler.keyring.key_by_id(state.kid)) + meta = MultipartMetadata( + version=3, + generation=state.generation, + upload_id=state.upload_id, + upload_bucket=state.bucket, + upload_key=state.key, + client_etag=etag, + parts=parts, + part_count=len(parts), + total_plaintext_size=sum(p.plaintext_size for p in parts), + wrapped_dek=wrapped, + kid=state.kid, + ) + # The object's immutable generation pointer becomes visible only at Complete. + # Persist its full decryption map first; failed Complete is safely retryable. + await save_multipart_metadata(client, state.bucket, state.key, meta) + await handler._complete_multipart_upload_with_retry( + client, state.bucket, state.key, state.upload_id, copies, parts + ) + location = f"{handler.settings.s3_endpoint}/{state.bucket}/{state.key}" + return Response( + content=xml_responses.complete_multipart(location, state.bucket, state.key, etag), + media_type="application/xml", + ) + + +async def copy_with_retry( + client, bucket, key, upload_id, number, source, byte_range, *, if_match=None +): + import asyncio + + from ..base import SOURCE_READ_ATTEMPTS, SOURCE_READ_BACKOFF_SEC, is_retryable_source_error + + for attempt in range(SOURCE_READ_ATTEMPTS): + try: + return await client.upload_part_copy( + bucket, + key, + upload_id, + number, + source, + byte_range, + **({"copy_source_if_match": if_match} if if_match else {}), + ) + except Exception as error: + if attempt + 1 == SOURCE_READ_ATTEMPTS or not is_retryable_source_error(error): + raise + await asyncio.sleep(SOURCE_READ_BACKOFF_SEC * 2**attempt) + + +async def cleanup_attempts(handler, client, state): + """Best-effort terminal cleanup; lifecycle expiry covers crashes and late writers.""" + try: + token = None + while True: + page = await client.list_objects_v2( + state.bucket, + prefix=f"{INTERNAL_PREFIX}attempts/{state.generation}/", + continuation_token=token, + ) + keys = [o["Key"] for o in page.get("Contents", [])] + if keys: + await client.delete_objects(state.bucket, [{"Key": key} for key in keys]) + if not page.get("IsTruncated"): + break + token = page["NextContinuationToken"] + except Exception as error: + import structlog + + structlog.get_logger(__name__).warning("STAGING_CLEANUP_FAILED", error=str(error)) + + +async def stage_ciphertext_copy(handler, client, state, part_number, source, head, segments, etag): + """Snapshot a whole encrypted source without changing its ciphertext/nonces.""" + if not 1 <= part_number <= 10000 or sum(p.plaintext_size for p in segments) > 5 * 1024**3: + raise S3Error.invalid_argument("Invalid part number or source exceeds 5 GiB") + stage_key = f"{INTERNAL_PREFIX}attempts/{state.generation}/{uuid.uuid4().hex}" + upload = await client.create_multipart_upload(state.bucket, stage_key) + stage_id = upload["UploadId"] + publishing = False + try: + total = head["ContentLength"] + count = max(1, math.ceil(total / (5 * 1024**3))) + step = math.ceil(total / count) + copied = [] + for start in range(0, total, step): + number = len(copied) + 1 + result = await copy_with_retry( + client, + state.bucket, + stage_key, + stage_id, + number, + source, + f"bytes={start}-{min(start + step, total) - 1}", + if_match=head["ETag"], + ) + copied.append({"PartNumber": number, "ETag": result["CopyPartResult"]["ETag"]}) + await client.complete_multipart_upload(state.bucket, stage_key, stage_id, copied) + # S3 ETags are opaque; keep a 128-bit token for multipart ETag composition. + if len(etag) != 32 or any(c not in "0123456789abcdef" for c in etag): + etag = hashlib.md5(etag.encode(), usedforsecurity=False).hexdigest() + part = PartMetadata( + part_number, + sum(p.plaintext_size for p in segments), + total, + etag, + etag, + internal_parts=[ + InternalPartMetadata(i, p.plaintext_size, p.ciphertext_size, "") + for i, p in enumerate(segments, 1) + ], + staging_key=stage_key, + ) + publishing = True + await handler.multipart_manager.add_part(state.bucket, state.key, state.upload_id, part) + return part + except BaseException: + with contextlib.suppress(Exception): + await handler._safe_abort(client, state.bucket, stage_key, stage_id) + if not publishing: + with contextlib.suppress(Exception): + await client.delete_object(state.bucket, stage_key) + raise diff --git a/s3proxy/handlers/multipart/upload_part.py b/s3proxy/handlers/multipart/upload_part.py index ae06e2e..3b94f41 100644 --- a/s3proxy/handlers/multipart/upload_part.py +++ b/s3proxy/handlers/multipart/upload_part.py @@ -17,7 +17,6 @@ from ... import crypto from ...client import S3Client, S3Credentials from ...errors import S3Error, raise_for_client_error, raise_for_exception -from ...signature import deferred_signature_required, verify_deferred_payload_hash from ...state import ( InternalPartMetadata, MultipartUploadState, @@ -98,144 +97,21 @@ async def handle_upload_part(self, request: Request, creds: S3Credentials) -> Re # Get upload state state = await self._get_or_recover_state(client, bucket, key, upload_id, part_num) - # Parse request info - content_encoding = request.headers.get("content-encoding", "") - content_sha = request.headers.get("x-amz-content-sha256", "") - try: - content_length = int(request.headers.get("content-length", "0")) - except ValueError: - content_length = 0 - - upload_start_time = time.monotonic() - logger.info( - "UPLOAD_PART_START", - bucket=bucket, - key=key, - upload_id=upload_id[:20] + "...", - part_number=part_num, - content_length_mb=f"{content_length / 1024 / 1024:.2f}MB", - ) - - # Determine encoding type and upload path. - cls = classify_upload(content_sha, content_encoding, content_length) - is_unsigned = cls.is_unsigned - is_streaming_sig = cls.is_streaming_sig - needs_chunked_decode = cls.needs_chunked_decode - is_large_signed = cls.is_large_signed - use_framed = cls.use_framed - - # Smallest internal part that bounds memory while staying within the - # per-client part-number allocation range (so we never collide and - # never buffer more than necessary). - internal_part_size = crypto.memory_bounded_part_size(content_length) - estimated_parts = max(1, -(-content_length // internal_part_size)) - logger.info( - "UPLOAD_PART_CONFIG", - bucket=bucket, - key=key, - part_number=part_num, - internal_part_size_mb=f"{internal_part_size / 1024 / 1024:.2f}MB", - estimated_internal_parts=estimated_parts, - is_unsigned=is_unsigned, - is_large_signed=is_large_signed, - is_streaming_sig=is_streaming_sig, - needs_chunked_decode=needs_chunked_decode, - upload_path="framed" if use_framed else "buffered", - ) - - # Per-client allocation: dense 1:1 for all-5MB uploads (ClickHouse 600+ - # parts), sparse ranges once a client part needs multiple internals (Scylla). - internal_part_start = await self.multipart_manager.allocate_internal_parts( - bucket, - key, - upload_id, - estimated_parts, - client_part_number=part_num, - ) - internal_part_end = internal_part_start + estimated_parts - 1 - logger.info( - "UPLOAD_PART_INTERNAL_RANGE", - bucket=bucket, - key=key, - part_number=part_num, - internal_part_start=internal_part_start, - internal_part_end=internal_part_end, - estimated_internal_parts=estimated_parts, - ) + if state.layout_version >= 3: + from ..objects.put import _iter_request_body + from .staged import stage_part - try: - # Known-length direct streams (unsigned or large signed, e.g. barman - # backups) can be uploaded frame-by-frame with O(frame) memory. - # aws-chunked / streaming-sig bodies don't know the size up front and - # keep the buffered path. - if use_framed: - result = await self._stream_and_upload_framed( - request, - client, - bucket, - key, - upload_id, - part_num, - state, - content_length, - internal_part_size, - internal_part_start, - estimated_parts, - ) - else: - result = await self._stream_and_upload( - request, - client, - bucket, - key, - upload_id, - part_num, - state, - content_sha, - content_length, - is_unsigned, - is_streaming_sig, - is_large_signed, - needs_chunked_decode, - internal_part_size, - internal_part_start, - ) - - # Late signature verification for large signed uploads - if deferred_signature_required(request): - verify_deferred_payload_hash( - request, request.app.state.verifier, result["computed_sha256"] - ) - elif is_large_signed and content_sha and result["computed_sha256"] != content_sha: - logger.warning( - "UPLOAD_PART_SHA256_MISMATCH", - bucket=bucket, - key=key, - part_num=part_num, - expected=content_sha, - computed=result["computed_sha256"], - ) - raise S3Error.signature_does_not_match("Signature verification failed") - - upload_duration = time.monotonic() - upload_start_time - logger.info( - "UPLOAD_PART_COMPLETE", - bucket=bucket, - key=key, - part_number=part_num, - plaintext_mb=f"{result['total_plaintext_size'] / 1024 / 1024:.2f}MB", - internal_parts=result["internal_parts_count"], - duration_sec=f"{upload_duration:.2f}", + decode = "aws-chunked" in request.headers.get( + "content-encoding", "" + ) or request.headers.get("x-amz-content-sha256", "").startswith("STREAMING-") + part = await stage_part( + self, request, client, state, part_num, _iter_request_body(request, decode) ) + return Response(headers={"ETag": f'"{part.md5}"'}) - return Response(headers={"ETag": f'"{result["client_etag"]}"'}) - - except S3Error: - raise - except ClientError as e: - return self._handle_client_error(e, bucket, key, part_num, upload_id) - except Exception as e: - return self._handle_generic_error(e, bucket, key, part_num, upload_id) + raise S3Error.invalid_request( + "Legacy in-flight uploads must be restarted after upgrade" + ) async def _upload_part_with_retry( self, @@ -689,7 +565,7 @@ async def _upload_internal_part_with_semaphore( try: # Encrypt - nonce = crypto.derive_part_nonce(upload_id, internal_part_num) + nonce = crypto.generate_nonce() ciphertext = crypto.encrypt(data, state.dek, nonce) plaintext_size = len(data) ciphertext_size = len(ciphertext) diff --git a/s3proxy/handlers/objects/get.py b/s3proxy/handlers/objects/get.py index 901f095..9bb8453 100644 --- a/s3proxy/handlers/objects/get.py +++ b/s3proxy/handlers/objects/get.py @@ -4,25 +4,22 @@ import base64 import contextlib from collections.abc import AsyncIterator, Awaitable, Callable -from itertools import accumulate from typing import Any import structlog from botocore.exceptions import ClientError from fastapi import Request, Response -from fastapi.responses import StreamingResponse from structlog.stdlib import BoundLogger -from ... import concurrency, crypto +from ... import crypto from ...client import S3Client, S3Credentials -from ...concurrency import MAX_BUFFER_SIZE from ...errors import S3Error from ...state import ( MultipartMetadata, calculate_part_range, - load_multipart_metadata, ) from ...streaming import STREAM_CHUNK_SIZE +from ...streaming.response import OwnedStreamingResponse from ...utils import format_http_date from ..base import BaseHandler @@ -49,7 +46,9 @@ async def handle_get_object(self, request: Request, creds: S3Credentials) -> Res # Get the effective ETag (client-etag for encrypted, S3 etag otherwise) metadata = head_resp.get("Metadata", {}) - effective_etag = self._get_effective_etag(metadata, head_resp.get("ETag", "")) + descriptor = await self._resolve_object(client, bucket, key, head_resp) + mp_meta = descriptor.multipart + effective_etag = descriptor.etag # Check conditional headers (inherited from BaseHandler) cond_response = self._check_conditional_headers( @@ -64,9 +63,9 @@ async def handle_get_object(self, request: Request, creds: S3Credentials) -> Res if cond_response: return cond_response - if meta := await load_multipart_metadata(client, bucket, key): + if (meta := mp_meta) is not None: response = await self._get_multipart( - client, bucket, key, meta, range_header, last_modified, creds + client, bucket, key, meta, range_header, last_modified, creds, head_resp ) else: response = await self._get_single( @@ -120,20 +119,35 @@ async def _stream_unencrypted( last_modified: str | None, ) -> Response: logger.info("GET_UNENCRYPTED", bucket=bucket, key=key) - resp = await client.get_object(bucket, key, range_header=range_header) + lease = self._client(client.credentials) + stream_client = await lease.__aenter__() + try: + resp = await stream_client.get_object(bucket, key, range_header=range_header) + except BaseException: + await lease.__aexit__(None, None, None) + raise s3_body = resp["Body"] - headers = self._build_response_headers(resp, last_modified) - async def stream_s3_body() -> AsyncIterator[bytes]: + async def stream_s3_body(): async with s3_body: while chunk := await s3_body.read(STREAM_CHUNK_SIZE): yield chunk + async def cleanup(): + try: + await s3_body.__aexit__(None, None, None) + finally: + await lease.__aexit__(None, None, None) + if "ContentRange" in resp: headers["Content-Range"] = resp["ContentRange"] - return StreamingResponse(stream_s3_body(), status_code=206, headers=headers) - return StreamingResponse(stream_s3_body(), headers=headers) + return OwnedStreamingResponse( + stream_s3_body(), + headers=headers, + status_code=206 if "ContentRange" in resp else 200, + cleanup=cleanup, + ) async def _decrypt_single_object( self, @@ -147,52 +161,41 @@ async def _decrypt_single_object( kid: str = "", ) -> Response: logger.info("GET_ENCRYPTED_SINGLE", bucket=bucket, key=key) - resp = await client.get_object(bucket, key) - content_length = resp.get("ContentLength", 0) - - # Encrypted decrypts buffer ciphertext + plaintext simultaneously. - # Acquire additional memory beyond the initial MAX_BUFFER_SIZE reservation. - additional = max(0, content_length * 2 - MAX_BUFFER_SIZE) - extra_reserved = 0 - try: - if additional > 0: - extra_reserved = await concurrency.try_acquire_memory(additional) + resp = await client.get_object(bucket, key, if_match=head_resp.get("ETag")) - wrapped_dek = base64.b64decode(wrapped_dek_b64) - async with resp["Body"] as body: - ciphertext = await body.read() - plaintext = crypto.decrypt_object(ciphertext, wrapped_dek, self.keyring.key_by_id(kid)) - del ciphertext - - content_type = head_resp.get("ContentType", "application/octet-stream") - cache_control = head_resp.get("CacheControl") - expires = head_resp.get("Expires") + wrapped_dek = base64.b64decode(wrapped_dek_b64) + dek = crypto.unwrap_key(wrapped_dek, self.keyring.key_by_id(kid)) + from ...streaming.authenticated import decrypt_to_file, file_range + # Authenticate to bounded disk storage before emitting any plaintext. This + # also handles old single-envelope objects larger than the memory budget. + spool, length = await decrypt_to_file(resp["Body"], dek) + try: + start, end = ( + self._parse_range(range_header, length) if range_header else (0, length - 1) + ) + headers = self._build_headers( + head_resp.get("ContentType", "application/octet-stream"), + end - start + 1, + last_modified, + head_resp.get("CacheControl"), + head_resp.get("Expires"), + ) if range_header: - start, end = self._parse_range(range_header, len(plaintext)) - headers = self._build_headers( - content_type=content_type, - content_length=end - start + 1, - last_modified=last_modified, - cache_control=cache_control, - expires=expires, - ) - headers["Content-Range"] = f"bytes {start}-{end}/{len(plaintext)}" - return Response( - content=plaintext[start : end + 1], status_code=206, headers=headers - ) + headers["Content-Range"] = f"bytes {start}-{end}/{length}" - headers = self._build_headers( - content_type=content_type, - content_length=len(plaintext), - last_modified=last_modified, - cache_control=cache_control, - expires=expires, + async def cleanup(): + spool.close() + + return OwnedStreamingResponse( + file_range(spool, start, end), + headers=headers, + cleanup=cleanup, + status_code=206 if range_header else 200, ) - return Response(content=plaintext, headers=headers) - finally: - if extra_reserved > 0: - await concurrency.release_memory(extra_reserved) + except BaseException: + spool.close() + raise async def _get_multipart( self, @@ -203,26 +206,44 @@ async def _get_multipart( range_header: str | None, last_modified: str | None, creds: S3Credentials, + head_resp: dict | None = None, ) -> Response: dek = crypto.unwrap_key(meta.wrapped_dek, self.keyring.key_by_id(meta.kid)) total = meta.total_plaintext_size start, end = self._parse_range(range_header, total) if range_header else (0, total - 1) parts = calculate_part_range(meta.parts, start, end) - # Build lookup: part_number -> (part_metadata, ciphertext_offset) - sorted_parts = sorted(meta.parts, key=lambda p: p.part_number) - offsets = [0, *accumulate(p.ciphertext_size for p in sorted_parts)] - part_info = {p.part_number: (p, offsets[i]) for i, p in enumerate(sorted_parts)} - # Get actual object size and content type - actual_size, content_type, cache_control, expires_val = await self._get_object_info( - client, bucket, key, meta - ) + head_resp = head_resp or await client.head_object(bucket, key) + content_type = head_resp.get("ContentType", "application/octet-stream") + cache_control = head_resp.get("CacheControl") + expires_val = head_resp.get("Expires") # Create stream generator - stream = self._create_multipart_stream( - creds, bucket, key, parts, part_info, dek, actual_size, start, end - ) + async def stream(): + from ...streaming.frames import plaintext_frames + + async with ( + self._client(creds) as stream_client, + contextlib.aclosing( + plaintext_frames( + stream_client, + bucket, + key, + meta, + dek, + start if range_header else None, + end if range_header else None, + if_match=head_resp.get("ETag"), + ciphertext_size=head_resp.get("ContentLength"), + ) + ) as plaintext, + ): + try: + async for chunk in plaintext: + yield chunk + except ClientError as error: + self._raise_s3_error(error, bucket, key) # Build response length = sum(e - s + 1 for _, s, e in parts) @@ -235,8 +256,12 @@ async def _get_multipart( ) if range_header: headers["Content-Range"] = f"bytes {start}-{end}/{total}" - return StreamingResponse(stream, status_code=206, headers=headers) - return StreamingResponse(stream, headers=headers) + return OwnedStreamingResponse(stream(), status_code=206, headers=headers) + if total == 0: + async for _ in stream(): + pass + return Response(headers=headers) + return OwnedStreamingResponse(stream(), headers=headers) async def _get_object_info( self, client: S3Client, bucket: str, key: str, meta: MultipartMetadata diff --git a/s3proxy/handlers/objects/misc.py b/s3proxy/handlers/objects/misc.py index 75b9d18..fcce474 100644 --- a/s3proxy/handlers/objects/misc.py +++ b/s3proxy/handlers/objects/misc.py @@ -4,8 +4,8 @@ import base64 import hashlib import xml.etree.ElementTree as ET +from dataclasses import replace from datetime import UTC, datetime -from urllib.parse import quote import structlog from botocore.exceptions import ClientError @@ -20,10 +20,15 @@ MultipartMetadata, PartMetadata, delete_multipart_metadata, - load_multipart_metadata, save_multipart_metadata, ) -from ...state.metadata import _internal_meta_key +from ...state.metadata import ( + FORMAT_KEY, + GENERATION_KEY, + generation_for, + multipart_etag, + multipart_headers, +) from ...utils import format_http_date, format_iso8601 from ...xml_utils import find_element, find_elements from ..base import BaseHandler @@ -50,8 +55,8 @@ async def handle_head_object(self, request: Request, creds: S3Credentials) -> Re last_modified_dt = resp.get("LastModified") # Get the effective ETag (client-etag for encrypted, S3 etag otherwise) - metadata = resp.get("Metadata", {}) - effective_etag = self._get_effective_etag(metadata, resp.get("ETag", "")) + descriptor = await self._resolve_object(client, bucket, key, resp) + effective_etag = descriptor.etag # Check conditional headers (inherited from BaseHandler) cond_response = self._check_conditional_headers( @@ -68,27 +73,10 @@ async def handle_head_object(self, request: Request, creds: S3Credentials) -> Re extra_headers = self._build_head_extra_headers(resp, last_modified) - if meta := await load_multipart_metadata(client, bucket, key): - headers = { - "Content-Length": str(meta.total_plaintext_size), - "Content-Type": resp.get("ContentType", "application/octet-stream"), - "ETag": f'"{ - hashlib.md5( - str(meta.total_plaintext_size).encode(), - usedforsecurity=False, - ).hexdigest() - }"', - **extra_headers, - } - return Response(headers=headers) - - size = self._get_plaintext_size(metadata, resp.get("ContentLength", 0)) - etag = self._get_effective_etag(metadata, resp.get("ETag", "")) - headers = { - "Content-Length": str(size), + "Content-Length": str(descriptor.plaintext_size), "Content-Type": resp.get("ContentType", "application/octet-stream"), - "ETag": f'"{etag}"', + "ETag": f'"{effective_etag}"', **extra_headers, } return Response(headers=headers) @@ -162,7 +150,10 @@ async def handle_copy_object(self, request: Request, creds: S3Credentials) -> Re if metadata_directive == "REPLACE": new_metadata = {} for hdr, val in request.headers.items(): - if hdr.lower().startswith("x-amz-meta-"): + if ( + hdr.lower().startswith("x-amz-meta-") + and hdr[11:] not in self._internal_meta_keys() + ): new_metadata[hdr[11:]] = val # Strip x-amz-meta- prefix logger.info( @@ -184,11 +175,12 @@ async def handle_copy_object(self, request: Request, creds: S3Credentials) -> Re src_key=src_key, error=str(e), ) - raise S3Error.no_such_key(src_key) from e + self._raise_s3_error(e, src_bucket, src_key) src_metadata = head_resp.get("Metadata", {}) src_wrapped_dek = src_metadata.get(self.settings.dektag_name) - src_multipart_meta = await load_multipart_metadata(client, src_bucket, src_key) + source = await self._resolve_object(client, src_bucket, src_key, head_resp) + src_multipart_meta = source.multipart if not src_wrapped_dek and not src_multipart_meta: # Not encrypted - pass through @@ -203,6 +195,7 @@ async def handle_copy_object(self, request: Request, creds: S3Credentials) -> Re metadata_directive, new_metadata, request, + head_resp, ) # Encrypted source. A plain COPY needs no re-encrypt: the ciphertext @@ -267,6 +260,7 @@ async def _copy_passthrough( metadata_directive: str, new_metadata: dict[str, str] | None, request: Request, + head_resp: dict, ) -> Response: logger.info( "COPY_PASSTHROUGH", @@ -285,9 +279,19 @@ async def _copy_passthrough( bucket, key, copy_source, - metadata=new_metadata, - metadata_directive=metadata_directive, - content_type=content_type, + metadata={ + **( + new_metadata or {} + if metadata_directive == "REPLACE" + else head_resp.get("Metadata", {}) + ), + FORMAT_KEY: "plain-v3", + }, + metadata_directive="REPLACE", + content_type=content_type or head_resp.get("ContentType"), + copy_source_if_match=head_resp.get("ETag"), + cache_control=head_resp.get("CacheControl") if metadata_directive == "COPY" else None, + expires=head_resp.get("Expires") if metadata_directive == "COPY" else None, tagging_directive=tagging_directive if tagging_directive != "COPY" else None, tagging=tagging, ) @@ -333,30 +337,39 @@ async def _copy_passthrough_encrypted( is_multipart=bool(src_multipart_meta), ) + metadata = dict(head_resp.get("Metadata", {})) + if src_multipart_meta: + if not src_multipart_meta.generation: + src_multipart_meta = replace( + src_multipart_meta, generation=generation_for(src_multipart_meta.wrapped_dek) + ) + metadata.update( + {FORMAT_KEY: "multipart-v3", GENERATION_KEY: src_multipart_meta.generation} + ) + await save_multipart_metadata(client, bucket, key, src_multipart_meta) + else: + metadata[FORMAT_KEY] = "single-v3" resp = await client.copy_object( bucket, key, copy_source, - metadata_directive="COPY", - content_type=content_type, + metadata=metadata, + metadata_directive="REPLACE", + content_type=content_type or head_resp.get("ContentType"), + cache_control=head_resp.get("CacheControl"), + expires=head_resp.get("Expires"), + copy_source_if_match=head_resp.get("ETag"), ) - # Multipart objects keep their part/frame map in a separate sidecar - # object; the destination needs its own copy or the read path can't - # reconstruct (and decrypt) it. - if src_multipart_meta: - await client.copy_object( - bucket, - _internal_meta_key(key), - f"{src_bucket}/{quote(_internal_meta_key(src_key), safe='/')}", - metadata_directive="COPY", - ) - # Encrypted objects report the plaintext md5 (client-etag), not the # ciphertext ETag, to match GET/HEAD and the re-encrypt path. src_metadata = head_resp.get("Metadata", {}) result = resp.get("CopyObjectResult", {}) - etag = src_metadata.get("client-etag") or str(result.get("ETag", "")).strip('"') + etag = ( + multipart_etag(src_multipart_meta) + if src_multipart_meta + else src_metadata.get("client-etag") or str(result.get("ETag", "")).strip('"') + ) last_modified = result.get("LastModified") if hasattr(last_modified, "isoformat"): last_modified = last_modified.isoformat().replace("+00:00", "Z") @@ -469,6 +482,7 @@ async def _copy_encrypted_inner( etag = hashlib.md5(plaintext, usedforsecurity=False).hexdigest() dest_metadata = { + FORMAT_KEY: "single-v3", self.settings.dektag_name: base64.b64encode(encrypted.wrapped_dek).decode(), self.settings.kidtag_name: dest_kid, "client-etag": etag, @@ -550,6 +564,7 @@ async def _copy_encrypted_streaming( wrapped_dek = crypto.wrap_key(dek, dest_kek) upload_metadata: dict[str, str] = { + **multipart_headers(wrapped_dek), self.settings.dektag_name: base64.b64encode(wrapped_dek).decode(), self.settings.kidtag_name: dest_kid, } @@ -593,27 +608,28 @@ async def _copy_encrypted_streaming( head_resp, ) + etag = hashlib.sha256(wrapped_dek).hexdigest() + await save_multipart_metadata( + client, + bucket, + key, + MultipartMetadata( + version=3, + generation=generation_for(wrapped_dek), + upload_id=upload_id, + client_etag=etag, + part_count=len(meta_parts), + total_plaintext_size=total_plaintext, + parts=meta_parts, + wrapped_dek=wrapped_dek, + kid=dest_kid, + ), + ) await client.complete_multipart_upload(bucket, key, upload_id, s3_parts) - except Exception: + except BaseException: await self._safe_abort(client, bucket, key, upload_id) raise - await save_multipart_metadata( - client, - bucket, - key, - MultipartMetadata( - version=2, - part_count=len(meta_parts), - total_plaintext_size=total_plaintext, - parts=meta_parts, - wrapped_dek=wrapped_dek, - kid=dest_kid, - ), - ) - - etag = hashlib.md5(str(total_plaintext).encode(), usedforsecurity=False).hexdigest() - logger.info( "COPY_ENCRYPTED_STREAMING_COMPLETE", src_bucket=src_bucket, @@ -716,7 +732,7 @@ async def _encrypt_and_upload_chunk( meta_parts: list[PartMetadata], total_plaintext: int, ) -> tuple[int, list[dict], list[PartMetadata], int]: - nonce = crypto.derive_part_nonce(upload_id, part_number) + nonce = crypto.generate_nonce() ciphertext = crypto.encrypt(chunk, dek, nonce) resp = await client.upload_part(bucket, key, upload_id, part_number, ciphertext) etag = resp["ETag"].strip('"') @@ -765,10 +781,20 @@ async def _iter_object_plaintext( yield chunk else: src_kid = head_resp.get("Metadata", {}).get(self.settings.kidtag_name, "") - plaintext = await self._download_encrypted_single( - client, src_bucket, src_key, src_wrapped_dek, src_kid - ) - yield plaintext + from contextlib import aclosing + + async with aclosing( + self._iter_single_plaintext( + client, + src_bucket, + src_key, + src_wrapped_dek, + src_kid, + if_match=head_resp.get("ETag"), + ) + ) as stream: + async for chunk in stream: + yield chunk async def handle_get_object_tagging(self, request: Request, creds: S3Credentials) -> Response: bucket, key = self._parse_path(request.url.path) diff --git a/s3proxy/handlers/objects/put.py b/s3proxy/handlers/objects/put.py index ed22513..0a4d98b 100644 --- a/s3proxy/handlers/objects/put.py +++ b/s3proxy/handlers/objects/put.py @@ -1,11 +1,13 @@ """PUT object operations with encryption support.""" +import asyncio import base64 import hashlib from collections.abc import AsyncIterator from typing import Any import structlog +from botocore.exceptions import ClientError from fastapi import Request, Response from starlette.requests import ClientDisconnect from structlog.stdlib import BoundLogger @@ -14,14 +16,14 @@ from ...client import S3Client, S3Credentials from ...disconnect import ClientDisconnectError from ...errors import S3Error -from ...signature import verify_deferred_payload_hash +from ...signature import verify_deferred_payload_hash, verify_payload_hash from ...state import ( MultipartMetadata, PartMetadata, save_multipart_metadata, ) +from ...state.metadata import FORMAT_KEY, generation_for, multipart_headers from ...streaming import decode_aws_chunked, decode_aws_chunked_stream -from ...utils import etag_matches from ..base import BaseHandler logger: BoundLogger = structlog.get_logger(__name__) @@ -52,29 +54,9 @@ class PutObjectMixin(BaseHandler): async def handle_put_object(self, request: Request, creds: S3Credentials) -> Response: bucket, key = self._parse_path(request.url.path) async with self._client(creds) as client: - # Check If-None-Match header (prevents overwriting existing objects) if_none_match = request.headers.get("if-none-match") - if if_none_match: - try: - head_resp = await client.head_object(bucket, key) - # Object exists - check if etag matches - if if_none_match.strip() == "*": - # * means fail if object exists at all - raise S3Error.precondition_failed( - "At least one of the pre-conditions you specified did not hold" - ) - # Check specific etag match - metadata = head_resp.get("Metadata", {}) - existing_etag = self._get_effective_etag(metadata, head_resp.get("ETag", "")) - if etag_matches(existing_etag, if_none_match): - raise S3Error.precondition_failed( - "At least one of the pre-conditions you specified did not hold" - ) - except S3Error: - raise - except Exception: - # Object doesn't exist - proceed with upload - pass + if if_none_match and if_none_match != "*": + raise S3Error.invalid_argument("If-None-Match on PUT must be '*'") content_type = request.headers.get("content-type", "application/octet-stream") content_sha = request.headers.get("x-amz-content-sha256", "") content_encoding = request.headers.get("content-encoding", "") @@ -91,7 +73,12 @@ async def handle_put_object(self, request: Request, creds: S3Credentials) -> Res needs_chunked_decode = "aws-chunked" in content_encoding or is_streaming_sig # Stream large uploads to avoid buffering - if is_unsigned or is_streaming_sig or content_length > crypto.MAX_BUFFER_SIZE: + if ( + is_unsigned + or is_streaming_sig + or content_length > crypto.MAX_BUFFER_SIZE + or "content-length" not in request.headers + ): logger.debug( "PUT_STREAMING", bucket=bucket, @@ -159,6 +146,7 @@ async def _put_buffered( if needs_chunked_decode: body = decode_aws_chunked(body) + verify_payload_hash(request, hashlib.sha256(body).hexdigest()) kid, kek = self.keyring.key_for(client.credentials.access_key) encrypted = crypto.encrypt_object(body, kek) logger.debug( @@ -176,6 +164,8 @@ async def _put_buffered( key, encrypted.ciphertext, metadata={ + **self._user_metadata(request), + FORMAT_KEY: "single-v3", self.settings.dektag_name: base64.b64encode(encrypted.wrapped_dek).decode(), self.settings.kidtag_name: kid, "client-etag": etag, @@ -185,6 +175,7 @@ async def _put_buffered( cache_control=cache_control, expires=expires, tagging=tagging, + **({"if_none_match": "*"} if request.headers.get("if-none-match") else {}), ) return Response(headers={"ETag": f'"{etag}"'}) @@ -209,6 +200,12 @@ async def _put_streaming( bucket, key, content_type=content_type, + metadata={ + **self._user_metadata(request), + **multipart_headers(wrapped_dek), + self.settings.dektag_name: base64.b64encode(wrapped_dek).decode(), + self.settings.kidtag_name: kid, + }, cache_control=cache_control, expires=expires, tagging=tagging, @@ -227,7 +224,7 @@ async def _put_streaming( async def upload_part(data: bytes) -> None: nonlocal part_num part_num += 1 - nonce = crypto.derive_part_nonce(upload_id, part_num) + nonce = crypto.generate_nonce() data_len = len(data) data_md5 = hashlib.md5(data, usedforsecurity=False).hexdigest() ciphertext = crypto.encrypt(data, dek, nonce) @@ -283,13 +280,9 @@ async def upload_part(data: bytes) -> None: # Verify SHA256 if provided, or deferred SigV4 after streaming hash if deferred_sig: - try: - verify_deferred_payload_hash( - request, request.app.state.verifier, sha256_hash.hexdigest() - ) - except S3Error: - await client.abort_multipart_upload(bucket, key, upload_id) - raise + verify_deferred_payload_hash( + request, request.app.state.verifier, sha256_hash.hexdigest() + ) elif expected_sha256 is not None: computed_sha256 = sha256_hash.hexdigest() if computed_sha256 != expected_sha256: @@ -301,19 +294,22 @@ async def upload_part(data: bytes) -> None: expected=expected_sha256, computed=computed_sha256, ) - await client.abort_multipart_upload(bucket, key, upload_id) raise S3Error.signature_does_not_match( f"SHA256 mismatch: {computed_sha256} != {expected_sha256}" ) # Complete upload - await client.complete_multipart_upload(bucket, key, upload_id, parts_complete) + if not parts_complete: + await upload_part(b"") await save_multipart_metadata( client, bucket, key, MultipartMetadata( - version=2, + version=3, + generation=generation_for(wrapped_dek), + upload_id=upload_id, + client_etag=md5_hash.hexdigest(), part_count=len(parts_meta), total_plaintext_size=total_plaintext_size, parts=parts_meta, @@ -322,6 +318,13 @@ async def upload_part(data: bytes) -> None: ), ) + await client.complete_multipart_upload( + bucket, + key, + upload_id, + parts_complete, + **({"if_none_match": "*"} if request.headers.get("if-none-match") else {}), + ) etag = md5_hash.hexdigest() logger.info( "PUT_STREAMING_COMPLETE", @@ -336,7 +339,8 @@ async def upload_part(data: bytes) -> None: except ClientDisconnectError, ClientDisconnect: await self._safe_abort(client, bucket, key, upload_id) raise ClientDisconnectError.raised() from None - except S3Error: + except S3Error, ClientError, asyncio.CancelledError: + await self._safe_abort(client, bucket, key, upload_id) raise except Exception as e: logger.error( diff --git a/s3proxy/request_handler.py b/s3proxy/request_handler.py index 4b34490..00ee2cc 100644 --- a/s3proxy/request_handler.py +++ b/s3proxy/request_handler.py @@ -2,6 +2,7 @@ from __future__ import annotations +import hashlib import os import time from urllib.parse import parse_qs @@ -26,11 +27,20 @@ ) from .request_context import bind_request, clear_request, get_request_context from .routing import RequestDispatcher +from .signature import verify_payload_hash +from .streaming.response import OwnedStreamingResponse pod_name = os.environ.get("HOSTNAME", "unknown") logger: BoundLogger = structlog.get_logger(__name__).bind(pod=pod_name) +async def _record_request_safely(*args): + try: + await record_request(*args) + except Exception as error: + logger.warning("REQUEST_METRICS_FAILED", error=str(error)) + + def _is_dashboard_path(request: Request, path: str) -> bool: """True if the path targets the dashboard (so it's excluded from stats). @@ -138,38 +148,40 @@ async def handle_proxy_request( # Check memory limit BEFORE reading body data - reject if at capacity reserved_memory = 0 - needs_limit = method in ("PUT", "POST", "GET") + needs_limit = method in ("PUT", "POST", "GET") and not request.headers.get("x-amz-copy-source") memory_limit = concurrency.get_memory_limit() - if memory_limit > 0 and needs_limit: - try: - content_length = int(request.headers.get("content-length", "0")) - except ValueError: - content_length = 0 - bind_request(method=method, path=path, query=query, content_length=content_length) - memory_needed = concurrency.estimate_memory_footprint(method, content_length) - - logger.info( - "REQUEST_ARRIVED - attempting to acquire memory", - memory_needed_mb=round(memory_needed / 1024 / 1024, 2), - active_mb=round(concurrency.get_active_memory() / 1024 / 1024, 2), - limit_mb=round(memory_limit / 1024 / 1024, 2), - method=method, - path=path, - content_length=content_length, - ) - reserved_memory = await concurrency.try_acquire_memory(memory_needed) - logger.info( - "MEMORY_RESERVED", - reserved_mb=round(reserved_memory / 1024 / 1024, 2), - active_mb=round(concurrency.get_active_memory() / 1024 / 1024, 2), - limit_mb=round(memory_limit / 1024 / 1024, 2), - method=method, - path=path, - ) - response = None try: + if memory_limit > 0 and needs_limit: + try: + content_length = int(request.headers.get("content-length", "0")) + except ValueError: + content_length = 0 + bind_request(method=method, path=path, query=query, content_length=content_length) + memory_needed = concurrency.estimate_memory_footprint(method, content_length) + if method in ("PUT", "POST") and "content-length" not in request.headers: + memory_needed = 4 * crypto.MAX_BUFFER_SIZE + + logger.info( + "REQUEST_ARRIVED - attempting to acquire memory", + memory_needed_mb=round(memory_needed / 1024 / 1024, 2), + active_mb=round(concurrency.get_active_memory() / 1024 / 1024, 2), + limit_mb=round(memory_limit / 1024 / 1024, 2), + method=method, + path=path, + content_length=content_length, + ) + reserved_memory = await concurrency.try_acquire_memory(memory_needed) + logger.info( + "MEMORY_RESERVED", + reserved_mb=round(reserved_memory / 1024 / 1024, 2), + active_mb=round(concurrency.get_active_memory() / 1024 / 1024, 2), + limit_mb=round(memory_limit / 1024 / 1024, 2), + method=method, + path=path, + ) + response = await _handle_proxy_request_impl(request, handler, verifier) if response is not None: status_code = response.status_code @@ -180,9 +192,63 @@ async def handle_proxy_request( # GETs accumulates frames and OOMs the pod while the limiter reads ~budget. # Hold the reservation for the whole stream lifetime so the limiter bounds # how many streaming GETs run at once (admission control). - if reserved_memory > 0 and isinstance(response, StreamingResponse): - response.body_iterator = _release_after_stream(response.body_iterator, reserved_memory) + if isinstance(response, StreamingResponse): + original = response + reserved = reserved_memory reserved_memory = 0 + stream_status = [status_code] + cleaned = False + + async def stream(): + try: + async for chunk in original.body_iterator: + yield chunk + except BaseException: + stream_status[0] = 500 + raise + finally: + try: + if hasattr(original.body_iterator, "aclose"): + await original.body_iterator.aclose() + finally: + await cleanup() + + async def cleanup(): + nonlocal cleaned + if cleaned: + return + cleaned = True + try: + if isinstance(original, OwnedStreamingResponse) and original.cleanup: + await original.cleanup() + finally: + await concurrency.release_memory(reserved) + if not _is_dashboard_path(request, path): + await _record_request_safely( + method, + path, + operation, + stream_status[0], + time.perf_counter() - start_time, + int(original.headers.get("content-length", "0")), + request.client.host if request.client else "", + ) + REQUESTS_IN_FLIGHT.labels(method=method).dec() + REQUEST_COUNT.labels( + method=method, operation=operation, status=stream_status[0] + ).inc() + REQUEST_DURATION.labels(method=method, operation=operation).observe( + time.perf_counter() - start_time + ) + + response = OwnedStreamingResponse( + stream(), + status_code=original.status_code, + headers=dict(original.headers), + background=original.background, + cleanup=cleanup, + on_error=lambda: stream_status.__setitem__(0, 500), + ) return response except HTTPException as e: status_code = e.status_code @@ -219,9 +285,10 @@ async def handle_proxy_request( clear_request() # Record metrics duration = time.perf_counter() - start_time - REQUESTS_IN_FLIGHT.labels(method=method).dec() - REQUEST_COUNT.labels(method=method, operation=operation, status=status_code).inc() - REQUEST_DURATION.labels(method=method, operation=operation).observe(duration) + if not isinstance(response, StreamingResponse): + REQUESTS_IN_FLIGHT.labels(method=method).dec() + REQUEST_COUNT.labels(method=method, operation=operation, status=status_code).inc() + REQUEST_DURATION.labels(method=method, operation=operation).observe(duration) try: if method == "GET" and response is not None: @@ -235,8 +302,10 @@ async def handle_proxy_request( # "/dashboard" (no trailing slash) doesn't match the mounted dashboard router and # falls through to this S3 catch-all, where it would otherwise be logged # as a phantom "dashboard" bucket. - if not _is_dashboard_path(request, path): - await record_request(method, path, operation, status_code, duration, size, client_ip) + if not isinstance(response, StreamingResponse) and not _is_dashboard_path(request, path): + await _record_request_safely( + method, path, operation, status_code, duration, size, client_ip + ) if reserved_memory > 0: await concurrency.release_memory(reserved_memory) @@ -260,14 +329,28 @@ async def _handle_proxy_request_impl( query = parse_qs(str(request.url.query), keep_blank_values=True) content_length = _parse_content_length(headers) - defer_sig = request.method in ("PUT", "POST") and _defer_signature_for_body( - headers, content_length, query + data_write = ( + request.method == "PUT" + and not headers.get("x-amz-copy-source") + and not any(k in query for k in ("tagging", "acl", "lifecycle", "policy", "cors")) + and "/" in request.url.path.strip("/") + ) + defer_sig = data_write and _defer_signature_for_body( + headers, + content_length if "content-length" in headers else crypto.MAX_BUFFER_SIZE + 1, + query, ) needs_body = request.method in ("PUT", "POST") and _needs_body_for_signature(headers, query) body = b"" if needs_body and not defer_sig: - body = await request.body() + chunks = bytearray() + async for chunk in request.stream(): + chunks.extend(chunk) + if len(chunks) > crypto.MAX_BUFFER_SIZE: + raise S3Error.invalid_request("Control request body exceeds 8 MiB") + body = bytes(chunks) + request._body = body if body: request.state.s3proxy_preloaded_body = body logger.debug( @@ -312,6 +395,17 @@ async def _handle_proxy_request_impl( raise S3Error.signature_does_not_match(error) raise S3Error.access_denied(error or "No credentials") + if request.method in ("PUT", "POST", "DELETE") and not data_write: + if headers.get("x-amz-content-sha256", "").startswith("STREAMING-"): + raise S3Error.invalid_request("Streaming encoding is only supported for data uploads") + chunks = bytearray() + async for chunk in request.stream(): + chunks.extend(chunk) + if len(chunks) > crypto.MAX_BUFFER_SIZE: + raise S3Error.invalid_request("Control request body exceeds 8 MiB") + request._body = bytes(chunks) + verify_payload_hash(request, hashlib.sha256(chunks).hexdigest()) + dispatcher = RequestDispatcher(handler) try: return await dispatcher.dispatch(request, verified_creds) diff --git a/s3proxy/signature.py b/s3proxy/signature.py index ca22d0f..0843b70 100644 --- a/s3proxy/signature.py +++ b/s3proxy/signature.py @@ -38,3 +38,20 @@ def verify_deferred_payload_hash( if error and "signature" in error.lower(): raise S3Error.signature_does_not_match(error) raise S3Error.access_denied(error or "Access Denied") + + +def verify_payload_hash(request: Request, payload_hash: str) -> None: + """Verify the body before publishing any new object or part state.""" + import hmac + + if deferred_signature_required(request): + verify_deferred_payload_hash(request, request.app.state.verifier, payload_hash) + return + expected = request.headers.get("x-amz-content-sha256", "") + if ( + expected + and expected != "UNSIGNED-PAYLOAD" + and not expected.startswith("STREAMING-") + and not hmac.compare_digest(expected, payload_hash) + ): + raise S3Error.signature_does_not_match("Payload SHA256 mismatch") diff --git a/s3proxy/state/attr_cache.py b/s3proxy/state/attr_cache.py index 99c1eb9..4dfdb8d 100644 --- a/s3proxy/state/attr_cache.py +++ b/s3proxy/state/attr_cache.py @@ -10,8 +10,10 @@ backend ETag, which invalidates the cached attributes without coordination. """ +import asyncio import hashlib from collections import OrderedDict +from contextlib import asynccontextmanager # ~100k entries of (bucket, key, etag) -> (size, etag) stays in the tens of # MB even with long backup keys — well inside the pod memory limit. @@ -30,8 +32,22 @@ def synthetic_multipart_etag(plaintext_size: int) -> str: class PlaintextAttrCache: def __init__(self, maxsize: int = _DEFAULT_MAXSIZE) -> None: self._maxsize = maxsize + self._locks = {} self._entries: OrderedDict[tuple[str, str, str], tuple[int, str]] = OrderedDict() + @asynccontextmanager + async def coalesce(self, bucket, key, backend_etag): + cache_key = (bucket, key, backend_etag) + entry = self._locks.setdefault(cache_key, [asyncio.Lock(), 0]) + entry[1] += 1 + try: + async with entry[0]: + yield + finally: + entry[1] -= 1 + if entry[1] == 0: + del self._locks[cache_key] + def get(self, bucket: str, key: str, backend_etag: str) -> tuple[int, str] | None: if not backend_etag: return None diff --git a/s3proxy/state/complete_lock.py b/s3proxy/state/complete_lock.py index dec0538..954b030 100644 --- a/s3proxy/state/complete_lock.py +++ b/s3proxy/state/complete_lock.py @@ -9,6 +9,7 @@ from __future__ import annotations import asyncio +import contextlib import os import uuid from collections.abc import AsyncIterator @@ -51,7 +52,7 @@ def __init__( self._ttl = ttl_seconds self._acquire_timeout = acquire_timeout_seconds self._poll_interval = poll_interval_seconds - self._memory_locks: dict[str, asyncio.Lock] = {} + self._memory_locks: dict[str, list] = {} self._memory_guard = asyncio.Lock() def _storage_key(self, bucket: str, key: str, upload_id: str) -> str: @@ -73,30 +74,16 @@ async def hold(self, bucket: str, key: str, upload_id: str) -> AsyncIterator[Non async def _memory_hold(self, bucket: str, key: str, upload_id: str) -> AsyncIterator[None]: lk = self._storage_key(bucket, key, upload_id) async with self._memory_guard: - lock = self._memory_locks.get(lk) - if lock is None: - lock = asyncio.Lock() - self._memory_locks[lk] = lock - - await lock.acquire() - logger.debug( - "COMPLETE_LOCK_ACQUIRED", - bucket=bucket, - key=key, - upload_id=upload_id[:20] + "..." if len(upload_id) > 20 else upload_id, - backend="memory", - ) + entry = self._memory_locks.setdefault(lk, [asyncio.Lock(), 0]) + entry[1] += 1 try: - yield + async with entry[0]: + yield finally: - lock.release() - logger.debug( - "COMPLETE_LOCK_RELEASED", - bucket=bucket, - key=key, - upload_id=upload_id[:20] + "..." if len(upload_id) > 20 else upload_id, - backend="memory", - ) + async with self._memory_guard: + entry[1] -= 1 + if entry[1] == 0: + del self._memory_locks[lk] @asynccontextmanager async def _redis_hold(self, bucket: str, key: str, upload_id: str) -> AsyncIterator[None]: @@ -130,17 +117,45 @@ async def _redis_hold(self, bucket: str, key: str, upload_id: str) -> AsyncItera await asyncio.sleep(self._poll_interval) + owner = asyncio.current_task() + lost = False + + async def renew(): + nonlocal lost + try: + while True: + await asyncio.sleep(max(0.1, self._ttl / 3)) + if not await self._renew_redis_lock(redis_key, token): + raise RuntimeError("Completion lease was lost") + except Exception: + lost = True + owner.cancel() + + renewal = asyncio.create_task(renew()) try: yield + except asyncio.CancelledError: + if lost: + raise S3Error.slow_down("Completion lease lost; retry the upload") from None + raise finally: + renewal.cancel() + with contextlib.suppress(asyncio.CancelledError): + await renewal await self._release_redis_lock(redis_key, token) - logger.debug( - "COMPLETE_LOCK_RELEASED", - bucket=bucket, - key=key, - upload_id=upload_id[:20] + "..." if len(upload_id) > 20 else upload_id, - backend="redis", - ) + + async def _renew_redis_lock(self, redis_key, token): + async with self._redis.pipeline(transaction=True) as pipe: + await pipe.watch(redis_key) + current = await pipe.get(redis_key) + current = current.decode() if isinstance(current, bytes) else current + if current != token: + await pipe.unwatch() + return False + pipe.multi() + pipe.expire(redis_key, self._ttl) + await pipe.execute() + return True async def _release_redis_lock(self, redis_key: str, token: str) -> None: import redis.asyncio as redis diff --git a/s3proxy/state/manager.py b/s3proxy/state/manager.py index 995ebf5..73fe95e 100644 --- a/s3proxy/state/manager.py +++ b/s3proxy/state/manager.py @@ -59,6 +59,9 @@ async def create_upload( upload_id: str, dek: bytes, kid: str = "", + *, + generation: str = "", + layout_version: int = 2, ) -> MultipartUploadState: """Create new upload state.""" state = MultipartUploadState( @@ -67,6 +70,8 @@ async def create_upload( key=key, upload_id=upload_id, kid=kid, + generation=generation, + layout_version=layout_version, ) sk = self._storage_key(bucket, key, upload_id) @@ -130,6 +135,31 @@ async def get_upload( ) return state + async def begin_write(self, bucket, key, upload_id, candidate_dek=None, candidate_kid=None): + """Freeze the upload DEK atomically before any concurrent attempt uses it. + + The first whole-object copy may select its source DEK. Once any writer + starts (even a failed one), no later request can change the key. + """ + + def updater(data): + state = deserialize_upload_state(data) + if state is None: + raise StateMissingError("Corrupt multipart state") + if not state.write_started: + if candidate_dek is not None and not state.parts: + state.dek = candidate_dek + state.kid = candidate_kid + state.write_started = True + return serialize_upload_state(state) + + data = await self._store.update( + self._storage_key(bucket, key, upload_id), updater, self._ttl + ) + if data is None: + raise StateMissingError("Upload state missing") + return deserialize_upload_state(data) + async def add_part( self, bucket: str, diff --git a/s3proxy/state/metadata.py b/s3proxy/state/metadata.py index d3452d3..1fde6b1 100644 --- a/s3proxy/state/metadata.py +++ b/s3proxy/state/metadata.py @@ -2,10 +2,13 @@ import base64 import gzip +import hashlib import structlog +from botocore.exceptions import ClientError from structlog.stdlib import BoundLogger +from ..errors import S3Error from .models import InternalPartMetadata, MultipartMetadata, PartMetadata from .serialization import json_dumps, json_loads @@ -35,6 +38,11 @@ def encode_multipart_metadata(meta: MultipartMetadata) -> str: """ data = { "v": meta.version, + "generation": meta.generation, + "upload_id": meta.upload_id, + "upload_bucket": meta.upload_bucket, + "upload_key": meta.upload_key, + "client_etag": meta.client_etag, "pc": meta.part_count, "ts": meta.total_plaintext_size, "dek": base64.b64encode(meta.wrapped_dek).decode(), @@ -63,6 +71,8 @@ def encode_multipart_metadata(meta: MultipartMetadata) -> str: } json_bytes = json_dumps(data) + if len(json_bytes) > MAX_METADATA_SIZE: + raise S3Error.invalid_request("Multipart metadata exceeds the supported size limit") compressed = gzip.compress(json_bytes) return base64.b64encode(compressed).decode() @@ -88,6 +98,11 @@ def decode_multipart_metadata(encoded: str) -> MultipartMetadata: return MultipartMetadata( version=data.get("v", 1), + generation=data.get("generation", ""), + upload_id=data.get("upload_id", ""), + upload_bucket=data.get("upload_bucket", ""), + upload_key=data.get("upload_key", ""), + client_etag=data.get("client_etag", ""), part_count=data.get("pc", 0), total_plaintext_size=data.get("ts", 0), wrapped_dek=base64.b64decode(data.get("dek", "")), @@ -121,10 +136,16 @@ async def persist_upload_state( upload_id: str, wrapped_dek: bytes, kid: str = "", + *, + layout_version: int = 2, ) -> None: """Persist DEK to S3 during upload (fallback for Redis failures).""" state_key = _internal_upload_key(key, upload_id) - data = {"dek": base64.b64encode(wrapped_dek).decode(), "kid": kid} + data = { + "dek": base64.b64encode(wrapped_dek).decode(), + "kid": kid, + "layout_version": layout_version, + } logger.info( "PERSIST_UPLOAD_STATE", @@ -174,6 +195,8 @@ async def load_upload_state( response = await s3_client.get_object(bucket, state_key) body = await response["Body"].read() data = json_loads(body) + if data.get("layout_version", 2) >= 3: + return None # Restart incomplete uploads; never guess accepted part state. wrapped_dek = base64.b64decode(data["dek"]) logger.info( @@ -230,7 +253,7 @@ async def save_multipart_metadata( meta: MultipartMetadata, ) -> None: """Save multipart metadata to S3.""" - meta_key = _internal_meta_key(key) + meta_key = generation_meta_key(meta.generation) if meta.generation else _internal_meta_key(key) encoded = encode_multipart_metadata(meta) logger.info( @@ -262,68 +285,93 @@ async def save_multipart_metadata( raise -async def load_multipart_metadata( - s3_client, - bucket: str, - key: str, -) -> MultipartMetadata | None: - """Load multipart metadata from S3. +FORMAT_KEY = "s3proxy-format" +GENERATION_KEY = "s3proxy-generation" - Checks the new internal prefix first, then falls back to legacy location. - """ - # Try new location first - meta_key = _internal_meta_key(key) - logger.debug("LOAD_METADATA", bucket=bucket, key=key, meta_key=meta_key) - try: - response = await s3_client.get_object(bucket, meta_key) - body = await response["Body"].read() - encoded = body.decode() - meta = decode_multipart_metadata(encoded) +def generation_for(wrapped_dek: bytes) -> str: + """The wrapped random per-upload DEK identifies an immutable generation.""" + return hashlib.sha256(wrapped_dek).hexdigest() - logger.info( - "METADATA_LOADED", - bucket=bucket, - key=key, - meta_key=meta_key, - part_count=meta.part_count, - total_size=meta.total_plaintext_size, - ) - return meta - except Exception as e: - logger.debug( - "METADATA_NOT_AT_NEW_LOCATION", - bucket=bucket, - key=key, - error=str(e), - ) +def generation_meta_key(generation: str) -> str: + if len(generation) != 64 or any(c not in "0123456789abcdef" for c in generation): + raise S3Error.internal_error("Invalid object generation") + return f"{INTERNAL_PREFIX}generations/{generation}.meta" - # Fall back to legacy location - legacy_key = f"{key}{META_SUFFIX_LEGACY}" - try: - response = await s3_client.get_object(bucket, legacy_key) - body = await response["Body"].read() - encoded = body.decode() - meta = decode_multipart_metadata(encoded) - logger.info( - "METADATA_LOADED_LEGACY", - bucket=bucket, - key=key, - legacy_key=legacy_key, - part_count=meta.part_count, - ) - return meta +def multipart_headers(wrapped_dek: bytes) -> dict[str, str]: + return {FORMAT_KEY: "multipart-v3", GENERATION_KEY: generation_for(wrapped_dek)} - except Exception as e: - logger.debug( - "NO_MULTIPART_METADATA", - bucket=bucket, - key=key, - error=str(e), - ) + +def multipart_etag(meta: MultipartMetadata) -> str: + # Legacy ETags were inconsistent across operations. Use the existing HEAD + # representation consistently for old objects; new manifests record identity. + return ( + meta.client_etag + or hashlib.md5(str(meta.total_plaintext_size).encode(), usedforsecurity=False).hexdigest() + ) + + +def is_not_found(error: ClientError) -> bool: + return str(error.response.get("Error", {}).get("Code")) in {"404", "NoSuchKey", "NotFound"} + + +async def load_multipart_metadata( + s3_client, bucket: str, key: str, head: dict | None = None +) -> MultipartMetadata | None: + """Resolve only the current generation; never turn backend failures into plaintext.""" + if head is None: + try: + head = await s3_client.head_object(bucket, key) + except ClientError as error: + if is_not_found(error): + return None + raise + metadata = head.get("Metadata", {}) + fmt = metadata.get(FORMAT_KEY) + if fmt in ("single-v3", "plain-v3") or (not fmt and "plaintext-size" in metadata): return None + if fmt and fmt != "multipart-v3": + raise S3Error.internal_error("Unsupported encrypted object format") + generation = metadata.get(GENERATION_KEY, "") if fmt else "" + keys = ( + [generation_meta_key(generation)] + if fmt + else [_internal_meta_key(key), f"{key}{META_SUFFIX_LEGACY}"] + ) + for meta_key in keys: + try: + response = await s3_client.get_object(bucket, meta_key) + stream = response["Body"] + body = bytearray() + async with stream: + while len(body) <= MAX_METADATA_SIZE: + chunk = await stream.read(min(65536, MAX_METADATA_SIZE + 1 - len(body))) + if not chunk: + break + body.extend(chunk) + if len(body) > MAX_METADATA_SIZE: + raise S3Error.internal_error("Metadata exceeds size limit") + meta = decode_multipart_metadata(body.decode()) + if generation and meta.generation != generation: + raise S3Error.internal_error("Metadata generation mismatch") + if generation: + sizes = [p.plaintext_size for p in meta.parts] + if ( + meta.part_count != len(meta.parts) + or sum(sizes) != meta.total_plaintext_size + or any(size < 0 for size in sizes) + or sum(p.ciphertext_size for p in meta.parts) != head.get("ContentLength") + ): + raise S3Error.internal_error("Inconsistent encryption metadata") + return meta + except ClientError as error: + if not is_not_found(error): + raise + if fmt: + raise S3Error.internal_error("Required encryption metadata is missing") + return None async def delete_multipart_metadata( diff --git a/s3proxy/state/models.py b/s3proxy/state/models.py index 99292f6..8b09082 100644 --- a/s3proxy/state/models.py +++ b/s3proxy/state/models.py @@ -33,6 +33,7 @@ class PartMetadata: md5: str = "" # Internal sub-parts for streaming uploads internal_parts: list[InternalPartMetadata] = field(default_factory=list) + staging_key: str = "" @dataclass(slots=True) @@ -52,6 +53,9 @@ class MultipartUploadState: total_plaintext_size: int = 0 next_internal_part_number: int = 1 # Next S3 part number to use kid: str = "" # Key id that wraps this upload's DEK ("" = default key) + generation: str = "" + layout_version: int = 2 + write_started: bool = False # True while every client part uses a single internal part (e.g. 5MB ClickHouse # shadow tars). Maps client part N → internal N so 600-part uploads stay under # S3's 10k part limit. Cleared on the first multi-internal client part (Scylla). @@ -76,6 +80,12 @@ class MultipartMetadata: wrapped_dek: bytes = b"" kid: str = "" # Key id that wrapped the DEK ("" = legacy/default key) + generation: str = "" + upload_id: str = "" + upload_bucket: str = "" + upload_key: str = "" + client_etag: str = "" + class StateMissingError(Exception): """Raised when upload state is missing from Redis during add_part.""" diff --git a/s3proxy/state/object.py b/s3proxy/state/object.py new file mode 100644 index 0000000..bf5e58d --- /dev/null +++ b/s3proxy/state/object.py @@ -0,0 +1,17 @@ +"""One interpretation of a backend object for reads, conditions, lists and copies.""" + +from dataclasses import dataclass + +from .models import MultipartMetadata + + +@dataclass(frozen=True, slots=True) +class ObjectDescriptor: + head: dict + multipart: MultipartMetadata | None + plaintext_size: int + etag: str + + @property + def generation(self) -> str: + return self.multipart.generation if self.multipart else "" diff --git a/s3proxy/state/serialization.py b/s3proxy/state/serialization.py index e383d40..445eb38 100644 --- a/s3proxy/state/serialization.py +++ b/s3proxy/state/serialization.py @@ -51,6 +51,9 @@ def serialize_upload_state(state: MultipartUploadState) -> bytes: "total_plaintext_size": state.total_plaintext_size, "next_internal_part_number": state.next_internal_part_number, "kid": state.kid, + "layout_version": state.layout_version, + "write_started": state.write_started, + "generation": state.generation, "deferred_copy_tail": base64.b64encode(state.deferred_copy_tail).decode() if state.deferred_copy_tail else "", @@ -62,6 +65,7 @@ def serialize_upload_state(state: MultipartUploadState) -> bytes: "ciphertext_size": p.ciphertext_size, "etag": p.etag, "md5": p.md5, + "staging_key": p.staging_key, "internal_parts": [ { "internal_part_number": ip.internal_part_number, @@ -129,6 +133,7 @@ def deserialize_upload_state(data: bytes) -> MultipartUploadState | None: ciphertext_size=p["ciphertext_size"], etag=p["etag"], md5=p.get("md5", ""), + staging_key=p.get("staging_key", ""), internal_parts=[ InternalPartMetadata( internal_part_number=ip["internal_part_number"], @@ -164,6 +169,9 @@ def deserialize_upload_state(data: bytes) -> MultipartUploadState | None: total_plaintext_size=obj.get("total_plaintext_size", 0), next_internal_part_number=obj.get("next_internal_part_number", 1), kid=obj.get("kid", ""), + layout_version=obj.get("layout_version", 2), + write_started=obj.get("write_started", bool(parts)), + generation=obj.get("generation", ""), deferred_copy_tail=base64.b64decode(obj["deferred_copy_tail"]) if obj.get("deferred_copy_tail") else b"", diff --git a/s3proxy/streaming/authenticated.py b/s3proxy/streaming/authenticated.py new file mode 100644 index 0000000..7fdbe65 --- /dev/null +++ b/s3proxy/streaming/authenticated.py @@ -0,0 +1,82 @@ +"""Bounded legacy GCM reads: authenticate fully before exposing plaintext.""" + +import asyncio +import os +import tempfile + +from cryptography.hazmat.primitives.ciphers import Cipher, algorithms, modes + +from .. import crypto +from ..errors import S3Error + + +class _SealedSpool: + """Spooled plaintext stays AES-CTR encrypted under a key held only in memory.""" + + def __init__(self): + self._file = tempfile.SpooledTemporaryFile(max_size=crypto.MAX_BUFFER_SIZE) # noqa: SIM115 -- response owns it + self._key = os.urandom(32) + self._iv = int.from_bytes(os.urandom(16), "big") + self._encryptor = self._cipher(0).encryptor() + self._decryptor = None + + def _cipher(self, block): + counter = (self._iv + block) % (1 << 128) + return Cipher(algorithms.AES(self._key), modes.CTR(counter.to_bytes(16, "big"))) + + def write(self, data): + self._file.write(self._encryptor.update(data)) + + def seek(self, offset): + self._file.seek(offset) + block, skip = divmod(offset, 16) + self._decryptor = self._cipher(block).decryptor() + self._decryptor.update(bytes(skip)) + + def read(self, size): + return self._decryptor.update(self._file.read(size)) + + def close(self): + self._file.close() + + +async def decrypt_to_file(body, dek): + spool = _SealedSpool() + # Roll large plaintext to disk without holding a second full plaintext copy. + pending = bytearray() + decryptor = None + length = 0 + try: + async with body: + while chunk := await body.read(1024 * 1024): + pending.extend(chunk) + if decryptor is None and len(pending) >= crypto.NONCE_SIZE: + nonce = bytes(pending[: crypto.NONCE_SIZE]) + del pending[: crypto.NONCE_SIZE] + decryptor = Cipher(algorithms.AES(dek), modes.GCM(nonce)).decryptor() + if decryptor and len(pending) > crypto.TAG_SIZE: + data = decryptor.update(bytes(pending[: -crypto.TAG_SIZE])) + del pending[: -crypto.TAG_SIZE] + length += len(data) + await asyncio.to_thread(spool.write, data) + if decryptor is None or len(pending) != crypto.TAG_SIZE: + raise S3Error.internal_error("Truncated encrypted object") + decryptor.finalize_with_tag(bytes(pending)) + return spool, length + except BaseException: + spool.close() + raise + + +async def file_range(spool, start, end): + try: + await asyncio.to_thread(spool.seek, start) + remaining = end - start + 1 + while remaining > 0: + chunk = await asyncio.to_thread(spool.read, min(1024 * 1024, remaining)) + if not chunk: + raise S3Error.internal_error("Truncated plaintext spool") + remaining -= len(chunk) + yield chunk + finally: + spool.close() diff --git a/s3proxy/streaming/chunked.py b/s3proxy/streaming/chunked.py index 5b82eca..1ddfaa0 100644 --- a/s3proxy/streaming/chunked.py +++ b/s3proxy/streaming/chunked.py @@ -1,132 +1,227 @@ -"""AWS chunked encoding utilities for streaming SigV4. - -This module handles the aws-chunked transfer encoding used by -AWS SDK v4 streaming uploads. - -Format: ;chunk-signature=\r\n\r\n...0;chunk-signature=\r\n -""" +"""Strict aws-chunked framing with incremental reads and SigV4 chunk validation.""" +import base64 +import hashlib +import hmac +import zlib from collections.abc import AsyncIterator, Iterator from fastapi import Request -# Streaming chunk size for reads/writes -STREAM_CHUNK_SIZE = 64 * 1024 # 64KB chunks for streaming +from ..errors import S3Error -# Safety limits for chunked decoding -_MAX_CHUNK_HEADER_SIZE = 4096 # Max header line (hex size + signature) -_MAX_CHUNK_SIZE = 64 * 1024 * 1024 # 64 MB max per chunk -_MAX_BUFFER_SIZE = 66 * 1024 * 1024 # Slightly above max chunk to hold chunk + framing +STREAM_CHUNK_SIZE = 64 * 1024 +_MAX_CHUNK_HEADER_SIZE = 4096 +_MAX_CHUNK_SIZE = 64 * 1024 * 1024 +SIGNED_PAYLOAD = "STREAMING-AWS4-HMAC-SHA256-PAYLOAD" +UNSIGNED_PAYLOAD_TRAILER = "STREAMING-UNSIGNED-PAYLOAD-TRAILER" -def _parse_chunk_size(header: bytes) -> int: - """Parse and validate chunk size from header bytes.""" - size_str = header.split(b";")[0].strip() - if not size_str: - raise ValueError("Empty chunk size") - chunk_size = int(size_str, 16) - if chunk_size < 0: - raise ValueError(f"Negative chunk size: {chunk_size}") - if chunk_size > _MAX_CHUNK_SIZE: - raise ValueError(f"Chunk size {chunk_size} exceeds maximum {_MAX_CHUNK_SIZE}") - return chunk_size +class _Crc32: + def __init__(self): + self.value = 0 + def update(self, data: bytes) -> None: + self.value = zlib.crc32(data, self.value) -def decode_aws_chunked(body: bytes) -> bytes: - """Decode aws-chunked transfer encoding from buffered body. - - Args: - body: Complete body with aws-chunked encoding - - Returns: - Decoded bytes without chunk headers - - Raises: - ValueError: If chunked encoding is malformed or truncated. - """ - result = bytearray() - pos = 0 - while pos < len(body): - header_end = body.find(b"\r\n", pos) - if header_end == -1: - raise ValueError("Truncated chunk: missing header terminator") - header = body[pos:header_end] - chunk_size = _parse_chunk_size(header) - if chunk_size == 0: - break - data_start = header_end + 2 - data_end = data_start + chunk_size - if data_end > len(body): - raise ValueError( - f"Truncated chunk: expected {chunk_size} bytes, " - f"only {len(body) - data_start} available" - ) - result.extend(body[data_start:data_end]) - pos = data_end + 2 - return bytes(result) - - -async def decode_aws_chunked_stream( - request: Request, -) -> AsyncIterator[bytes]: - """Decode aws-chunked encoding from streaming request. - - Yields decoded data chunks without buffering entire body. - Memory-efficient for large uploads. - - Args: - request: FastAPI request with aws-chunked body - - Yields: - Decoded data chunks - - Raises: - ValueError: If buffer exceeds safety limits or encoding is malformed. - """ - buffer = bytearray() - - async for raw_chunk in request.stream(): - buffer.extend(raw_chunk) - - if len(buffer) > _MAX_BUFFER_SIZE: - raise ValueError( - f"Chunked decode buffer ({len(buffer)} bytes) exceeds " - f"maximum ({_MAX_BUFFER_SIZE} bytes)" - ) + def digest(self) -> bytes: + return self.value.to_bytes(4, "big") - while True: - header_end = buffer.find(b"\r\n") - if header_end == -1: - if len(buffer) > _MAX_CHUNK_HEADER_SIZE: - raise ValueError(f"Chunk header exceeds {_MAX_CHUNK_HEADER_SIZE} bytes") - break - header = buffer[:header_end] - chunk_size = _parse_chunk_size(header) +_TRAILER_CHECKSUMS = { + "x-amz-checksum-crc32": _Crc32, + "x-amz-checksum-sha1": lambda: hashlib.sha1(usedforsecurity=False), + "x-amz-checksum-sha256": hashlib.sha256, +} - if chunk_size == 0: - return - data_start = header_end + 2 - data_end = data_start + chunk_size - trailing_end = data_end + 2 +class ChunkDecoder: + """A bounded framing state machine; EOF is valid only after the terminal chunk.""" - if len(buffer) < trailing_end: - break + def __init__(self, validate=None, trailer=False): + self.buffer = bytearray() + self.remaining = None + self.terminal = False + self.done = False + self.validate = validate + self.digest = None + self.signature = "" + self.trailers = {} if trailer else None + self.trailer_size = 0 - yield bytes(buffer[data_start:data_end]) - del buffer[:trailing_end] + def feed(self, data: bytes) -> Iterator[bytes]: + self.buffer.extend(data) + while True: + if self.done: + if self.buffer: + raise ValueError("Unexpected bytes after terminal chunk") + return + if self.remaining is None: + end = self.buffer.find(b"\r\n") + if end < 0: + if len(self.buffer) > _MAX_CHUNK_HEADER_SIZE: + raise ValueError("Chunk header too large") + return + if end > _MAX_CHUNK_HEADER_SIZE: + raise ValueError("Chunk header too large") + header = bytes(self.buffer[:end]) + del self.buffer[: end + 2] + size, *extensions = header.split(b";") + if not size or any(c not in b"0123456789abcdefABCDEF" for c in size): + raise ValueError("Invalid chunk size") + self.remaining = int(size, 16) + if self.remaining > _MAX_CHUNK_SIZE: + raise ValueError("Chunk size exceeds limit") + self.terminal = self.remaining == 0 + self.digest = hashlib.sha256() + self.signature = "" + for extension in extensions: + if extension.startswith(b"chunk-signature=") and not self.signature: + self.signature = extension.split(b"=", 1)[1].decode("ascii") + else: + raise ValueError("Unsupported chunk extension") + if self.remaining: + if not self.buffer: + return + count = min(self.remaining, len(self.buffer), STREAM_CHUNK_SIZE) + chunk = bytes(self.buffer[:count]) + del self.buffer[:count] + self.remaining -= count + self.digest.update(chunk) + yield chunk + continue + if self.terminal and self.trailers is not None: + end = self.buffer.find(b"\r\n") + if end < 0 or self.trailer_size + end > _MAX_CHUNK_HEADER_SIZE: + if self.trailer_size + len(self.buffer) > _MAX_CHUNK_HEADER_SIZE: + raise ValueError("Trailer too large") + return + line = bytes(self.buffer[:end]) + del self.buffer[: end + 2] + self.trailer_size += end + 2 + if line: + name, sep, value = line.partition(b":") + if not sep: + raise ValueError("Invalid trailer") + self.trailers[name.strip().lower().decode("ascii")] = value.strip().decode( + "ascii" + ) + else: + self.done = True + continue + if len(self.buffer) < 2: + return + if self.buffer[:2] != b"\r\n": + raise ValueError("Missing chunk data terminator") + del self.buffer[:2] + if self.validate: + self.validate(self.signature, self.digest.hexdigest()) + elif self.signature: + raise ValueError("Signed chunks require signature verification") + if self.terminal: + self.done = True + self.remaining = None + + def finish(self): + if not self.done or self.buffer: + raise ValueError("Truncated aws-chunked body") -def chunked(data: bytes, size: int) -> Iterator[tuple[int, bytes]]: - """Split data into numbered chunks for multipart upload. +def decode_aws_chunked(body: bytes) -> bytes: + decoder = ChunkDecoder() + result = b"".join(decoder.feed(body)) + decoder.finish() + return result + + +def _chunk_validator(request: Request): + mode = request.headers.get("x-amz-content-sha256", "") + if mode.startswith("STREAMING-") and mode not in (SIGNED_PAYLOAD, UNSIGNED_PAYLOAD_TRAILER): + raise S3Error.invalid_request("Unsupported streaming signature/trailer format") + if mode != SIGNED_PAYLOAD: + return None + from ..client import ParsedRequest + from ..client.verifier import _derive_signing_key + + verifier = request.app.state.verifier + parsed = ParsedRequest( + method=request.method, + bucket="", + key="", + query_params={}, + headers=dict(request.headers), + body=b"", + ) + auth = verifier._parse_header_auth(parsed, request.headers.get("authorization", "")) + if auth.error or auth.credentials is None: + raise S3Error.signature_does_not_match("Invalid streaming authorization") + signing_key = _derive_signing_key( + auth.credentials.secret_key, auth.date_stamp, auth.region, auth.service + ) + scope = f"{auth.date_stamp}/{auth.region}/{auth.service}/aws4_request" + previous = auth.signature + + def validate(signature, digest): + nonlocal previous + message = "\n".join( + [ + "AWS4-HMAC-SHA256-PAYLOAD", + auth.amz_date, + scope, + previous, + hashlib.sha256(b"").hexdigest(), + digest, + ] + ) + expected = hmac.new(signing_key, message.encode(), hashlib.sha256).hexdigest() + if not hmac.compare_digest(expected, signature): + raise S3Error.signature_does_not_match("Invalid streaming chunk signature") + previous = signature + + return validate + + +def _trailer_checksum(request: Request): + if request.headers.get("x-amz-content-sha256") != UNSIGNED_PAYLOAD_TRAILER: + return None + name = request.headers.get("x-amz-trailer", "").strip().lower() + factory = _TRAILER_CHECKSUMS.get(name) + if factory is None: + raise S3Error.invalid_request(f"Unsupported trailing checksum: {name or 'missing'}") + return name, factory() + + +async def decode_aws_chunked_stream(request: Request) -> AsyncIterator[bytes]: + trailer = _trailer_checksum(request) + decoder = ChunkDecoder(_chunk_validator(request), trailer=trailer is not None) + decoded = 0 + try: + async for raw in request.stream(): + # Bound parser buffering even if the ASGI server delivers a large block. + for offset in range(0, len(raw), STREAM_CHUNK_SIZE): + for chunk in decoder.feed(raw[offset : offset + STREAM_CHUNK_SIZE]): + decoded += len(chunk) + if trailer: + trailer[1].update(chunk) + yield chunk + decoder.finish() + expected = request.headers.get("x-amz-decoded-content-length") + if expected is not None and int(expected) != decoded: + raise ValueError("Decoded content length mismatch") + if trailer: + name, checksum = trailer + if set(decoder.trailers) != {name}: + raise ValueError(f"Expected exactly the {name} trailer") + try: + received = base64.b64decode(decoder.trailers[name], validate=True) + except ValueError as error: + raise S3Error.invalid_digest(name) from error + if not hmac.compare_digest(received, checksum.digest()): + raise S3Error.bad_digest(name) + except (ValueError, UnicodeError) as error: + raise S3Error.bad_request(str(error)) from error - Args: - data: Data to split - size: Chunk size in bytes - Yields: - (part_number, chunk) tuples starting from part 1 - """ +def chunked(data: bytes, size: int) -> Iterator[tuple[int, bytes]]: for i in range(0, len(data), size): yield i // size + 1, data[i : i + size] diff --git a/s3proxy/streaming/frames.py b/s3proxy/streaming/frames.py new file mode 100644 index 0000000..eb998d7 --- /dev/null +++ b/s3proxy/streaming/frames.py @@ -0,0 +1,127 @@ +"""Read contiguous ciphertext ranges while authenticating one frame at a time.""" + +import asyncio +import contextlib + +from .. import crypto +from ..errors import S3Error + + +async def read_frames(client, bucket, key, frames, *, if_match=None): + # Frames are (ciphertext_offset, ciphertext_size, plaintext_slice_start, end). + # The window limits retry scope, not buffering: only one frame is accumulated. + index = 0 + while index < len(frames): + end_index = index + 1 + end = frames[index][0] + frames[index][1] + while end_index < len(frames): + offset, size, _, _ = frames[end_index] + if offset != end or offset + size - frames[index][0] > 64 * 1024**2: + break + end += size + end_index += 1 + from ..handlers.base import ( + SOURCE_READ_ATTEMPTS, + SOURCE_READ_BACKOFF_SEC, + is_retryable_source_error, + ) + + attempt = 0 + while index < end_index: + response = None + try: + response = await client.get_object( + bucket, + key, + f"bytes={frames[index][0]}-{end - 1}", + **({"if_match": if_match} if if_match else {}), + ) + body = response["Body"] + async with body: + while index < end_index: + _, size, start, stop = frames[index] + ciphertext = bytearray(size) + received = 0 + while received < size: + chunk = await body.read(min(1024**2, size - received)) + if not chunk: + raise EOFError("Truncated ciphertext range") + ciphertext[received : received + len(chunk)] = chunk + received += len(chunk) + yield ciphertext, start, stop + index += 1 + attempt = 0 + except Exception as error: + attempt += 1 + if attempt >= SOURCE_READ_ATTEMPTS or not ( + isinstance(error, EOFError) or is_retryable_source_error(error) + ): + raise + await asyncio.sleep(SOURCE_READ_BACKOFF_SEC * 2 ** (attempt - 1)) + + +async def plaintext_frames( + client, bucket, key, meta, dek, start=None, end=None, *, if_match=None, ciphertext_size=None +): + frames = [] + pt_offset = 0 + ct_offset = 0 + for part in sorted(meta.parts, key=lambda p: p.part_number): + segments = part.internal_parts or [part] + for segment_number, segment in enumerate(segments, 1): + for size in crypto.ciphertext_frame_byte_sizes( + segment.plaintext_size, segment.ciphertext_size + ): + plaintext_size = size - crypto.ENCRYPTION_OVERHEAD + if start is None or (pt_offset + plaintext_size > start and pt_offset <= end): + if ciphertext_size is not None and ct_offset + size > ciphertext_size: + raise S3Error.invalid_range( + f"Metadata corruption: part {part.part_number}, internal part " + f"{segment_number} exceeds object size {ciphertext_size}" + ) + left = max(0, start - pt_offset) if start is not None else 0 + right = ( + min(plaintext_size, end - pt_offset + 1) + if end is not None + else plaintext_size + ) + frames.append((ct_offset, size, left, right)) + pt_offset += plaintext_size + ct_offset += size + # Old objects may contain a single GCM seal larger than the modern frame. + # Authenticate those seals to a bounded spool before releasing any plaintext. + index = 0 + while index < len(frames): + offset, size, left, right = frames[index] + if size > (crypto.FRAME_PLAINTEXT_SIZE + crypto.ENCRYPTION_OVERHEAD): + from .authenticated import decrypt_to_file, file_range + + response = await client.get_object( + bucket, + key, + f"bytes={offset}-{offset + size - 1}", + **({"if_match": if_match} if if_match else {}), + ) + spool, _ = await decrypt_to_file(response["Body"], dek) + try: + async with contextlib.aclosing(file_range(spool, left, right - 1)) as stream: + async for chunk in stream: + yield chunk + finally: + spool.close() + index += 1 + continue + stop = index + 1 + while stop < len(frames) and frames[stop][1] <= ( + crypto.FRAME_PLAINTEXT_SIZE + crypto.ENCRYPTION_OVERHEAD + ): + stop += 1 + async with contextlib.aclosing( + read_frames(client, bucket, key, frames[index:stop], if_match=if_match) + ) as reader: + async for ciphertext, left, right in reader: + plaintext = crypto.decrypt(ciphertext, dek) + for offset in range(left, right, 1024**2): + yield plaintext[offset : min(offset + 1024**2, right)] + del plaintext, ciphertext + index = stop diff --git a/s3proxy/streaming/response.py b/s3proxy/streaming/response.py new file mode 100644 index 0000000..46877a0 --- /dev/null +++ b/s3proxy/streaming/response.py @@ -0,0 +1,27 @@ +"""Streaming responses whose resources also close on ASGI send failures.""" + +import anyio +from fastapi.responses import StreamingResponse + + +class OwnedStreamingResponse(StreamingResponse): + def __init__(self, *args, cleanup=None, on_error=None, **kwargs): + super().__init__(*args, **kwargs) + self.cleanup = cleanup + self.on_error = on_error + + async def __call__(self, scope, receive, send): + try: + await super().__call__(scope, receive, send) + except BaseException: + if self.on_error is not None: + self.on_error() + raise + finally: + with anyio.CancelScope(shield=True): + try: + if hasattr(self.body_iterator, "aclose"): + await self.body_iterator.aclose() + finally: + if self.cleanup is not None: + await self.cleanup() diff --git a/tests/conftest.py b/tests/conftest.py index 62396d2..c757411 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -9,6 +9,7 @@ import fakeredis.aioredis import pytest +from botocore.exceptions import ClientError # Set environment variables before importing s3proxy modules os.environ.setdefault("S3PROXY_HOST", "http://localhost:9000") @@ -207,7 +208,9 @@ async def put_object( } return {"ETag": f'"{hashlib.md5(body).hexdigest()}"'} - async def get_object(self, bucket: str, key: str, range_header: str | None = None) -> dict: + async def get_object( + self, bucket: str, key: str, range_header: str | None = None, if_match: str | None = None + ) -> dict: """Retrieve an object.""" self.call_history.append( ("get_object", {"bucket": bucket, "key": key, "range": range_header}) @@ -341,6 +344,7 @@ async def copy_object( metadata: dict[str, str] | None = None, metadata_directive: str = "COPY", content_type: str | None = None, + **kwargs, ) -> dict: """Copy an object.""" self.call_history.append( @@ -413,6 +417,8 @@ async def create_multipart_upload(self, bucket: str, key: str, **kwargs) -> dict "Bucket": bucket, "Key": key, "Parts": {}, + "Metadata": kwargs.get("metadata", {}), + "ContentType": kwargs.get("content_type", "application/octet-stream"), "Initiated": datetime.now(UTC), } return {"UploadId": upload_id} @@ -466,8 +472,8 @@ async def complete_multipart_upload( etag = hashlib.md5(body).hexdigest() self.objects[self._key(bucket, key)] = { "Body": body, - "Metadata": {}, - "ContentType": "application/octet-stream", + "Metadata": upload.get("Metadata", {}), + "ContentType": upload.get("ContentType", "application/octet-stream"), "ContentLength": len(body), "ETag": etag, "LastModified": datetime.now(UTC), @@ -652,6 +658,7 @@ async def upload_part_copy( part_number: int, copy_source: str, copy_source_range: str | None = None, + copy_source_if_match: str | None = None, ) -> dict: """Copy a part from another object.""" self.call_history.append( @@ -704,9 +711,9 @@ async def upload_part_copy( def _not_found_error(self, key: str): """Create a NoSuchKey error.""" - error = Exception(f"NoSuchKey: {key}") - error.response = {"Error": {"Code": "NoSuchKey", "Message": f"Key not found: {key}"}} - return error + return ClientError( + {"Error": {"Code": "NoSuchKey", "Message": f"Key not found: {key}"}}, "GetObject" + ) def _bucket_not_found_error(self, bucket: str): """Create a NoSuchBucket error.""" diff --git a/tests/docker-compose.oom.yml b/tests/docker-compose.oom.yml index 6397005..bfd405a 100644 --- a/tests/docker-compose.oom.yml +++ b/tests/docker-compose.oom.yml @@ -1,6 +1,9 @@ # OOM tests upload 5GB+ into MinIO (e.g. 20 concurrent 256MB PUTs). # The default compose caps MinIO at 4g tmpfs for integration shards; drop # that cap here so MinIO uses disk-backed storage like before #117. +# The memory-usage shard also uses this override: immutable staging adds +# temporary copies to its multi-gigabyte workload. Proxy RSS limits and +# assertions stay unchanged; this removes only the backend storage cap. services: minio: tmpfs: !reset null diff --git a/tests/docker-compose.yml b/tests/docker-compose.yml index 7412f8c..84db799 100644 --- a/tests/docker-compose.yml +++ b/tests/docker-compose.yml @@ -6,7 +6,7 @@ services: - "6379:6379" minio: - image: minio/minio + image: pgsty/minio container_name: s3proxy-test-minio ports: - "9000:9000" diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index a51fadc..278e198 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -80,7 +80,7 @@ def minio_backend(*, isolated: bool = False) -> Generator[str]: "MINIO_ROOT_USER=minioadmin", "-e", "MINIO_ROOT_PASSWORD=minioadmin", - "minio/minio", + "pgsty/minio", "server", "/data", ], @@ -136,7 +136,7 @@ def run_s3proxy( '[{"access_key":"minioadmin","secret_key":"minioadmin",' '"kek":"test-encryption-key-32-bytes!!"}]' ), - "S3PROXY_HOST": "http://localhost:9000", + "S3PROXY_HOST": os.environ.get("S3PROXY_TEST_BACKEND", "http://localhost:9000"), "S3PROXY_REGION": "us-east-1", "S3PROXY_PORT": str(port), "S3PROXY_NO_TLS": "true", diff --git a/tests/integration/passthrough_verify.py b/tests/integration/passthrough_verify.py index 3bfbd61..e090586 100644 --- a/tests/integration/passthrough_verify.py +++ b/tests/integration/passthrough_verify.py @@ -137,7 +137,9 @@ def upload_multipart(ctx: RunContext, key: str, size: int) -> None: UploadId=upload_id, MultipartUpload={"Parts": parts}, ) - meta = f".s3proxy-internal/{key}.meta" + head = ctx.raw.head_object(Bucket=ctx.bucket, Key=key) + generation = head["Metadata"]["s3proxy-generation"] + meta = f".s3proxy-internal/generations/{generation}.meta" assert ctx.raw.head_object(Bucket=ctx.bucket, Key=meta)["ContentLength"] > 0 @@ -191,7 +193,13 @@ def _poll() -> None: def load_sidecar(ctx: RunContext, key: str): - meta_key = f".s3proxy-internal/{key}.meta" + head = ctx.raw.head_object(Bucket=ctx.bucket, Key=key) + generation = head.get("Metadata", {}).get("s3proxy-generation") + meta_key = ( + f".s3proxy-internal/generations/{generation}.meta" + if generation + else f".s3proxy-internal/{key}.meta" + ) raw = ctx.raw.get_object(Bucket=ctx.bucket, Key=meta_key)["Body"].read() return decode_multipart_metadata(raw.decode()) @@ -311,7 +319,7 @@ def check_reencrypt_control(ctx: RunContext, source: str, dest: str, size: int) lambda: upload_part_copy(ctx, dest, source, byte_range=f"bytes=0-{partial_end}"), ) ctx.ok("encrypts partial range", enc >= (partial_end + 1) * 0.5, f"{enc / MB:.0f}MB") - ctx.ok("high peak memory", peak >= CHUNK_PEAK * 0.5, f"{peak / MB:.2f}MB") + ctx.ok("bounded re-encryption memory", peak <= 32 * MB, f"{peak / MB:.2f}MB") def check_scylla_manifest_full_range_passthrough( diff --git a/tests/integration/test_copy_passthrough.py b/tests/integration/test_copy_passthrough.py index 04df925..b9188f4 100644 --- a/tests/integration/test_copy_passthrough.py +++ b/tests/integration/test_copy_passthrough.py @@ -17,7 +17,7 @@ from s3proxy.handlers import S3ProxyHandler from s3proxy.state import MultipartStateManager -from s3proxy.state.metadata import _internal_meta_key +from s3proxy.state.metadata import GENERATION_KEY, generation_meta_key BUCKET = "backups" @@ -109,7 +109,9 @@ async def test_multipart_object_copy_copies_sidecar_and_roundtrips(settings, moc body = b"m" * (256 * 1024) # streamed as multiple parts -> real sidecar await handler.handle_put_object(_stream_put_request(f"/{BUCKET}/sst/big.db", body), credentials) # sanity: a multipart sidecar exists for the source - assert mock_s3._key(BUCKET, _internal_meta_key("sst/big.db")) in mock_s3.objects + source_head = await mock_s3.head_object(BUCKET, "sst/big.db") + manifest = generation_meta_key(source_head["Metadata"][GENERATION_KEY]) + assert mock_s3._key(BUCKET, manifest) in mock_s3.objects mark = len(mock_s3.call_history) await handler.handle_copy_object( @@ -119,7 +121,8 @@ async def test_multipart_object_copy_copies_sidecar_and_roundtrips(settings, moc copied = _keys_touched(during, "copy_object") assert "sst/big.db.snap" in copied # assembled ciphertext, server-side - assert _internal_meta_key("sst/big.db.snap") in copied # sidecar, server-side + dest_head = await mock_s3.head_object(BUCKET, "sst/big.db.snap") + assert dest_head["Metadata"][GENERATION_KEY] == source_head["Metadata"][GENERATION_KEY] assert "sst/big.db" not in _keys_touched(during, "get_object") # no bulk download assert "sst/big.db.snap" not in _keys_touched(during, "put_object") # no re-upload diff --git a/tests/integration/test_elasticsearch_range_scenario.py b/tests/integration/test_elasticsearch_range_scenario.py index 720f91b..e941c8a 100644 --- a/tests/integration/test_elasticsearch_range_scenario.py +++ b/tests/integration/test_elasticsearch_range_scenario.py @@ -16,6 +16,7 @@ MultipartMetadata, PartMetadata, ) +from tests.conftest import MockS3Response @pytest.fixture @@ -105,7 +106,7 @@ async def test_elasticsearch_backup_range_error(self, handler, settings, kek): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() @@ -219,7 +220,7 @@ def get_object_side_effect(bucket, key, range_header=None): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() @@ -283,32 +284,12 @@ async def test_successful_3_part_fetch(self, handler, settings, kek): return_value={"ContentLength": total_ciphertext_size, "LastModified": None} ) - # Mock get_object to return the correct ciphertext for each range - def get_object_side_effect(bucket, key, range_header=None): - if range_header: - # Parse range to determine which part to return - range_str = range_header.replace("bytes=", "") - start, end = map(int, range_str.split("-")) + # A real body honors bounded reads and each GET's requested range. + ciphertext = b"".join(part["ciphertext"] for part in internal_parts_data) - # Find which internal part this range corresponds to - current_offset = 0 - for part_data in internal_parts_data: - part_size = part_data["meta"].ciphertext_size - if start >= current_offset and start < current_offset + part_size: - # This is the right part - mock_body = AsyncMock() - mock_body.read = AsyncMock(return_value=part_data["ciphertext"]) - mock_body.__aenter__ = AsyncMock(return_value=mock_body) - mock_body.__aexit__ = AsyncMock(return_value=None) - return {"Body": mock_body} - current_offset += part_size - - # Default mock - mock_body = AsyncMock() - mock_body.read = AsyncMock(return_value=b"") - mock_body.__aenter__ = AsyncMock(return_value=mock_body) - mock_body.__aexit__ = AsyncMock(return_value=None) - return {"Body": mock_body} + async def get_object_side_effect(bucket, key, range_header=None, **kwargs): + start, end = map(int, range_header.removeprefix("bytes=").split("-")) + return {"Body": MockS3Response(ciphertext[start : end + 1])} mock_client.get_object = AsyncMock(side_effect=get_object_side_effect) @@ -337,7 +318,7 @@ def get_object_side_effect(bucket, key, range_header=None): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() diff --git a/tests/integration/test_entity_too_small_errors.py b/tests/integration/test_entity_too_small_errors.py index 30a522f..d0240c6 100644 --- a/tests/integration/test_entity_too_small_errors.py +++ b/tests/integration/test_entity_too_small_errors.py @@ -17,6 +17,10 @@ class TestEntityTooSmallHandling: async def test_complete_with_missing_part_rejected(self, handler, settings): """Test that CompleteMultipartUpload fails when client requests non-existent parts.""" mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) @@ -82,6 +86,10 @@ async def test_complete_with_missing_part_rejected(self, handler, settings): async def test_entity_too_small_with_small_parts(self, handler, settings): """Test EntityTooSmall error when multiple parts are < 5MB.""" mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) diff --git a/tests/integration/test_generation_roundtrip.py b/tests/integration/test_generation_roundtrip.py new file mode 100644 index 0000000..3caa68a --- /dev/null +++ b/tests/integration/test_generation_roundtrip.py @@ -0,0 +1,156 @@ +"""Real S3/HTTP coverage for v3 generation publication and staged assembly.""" + +import os +import uuid + +import boto3 +import pytest +from botocore.config import Config +from botocore.exceptions import ClientError + +from .conftest import _find_free_port, minio_backend, run_s3proxy + +pytestmark = pytest.mark.e2e + + +@pytest.mark.parametrize( + "redis_url", + [""] + + ([os.environ["S3PROXY_TEST_REDIS_URL"]] if os.environ.get("S3PROXY_TEST_REDIS_URL") else []), +) +def test_generation_roundtrip(redis_url): + with ( + minio_backend() as backend, + run_s3proxy( + _find_free_port(), + S3PROXY_HOST=backend, + S3PROXY_MEMORY_LIMIT_MB="64", + S3PROXY_REDIS_URL=redis_url, + log_output=True, + ) as (endpoint, _), + ): + client = boto3.client( + "s3", + endpoint_url=endpoint, + aws_access_key_id="minioadmin", + aws_secret_access_key="minioadmin", + region_name="us-east-1", + config=Config( + s3={"addressing_style": "path"}, + request_checksum_calculation="when_required", + response_checksum_validation="when_required", + ), + ) + bucket = "generation-" + uuid.uuid4().hex[:16] + client.create_bucket(Bucket=bucket) + upload = client.create_multipart_upload( + Bucket=bucket, Key="object", Metadata={"owner": "test"} + )["UploadId"] + tail = client.upload_part( + Bucket=bucket, Key="object", UploadId=upload, PartNumber=2, Body=b"tail" + ) + data = b"A" * (9 * 1024**2) + first = client.upload_part( + Bucket=bucket, Key="object", UploadId=upload, PartNumber=1, Body=data + ) + pending = client.list_parts(Bucket=bucket, Key="object", UploadId=upload, MaxParts=1) + assert pending["IsTruncated"] and pending["Parts"][0]["Size"] == len(data) + assert pending["Parts"][0]["ETag"] == first["ETag"] + second_page = client.list_parts( + Bucket=bucket, Key="object", UploadId=upload, PartNumberMarker=1 + ) + assert second_page["Parts"][0]["Size"] == 4 + parts = [{"PartNumber": 1, "ETag": first["ETag"]}, {"PartNumber": 2, "ETag": tail["ETag"]}] + complete = client.complete_multipart_upload( + Bucket=bucket, Key="object", UploadId=upload, MultipartUpload={"Parts": parts} + ) + head = client.head_object(Bucket=bucket, Key="object") + get = client.get_object(Bucket=bucket, Key="object") + assert get["Body"].read() == data + b"tail" + assert head["Metadata"] == {"owner": "test"} + assert head["ETag"] == get["ETag"] == complete["ETag"] + listed = client.list_objects_v2(Bucket=bucket)["Contents"] + assert len(listed) == 1 and listed[0]["ETag"] == complete["ETag"] + ranged = client.get_object(Bucket=bucket, Key="object", Range=f"bytes={len(data) - 2}-") + assert ranged["Body"].read() == b"AAtail" + client.complete_multipart_upload( + Bucket=bucket, Key="object", UploadId=upload, MultipartUpload={"Parts": parts} + ) + with pytest.raises(ClientError): + client.complete_multipart_upload( + Bucket=bucket, + Key="object", + UploadId="wrong-upload", + MultipartUpload={"Parts": parts}, + ) + client.copy_object( + Bucket=bucket, Key="native", CopySource={"Bucket": bucket, "Key": "object"} + ) + assert client.get_object(Bucket=bucket, Key="native")["Body"].read() == data + b"tail" + assert client.head_object(Bucket=bucket, Key="native")["Metadata"] == {"owner": "test"} + client.put_object(Bucket=bucket, Key="object", Body=b"new", Metadata={"owner": "new"}) + assert client.get_object(Bucket=bucket, Key="object")["Body"].read() == b"new" + with pytest.raises(ClientError) as error: + client.put_object(Bucket=bucket, Key="object", Body=b"rejected", IfNoneMatch="*") + assert error.value.response["ResponseMetadata"]["HTTPStatusCode"] == 412 + assert client.get_object(Bucket=bucket, Key="object")["Body"].read() == b"new" + # Exercise the shared staging pipeline for server-side copy parts. + upload = client.create_multipart_upload(Bucket=bucket, Key="copy")["UploadId"] + part = client.upload_part_copy( + Bucket=bucket, + Key="copy", + UploadId=upload, + PartNumber=1, + CopySource={"Bucket": bucket, "Key": "object"}, + ) + client.complete_multipart_upload( + Bucket=bucket, + Key="copy", + UploadId=upload, + MultipartUpload={"Parts": [{"PartNumber": 1, "ETag": part["CopyPartResult"]["ETag"]}]}, + ) + assert client.get_object(Bucket=bucket, Key="copy")["Body"].read() == b"new" + + # A raw source copied over a previously encrypted destination must ignore old sidecars. + raw = boto3.client( + "s3", + endpoint_url=backend, + aws_access_key_id="minioadmin", + aws_secret_access_key="minioadmin", + region_name="us-east-1", + ) + raw.put_object(Bucket=bucket, Key="raw", Body=b"plain") + client.copy_object(Bucket=bucket, Key="native", CopySource={"Bucket": bucket, "Key": "raw"}) + assert client.get_object(Bucket=bucket, Key="native")["Body"].read() == b"plain" + # Fault injection against real HTTP at the buffering boundary. + import hashlib + + import requests + from botocore.auth import S3SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + + for length in (8 * 1024**2 - 1, 8 * 1024**2, 8 * 1024**2 + 1): + good = b"a" * length + signed = AWSRequest( + method="PUT", + url=f"{endpoint}/{bucket}/object", + data=good, + headers={"x-amz-content-sha256": hashlib.sha256(good).hexdigest()}, + ) + S3SigV4Auth(Credentials("minioadmin", "minioadmin"), "s3", "us-east-1").add_auth(signed) + result = requests.put( + signed.url, headers=dict(signed.headers), data=b"b" * length, timeout=15 + ) + assert result.status_code == 403 + assert client.get_object(Bucket=bucket, Key="object")["Body"].read() == b"new" + with pytest.raises(ClientError, match="BucketNotEmpty"): + client.delete_bucket(Bucket=bucket) + backend_keys = [o["Key"] for o in raw.list_objects_v2(Bucket=bucket)["Contents"]] + assert any(k.startswith(".s3proxy-internal/generations/") for k in backend_keys) + assert client.get_object(Bucket=bucket, Key="native")["Body"].read() == b"plain" + for obj in client.list_objects_v2(Bucket=bucket)["Contents"]: + client.delete_object(Bucket=bucket, Key=obj["Key"]) + client.delete_bucket(Bucket=bucket) + with pytest.raises(ClientError): + raw.head_bucket(Bucket=bucket) diff --git a/tests/integration/test_memory_usage.py b/tests/integration/test_memory_usage.py index d9e24fc..e154742 100644 --- a/tests/integration/test_memory_usage.py +++ b/tests/integration/test_memory_usage.py @@ -55,7 +55,7 @@ def s3proxy_with_memory_limit(self): '[{"access_key":"minioadmin","secret_key":"minioadmin",' '"kek":"test-encryption-key-32-bytes!!"}]' ), - "S3PROXY_HOST": "http://localhost:9000", + "S3PROXY_HOST": os.environ.get("S3PROXY_TEST_BACKEND", "http://localhost:9000"), "S3PROXY_REGION": "us-east-1", "S3PROXY_PORT": str(port), "S3PROXY_NO_TLS": "true", @@ -109,7 +109,7 @@ def s3proxy_with_short_backpressure(self): '[{"access_key":"minioadmin","secret_key":"minioadmin",' '"kek":"test-encryption-key-32-bytes!!"}]' ), - "S3PROXY_HOST": "http://localhost:9000", + "S3PROXY_HOST": os.environ.get("S3PROXY_TEST_BACKEND", "http://localhost:9000"), "S3PROXY_REGION": "us-east-1", "S3PROXY_PORT": str(port), "S3PROXY_NO_TLS": "true", @@ -170,14 +170,28 @@ def stress_bucket(self, stress_client): with contextlib.suppress(stress_client.exceptions.BucketAlreadyOwnedByYou): stress_client.create_bucket(Bucket=bucket) yield bucket - try: - response = stress_client.list_objects_v2(Bucket=bucket) - if "Contents" in response: - objects = [{"Key": obj["Key"]} for obj in response["Contents"]] - stress_client.delete_objects(Bucket=bucket, Delete={"Objects": objects}) - stress_client.delete_bucket(Bucket=bucket) - except Exception: - pass + # Delete only this test's unique bucket, including hidden manifests and + # unfinished staging uploads. Proxy LIST deliberately hides these keys. + with contextlib.closing( + boto3.client( + "s3", + endpoint_url=os.environ.get("S3PROXY_TEST_BACKEND", "http://localhost:9000"), + aws_access_key_id="minioadmin", + aws_secret_access_key="minioadmin", + region_name="us-east-1", + ) + ) as raw: + for page in raw.get_paginator("list_multipart_uploads").paginate(Bucket=bucket): + for upload in page.get("Uploads", []): + raw.abort_multipart_upload( + Bucket=bucket, Key=upload["Key"], UploadId=upload["UploadId"] + ) + for page in raw.get_paginator("list_objects_v2").paginate(Bucket=bucket): + objects = [{"Key": obj["Key"]} for obj in page.get("Contents", [])] + if objects: + result = raw.delete_objects(Bucket=bucket, Delete={"Objects": objects}) + assert not result.get("Errors"), result + raw.delete_bucket(Bucket=bucket) def test_backpressure_queues_concurrent_uploads(self, s3proxy_with_memory_limit, stress_bucket): """Verify backpressure queues excess requests instead of rejecting them. diff --git a/tests/integration/test_multipart_range_validation.py b/tests/integration/test_multipart_range_validation.py index c91e45b..a522386 100644 --- a/tests/integration/test_multipart_range_validation.py +++ b/tests/integration/test_multipart_range_validation.py @@ -9,6 +9,7 @@ from s3proxy.errors import S3Error from s3proxy.handlers.objects import ObjectHandlerMixin from s3proxy.state import InternalPartMetadata, MultipartMetadata, PartMetadata +from tests.conftest import MockS3Response @pytest.fixture @@ -97,7 +98,7 @@ async def test_invalid_range_detected_before_fetch( # Mock load_multipart_metadata to return our test metadata with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): # Mock credentials @@ -155,7 +156,7 @@ async def test_handles_s3_invalid_range_error(self, handler, settings, kek): invalid_range_error = ClientError(error_response, "GetObject") mock_client.head_object = AsyncMock( - return_value={"ContentLength": 100, "LastModified": None} + return_value={"ContentLength": 1028, "LastModified": None} ) mock_client.get_object = AsyncMock(side_effect=invalid_range_error) @@ -194,7 +195,7 @@ async def test_handles_s3_invalid_range_error(self, handler, settings, kek): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() @@ -209,10 +210,8 @@ async def test_handles_s3_invalid_range_error(self, handler, settings, kek): pass # Verify error message is helpful - assert ( - "metadata corruption" in str(exc_info.value).lower() - or "cannot read" in str(exc_info.value).lower() - ) + assert exc_info.value.code == "InvalidRange" + mock_client.get_object.assert_awaited_once() @pytest.mark.asyncio async def test_valid_range_succeeds(self, handler, settings, kek): @@ -234,10 +233,7 @@ async def test_valid_range_succeeds(self, handler, settings, kek): ) # Mock get_object to return ciphertext - mock_body = AsyncMock() - mock_body.read = AsyncMock(return_value=ciphertext) - mock_body.__aenter__ = AsyncMock(return_value=mock_body) - mock_body.__aexit__ = AsyncMock(return_value=None) + mock_body = MockS3Response(ciphertext) mock_client.get_object = AsyncMock( return_value={"Body": mock_body, "ContentType": "application/octet-stream"} ) @@ -277,7 +273,7 @@ async def test_valid_range_succeeds(self, handler, settings, kek): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() @@ -290,6 +286,7 @@ async def test_valid_range_succeeds(self, handler, settings, kek): # Verify response is valid assert response is not None assert response.status_code == 200 + assert b"".join([chunk async for chunk in response.body_iterator]) == plaintext @pytest.mark.asyncio async def test_multiple_internal_parts_validation(self, handler, settings, kek): @@ -349,7 +346,7 @@ async def test_multiple_internal_parts_validation(self, handler, settings, kek): mock_request.headers = {} with patch( - "s3proxy.handlers.objects.get.load_multipart_metadata", + "s3proxy.state.metadata.load_multipart_metadata", return_value=meta, ): creds = Mock() diff --git a/tests/integration/test_part_ordering.py b/tests/integration/test_part_ordering.py index 5081f88..eb739e8 100644 --- a/tests/integration/test_part_ordering.py +++ b/tests/integration/test_part_ordering.py @@ -3,6 +3,7 @@ from unittest.mock import AsyncMock, MagicMock, patch import pytest +from botocore.exceptions import ClientError from s3proxy.client import S3Credentials from s3proxy.handlers import S3ProxyHandler @@ -109,10 +110,14 @@ async def test_out_of_order_client_parts_sorted_internally(self, manager, settin # Mock S3 client mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) - mock_client.complete_multipart_upload = AsyncMock() + mock_client.complete_multipart_upload = AsyncMock(return_value={"ETag": "backend-etag"}) mock_client.head_object = AsyncMock(return_value={"ContentLength": 3156}) creds = S3Credentials( @@ -194,10 +199,14 @@ async def test_sequential_parts_remain_sorted(self, manager, settings): # Mock S3 client mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) - mock_client.complete_multipart_upload = AsyncMock() + mock_client.complete_multipart_upload = AsyncMock(return_value={"ETag": "backend-etag"}) mock_client.head_object = AsyncMock(return_value={"ContentLength": 3156}) creds = S3Credentials( diff --git a/tests/integration/test_partial_complete_fix.py b/tests/integration/test_partial_complete_fix.py index a67d206..08ff614 100644 --- a/tests/integration/test_partial_complete_fix.py +++ b/tests/integration/test_partial_complete_fix.py @@ -3,6 +3,7 @@ from unittest.mock import AsyncMock, Mock, patch import pytest +from botocore.exceptions import ClientError from s3proxy import crypto from s3proxy.state import InternalPartMetadata, PartMetadata @@ -19,6 +20,10 @@ async def test_complete_with_subset_of_parts(self, handler, settings): completes with 3 parts, the metadata should only reference the 3 completed parts. """ mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) @@ -50,7 +55,7 @@ async def test_complete_with_subset_of_parts(self, handler, settings): await handler.multipart_manager.add_part("bucket", "key", "upload-123", part) # Mock S3 client responses - mock_client.complete_multipart_upload = AsyncMock() + mock_client.complete_multipart_upload = AsyncMock(return_value={"ETag": "backend-etag"}) mock_client.head_object = AsyncMock( return_value={"ContentLength": 3 * 1028} # Only 3 parts completed ) @@ -126,6 +131,10 @@ async def capture_save(client, bucket, key, meta): async def test_complete_logs_size_mismatch(self, handler, settings): """Test that size mismatches are logged but don't fail the upload.""" mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) @@ -154,7 +163,7 @@ async def test_complete_logs_size_mismatch(self, handler, settings): ) await handler.multipart_manager.add_part("bucket", "key", "upload-123", part) - mock_client.complete_multipart_upload = AsyncMock() + mock_client.complete_multipart_upload = AsyncMock(return_value={"ETag": "backend-etag"}) # Return size that doesn't match our metadata (simulate S3 corruption or issue) mock_client.head_object = AsyncMock( return_value={"ContentLength": 9999} # Wrong size @@ -194,6 +203,10 @@ async def test_complete_with_no_parts_fails(self, handler, settings): from s3proxy.errors import S3Error mock_client = AsyncMock() + mock_client.head_object = AsyncMock(return_value={"ContentLength": 0}) + mock_client.get_object = AsyncMock( + side_effect=ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + ) # Make mock_client an async context manager mock_client.__aenter__ = AsyncMock(return_value=mock_client) mock_client.__aexit__ = AsyncMock(return_value=None) diff --git a/tests/integration/test_upload_part_copy_passthrough_e2e.py b/tests/integration/test_upload_part_copy_passthrough_e2e.py index e598579..36fd6bc 100644 --- a/tests/integration/test_upload_part_copy_passthrough_e2e.py +++ b/tests/integration/test_upload_part_copy_passthrough_e2e.py @@ -80,7 +80,7 @@ def test_range_copy_reencrypts_with_high_memory(self, passthrough_env): check_reencrypt_control(ctx, "sst/source-96mb.bin", "sst/dest-d.bin", QUICK_SIZE) assert proc.poll() is None _assert_check(ctx, "encrypts partial range") - _assert_check(ctx, "high peak memory") + _assert_check(ctx, "bounded re-encryption memory") def test_scylla_manifest_full_range_passthrough(self, passthrough_env): ctx, proc, _ = passthrough_env @@ -143,4 +143,4 @@ def test_passthrough_encrypt_delta_vs_reencrypt(self, passthrough_env): assert enc_pt <= 1 * MB assert enc_re >= (partial_end + 1) * 0.5 assert peak_pt <= CHUNK_PEAK * 0.5 - assert peak_re >= CHUNK_PEAK * 0.5 + assert peak_re <= 32 * MB diff --git a/tests/unit/test_complete_multipart_retry.py b/tests/unit/test_complete_multipart_retry.py index 2d30d88..ac2ec70 100644 --- a/tests/unit/test_complete_multipart_retry.py +++ b/tests/unit/test_complete_multipart_retry.py @@ -146,10 +146,19 @@ async def test_does_not_retry_non_retryable_error(handler): @pytest.mark.asyncio -async def test_recovers_via_head_object_when_prior_attempt_already_finished(handler): +async def test_recovers_via_head_object_when_prior_attempt_already_finished(handler, monkeypatch): """The exact prod failure: backend finished the assembly, the client only saw the error, and a naive retry would otherwise report a false failure.""" client = _FlakyCompleteClient(phantom_success_on_first=True) + from unittest.mock import AsyncMock + + from s3proxy.state import MultipartMetadata + + monkeypatch.setattr( + lifecycle, + "load_multipart_metadata", + AsyncMock(return_value=MultipartMetadata(upload_id="upload-1")), + ) resp = await handler._complete_multipart_upload_with_retry( client, "bucket", "key", "upload-1", S3_PARTS, COMPLETED_PARTS diff --git a/tests/unit/test_complete_upload_lock.py b/tests/unit/test_complete_upload_lock.py index 9e6c758..1efb353 100644 --- a/tests/unit/test_complete_upload_lock.py +++ b/tests/unit/test_complete_upload_lock.py @@ -223,6 +223,7 @@ async def test_complete_lock_idempotent_when_peer_already_finished( key, MultipartMetadata( version=2, + upload_id=upload_id, part_count=2, total_plaintext_size=len(chunk1) + len(chunk2), parts=parts, @@ -283,3 +284,20 @@ async def test_complete_lock_redis_acquire_timeout_raises(mock_redis): pass assert getattr(exc.value, "code", None) == "SlowDown" + + +@pytest.mark.asyncio +async def test_lease_is_renewed_and_registry_released(): + from fakeredis.aioredis import FakeRedis + + redis = FakeRedis() + lock = CompleteUploadLock(redis_client=redis, ttl_seconds=1) + async with lock.hold("bucket", "key", "upload"): + await asyncio.sleep(1.2) + assert await redis.exists(lock._redis_key("bucket", "key", "upload")) + assert not await redis.exists(lock._redis_key("bucket", "key", "upload")) + local = CompleteUploadLock() + async with local.hold("bucket", "key", "upload"): + assert len(local._memory_locks) == 1 + assert not local._memory_locks + await redis.aclose() diff --git a/tests/unit/test_concurrency_limit.py b/tests/unit/test_concurrency_limit.py index f7fa4ca..5c6dcb0 100644 --- a/tests/unit/test_concurrency_limit.py +++ b/tests/unit/test_concurrency_limit.py @@ -375,7 +375,7 @@ def test_estimate_memory_footprint_get(self): import s3proxy.concurrency as concurrency_module footprint = concurrency_module.estimate_memory_footprint("GET", 0) - assert footprint == concurrency_module.MAX_BUFFER_SIZE + assert footprint == 4 * concurrency_module.MAX_BUFFER_SIZE def test_estimate_memory_footprint_head(self): """HEAD should return 0 (bypass).""" diff --git a/tests/unit/test_dashboard_encryption.py b/tests/unit/test_dashboard_encryption.py index b256148..5fc2992 100644 --- a/tests/unit/test_dashboard_encryption.py +++ b/tests/unit/test_dashboard_encryption.py @@ -15,6 +15,7 @@ async def test_sidecar_present_means_encrypted(mock_s3) -> None: bucket, key = "scylla-backups", "backup/sst/me-big-Data.db" + await mock_s3.put_object(bucket, key, b"ciphertext") await save_multipart_metadata( mock_s3, bucket, diff --git a/tests/unit/test_disconnect_upload.py b/tests/unit/test_disconnect_upload.py index 538116a..569d3ee 100644 --- a/tests/unit/test_disconnect_upload.py +++ b/tests/unit/test_disconnect_upload.py @@ -1,5 +1,6 @@ """Client disconnect during streaming upload must abort and stop reading.""" +from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock import pytest @@ -42,6 +43,7 @@ async def stream(self): def _handler() -> PutObjectMixin: h = PutObjectMixin.__new__(PutObjectMixin) + h.settings = SimpleNamespace(dektag_name="isec", kidtag_name="isec-kid") h.keyring = MagicMock() h.keyring.key_for.return_value = ("kid1", b"0" * 32) return h diff --git a/tests/unit/test_generation_writes.py b/tests/unit/test_generation_writes.py new file mode 100644 index 0000000..954658f --- /dev/null +++ b/tests/unit/test_generation_writes.py @@ -0,0 +1,346 @@ +"""Regression coverage for generation publication and immutable part attempts.""" + +import hashlib +import xml.etree.ElementTree as ET +from unittest.mock import AsyncMock + +import pytest +from botocore.exceptions import ClientError +from fastapi import Request + +from s3proxy import crypto +from s3proxy.errors import S3Error +from s3proxy.handlers import S3ProxyHandler +from s3proxy.state import MultipartStateManager +from s3proxy.state.metadata import load_multipart_metadata +from s3proxy.streaming.chunked import decode_aws_chunked_stream + + +def request(method="PUT", body=b"", headers=None, query=""): + headers = dict(headers or {}) + headers.setdefault("content-length", str(len(body))) + + async def receive(): + return {"type": "http.request", "body": body, "more_body": False} + + return Request( + { + "type": "http", + "method": method, + "path": "/bucket/key", + "raw_path": b"/bucket/key", + "query_string": query.encode(), + "headers": [(k.encode(), v.encode()) for k, v in headers.items()], + }, + receive, + ) + + +@pytest.fixture +def proxy(settings, mock_s3): + handler = S3ProxyHandler(settings, {}, MultipartStateManager()) + handler._client = lambda _: mock_s3 + return handler + + +async def consume(response): + if hasattr(response, "body_iterator"): + try: + return b"".join([chunk async for chunk in response.body_iterator]) + finally: + if getattr(response, "cleanup", None): + await response.cleanup() + return response.body + + +async def create(proxy, credentials): + response = await proxy.handle_create_multipart_upload(request("POST"), credentials) + return next(e.text for e in ET.fromstring(response.body).iter() if e.tag.endswith("UploadId")) + + +def completion(upload, parts): + body = ( + "" + + "".join( + f"{n}{etag}" for n, etag in parts + ) + + "" + ) + return request("POST", body.encode(), query=f"uploadId={upload}") + + +async def put_part(proxy, credentials, upload, number, data, sha=None): + return await proxy.handle_upload_part( + request( + body=data, + headers={"x-amz-content-sha256": sha or "UNSIGNED-PAYLOAD"}, + query=f"uploadId={upload}&partNumber={number}", + ), + credentials, + ) + + +async def test_wrong_hash_cannot_replace_buffered_object(proxy, credentials, mock_s3): + await proxy.handle_put_object(request(body=b"before"), credentials) + before = mock_s3.objects["bucket/key"]["Body"] + with pytest.raises(S3Error): + await proxy.handle_put_object( + request( + body=b"after", + headers={"x-amz-content-sha256": hashlib.sha256(b"different").hexdigest()}, + ), + credentials, + ) + assert mock_s3.objects["bucket/key"]["Body"] == before + + +async def test_multipart_to_buffered_overwrite(proxy, credentials): + await proxy.handle_put_object( + request(body=b"old-content", headers={"x-amz-content-sha256": "UNSIGNED-PAYLOAD"}), + credentials, + ) + await proxy.handle_put_object(request(body=b"new"), credentials) + head = await proxy.handle_head_object(request("HEAD"), credentials) + assert head.headers["content-length"] == "3" + assert await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"new" + + +async def test_unknown_complete_does_not_succeed_for_existing_object(proxy, credentials): + await proxy.handle_put_object( + request(body=b"old", headers={"x-amz-content-sha256": "UNSIGNED-PAYLOAD"}), credentials + ) + with pytest.raises(S3Error): + await proxy.handle_complete_multipart_upload( + completion("never-created", [(1, "bad")]), credentials + ) + + +async def test_rejected_part_preserves_accepted_attempt(proxy, credentials): + upload = await create(proxy, credentials) + good = await put_part(proxy, credentials, upload, 1, b"before") + old = (await proxy.multipart_manager.get_upload("bucket", "key", upload)).parts[1] + with pytest.raises(S3Error): + await put_part( + proxy, credentials, upload, 1, b"after", hashlib.sha256(b"wrong").hexdigest() + ) + current = (await proxy.multipart_manager.get_upload("bucket", "key", upload)).parts[1] + assert current.staging_key == old.staging_key + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, good.headers["etag"])]), credentials + ) + assert await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"before" + + +async def test_out_of_order_parts_with_short_logical_tail(proxy, credentials): + upload = await create(proxy, credentials) + tail = await put_part(proxy, credentials, upload, 2, b"tail") + data = b"A" * (9 * 1024**2) + first = await put_part(proxy, credentials, upload, 1, data) + complete_request = completion(upload, [(1, first.headers["etag"]), (2, tail.headers["etag"])]) + response = await proxy.handle_complete_multipart_upload(complete_request, credentials) + assert response.status_code == 200 + get = await proxy.handle_get_object(request("GET"), credentials) + head = await proxy.handle_head_object(request("HEAD"), credentials) + assert get.headers["etag"] == head.headers["etag"] + assert await consume(get) == data + b"tail" + assert ( + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, first.headers["etag"]), (2, tail.headers["etag"])]), credentials + ) + ).status_code == 200 + + +async def test_metadata_failure_never_becomes_plaintext(proxy, credentials, mock_s3, monkeypatch): + await proxy.handle_put_object( + request(body=b"old", headers={"x-amz-content-sha256": "UNSIGNED-PAYLOAD"}), credentials + ) + monkeypatch.setattr( + mock_s3, + "get_object", + AsyncMock(side_effect=ClientError({"Error": {"Code": "ServiceUnavailable"}}, "GetObject")), + ) + with pytest.raises(ClientError): + await load_multipart_metadata(mock_s3, "bucket", "key") + + +async def test_sidecar_failure_does_not_publish_object(proxy, credentials, mock_s3, monkeypatch): + original = mock_s3.put_object + + async def fail_metadata(bucket, key, *args, **kwargs): + if "generations/" in key: + raise RuntimeError("injected metadata failure") + return await original(bucket, key, *args, **kwargs) + + monkeypatch.setattr(mock_s3, "put_object", fail_metadata) + with pytest.raises(S3Error): + await proxy.handle_put_object( + request(body=b"new", headers={"x-amz-content-sha256": "UNSIGNED-PAYLOAD"}), credentials + ) + assert "bucket/key" not in mock_s3.objects + + +@pytest.mark.parametrize("body", [b"3\r\nabc\r\n5\r\nxy", b"3\r\nabcXX0\r\n\r\n", b"0\r\n"]) +async def test_truncated_or_malformed_chunks_rejected(body): + with pytest.raises(S3Error): + _ = [chunk async for chunk in decode_aws_chunked_stream(request(body=body))] + + +def test_new_frame_attempts_do_not_reuse_nonce(): + dek = crypto.generate_dek() + first = crypto.encrypt_frame(b"AAAA", dek, "upload", 1, 0) + second = crypto.encrypt_frame(b"BBBB", dek, "upload", 1, 0) + assert first[:12] != second[:12] + assert crypto.decrypt(first, dek) == b"AAAA" + + +async def test_cancelled_publication_keeps_possibly_accepted_stage( + proxy, credentials, mock_s3, monkeypatch +): + import asyncio + + upload = await create(proxy, credentials) + original = proxy.multipart_manager.add_part + + async def committed_then_cancelled(*args): + await original(*args) + raise asyncio.CancelledError + + monkeypatch.setattr(proxy.multipart_manager, "add_part", committed_then_cancelled) + with pytest.raises(asyncio.CancelledError): + await put_part(proxy, credentials, upload, 1, b"accepted") + state = await proxy.multipart_manager.get_upload("bucket", "key", upload) + assert f"bucket/{state.parts[1].staging_key}" in mock_s3.objects + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, state.parts[1].md5)]), credentials + ) + assert await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"accepted" + + +async def test_completed_copy_does_not_prove_destination_upload_identity(proxy, credentials): + upload = await create(proxy, credentials) + part = await put_part(proxy, credentials, upload, 1, b"source") + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, part.headers["etag"])]), credentials + ) + copy = request(headers={"x-amz-copy-source": "/bucket/key"}) + copy.scope["path"] = "/bucket/destination" + await proxy.handle_copy_object(copy, credentials) + complete = completion(upload, [(1, part.headers["etag"])]) + complete.scope["path"] = "/bucket/destination" + with pytest.raises(S3Error): + await proxy.handle_complete_multipart_upload(complete, credentials) + + +async def test_abort_removes_accepted_and_replaced_attempts(proxy, credentials, mock_s3): + upload = await create(proxy, credentials) + await put_part(proxy, credentials, upload, 1, b"first") + await put_part(proxy, credentials, upload, 1, b"replacement") + assert sum("/attempts/" in k for k in mock_s3.objects) == 2 + await proxy.handle_abort_multipart_upload( + request("DELETE", query=f"uploadId={upload}"), credentials + ) + assert not any("/attempts/" in k for k in mock_s3.objects) + + +async def test_legacy_active_copy_is_rejected(proxy, credentials): + await proxy.multipart_manager.create_upload("bucket", "key", "legacy", crypto.generate_dek()) + with pytest.raises(S3Error, match="Legacy in-flight"): + await proxy.handle_upload_part_copy( + request(query="uploadId=legacy&partNumber=1"), credentials + ) + + +async def test_failed_complete_keeps_upload_retryable(proxy, credentials, monkeypatch): + upload = await create(proxy, credentials) + part = await put_part(proxy, credentials, upload, 1, b"retry-me") + import s3proxy.handlers.multipart.staged as staged + + save = staged.save_multipart_metadata + with monkeypatch.context() as patch: + patch.setattr( + staged, "save_multipart_metadata", AsyncMock(side_effect=RuntimeError("failed")) + ) + with pytest.raises(RuntimeError): + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, part.headers["etag"])]), credentials + ) + assert staged.save_multipart_metadata is save + assert await proxy.multipart_manager.get_upload("bucket", "key", upload) is not None + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, part.headers["etag"])]), credentials + ) + assert await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"retry-me" + + +@pytest.mark.parametrize("copy_first", [True, False]) +async def test_key_selection_cannot_change_after_first_writer( + proxy, credentials, mock_s3, copy_first +): + source = request(body=b"source") + source.scope["path"] = "/bucket/source" + await proxy.handle_put_object(source, credentials) + upload = await create(proxy, credentials) + state = await proxy.multipart_manager.get_upload("bucket", "key", upload) + original_dek = state.dek + if not copy_first: + await put_part(proxy, credentials, upload, 1, b"initial") + copy = request( + headers={"x-amz-copy-source": "/bucket/source"}, query=f"uploadId={upload}&partNumber=1" + ) + response = await proxy.handle_upload_part_copy(copy, credentials) + await consume(response) + state = await proxy.multipart_manager.get_upload("bucket", "key", upload) + chosen = state.dek + if not copy_first: + assert chosen == original_dek + else: + assert chosen != original_dek + assert ( + mock_s3.objects[f"bucket/{state.parts[1].staging_key}"]["Body"] + == mock_s3.objects["bucket/source"]["Body"] + ) + replacement = await put_part(proxy, credentials, upload, 1, b"replacement") + assert (await proxy.multipart_manager.get_upload("bucket", "key", upload)).dek == chosen + await proxy.handle_complete_multipart_upload( + completion(upload, [(1, replacement.headers["etag"])]), credentials + ) + assert ( + await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"replacement" + ) + + +async def test_empty_streaming_put_roundtrip(proxy, credentials): + await proxy.handle_put_object( + request(body=b"", headers={"x-amz-content-sha256": "UNSIGNED-PAYLOAD"}), credentials + ) + assert await consume(await proxy.handle_get_object(request("GET"), credentials)) == b"" + + +@pytest.mark.parametrize("blocker", [None, "object", "upload"]) +async def test_delete_bucket_purges_only_orphaned_proxy_metadata( + proxy, mock_s3, credentials, monkeypatch, blocker +): + await mock_s3.create_bucket("bucket") + await mock_s3.put_object("bucket", f".s3proxy-internal/generations/{'a' * 64}.meta", b"m") + if blocker == "object": + await mock_s3.put_object("bucket", "visible", b"x") + if blocker == "upload": + await mock_s3.create_multipart_upload("bucket", "pending") + delete_bucket = mock_s3.delete_bucket + + async def strict_delete_bucket(bucket): + if any(k.startswith(f"{bucket}/") for k in mock_s3.objects): + raise ClientError({"Error": {"Code": "BucketNotEmpty"}}, "DeleteBucket") + return await delete_bucket(bucket) + + monkeypatch.setattr(mock_s3, "delete_bucket", strict_delete_bucket) + req = request(method="DELETE") + req.scope["path"] = "/bucket" + if blocker: + with pytest.raises(S3Error) as error: + await proxy.handle_delete_bucket(req, credentials) + assert error.value.code == "BucketNotEmpty" + assert any("/generations/" in k for k in mock_s3.objects) + else: + assert (await proxy.handle_delete_bucket(req, credentials)).status_code == 204 + assert not mock_s3.objects and "bucket" not in mock_s3.buckets diff --git a/tests/unit/test_list_multipart_plaintext_size.py b/tests/unit/test_list_multipart_plaintext_size.py index 981fa58..c70e210 100644 --- a/tests/unit/test_list_multipart_plaintext_size.py +++ b/tests/unit/test_list_multipart_plaintext_size.py @@ -33,6 +33,7 @@ ) from s3proxy.state.attr_cache import PlaintextAttrCache from s3proxy.state.metadata import _internal_meta_key, persist_upload_state +from tests.conftest import MockS3Response as _Body INTERNAL_PREFIX = ".s3proxy-internal/" @@ -47,15 +48,8 @@ def fresh_cache(monkeypatch): plaintext_attr_cache.clear() -class _Body: - def __init__(self, data: bytes) -> None: - self._data = data - - async def read(self) -> bytes: - return self._data - - class FakeHandler: + _resolve_object = BucketHandlerMixin._resolve_object _process_list_objects = BucketHandlerMixin._process_list_objects _list_entry = staticmethod(BucketHandlerMixin._list_entry) @@ -90,7 +84,9 @@ async def get_object(self, bucket, key): if key in self.sidecars: encoded = encode_multipart_metadata(self.sidecars[key]) return {"Body": _Body(encoded.encode())} - raise KeyError(key) + from botocore.exceptions import ClientError + + raise ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") def _obj(key, size, etag="backend-etag"): @@ -177,17 +173,33 @@ def test_cache_scoped_to_backend_etag(): assert client.head_calls == head_calls + 1 -def test_failed_head_falls_back_and_does_not_cache(fresh_cache): +def test_failed_head_propagates_and_does_not_cache(fresh_cache): handler = FakeHandler() client = FakeClient(fail_head_key="broken.db") + with pytest.raises(RuntimeError, match="backend HEAD failed"): + asyncio.run(handler._process_list_objects(client, "bucket", [_obj("broken.db", size=555)])) + assert len(fresh_cache) == 0 - result = asyncio.run( - handler._process_list_objects(client, "bucket", [_obj("broken.db", size=555)]) - ) - assert result[0]["size"] == 555 - assert result[0]["etag"] == "backend-etag" - assert len(fresh_cache) == 0 +@pytest.mark.asyncio +async def test_concurrent_listing_coalesces_metadata_lookups(): + handler = FakeHandler() + client = FakeClient(metadata={"small.txt": {"plaintext-size": "42", "client-etag": "abc"}}) + original = client.head_object + + async def delayed(*args): + await asyncio.sleep(0.01) + return await original(*args) + + client.head_object = delayed + results = await asyncio.gather( + *[ + handler._process_list_objects(client, "bucket", [_obj("small.txt", size=100)]) + for _ in range(8) + ] + ) + assert all(r[0]["size"] == 42 for r in results) + assert client.head_calls == 1 def test_cache_evicts_least_recently_used(): diff --git a/tests/unit/test_list_objects_parallel.py b/tests/unit/test_list_objects_parallel.py index 2160633..185a1d3 100644 --- a/tests/unit/test_list_objects_parallel.py +++ b/tests/unit/test_list_objects_parallel.py @@ -8,12 +8,15 @@ import asyncio import datetime as dt +from botocore.exceptions import ClientError + from s3proxy.handlers.buckets import LIST_HEAD_CONCURRENCY, BucketHandlerMixin INTERNAL_PREFIX = "s3proxy-internal/" class FakeHandler: + _resolve_object = BucketHandlerMixin._resolve_object _process_list_objects = BucketHandlerMixin._process_list_objects _list_entry = staticmethod(BucketHandlerMixin._list_entry) @@ -42,7 +45,7 @@ async def head_object(self, bucket, key): try: await asyncio.sleep(0.02) # simulate backend round-trip if key == self.fail_key: - raise RuntimeError("backend HEAD failed") + raise ClientError({"Error": {"Code": "NoSuchKey"}}, "HeadObject") return {"Metadata": {"plaintext-size": "111", "client-etag": f"etag-{key}"}} finally: self.handler.inflight -= 1 diff --git a/tests/unit/test_list_objects_v1_via_v2.py b/tests/unit/test_list_objects_v1_via_v2.py index 778dec2..0ab1c2d 100644 --- a/tests/unit/test_list_objects_v1_via_v2.py +++ b/tests/unit/test_list_objects_v1_via_v2.py @@ -54,7 +54,7 @@ async def list_objects_v2( return self.resp async def head_object(self, bucket, key): - return {"Metadata": {}} + return {"Metadata": {"s3proxy-format": "plain-v3"}} class _Handler(BucketHandlerMixin): diff --git a/tests/unit/test_memory_concurrency.py b/tests/unit/test_memory_concurrency.py index 2997c72..b6c3bd5 100644 --- a/tests/unit/test_memory_concurrency.py +++ b/tests/unit/test_memory_concurrency.py @@ -77,7 +77,7 @@ def test_get_uses_fixed_buffer(self): import s3proxy.concurrency as concurrency_module footprint = concurrency_module.estimate_memory_footprint("GET", 0) - assert footprint == concurrency_module.MAX_BUFFER_SIZE + assert footprint == 4 * concurrency_module.MAX_BUFFER_SIZE def test_head_delete_bypass(self): """HEAD and DELETE reserve 0 (no buffering, bypass limit).""" diff --git a/tests/unit/test_request_handler_body.py b/tests/unit/test_request_handler_body.py index 0c045fe..22ffcf7 100644 --- a/tests/unit/test_request_handler_body.py +++ b/tests/unit/test_request_handler_body.py @@ -132,6 +132,11 @@ async def test_small_put_without_header_loads_body_once(): payload = b"x" * (4 * MB) request = _make_request(content_length=len(payload)) request.body = AsyncMock(return_value=payload) + + async def chunks(): + yield payload + + request.stream = MagicMock(side_effect=chunks) verifier = MagicMock() verifier.verify = MagicMock(return_value=(True, MagicMock(), "")) @@ -139,7 +144,8 @@ async def test_small_put_without_header_loads_body_once(): dispatcher_cls.return_value.dispatch = AsyncMock(return_value=None) await _handle_proxy_request_impl(request, MagicMock(), verifier) - request.body.assert_awaited_once() + request.stream.assert_called_once() + request.body.assert_not_awaited() assert request.state.s3proxy_preloaded_body == payload verifier.verify.assert_called_once() diff --git a/tests/unit/test_scylla_deferred_put_memory.py b/tests/unit/test_scylla_deferred_put_memory.py index 77841bd..b1794b9 100644 --- a/tests/unit/test_scylla_deferred_put_memory.py +++ b/tests/unit/test_scylla_deferred_put_memory.py @@ -10,6 +10,7 @@ import hashlib import tracemalloc +from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock import pytest @@ -61,6 +62,7 @@ async def stream(self): def _handler() -> PutObjectMixin: h = PutObjectMixin.__new__(PutObjectMixin) + h.settings = SimpleNamespace(dektag_name="isec", kidtag_name="isec-kid") h.keyring = MagicMock() h.keyring.key_for.return_value = ("kid1", b"0" * 32) return h diff --git a/tests/unit/test_source_read_retry.py b/tests/unit/test_source_read_retry.py index 341578b..bfd4049 100644 --- a/tests/unit/test_source_read_retry.py +++ b/tests/unit/test_source_read_retry.py @@ -322,7 +322,7 @@ async def test_passthrough_copy_survives_transient_backend_failures( upload_id = resp_create["UploadId"] await manager.create_upload(BUCKET, "sst/big.db.snap", upload_id, crypto.generate_dek(), kid) - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/big.db.snap", f"/{BUCKET}/sst/big.db", diff --git a/tests/unit/test_streaming_resources.py b/tests/unit/test_streaming_resources.py new file mode 100644 index 0000000..a354a14 --- /dev/null +++ b/tests/unit/test_streaming_resources.py @@ -0,0 +1,268 @@ +"""Failure-path and backend round-trip regressions for the shared services.""" + +import asyncio +import io +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import pytest +from botocore.httpchecksum import ( + AwsChunkedWrapper, + Crc32Checksum, + Sha1Checksum, + Sha256Checksum, +) +from cryptography.exceptions import InvalidTag +from starlette.requests import ClientDisconnect + +from s3proxy import crypto +from s3proxy.client import SigV4Verifier +from s3proxy.client.pool import S3ClientPool +from s3proxy.errors import S3Error +from s3proxy.state import MultipartMetadata, PartMetadata +from s3proxy.streaming.authenticated import decrypt_to_file, file_range +from s3proxy.streaming.chunked import decode_aws_chunked_stream +from s3proxy.streaming.frames import plaintext_frames +from s3proxy.streaming.response import OwnedStreamingResponse +from tests.conftest import MockS3Response +from tests.unit.test_generation_writes import request + + +async def test_pool_reuses_and_isolates_credentials(settings, credentials, monkeypatch): + instances = [] + + class Client: + def __init__(self, *args): + self.closed = False + instances.append(self) + + async def __aenter__(self): + return self + + async def __aexit__(self, *args): + self.closed = True + + monkeypatch.setattr("s3proxy.client.pool.S3Client", Client) + pool = S3ClientPool(settings, max_clients=1) + async with pool.acquire(credentials) as first: + async with pool.acquire(credentials) as second: + assert first is second + other = SimpleNamespace(access_key="other", secret_key="secret", region="us-east-1") + with pytest.raises(S3Error): + async with pool.acquire(other): + pass + closing = asyncio.create_task(pool.close()) + await asyncio.sleep(0) + assert not closing.done() + assert not first.closed + await asyncio.wait_for(closing, 1) + assert first.closed and len(instances) == 1 + + +async def test_response_closes_resources_when_headers_fail(): + cleanup = AsyncMock() + failed = [] + + async def body(): + pytest.fail("The body must not start when headers fail") + yield b"" + + async def send(message): + raise OSError("disconnected") + + response = OwnedStreamingResponse(body(), cleanup=cleanup, on_error=lambda: failed.append(True)) + with pytest.raises(ClientDisconnect): + await response({"type": "http", "asgi": {"spec_version": "2.4"}}, AsyncMock(), send) + cleanup.assert_awaited_once() + assert failed == [True] + + +async def test_contiguous_frames_use_one_backend_request(mock_s3): + dek = crypto.generate_dek() + plaintext = [bytes([n]) * 32768 for n in range(8)] + ciphertext = [crypto.encrypt(p, dek) for p in plaintext] + await mock_s3.put_object("bucket", "key", b"".join(ciphertext)) + meta = MultipartMetadata( + parts=[ + PartMetadata(i, len(p), len(c), "") + for i, (p, c) in enumerate(zip(plaintext, ciphertext, strict=True), 1) + ] + ) + result = b"".join([p async for p in plaintext_frames(mock_s3, "bucket", "key", meta, dek)]) + assert result == b"".join(plaintext) + assert len([c for c in mock_s3.call_history if c[0] == "get_object"]) == 1 + + +async def test_truncated_range_resumes_at_unpublished_frame(mock_s3, monkeypatch): + dek = crypto.generate_dek() + data = [b"a" * 1024, b"b" * 1024] + seals = [crypto.encrypt(p, dek) for p in data] + await mock_s3.put_object("bucket", "key", b"".join(seals)) + original = mock_s3.get_object + calls = [] + + async def truncated(bucket, key, byte_range, **kwargs): + calls.append(byte_range) + response = await original(bucket, key, byte_range, **kwargs) + if len(calls) == 1: + response["Body"] = MockS3Response(seals[0] + seals[1][:5]) + return response + + monkeypatch.setattr(mock_s3, "get_object", truncated) + monkeypatch.setattr("s3proxy.handlers.base.SOURCE_READ_BACKOFF_SEC", 0) + meta = MultipartMetadata( + parts=[ + PartMetadata(i, len(p), len(c), "") + for i, (p, c) in enumerate(zip(data, seals, strict=True), 1) + ] + ) + assert b"".join( + [p async for p in plaintext_frames(mock_s3, "bucket", "key", meta, dek)] + ) == b"".join(data) + assert calls[1].startswith(f"bytes={len(seals[0])}-") + + +async def test_large_legacy_seal_is_authenticated_before_plaintext(mock_s3): + dek = crypto.generate_dek() + data = b"legacy" * (2 * 1024**2) + ciphertext = crypto.encrypt(data, dek) + await mock_s3.put_object("bucket", "key", ciphertext) + meta = MultipartMetadata(parts=[PartMetadata(1, len(data), len(ciphertext), "")]) + assert ( + b"".join([p async for p in plaintext_frames(mock_s3, "bucket", "key", meta, dek)]) == data + ) + with pytest.raises(InvalidTag): + await decrypt_to_file(MockS3Response(ciphertext[:-1] + bytes([ciphertext[-1] ^ 1])), dek) + + +# Public AWS SigV4 test vectors, including the signature on the terminal chunk: +# https://docs.aws.amazon.com/AmazonS3/latest/developerguide/sigv4-streaming.html +@pytest.mark.parametrize("tamper", [False, True]) +async def test_aws_published_chunk_signature_vector(credentials, tamper): + signatures = [ + "ad80c730a21e5b8d04586a2213dd63b9a0e99e0e2307b0ade35a65485a288648", + "0055627c9e194cb4542bae2aa5492e3c1575bbb81b612b7d234b86a503ef5497", + "b6c6ea8a5354eaf15b3cb7646744f4275b71ea724fed81ceb9323e279d449df9", + ] + body = b"".join( + f"{n:x};chunk-signature={sig}\r\n".encode() + b"a" * n + b"\r\n" + for n, sig in zip([65536, 1024, 0], signatures, strict=True) + ) + if tamper: + body = body.replace(b"aaaa", b"baaa", 1) + req = request( + body=body, + headers={ + "x-amz-content-sha256": "STREAMING-AWS4-HMAC-SHA256-PAYLOAD", + "x-amz-date": "20130524T000000Z", + "x-amz-decoded-content-length": "66560", + "authorization": "AWS4-HMAC-SHA256 " + "Credential=AKIAIOSFODNN7EXAMPLE/20130524/us-east-1/s3/aws4_request," + "SignedHeaders=host;x-amz-date," + "Signature=4f232c4386841ef735655705268965c44a0e4690baa4adea153f7db9fa80a0a9", + }, + ) + req.scope["app"] = SimpleNamespace( + state=SimpleNamespace( + verifier=SigV4Verifier({credentials.access_key: credentials.secret_key}) + ) + ) + if tamper: + with pytest.raises(S3Error): + _ = [part async for part in decode_aws_chunked_stream(req)] + else: + assert b"".join([part async for part in decode_aws_chunked_stream(req)]) == b"a" * 66560 + + +async def test_spilled_plaintext_is_encrypted_at_rest(): + dek = crypto.generate_dek() + data = b"secret-plaintext" * (crypto.MAX_BUFFER_SIZE // 16 + 4096) + spool, length = await decrypt_to_file(MockS3Response(crypto.encrypt(data, dek)), dek) + assert length == len(data) + assert spool._file._rolled + spool._file.seek(0) + assert b"secret-plaintext" not in spool._file.read() + start, end = 12345, 12345 + 3 * 1024 * 1024 + assert b"".join([c async for c in file_range(spool, start, end)]) == data[start : end + 1] + + +_CHECKSUMS = { + "x-amz-checksum-crc32": Crc32Checksum, + "x-amz-checksum-sha1": Sha1Checksum, + "x-amz-checksum-sha256": Sha256Checksum, +} + + +def _trailer_request(data, name="x-amz-checksum-crc32", body=None, mode=None): + if body is None: + body = b"".join( + AwsChunkedWrapper( + io.BytesIO(data), + checksum_cls=_CHECKSUMS[name], + checksum_name=name, + chunk_size=64 * 1024, + ) + ) + return request( + body=body, + headers={ + "x-amz-content-sha256": mode or "STREAMING-UNSIGNED-PAYLOAD-TRAILER", + "content-encoding": "aws-chunked", + "x-amz-trailer": name, + "x-amz-decoded-content-length": str(len(data)), + }, + ) + + +async def _decode(req): + return b"".join([part async for part in decode_aws_chunked_stream(req)]) + + +@pytest.mark.parametrize("name", list(_CHECKSUMS)) +async def test_unsigned_trailer_checksum_is_verified(name): + data = bytes(range(256)) * 1000 + assert await _decode(_trailer_request(data, name)) == data + + +async def test_unsigned_trailer_checksum_mismatch_is_rejected(): + data = b"a" * 100_000 + body = b"".join( + AwsChunkedWrapper( + io.BytesIO(data), checksum_cls=Crc32Checksum, checksum_name="x-amz-checksum-crc32" + ) + ) + with pytest.raises(S3Error) as error: + await _decode(_trailer_request(data, body=body.replace(b"aaaa", b"baaa", 1))) + assert error.value.code == "BadDigest" + + +@pytest.mark.parametrize( + "body", + [b"5\r\nhello\r\n0\r\n\r\n", b"5\r\nhello\r\n0\r\nx-amz-checksum-crc32:NhCmhg=="], +) +async def test_unsigned_trailer_must_be_complete(body): + with pytest.raises(S3Error): + await _decode(_trailer_request(b"hello", body=body)) + + +async def test_unsupported_trailer_algorithm_is_rejected_before_reading(): + body = b"".join( + AwsChunkedWrapper( + io.BytesIO(b"x"), checksum_cls=Crc32Checksum, checksum_name="x-amz-checksum-crc32" + ) + ) + req = _trailer_request(b"x", body=body) + req.scope["headers"] = [ + (k, b"x-amz-checksum-crc32c" if k == b"x-amz-trailer" else v) + for k, v in req.scope["headers"] + ] + with pytest.raises(S3Error) as error: + await _decode(req) + assert error.value.code == "InvalidRequest" + + +async def test_signed_trailer_mode_is_still_rejected(): + req = _trailer_request(b"x", mode="STREAMING-AWS4-HMAC-SHA256-PAYLOAD-TRAILER") + with pytest.raises(S3Error) as error: + await _decode(req) + assert error.value.code == "InvalidRequest" diff --git a/tests/unit/test_upload_part_copy_keepalive.py b/tests/unit/test_upload_part_copy_keepalive.py index 509ea07..948360f 100644 --- a/tests/unit/test_upload_part_copy_keepalive.py +++ b/tests/unit/test_upload_part_copy_keepalive.py @@ -134,7 +134,7 @@ async def slow_copy(*args, **kwargs): mock_s3.upload_part_copy = slow_copy - resp = await handler.handle_upload_part_copy(_copy_part_request(upload_id), credentials) + resp = await handler._copy_part_impl(_copy_part_request(upload_id), credentials) assert resp.status_code == 200 stream = resp.body_iterator @@ -173,7 +173,7 @@ async def failing_copy(*args, **kwargs): mock_s3.upload_part_copy = failing_copy - resp = await handler.handle_upload_part_copy(_copy_part_request(upload_id), credentials) + resp = await handler._copy_part_impl(_copy_part_request(upload_id), credentials) body = b"".join([c async for c in resp.body_iterator]) assert resp.status_code == 200 @@ -210,7 +210,7 @@ async def hanging_copy(*args, **kwargs): mock_s3.upload_part_copy = hanging_copy - resp = await handler.handle_upload_part_copy(_copy_part_request(upload_id), credentials) + resp = await handler._copy_part_impl(_copy_part_request(upload_id), credentials) stream = resp.body_iterator assert await anext(stream) == b" " await asyncio.wait_for(started.wait(), 1) @@ -249,7 +249,7 @@ async def tracking_copy(*args, **kwargs): mock_s3.upload_part_copy = tracking_copy - resp = await handler.handle_upload_part_copy(_copy_part_request(upload_id), credentials) + resp = await handler._copy_part_impl(_copy_part_request(upload_id), credentials) body = b"".join([c async for c in resp.body_iterator]) assert ET.fromstring(body).tag.endswith("CopyPartResult") @@ -291,7 +291,7 @@ async def tracked_get(*args, **kwargs): mock_s3.upload_part_copy = slow_copy mock_s3.get_object = tracked_get - resp = await handler.handle_upload_part_copy(_copy_part_request(upload_id), credentials) + resp = await handler._copy_part_impl(_copy_part_request(upload_id), credentials) b"".join([c async for c in resp.body_iterator]) assert first_md5_read is not None and last_copy_done is not None @@ -315,7 +315,7 @@ async def test_hybrid_tail_roundtrip_with_parallel_segments( range_end = 4 * frame_size + frame_size // 2 - 1 # ends mid 5th frame assert range_end + 1 > crypto.STREAMING_THRESHOLD - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request(upload_id, copy_source_range=f"bytes=0-{range_end}"), credentials, ) diff --git a/tests/unit/test_upload_part_copy_passthrough.py b/tests/unit/test_upload_part_copy_passthrough.py index 7253b75..d67268c 100644 --- a/tests/unit/test_upload_part_copy_passthrough.py +++ b/tests/unit/test_upload_part_copy_passthrough.py @@ -123,7 +123,7 @@ async def test_multipart_encrypted_upload_part_copy_is_server_side_passthrough( await manager.create_upload(BUCKET, "sst/big.db.snap", upload_id, dst_dek, kid) mark = len(mock_s3.call_history) - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/big.db.snap", f"/{BUCKET}/sst/big.db", @@ -205,7 +205,7 @@ async def test_upload_part_copy_passthrough_roundtrips_via_get( ) upload_id = _extract_upload_id(create_resp.body) - copy_resp = await handler.handle_upload_part_copy( + copy_resp = await handler._copy_part_impl( _copy_part_request(f"/{BUCKET}/sst/dest.db", f"/{BUCKET}/sst/source.db", upload_id), credentials, ) @@ -271,7 +271,7 @@ async def test_range_copy_still_reencrypts(mock_s3, settings, manager, credentia req.headers["x-amz-copy-source-range"] = f"bytes=0-{crypto.MAX_BUFFER_SIZE - 1}" mark = len(mock_s3.call_history) - await _read(await handler.handle_upload_part_copy(req, credentials)) + await _read(await handler._copy_part_impl(req, credentials)) during = mock_s3.call_history[mark:] assert not any(c[0] == "upload_part_copy" for c in during) @@ -527,7 +527,7 @@ async def test_scylla_prod_shape_range_smaller_than_metadata_uses_passthrough( scylla_range = f"bytes=0-{range_end}" mark = len(mock_s3.call_history) - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/big-Data.db.sm_manifest", f"/{BUCKET}/sst/big-Data.db", @@ -667,7 +667,7 @@ async def test_two_part_hybrid_defer_tail_completes_fast( ) mark = len(mock_s3.call_history) - resp1 = await handler.handle_upload_part_copy( + resp1 = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/small-Data.db.sm_manifest", f"/{BUCKET}/sst/small-Data.db", @@ -692,7 +692,7 @@ async def test_two_part_hybrid_defer_tail_completes_fast( part2_start = part1_range_end + 1 mark2 = len(mock_s3.call_history) - resp2 = await handler.handle_upload_part_copy( + resp2 = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/small-Data.db.sm_manifest", f"/{BUCKET}/sst/small-Data.db", @@ -794,7 +794,7 @@ async def test_two_part_defer_tail_total_plaintext_not_double_counted( (1, f"bytes=0-{part1_range_end}"), (2, f"bytes={part1_range_end + 1}-{total - 1}"), ): - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/{dest_key}", f"/{BUCKET}/sst/big-Data.db", @@ -852,7 +852,7 @@ async def test_defer_tail_flushed_on_complete_keeps_part_accounting( upload_id = resp_create["UploadId"] await handler.multipart_manager.create_upload(BUCKET, dest_key, upload_id, src_dek, kid) - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/{dest_key}", f"/{BUCKET}/sst/solo-Data.db", @@ -1029,7 +1029,7 @@ async def test_scylla_manifest_full_range_uses_passthrough_not_streaming( full_range = f"bytes=0-{len(src_plaintext) - 1}" mark = len(mock_s3.call_history) - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/big-Data.db.sm_manifest", f"/{BUCKET}/sst/big-Data.db", @@ -1089,7 +1089,7 @@ async def blocked_streaming(*args, **kwargs): handler._streaming_copy_part = blocked_streaming # type: ignore[method-assign] async def run_passthrough(): - resp = await handler.handle_upload_part_copy( + resp = await handler._copy_part_impl( _copy_part_request( f"/{BUCKET}/sst/big-Data.db.sm_manifest", f"/{BUCKET}/sst/big-Data.db", @@ -1155,7 +1155,9 @@ async def test_single_segment_passthrough_complete_presents_backend_etag( import xml.etree.ElementTree as ET client_etag = ET.fromstring(await _read(copy_resp)).find("{*}ETag").text.strip('"') - backend_etag = mock_s3.multipart_uploads[upload_id]["Parts"][1]["ETag"] + state = await handler.multipart_manager.get_upload(BUCKET, "single/dst.bin", upload_id) + backend_etag = mock_s3.objects[f"{BUCKET}/{state.parts[1].staging_key}"]["ETag"].strip('"') + assert not mock_s3.multipart_uploads[upload_id]["Parts"] assert client_etag == hashlib.md5(plaintext, usedforsecurity=False).hexdigest() assert client_etag != backend_etag diff --git a/uv.lock b/uv.lock index 01ddac9..9a8986d 100644 --- a/uv.lock +++ b/uv.lock @@ -235,15 +235,15 @@ wheels = [ [[package]] name = "boto3-stubs" -version = "1.43.83" +version = "1.43.98" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "botocore-stubs" }, { name = "types-s3transfer" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/4a/2b/787cd161c23c23e97c5d13ec7992a26198f5cbe4a97ea4e7428b21b83a77/boto3_stubs-1.43.83.tar.gz", hash = "sha256:ac8f35cea6c125298e30516255cf616ba61285df5188ac4ecfd38d9506346c99", size = 104584, upload-time = "2026-08-28T19:47:22.251Z" } +sdist = { url = "https://files.pythonhosted.org/packages/0a/6f/523d9dafeabd24376f59eacd4743b71fd0493cc575cf7974b26d42b5596c/boto3_stubs-1.43.98.tar.gz", hash = "sha256:43b3b183b55a2ef17780da65fd7fb66456cb76727181a4edecde104053c06152", size = 104596, upload-time = "2026-09-18T22:36:24.662Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b4/53/e484f5ab8314efc542dcd6532458275add6f8d438a277fe212dc1f038673/boto3_stubs-1.43.83-py3-none-any.whl", hash = "sha256:40e914886ded99be487d8d53b049f36a52ab0ca49715e392d6f49705413c57c0", size = 71512, upload-time = "2026-08-28T19:47:14.666Z" }, + { url = "https://files.pythonhosted.org/packages/41/c3/e276ea9fda323ef883638cadc11d2b682460e7e0ffd095a534e119381345/boto3_stubs-1.43.98-py3-none-any.whl", hash = "sha256:3ba0f125c14c1c0dc37b027ca74e33dfb43c5939c8812baa0124fd62fff29e90", size = 71512, upload-time = "2026-09-18T22:36:19.288Z" }, ] [package.optional-dependencies] @@ -481,15 +481,15 @@ wheels = [ [[package]] name = "fakeredis" -version = "2.37.1" +version = "2.38.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "redis" }, { name = "sortedcontainers" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/95/5e/fce3355d20d37dc18f7f4c422653ac69cabffa019afaaf607e26704fb829/fakeredis-2.37.1.tar.gz", hash = "sha256:9045851b0a9fe56312696aadc82435141aa43a193cab462d372c8fb583a7c087", size = 238182, upload-time = "2026-08-18T15:22:01.676Z" } +sdist = { url = "https://files.pythonhosted.org/packages/bf/6b/ad7db311fea3a62f7e359720cc8c064f06f8f0006e65272f50d3a23d37f6/fakeredis-2.38.0.tar.gz", hash = "sha256:d2abfd24652f86501044499bf08c9d639db050f695eefc06b8b8b6f0bb24dbd6", size = 271116, upload-time = "2026-09-08T21:15:04.806Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/7b/e0/fe1c760abd9e9cb5dfbce057b6582cb1ac36dfd8b5ce202e1445605c922d/fakeredis-2.37.1-py3-none-any.whl", hash = "sha256:f15c41be151c1e9206416dece764369a4dedb6b1341df3734c3c2c000e405508", size = 152738, upload-time = "2026-08-18T15:22:00.1Z" }, + { url = "https://files.pythonhosted.org/packages/ec/8e/9a80a8808e8a723aacb3a5d9df07890b46b363a056b372d3cd2e655259fe/fakeredis-2.38.0-py3-none-any.whl", hash = "sha256:d9fb0518c4eaa35f1f2c94df6b4a4c97ff3ca9f43d6cc8112317a9037d244301", size = 167360, upload-time = "2026-09-08T21:15:03.219Z" }, ] [[package]] @@ -560,15 +560,15 @@ wheels = [ [[package]] name = "h2" -version = "4.3.0" +version = "4.4.1" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "hpack" }, { name = "hyperframe" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/1d/17/afa56379f94ad0fe8defd37d6eb3f89a25404ffc71d4d848893d270325fc/h2-4.3.0.tar.gz", hash = "sha256:6c59efe4323fa18b47a632221a1888bd7fde6249819beda254aeca909f221bf1", size = 2152026, upload-time = "2025-08-23T18:12:19.778Z" } +sdist = { url = "https://files.pythonhosted.org/packages/e7/85/7c366e69d84c17bb778fe41419e1fbcce3033d5b7ce29bbffff0a98b859f/h2-4.4.1.tar.gz", hash = "sha256:4e866ffb1a869ae14dd9b5e6beb5c24a13da0495ad72b65925ded182521c1516", size = 2157281, upload-time = "2026-08-03T11:45:09.509Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/69/b2/119f6e6dcbd96f9069ce9a2665e0146588dc9f88f29549711853645e736a/h2-4.3.0-py3-none-any.whl", hash = "sha256:c438f029a25f7945c69e0ccf0fb951dc3f73a5f6412981daee861431b70e2bdd", size = 61779, upload-time = "2025-08-23T18:12:17.779Z" }, + { url = "https://files.pythonhosted.org/packages/7e/22/e85faf23bd72a92d1921e37d674ca56eb298a3c8be31fdecef0ff2b3aaac/h2-4.4.1-py3-none-any.whl", hash = "sha256:0e25f1462b23c9cb82d9eb02e28bc706dac2a68cb457c6a0d74d63c8a2a5d0e6", size = 62636, upload-time = "2026-08-03T11:44:59.164Z" }, ] [[package]] @@ -607,11 +607,11 @@ wheels = [ [[package]] name = "hpack" -version = "4.1.0" +version = "4.2.0" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/2c/48/71de9ed269fdae9c8057e5a4c0aa7402e8bb16f2c6e90b3aa53327b113f8/hpack-4.1.0.tar.gz", hash = "sha256:ec5eca154f7056aa06f196a557655c5b009b382873ac8d1e66e79e87535f1dca", size = 51276, upload-time = "2025-01-22T21:44:58.347Z" } +sdist = { url = "https://files.pythonhosted.org/packages/26/5b/fcabf6028144a8723726318b07a32c2f3314acdff6265743cf08a344b18e/hpack-4.2.0.tar.gz", hash = "sha256:0895cfa3b5531fc65fe439c05eb65144f123bf7a394fcaa56aa423548d8e45c0", size = 51300, upload-time = "2026-06-23T18:34:46.667Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/07/c6/80c95b1b2b94682a72cbdbfb85b81ae2daffa4291fbfa1b1464502ede10d/hpack-4.1.0-py3-none-any.whl", hash = "sha256:157ac792668d995c657d93111f46b4535ed114f0c9c8d672271bbec7eae1b496", size = 34357, upload-time = "2025-01-22T21:44:56.92Z" }, + { url = "https://files.pythonhosted.org/packages/71/b4/4a9fcfb2aef6ba44d9073ecd301443aa00b3dac95de5619f2a7de7ec8a91/hpack-4.2.0-py3-none-any.whl", hash = "sha256:858ac0b02280fa582b5080d68db0899c62a80375e0e5413a74970c5e518b6986", size = 34246, upload-time = "2026-06-23T18:34:45.472Z" }, ] [[package]] @@ -1279,27 +1279,27 @@ wheels = [ [[package]] name = "ruff" -version = "0.16.5" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/f3/85/c8e12473c93018f92d19dd988a294202e1c27426c47ec4de53ffb847b8d8/ruff-0.16.5.tar.gz", hash = "sha256:1b88500f9ffbcab3dedb0082c9f9492e91ec3d618aac1236a3e0189938f7040b", size = 4912003, upload-time = "2026-08-27T16:34:18.258Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/b6/77c90a970fe2dae17a723acbd011043ea97c98d7deacccefdc4ba74ec512/ruff-0.16.5-py3-none-linux_armv6l.whl", hash = "sha256:12e5f673e774c35fbb62f288809c7653b73445f8ecec6b6063fd6ea3521aa14b", size = 10011941, upload-time = "2026-08-27T16:33:41.287Z" }, - { url = "https://files.pythonhosted.org/packages/4b/46/6cf67cf6411885a1d6f7f6d801682f155536a85176d10b605e2ceffed8bd/ruff-0.16.5-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:eda58a5802de40e7ed5b32b64e0b32539338cc6fcd2c78f61e3ad6a0d79f51c3", size = 10204049, upload-time = "2026-08-27T16:33:44.056Z" }, - { url = "https://files.pythonhosted.org/packages/46/fd/c8720ca7a090abf0c2fef4abe8a5ef6e5127ed15196d8886ff75a2b370e2/ruff-0.16.5-py3-none-macosx_11_0_arm64.whl", hash = "sha256:c5ae9a7b9a8875131f40f8fe967cc86abf899779efd663cb7ce3d572d01da7eb", size = 9809037, upload-time = "2026-08-27T16:33:46.257Z" }, - { url = "https://files.pythonhosted.org/packages/43/45/a684caacdedaca180f52bacccc40bf0789d2c5a7c75f25324853e9eaedb5/ruff-0.16.5-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:7b719b0a1f4d59710d283ab2965f621684a108a9e41da622e3b23f0326cd0025", size = 9964129, upload-time = "2026-08-27T16:33:48.352Z" }, - { url = "https://files.pythonhosted.org/packages/9e/f2/5d2bcdaca6b5b93d1b4dfc166cd2aebf7680143a1b38a28759df13a94d31/ruff-0.16.5-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:2298f2780ed1be0c5cb1361e32ab7b1467f3cce7dabe101d2210a314f2fe42e9", size = 9821518, upload-time = "2026-08-27T16:33:50.57Z" }, - { url = "https://files.pythonhosted.org/packages/aa/ff/011cce29accf9257d5974145b733fc653a37985ed6825413a3987cefbfe0/ruff-0.16.5-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:258f29035a2dd021e7861e631b227a5b3f14e50c1184c9a6a122c5f4576154d7", size = 10534835, upload-time = "2026-08-27T16:33:52.522Z" }, - { url = "https://files.pythonhosted.org/packages/d7/5a/f0cf109bada9bba0e96c90c21c9f9251803f57225c32d293327a03c710d6/ruff-0.16.5-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b9a4f0432966834019c74d1b7e5c51224305d7713f3d7faf3e7451f1a3be3cde", size = 11252550, upload-time = "2026-08-27T16:33:54.521Z" }, - { url = "https://files.pythonhosted.org/packages/63/4d/1d481aaea2046c6a7ed7c291f9004c669cce3c087b6b376ed5b08271e3fe/ruff-0.16.5-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:b5eb3a8c3d0ade9cea42b591fd530368e8798380e30e0a308b85a5cf718f09ea", size = 10777949, upload-time = "2026-08-27T16:33:56.88Z" }, - { url = "https://files.pythonhosted.org/packages/ee/34/ee245ca55f64443233034b3d02b03236b19242004281247c079390b7facd/ruff-0.16.5-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ef0f69e191a13a3c9816f63163c88790cb12cd157bbbb384e9c44745702ab105", size = 10311656, upload-time = "2026-08-27T16:33:59.12Z" }, - { url = "https://files.pythonhosted.org/packages/a7/4d/c33a333e341c0a2b96c715b52d89a606f5a34cd4ac493cd9b8d0187186b8/ruff-0.16.5-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:0eeab41fbea2c42f98dfb9822cdccda9d24ba38d49f6dc945b5c236d48f0ef29", size = 10532125, upload-time = "2026-08-27T16:34:01.166Z" }, - { url = "https://files.pythonhosted.org/packages/30/e1/a64cef78b40192497bb98a27a8aa8f2c98ee9ee15bc97f7712d94ef32937/ruff-0.16.5-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:f0768e9df4300713fff30733c87575f68b6f1d8de41184e505b7fdd9c0c95eaf", size = 10097648, upload-time = "2026-08-27T16:34:03.16Z" }, - { url = "https://files.pythonhosted.org/packages/cc/4e/4cdc9ed3c3e109d2f71e62572a37457298d7bc7501ec3138babb7ed32bbd/ruff-0.16.5-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:95cc70cdc7aa80c338de356279d2adbeb2de0f520b9ecd8aba75b94e95e02f91", size = 9829344, upload-time = "2026-08-27T16:34:05.134Z" }, - { url = "https://files.pythonhosted.org/packages/39/4a/31ed35ce31729955fc583ee0d176d6e784c1290cb0b0a75cb2134c1ab72a/ruff-0.16.5-py3-none-musllinux_1_2_i686.whl", hash = "sha256:d185c8398ded1bfd91c0c2cb258346307571eccc473a8490af8c3977399c384a", size = 10277117, upload-time = "2026-08-27T16:34:07.425Z" }, - { url = "https://files.pythonhosted.org/packages/a8/a0/60356d86687b4b666d593df213f4dc3041750d024cb7bf2cfa81cfd65c2e/ruff-0.16.5-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fb8e3a3c4c6a784150a7ced53b015f4b253fc2bf97a610886419ead64b4756ef", size = 10711653, upload-time = "2026-08-27T16:34:09.712Z" }, - { url = "https://files.pythonhosted.org/packages/ed/20/656d67f5b25ca9bda4e02b1de25867b2954e1d19e03648060f167ad0f4cc/ruff-0.16.5-py3-none-win32.whl", hash = "sha256:288b0a5f080492fe5635db849f9e2e84aa3cce7b7f0e955997d416c507c76a26", size = 10034250, upload-time = "2026-08-27T16:34:11.8Z" }, - { url = "https://files.pythonhosted.org/packages/5b/42/ee8e68a207b9127fcde6c3d7e197def432f346cb1af159e1fa14ca0d1cdc/ruff-0.16.5-py3-none-win_amd64.whl", hash = "sha256:ddc6385fb2137f616357ca03d6c74f4be987f80fed4008566b754f6032b8546f", size = 10516714, upload-time = "2026-08-27T16:34:13.963Z" }, - { url = "https://files.pythonhosted.org/packages/73/e3/7df5a396e445b9ba49ce9a9437439a4d80042c61c0ade199abf8d16de1ac/ruff-0.16.5-py3-none-win_arm64.whl", hash = "sha256:a64abe90968719b851bb7cedffaa8753fbdbdadab483089682db623f3edc587e", size = 10391564, upload-time = "2026-08-27T16:34:16.064Z" }, +version = "0.16.8" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ba/78/449cb84790bd5cc3823b2652ee405a4558856e5c4195aee3a16bf7b3eb5d/ruff-0.16.8.tar.gz", hash = "sha256:9247bf92b5f04d825c8639a4fe423ec2e4222acd9222e58412b0dab7e442798b", size = 4938814, upload-time = "2026-09-16T15:54:46.688Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ac/25/6071aabc530e9be7e2c195e8fe3f7aea2735405b6cf447212832d7811831/ruff-0.16.8-py3-none-linux_armv6l.whl", hash = "sha256:6ffbd6d87383c1edf5f6fa890f10200950240d7c1a16052a19a09d3a2307dd38", size = 10048966, upload-time = "2026-09-16T15:53:57.605Z" }, + { url = "https://files.pythonhosted.org/packages/54/98/07f90ecbc74dd5fb5764f11f2bc774d6a7cffef92d2ff5f5b4e9e23c754e/ruff-0.16.8-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:42ed6b878ed61e3acca92f2730a17acff39286944ea82398544696366a6f925e", size = 10165498, upload-time = "2026-09-16T15:54:01.14Z" }, + { url = "https://files.pythonhosted.org/packages/fe/1f/e6a712e3b47cad4a40600134105ed193cb773f618a42eb7ba323cb812cc0/ruff-0.16.8-py3-none-macosx_11_0_arm64.whl", hash = "sha256:7ea781c7f2afba8c6a505ea0fb3f994020249e0c450635f5381286fea6b46170", size = 9830004, upload-time = "2026-09-16T15:54:03.998Z" }, + { url = "https://files.pythonhosted.org/packages/23/f2/311a08776d75d81c7676e20b6b020ae63cbe881fcdc7a8dd64e6e18bdd93/ruff-0.16.8-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:8efeae3bbe414a5efefda11a792dfb51ef90ac48d50c4830de2f644caf3e8659", size = 9986558, upload-time = "2026-09-16T15:54:06.804Z" }, + { url = "https://files.pythonhosted.org/packages/f3/ed/37b6cb3d3ba8c73e68ae3eb1d502383beb5aa05a582bb7bb3a922f929f54/ruff-0.16.8-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:3a79b795469fef7fc6e908b218eed2eb17332afd85031db6480dc864560e69b2", size = 9877332, upload-time = "2026-09-16T15:54:09.552Z" }, + { url = "https://files.pythonhosted.org/packages/22/cc/40873a8f36ad084cc540d55fcca7077264d5b13b24659e9180c176fb2b08/ruff-0.16.8-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:3fdc5563cdc50555e6fba39322850860e9267c1b3d12c26a74729d8604c3c812", size = 10507125, upload-time = "2026-09-16T15:54:12.152Z" }, + { url = "https://files.pythonhosted.org/packages/c3/e4/fc91a642b78ccbab6b9477720f3644ae7a10a9bcce69a934679cd64f62bc/ruff-0.16.8-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:34508983c70665578dab88f5223d8e6228307e1135398ca8bfc8b7e9501e282b", size = 11336694, upload-time = "2026-09-16T15:54:15.489Z" }, + { url = "https://files.pythonhosted.org/packages/c2/3d/bbd2a9a600a4e73dc3e7548a249c8d1671273464b55822c6fae50f602dff/ruff-0.16.8-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:644bb578569e0ffc575741232bd385dacdd6fbe123f1a729e7a225f54aa3957f", size = 10774448, upload-time = "2026-09-16T15:54:18.16Z" }, + { url = "https://files.pythonhosted.org/packages/1a/41/d83af9879a7b6e8bf5fe16b1da0b134049d2f5d3afac12defb0897cb84bd/ruff-0.16.8-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:15e7d226246961db9235098333caa13063906d3851136b84c2900b82f5daa1df", size = 10323796, upload-time = "2026-09-16T15:54:20.743Z" }, + { url = "https://files.pythonhosted.org/packages/f5/2c/cefd07bfe914b84943ea769ade8d607bd22750b965d3228eefd7cebd15d0/ruff-0.16.8-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:a2bf6bc3e9ebdd4449abc6f06cf64b98051a2c61cf94d2fe9596518c881f1a1e", size = 10514115, upload-time = "2026-09-16T15:54:23.497Z" }, + { url = "https://files.pythonhosted.org/packages/f3/9d/76a2e26c79a23be6e6e3664c57bec9e9fc8de155cfb9e4b67ea91b64f9d7/ruff-0.16.8-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:6ca111ba0849539165e9e59d2b442542f3c1e8060ebbdea82494f1ffbccb1e1f", size = 10072582, upload-time = "2026-09-16T15:54:26.185Z" }, + { url = "https://files.pythonhosted.org/packages/2e/d4/f42edddb39668af1a559ceafa3823aedd65633a48dc9768e775485faa2c1/ruff-0.16.8-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:359a1e5b495448ee1e91018064382ebc86f90e8aac2fed222c7d0e4e8df85fd2", size = 9879644, upload-time = "2026-09-16T15:54:29.278Z" }, + { url = "https://files.pythonhosted.org/packages/f8/d4/913e3195d95e0378786c6656945c865f534a3560e29139da4882aff630d1/ruff-0.16.8-py3-none-musllinux_1_2_i686.whl", hash = "sha256:59e8f5681349474110b24d62e93cfda6593f5fa3473446ca3705200cac1a08b9", size = 10231569, upload-time = "2026-09-16T15:54:32.036Z" }, + { url = "https://files.pythonhosted.org/packages/2b/c4/8aa6ea0bdcedbd1bf87397e2fc4ed8406448ea5842f8660bc6e5f163039d/ruff-0.16.8-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:efa3e7a16d1baaa79957888dfdf8be9ef2e44db81cb032af06d76632ab59e773", size = 10663666, upload-time = "2026-09-16T15:54:34.838Z" }, + { url = "https://files.pythonhosted.org/packages/3d/02/7f10ef4700bc223c30a3fdd10631a29830c45524b810a3c7ed947af64591/ruff-0.16.8-py3-none-win32.whl", hash = "sha256:55793ba85c69921e89be061426d91a78652d6e50317c962240922747a4eb713f", size = 10093472, upload-time = "2026-09-16T15:54:37.47Z" }, + { url = "https://files.pythonhosted.org/packages/1e/5d/a509c07d714b6da88f2c518b4637cf6f1d46b074be8f0f1e5fb9ff5126fe/ruff-0.16.8-py3-none-win_amd64.whl", hash = "sha256:a6b85621fd3c81e31fc5f5add09c9c078b430db3595ca632efafdec9e64ebfaa", size = 10586899, upload-time = "2026-09-16T15:54:40.488Z" }, + { url = "https://files.pythonhosted.org/packages/fe/a0/50787329e4f20bf9dc9f6230015d46ec69c51a97ace5bc202dae4755365d/ruff-0.16.8-py3-none-win_arm64.whl", hash = "sha256:d075e820af612102ce217f07cc93e69f9490b10ec13ea85fa87bd03d996cef8a", size = 10386316, upload-time = "2026-09-16T15:54:43.332Z" }, ] [[package]] @@ -1343,10 +1343,10 @@ dev = [ requires-dist = [ { name = "aioboto3", specifier = ">=15.5.0" }, { name = "boto3", specifier = ">=1.40.0" }, - { name = "boto3-stubs", extras = ["s3"], marker = "extra == 'dev'", specifier = ">=1.43.83" }, + { name = "boto3-stubs", extras = ["s3"], marker = "extra == 'dev'", specifier = ">=1.43.98" }, { name = "botocore", specifier = ">=1.40.0" }, { name = "cryptography", specifier = ">=50.0.1" }, - { name = "fakeredis", marker = "extra == 'dev'", specifier = ">=2.37.1" }, + { name = "fakeredis", marker = "extra == 'dev'", specifier = ">=2.38.0" }, { name = "fastapi", specifier = ">=0.141.1" }, { name = "httpx", extras = ["http2"], specifier = ">=0.28.1" }, { name = "moto", extras = ["s3"], marker = "extra == 'dev'", specifier = ">=5.2.3" }, @@ -1363,9 +1363,9 @@ requires-dist = [ { name = "python-multipart", specifier = ">=0.0.32" }, { name = "redis", extras = ["hiredis"], specifier = ">=8.1.0" }, { name = "requests", marker = "extra == 'dev'", specifier = ">=2.34.2" }, - { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.16.5" }, + { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.16.8" }, { name = "structlog", specifier = ">=26.1.0" }, - { name = "uvicorn", extras = ["standard"], specifier = ">=0.52.4" }, + { name = "uvicorn", extras = ["standard"], specifier = ">=0.53.0" }, { name = "uvloop", marker = "sys_platform != 'win32'", specifier = ">=0.22.0" }, ] provides-extras = ["dev"] @@ -1471,15 +1471,15 @@ wheels = [ [[package]] name = "uvicorn" -version = "0.52.4" +version = "0.53.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "click" }, { name = "h11" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/f2/0f/3f86e61397dd33bf2ccf28188c40db6a740658aeebbbf6e7dbc101a1f487/uvicorn-0.52.4.tar.gz", hash = "sha256:73acfee47a0b133c5de13d219492d62d8a31e935f4fe6e41a232451a15379f86", size = 100627, upload-time = "2026-08-19T06:27:41.821Z" } +sdist = { url = "https://files.pythonhosted.org/packages/5d/ad/04bbb797c84fc1f26cb171f7394716f4865ffb8d8c5e1eef42565c2dfa6b/uvicorn-0.53.0.tar.gz", hash = "sha256:a9356f0cb89b3b8621529c5d5eebd69bfe154f4c3f68b4cf2de47e45fa855c2e", size = 110881, upload-time = "2026-09-14T07:44:23.815Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f1/79/4a20b54ab0491485ccd8c077db2d39187c7f12b3e15485d38a7be37c81b4/uvicorn-0.52.4-py3-none-any.whl", hash = "sha256:f86e41a149d7d05a9969337e3946a9c171c06a5d42680896daaba624aeac8da1", size = 79871, upload-time = "2026-08-19T06:27:40.36Z" }, + { url = "https://files.pythonhosted.org/packages/76/18/0eea75741ee812e9f598b687619ce2454f6c3a1c5cd21ea990ec6bd26f45/uvicorn-0.53.0-py3-none-any.whl", hash = "sha256:e8dca71ec86dce5f04e333f0d56cdedf942446e6643b9cea1af0d6d3a02cb03e", size = 87081, upload-time = "2026-09-14T07:44:22.179Z" }, ] [package.optional-dependencies]