From ef1d9bafdf17f6a41b7c4931c0042aef6fca050f Mon Sep 17 00:00:00 2001 From: Alessandro De Blasis Date: Thu, 1 Oct 2026 06:55:53 +0300 Subject: [PATCH 1/3] feat(engine): crypto.md5_hex_concat total builtin for S3 MPU ETags --- internal/starlark/cryptomod.go | 53 +++++++++- internal/starlark/cryptomod_test.go | 146 ++++++++++++++++++++++++++++ 2 files changed, 198 insertions(+), 1 deletion(-) diff --git a/internal/starlark/cryptomod.go b/internal/starlark/cryptomod.go index cfc2aa1e..be4af43c 100644 --- a/internal/starlark/cryptomod.go +++ b/internal/starlark/cryptomod.go @@ -37,7 +37,11 @@ var cryptoModule = &starlarkstruct.Module{ // md5 exists for PROTOCOL CHECKSUMS only (SQS MD5OfMessage etc.) — // providers validate it client-side; it is not a security primitive // and adapters must not use it for signatures. - "md5": sk.NewBuiltin("crypto.md5", md5Hash), + "md5": sk.NewBuiltin("crypto.md5", md5Hash), + // md5_hex_concat exists for S3 MPU ETag composition only + // (MD5 over concatenated part-MD5 binaries; handler appends -N). + // Same compat-checksum charter as md5: never auth/integrity. + "md5_hex_concat": sk.NewBuiltin("crypto.md5_hex_concat", md5HexConcat), "base64_encode": sk.NewBuiltin("crypto.base64_encode", base64Encode), "base64_decode": sk.NewBuiltin("crypto.base64_decode", base64Decode), "base64url_encode": sk.NewBuiltin("crypto.base64url_encode", base64urlEncode), @@ -136,6 +140,53 @@ func sha256Hash(_ *sk.Thread, b *sk.Builtin, args sk.Tuple, kwargs []sk.Tuple) ( return sk.String(out), nil } +// md5HexConcat implements crypto.md5_hex_concat(hex_list) -> hex | None. +// It hex-decodes each element (case-insensitive, normalized lower), concats +// the binaries, and returns hex(md5(concat)). The handler appends "-N". +// +// Total on any shape mismatch (None, never Error): handlers have no +// try/except and a Starlark Error surfaces as an unhandled 500 +// (adapter_dispatch.go:155-159), so arity, kwarg, type, length, hex, empty, +// and >10000 mismatches all return None. Manual arg inspection (not +// UnpackArgs) keeps extra kwargs and wrong arity total. +func md5HexConcat(_ *sk.Thread, _ *sk.Builtin, args sk.Tuple, kwargs []sk.Tuple) (sk.Value, error) { + if len(kwargs) != 0 { + return sk.None, nil + } + if len(args) != 1 { + return sk.None, nil + } + lst, ok := args[0].(*sk.List) + if !ok { + return sk.None, nil + } + n := lst.Len() + if n == 0 || n > 10000 { + return sk.None, nil + } + raw := make([]byte, 0, n*16) + it := lst.Iterate() + defer it.Done() + var v sk.Value + for it.Next(&v) { + s, ok := v.(sk.String) + if !ok { + return sk.None, nil + } + str := string(s) + if len(str) != 32 { + return sk.None, nil + } + b, err := hex.DecodeString(strings.ToLower(str)) + if err != nil || len(b) != 16 { + return sk.None, nil + } + raw = append(raw, b...) + } + sum := md5.Sum(raw) + return sk.String(hex.EncodeToString(sum[:])), nil +} + func base64Encode(_ *sk.Thread, b *sk.Builtin, args sk.Tuple, kwargs []sk.Tuple) (sk.Value, error) { var data string if err := sk.UnpackArgs(b.Name(), args, kwargs, "data", &data); err != nil { diff --git a/internal/starlark/cryptomod_test.go b/internal/starlark/cryptomod_test.go index d8f6a6cf..be803fbc 100644 --- a/internal/starlark/cryptomod_test.go +++ b/internal/starlark/cryptomod_test.go @@ -2,9 +2,12 @@ package starlark import ( "crypto/hmac" + "crypto/md5" "crypto/sha1" "crypto/sha256" "encoding/base64" + "encoding/hex" + "strings" "testing" sk "go.starlark.net/starlark" @@ -106,3 +109,146 @@ func bytesToHex(b []byte) string { } return string(out) } + +func callMD5HexConcatRaw(t *testing.T, args []sk.Value, kwargs []sk.Tuple) (sk.Value, error) { + t.Helper() + fn, ok := cryptoModule.Members["md5_hex_concat"] + if !ok { + t.Fatalf("crypto.md5_hex_concat not found") + } + return sk.Call(new(sk.Thread), fn, sk.Tuple(args), kwargs) +} + +func md5HexConcatWant(hexes []string) string { + raw := make([]byte, 0, len(hexes)*16) + for _, h := range hexes { + b, err := hex.DecodeString(strings.ToLower(h)) + if err != nil { + panic(err) + } + raw = append(raw, b...) + } + sum := md5.Sum(raw) + return hex.EncodeToString(sum[:]) +} + +func TestMD5HexConcat(t *testing.T) { + // Single valid 32-hex → hex out (md5("hello") digest as input element). + single := "5d41402abc4b2a76b9719d911017c592" + got, err := callMD5HexConcatRaw(t, + []sk.Value{sk.NewList([]sk.Value{sk.String(single)})}, nil) + if err != nil { + t.Fatalf("md5_hex_concat single: unexpected error: %v", err) + } + if got == sk.None { + t.Fatal("md5_hex_concat single: want hex, got None") + } + if want := md5HexConcatWant([]string{single}); string(got.(sk.String)) != want { + t.Errorf("md5_hex_concat single = %q, want %q", got, want) + } + + // Uppercase accepted, normalized to same lowercase result. + upper := "5D41402ABC4B2A76B9719D911017C592" + gotUpper, err := callMD5HexConcatRaw(t, + []sk.Value{sk.NewList([]sk.Value{sk.String(upper)})}, nil) + if err != nil { + t.Fatalf("md5_hex_concat upper: unexpected error: %v", err) + } + if gotUpper == sk.None { + t.Fatal("md5_hex_concat upper: want hex, got None") + } + if string(gotUpper.(sk.String)) != string(got.(sk.String)) { + t.Errorf("md5_hex_concat upper = %q, want %q (lowercased)", gotUpper, got) + } + + // Multi-element vector: concat binary then md5. + multi := []string{single, "d41d8cd98f00b204e9800998ecf8427e"} + gotMulti, err := callMD5HexConcatRaw(t, + []sk.Value{sk.NewList([]sk.Value{sk.String(multi[0]), sk.String(multi[1])})}, nil) + if err != nil { + t.Fatalf("md5_hex_concat multi: unexpected error: %v", err) + } + if want := md5HexConcatWant(multi); string(gotMulti.(sk.String)) != want { + t.Errorf("md5_hex_concat multi = %q, want %q", gotMulti, want) + } + + // Every shape mismatch must be total (None, never Error). + noneCases := map[string]sk.Value{ + "empty list": sk.NewList(nil), + "None top": sk.None, + "string top": sk.String(single), + "int top": sk.MakeInt(1), + "dict top": sk.NewDict(0), + "tuple top": sk.Tuple{sk.String(single)}, + "non-string": sk.NewList([]sk.Value{sk.MakeInt(1)}), + "none elem": sk.NewList([]sk.Value{sk.None}), + "empty string": sk.NewList([]sk.Value{sk.String("")}), + "short non-32": sk.NewList([]sk.Value{sk.String("abc")}), + "long 33": sk.NewList([]sk.Value{sk.String(single + "0")}), + "64-hex": sk.NewList([]sk.Value{sk.String(strings.Repeat("0", 64))}), + "non-hex": sk.NewList([]sk.Value{sk.String(strings.Repeat("z", 32))}), + "ws-padded": sk.NewList([]sk.Value{sk.String(" " + single + " ")}), + "ws-tab-padded": sk.NewList([]sk.Value{sk.String("\t" + single)}), + "one bad elem": sk.NewList([]sk.Value{ + sk.String(single), sk.String(strings.Repeat("z", 32)), + }), + } + for name, arg := range noneCases { + v, err := callMD5HexConcatRaw(t, []sk.Value{arg}, nil) + if err != nil { + t.Errorf("md5_hex_concat %s: want None with nil error, got error %v", name, err) + continue + } + if v != sk.None { + t.Errorf("md5_hex_concat %s = %v, want None", name, v) + } + } + + // >10000 elements → None. + big := make([]sk.Value, 10001) + for i := range big { + big[i] = sk.String(single) + } + v, err := callMD5HexConcatRaw(t, []sk.Value{sk.NewList(big)}, nil) + if err != nil { + t.Fatalf("md5_hex_concat >10000: unexpected error: %v", err) + } + if v != sk.None { + t.Errorf("md5_hex_concat >10000 = %v, want None", v) + } + // Exactly 10000 valid elements must still hash (boundary). + boundary := make([]sk.Value, 10000) + for i := range boundary { + boundary[i] = sk.String(single) + } + v, err = callMD5HexConcatRaw(t, []sk.Value{sk.NewList(boundary)}, nil) + if err != nil { + t.Fatalf("md5_hex_concat 10000: unexpected error: %v", err) + } + if v == sk.None { + t.Error("md5_hex_concat 10000 valid: want hex, got None") + } + + // Extra kwargs → None (total, not an UnpackArgs error). + v, err = callMD5HexConcatRaw(t, + []sk.Value{sk.NewList([]sk.Value{sk.String(single)})}, + []sk.Tuple{{sk.String("extra"), sk.String("1")}}) + if err != nil { + t.Errorf("md5_hex_concat extra kwarg: want None with nil error, got error %v", err) + } else if v != sk.None { + t.Errorf("md5_hex_concat extra kwarg = %v, want None", v) + } + + // Wrong positional arity → None (total, not an error). + for _, args := range [][]sk.Value{ + nil, + {sk.NewList([]sk.Value{sk.String(single)}), sk.NewList(nil)}, + } { + v, err := callMD5HexConcatRaw(t, args, nil) + if err != nil { + t.Errorf("md5_hex_concat arity %d: want None with nil error, got error %v", len(args), err) + } else if v != sk.None { + t.Errorf("md5_hex_concat arity %d = %v, want None", len(args), v) + } + } +} From 718fc5cea9025d584c89186b02653923aa06ea57 Mon Sep 17 00:00:00 2001 From: Alessandro De Blasis Date: Thu, 1 Oct 2026 07:01:03 +0300 Subject: [PATCH 2/3] fix(s3): MD5 ETags single+multipart via md5_hex_concat --- adapters/aws-s3-style/scripts/lib.star | 51 ++++++++++++++++------ adapters/aws-s3-style/scripts/objects.star | 14 +++--- internal/engine/aws_s3_style_test.go | 29 +++++++++--- 3 files changed, 67 insertions(+), 27 deletions(-) diff --git a/adapters/aws-s3-style/scripts/lib.star b/adapters/aws-s3-style/scripts/lib.star index 9648d64b..fd8e5ce9 100644 --- a/adapters/aws-s3-style/scripts/lib.star +++ b/adapters/aws-s3-style/scripts/lib.star @@ -739,8 +739,9 @@ def _find_object(bucket, key): # _upsert_object writes an object's content bytes (reusing the existing # blob id when overwriting, so the blob store has one file per object) and # refreshes its metadata doc. Returns nothing; the ETag is derived by the -# caller (it differs for simple vs multipart uploads). -def _upsert_object(bucket, key, raw, ct, etag): +# caller (it differs for simple vs multipart uploads). meta is the +# user-metadata map (x-amz-meta-*); {} when none. +def _upsert_object(bucket, key, raw, ct, etag, meta): oc = store_collection("objects") bid = "" obj_id = "" @@ -758,6 +759,7 @@ def _upsert_object(bucket, key, raw, ct, etag): "bid": bid, "contentType": ct, "etag": etag, + "metadata": meta, "lastModified": _unix_to_iso8601(now_unix), "lastModifiedUnix": now_unix, "size": len(raw), @@ -787,10 +789,10 @@ def _upsert_object(bucket, key, raw, ct, etag): # a non-ascending part list → 400 InvalidPartOrder. # - Completion assembles the parts, in ascending part-number order, # into the object; abort discards every part and creates nothing. -# - Documented deviations: part ETags are SHA-256 digests (the crypto -# module has no MD5), the multipart object ETag is -# sha256(concat part etags)-N, and the 5 MiB minimum part size is -# NOT enforced so small chunks can be exercised in tests. +# - Documented deviations: part ETags are MD5 digests (like real S3), +# the multipart object ETag is md5(concat part-md5 binaries)-N, and +# the 5 MiB minimum part size is NOT enforced so small chunks can be +# exercised in tests. # Real S3 allows part numbers 1..10k (assembled to keep digit runs short). _MPU_MAX_PART_NUMBER = 10 * 1000 @@ -871,7 +873,7 @@ def _mpu_create(req, bucket, key): # _mpu_upload_part handles PUT /{bucket}/{key}?partNumber=N&uploadId=... — # stores the part bytes (out-of-order and re-uploads both fine) and returns -# the part ETag (SHA-256 of the verbatim part bytes). +# the part ETag (MD5 of the verbatim part bytes, like real S3). def _mpu_upload_part(req, bucket, key): part_raw = _query_val(req, "partNumber") upload_id = _query_val(req, "uploadId") @@ -896,7 +898,7 @@ def _mpu_upload_part(req, bucket, key): raw = req.get("raw_body", "") if raw == None: raw = "" - etag = crypto.sha256(raw) + etag = crypto.md5(raw) # One blob per (upload, part number); re-uploading a part overwrites it. bid = upload_id + "_p" + str(n) @@ -973,6 +975,16 @@ def _strip_quotes(s): out = out + s[i] return out +# _strip_weak normalizes a Complete ETag for comparison: strips one W/ +# prefix, removes quotes, lowercases (uppercase hex accepted, like real +# S3; Complete is strict otherwise — no whitespace trim). +def _strip_weak(s): + if s == None: + return "" + if _has_prefix(s, "W/"): + s = s[2:] + return _strip_quotes(s).lower() + # _mpu_parse_complete parses the CompleteMultipartUpload XML body into an # ordered [(part_number, etag), ...] list, or None when malformed. def _mpu_parse_complete(raw): @@ -1037,32 +1049,45 @@ def _mpu_complete(req, bucket, key): prev = n # Every listed part must exist with a matching ETag (real S3: InvalidPart). + # Both sides are normalized (W/ prefix, quotes, case) before comparing; + # an empty request ETag never matches (always checked, no bypass). for entry in listed: n = entry[0] etag_req = entry[1] row = stored.get(n, None) if row == None: return _xml_error("InvalidPart", "One or more of the specified parts could not be found. The part may not have been uploaded, or the specified entity tag may not match the part's entity tag.", "/" + bucket + "/" + key, 400) - if etag_req != "" and etag_req != row.get("etag", ""): + if _strip_weak(etag_req) != _strip_weak(row.get("etag", "")): return _xml_error("InvalidPart", "One or more of the specified parts could not be found. The part may not have been uploaded, or the specified entity tag may not match the part's entity tag.", "/" + bucket + "/" + key, 400) # Assemble: concatenate the part blobs in ascending part-number order. b = store_blob("s3-objects") full = "" - concat_etags = "" + hexes = [] for entry in listed: row = stored[entry[0]] content = b.get(row.get("bid", "")) if content == None: content = "" full = full + content - concat_etags = concat_etags + row.get("etag", "") - etag = crypto.sha256(concat_etags) + "-" + str(len(listed)) + hexes.append(_strip_weak(row.get("etag", ""))) + # Guarded pre-check (type first — never bare len() on a non-string, + # which would 500): any shape mismatch → 400 InvalidPart without + # calling the builtin. The builtin is total too (None on mismatch). + if len(hexes) == 0 or len(hexes) > 10000: + return _xml_error("InvalidPart", "One or more of the specified parts could not be found. The part may not have been uploaded, or the specified entity tag may not match the part's entity tag.", "/" + bucket + "/" + key, 400) + for h in hexes: + if type(h) != "string" or len(h) != 32 or not _is_hex(h): + return _xml_error("InvalidPart", "One or more of the specified parts could not be found. The part may not have been uploaded, or the specified entity tag may not match the part's entity tag.", "/" + bucket + "/" + key, 400) + digest = crypto.md5_hex_concat(hexes) + if digest == None: + return _xml_error("InvalidPart", "One or more of the specified parts could not be found. The part may not have been uploaded, or the specified entity tag may not match the part's entity tag.", "/" + bucket + "/" + key, 400) + etag = digest + "-" + str(len(hexes)) ct = upload.get("contentType", "application/octet-stream") if ct == None or ct == "": ct = "application/octet-stream" - _upsert_object(bucket, key, full, ct, etag) + _upsert_object(bucket, key, full, ct, etag, {}) _mpu_discard(upload_id) store_collection("mpu_uploads").delete(upload_id) diff --git a/adapters/aws-s3-style/scripts/objects.star b/adapters/aws-s3-style/scripts/objects.star index b509f6d5..43db9703 100644 --- a/adapters/aws-s3-style/scripts/objects.star +++ b/adapters/aws-s3-style/scripts/objects.star @@ -18,12 +18,12 @@ # scripts/lib.star. The POST multipart entry point lives in # scripts/multipart.star. -# _etag derives the object ETag from the content itself: the SHA-256 hex -# digest of the raw body (real S3 uses the MD5 digest for non-multipart -# uploads; the crypto module has no MD5, so the stronger digest is used — -# documented deviation). Returned/stored unquoted; rendered quoted. +# _etag derives the object ETag from the content itself: the MD5 hex +# digest of the raw body, like real S3 for non-multipart uploads. +# (MD5 here is a compat checksum only, never auth/integrity.) +# Returned/stored unquoted; rendered quoted. def _etag(raw): - return crypto.sha256(raw) + return crypto.md5(raw) # _obj_last_modified_rfc1123 renders the stored upload time as an RFC # 1123 Last-Modified header value (falls back to the current clock for @@ -84,10 +84,10 @@ def on_put_object(req): if ct == None: ct = "application/octet-stream" - # Content-derived ETag (SHA-256 of the verbatim bytes); the write path + # Content-derived ETag (MD5 of the verbatim bytes); the write path # (blob + metadata doc) is shared with CompleteMultipartUpload. etag = _etag(raw) - _upsert_object(bucket, key, raw, ct, etag) + _upsert_object(bucket, key, raw, ct, etag, {}) return respond(200, "", { "ETag": '"' + etag + '"', diff --git a/internal/engine/aws_s3_style_test.go b/internal/engine/aws_s3_style_test.go index bc72b33b..ca8047c6 100644 --- a/internal/engine/aws_s3_style_test.go +++ b/internal/engine/aws_s3_style_test.go @@ -4,6 +4,7 @@ import ( "bytes" "context" "crypto/hmac" + "crypto/md5" "crypto/sha256" "encoding/hex" "fmt" @@ -37,6 +38,11 @@ func awsSHA256Hex(b []byte) string { return hex.EncodeToString(sum[:]) } +func awsMD5Hex(b []byte) string { + sum := md5.Sum(b) + return hex.EncodeToString(sum[:]) +} + func awsHMACSHA256(key, data []byte) []byte { h := hmac.New(sha256.New, key) h.Write(data) @@ -247,12 +253,21 @@ func TestAwsS3StyleAdapter(t *testing.T) { if status != 200 { t.Fatalf("put object -> status %d, want 200", status) } - // ETag is content-derived: the quoted SHA-256 hex digest of the bytes. - wantETag := `"` + awsSHA256Hex([]byte(uploadContent)) + `"` + // ETag is content-derived: the quoted MD5 hex digest of the bytes. + wantETag := `"` + awsMD5Hex([]byte(uploadContent)) + `"` if etag != wantETag { t.Fatalf("put object ETag = %q, want %q", etag, wantETag) } + // md5("hello") = 5d41402abc4b2a76b9719d911017c592 (S3 compat checksum). + helloETag, status := s3PutETag(t, base+"/mybucket/hello.txt", []byte("hello"), now) + if status != 200 { + t.Fatalf("put hello -> status %d, want 200", status) + } + if helloETag != `"5d41402abc4b2a76b9719d911017c592"` { + t.Fatalf("put hello ETag = %q, want %q", helloETag, `"5d41402abc4b2a76b9719d911017c592"`) + } + // ===== ListObjectsV2 shows the uploaded object (STATEFUL) ===== body, status = s3Get(t, base+"/mybucket?list-type=2", now) @@ -579,7 +594,7 @@ func s3Delete(t *testing.T, rawurl string, at time.Time) *http.Response { // // - POST ?uploads → 200 InitiateMultipartUploadResult with an UploadId // - UploadPart (out of order: 3, then 1, then 2) → per-part ETags that -// equal the quoted SHA-256 of the part bytes +// equal the quoted MD5 of the part bytes // - ListParts → parts in ascending order with max-parts / // part-number-marker paging // - CompleteMultipartUpload with a missing part → 400 InvalidPart @@ -652,7 +667,7 @@ func TestAwsS3StyleMultipartUpload(t *testing.T) { if st != 200 { t.Fatalf("upload part %d -> %d", tc.n, st) } - want := `"` + awsSHA256Hex(tc.data) + `"` + want := `"` + awsMD5Hex(tc.data) + `"` if etag != want { t.Fatalf("upload part %d ETag = %q, want %q", tc.n, etag, want) } @@ -705,7 +720,7 @@ func TestAwsS3StyleMultipartUpload(t *testing.T) { } // ===== Complete: wrong part ETag → 400 InvalidPart ===== - wrongEtag := s3CompleteBody([][2]string{{"1", etags[1]}, {"2", strings.Repeat("0", 64)}, {"3", etags[3]}}) + wrongEtag := s3CompleteBody([][2]string{{"1", etags[1]}, {"2", strings.Repeat("0", 32)}, {"3", etags[3]}}) body, status = s3Post(t, partURL, []byte(wrongEtag), now) if status != 400 || !strings.Contains(body, "InvalidPart") { t.Fatalf("complete with wrong etag -> %d %q, want 400 InvalidPart", status, body) @@ -858,8 +873,8 @@ func TestAWSS3StyleBinaryRoundTrip(t *testing.T) { if status != 200 { t.Fatalf("put binary -> %d", status) } - if etag != `"`+awsSHA256Hex(bin)+`"` { - t.Fatalf("binary ETag = %q, want quoted sha256 of the bytes", etag) + if etag != `"`+awsMD5Hex(bin)+`"` { + t.Fatalf("binary ETag = %q, want quoted md5 of the bytes", etag) } got, status := s3Get(t, base+"/mybucket/bin.dat", now) From 855cd2c0fca8df5044388de93f6d963b9741676b Mon Sep 17 00:00:00 2001 From: Alessandro De Blasis Date: Thu, 1 Oct 2026 10:14:22 +0300 Subject: [PATCH 3/3] docs: record S3 MD5 ETag change and breaking migration --- CHANGELOG.md | 12 ++++++++++++ CONFORMANCE.md | 2 +- adapters/aws-s3-style/README.md | 21 +++++++++------------ conformance/matrix.json | 2 +- conformance/matrix.yaml | 2 +- 5 files changed, 24 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 542a6f0d..36f08cb9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,18 @@ All notable changes to **stunt** are documented here. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +### Adapters + +- **BREAKING (test double): aws-s3-style ETags are now real MD5.** + Single-object and part ETags are the MD5 hex of the verbatim bytes and + multipart objects use `md5(concat part-md5 binaries)-N`, like real S3. + Fixtures pinning the old 64-hex SHA-256 ETags must be regenerated; abort + and re-upload any in-flight multipart upload, since + `CompleteMultipartUpload` rejects mixed old and new part ETags with + `400 InvalidPart`. MD5 is a compat checksum here, never auth or integrity. + ## [0.52.0] — 2026-08-24 The conformance campaign: every real API adapter now carries a real test diff --git a/CONFORMANCE.md b/CONFORMANCE.md index b43255d3..283860c7 100644 --- a/CONFORMANCE.md +++ b/CONFORMANCE.md @@ -2600,7 +2600,7 @@ behavior notes live in each adapter's README. **Deviations** (5) -- ETags are SHA-256-based (real S3 uses MD5); multipart ETag is sha256(etags)-N +- ETags are MD5 hex (multipart MD5(binary-concat)-N) - Multipart 5 MiB minimum part size not enforced (small parts allowed) - DELETE of a missing bucket is an idempotent 204 (real S3: 404 NoSuchBucket) - SigV4 canonical URI/query rebuilt from decoded values — duplicates indistinguishable diff --git a/adapters/aws-s3-style/README.md b/adapters/aws-s3-style/README.md index a25c51c9..f89eb929 100644 --- a/adapters/aws-s3-style/README.md +++ b/adapters/aws-s3-style/README.md @@ -60,7 +60,7 @@ The real S3 multipart upload protocol, stateful on the object store: | Method | Route | Description | |--------|-------|-------------| | POST | `/{bucket}/{key}?uploads` | CreateMultipartUpload → XML `` | -| PUT | `/{bucket}/{key}?partNumber=N&uploadId=...` | UploadPart → `ETag` header (quoted SHA-256 of the part bytes) | +| PUT | `/{bucket}/{key}?partNumber=N&uploadId=...` | UploadPart → `ETag` header (quoted MD5 hex of the part bytes) | | GET | `/{bucket}/{key}?uploadId=...` | ListParts XML (`max-parts` / `part-number-marker` paging) | | POST | `/{bucket}/{key}?uploadId=...` | CompleteMultipartUpload (XML body listing the parts) | | DELETE | `/{bucket}/{key}?uploadId=...` | AbortMultipartUpload → 204 | @@ -83,10 +83,8 @@ Semantics enforced like the real service: real paging (`max-parts`, default 1000; `part-number-marker` → ``/``). -Documented deviations: part ETags and the assembled object's ETag are -SHA-256 based (`sha256(concat part etags)-N` for the multipart object, like -real S3's `md5(md5s)-N` shape), and the 5 MiB minimum part size is **not** -enforced so small chunks can be exercised in local tests. +Documented deviations: the 5 MiB minimum part size is **not** enforced so +small chunks can be exercised in local tests. ListObjectsV2 also honors the real S3 list params: @@ -190,10 +188,9 @@ unknown access key yields `InvalidAccessKeyId`; a stale `x-amz-date` yields ### Clock-derived response data -- **ETag** is content-derived: the quoted SHA-256 hex digest of the object's - verbatim bytes (real S3 uses the MD5 digest for non-multipart uploads; the - engine's crypto module has no MD5, so the stronger digest is used — a - documented deviation). +- **ETag** is content-derived: the quoted MD5 hex digest of the object's + verbatim bytes, as in real S3. Multipart objects use + `md5(concat part-md5 binaries)-N`. - **Last-Modified** (GET/HEAD headers, RFC 1123) and `` (XML, ISO 8601 with milliseconds) derive from the engine clock at upload time. - Bucket `` derives from the clock as well. @@ -223,7 +220,7 @@ curl "http://localhost:PORT/mybucket?list-type=2" # Paginated listing curl "http://localhost:PORT/mybucket?list-type=2&max-keys=10&continuation-token=" -# Binary round-trip (bytes stored verbatim, ETag = quoted sha256 of the bytes) +# Binary round-trip (bytes stored verbatim, ETag = quoted MD5 of the bytes) curl -X PUT "http://localhost:PORT/mybucket/photo.jpg" \ -H "Authorization: AWS4-HMAC-SHA256 ..." \ -H "Content-Type: image/jpeg" \ @@ -235,11 +232,11 @@ curl -X POST "http://localhost:PORT/mybucket/big.bin?uploads" -H "Authorization: -H "x-amz-date: 20260120T000000Z" # → ...mpu_1 curl -X PUT "http://localhost:PORT/mybucket/big.bin?partNumber=1&uploadId=mpu_1" \ - -H "Authorization: ..." --data-binary @part1.bin # → ETag: "sha256-of-part1" + -H "Authorization: ..." --data-binary @part1.bin # → ETag: "md5-of-part1" curl -X POST "http://localhost:PORT/mybucket/big.bin?uploadId=mpu_1" \ -H "Authorization: ..." \ -d '1"..."' -# → ..."sha256-of-etags-1"... +# → ..."md5-of-part-md5s-1"... ``` ## Error responses diff --git a/conformance/matrix.json b/conformance/matrix.json index af35391f..de608dea 100644 --- a/conformance/matrix.json +++ b/conformance/matrix.json @@ -9380,7 +9380,7 @@ "No ListMultipartUploads (GET /{bucket}?uploads)" ], "deviations": [ - "ETags are SHA-256-based (real S3 uses MD5); multipart ETag is sha256(etags)-N", + "ETags are MD5 hex (multipart MD5(binary-concat)-N)", "Multipart 5 MiB minimum part size not enforced (small parts allowed)", "DELETE of a missing bucket is an idempotent 204 (real S3: 404 NoSuchBucket)", "SigV4 canonical URI/query rebuilt from decoded values — duplicates indistinguishable", diff --git a/conformance/matrix.yaml b/conformance/matrix.yaml index 9245a04a..f8c57f16 100644 --- a/conformance/matrix.yaml +++ b/conformance/matrix.yaml @@ -149,7 +149,7 @@ adapters: - "No AssumeRoleWithSAML" aws-s3-style: deviations: - - "ETags are SHA-256-based (real S3 uses MD5); multipart ETag is sha256(etags)-N" + - "ETags are MD5 hex (multipart MD5(binary-concat)-N)" - "Multipart 5 MiB minimum part size not enforced (small parts allowed)" - "DELETE of a missing bucket is an idempotent 204 (real S3: 404 NoSuchBucket)" - "SigV4 canonical URI/query rebuilt from decoded values — duplicates indistinguishable"