From 91a694c15cdc6c2a7af88270b3415a2cb2bc23ed Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Tue, 9 Jun 2026 17:47:59 -0700 Subject: [PATCH 1/7] WIP merkle tree --- pkg/dotc1z/engine/pebble/adapter.go | 12 + .../engine/pebble/adapter_clone_sync.go | 2 + pkg/dotc1z/engine/pebble/cleanup.go | 2 + pkg/dotc1z/engine/pebble/grants.go | 35 +- pkg/dotc1z/engine/pebble/index_migrations.go | 68 +- pkg/dotc1z/engine/pebble/keys.go | 136 ++++ pkg/dotc1z/engine/pebble/merkle.go | 632 ++++++++++++++++++ pkg/dotc1z/engine/pebble/merkle_test.go | 484 ++++++++++++++ pkg/synccompactor/pebble/bucket_plans.go | 14 + 9 files changed, 1383 insertions(+), 2 deletions(-) create mode 100644 pkg/dotc1z/engine/pebble/merkle.go create mode 100644 pkg/dotc1z/engine/pebble/merkle_test.go diff --git a/pkg/dotc1z/engine/pebble/adapter.go b/pkg/dotc1z/engine/pebble/adapter.go index 23507c8ec..ea822eec7 100644 --- a/pkg/dotc1z/engine/pebble/adapter.go +++ b/pkg/dotc1z/engine/pebble/adapter.go @@ -228,6 +228,18 @@ func (a *Adapter) EndSync(ctx context.Context) error { zap.Error(err), ) } + // Build the per-entitlement grant merkle trees at seal time, after + // all grants are written and while still on the fresh-sync NoSync + // path (the EndFreshSync flush below hardens the nodes). Non-fatal, + // like the stats sidecar: a missing/stale tree only forces a diff + // consumer onto the on-demand ComputeBucketHash fold, and the + // on-Open migration backfills it next time the file opens writable. + if err := a.engine.BuildAllMerkleTrees(ctx, existing.GetSyncId()); err != nil { + ctxzap.Extract(ctx).Warn("pebble: build grant merkle trees failed; grant-diff callers will fall back to on-demand index folds until the next Open backfills them", + zap.String("sync_id", existing.GetSyncId()), + zap.Error(err), + ) + } // Single flush + WAL fsync at sync end. This is the durability // boundary — counterpart to MarkFreshSync at StartNewSync. After // this returns, all writes from the sync are on disk. diff --git a/pkg/dotc1z/engine/pebble/adapter_clone_sync.go b/pkg/dotc1z/engine/pebble/adapter_clone_sync.go index b0791cee0..9698f62d3 100644 --- a/pkg/dotc1z/engine/pebble/adapter_clone_sync.go +++ b/pkg/dotc1z/engine/pebble/adapter_clone_sync.go @@ -108,6 +108,8 @@ func cloneSync( {GrantByPrincipalSyncLowerBound(syncIDBytes), GrantByPrincipalSyncUpperBound(syncIDBytes)}, {GrantByPrincipalResourceTypeSyncLowerBound(syncIDBytes), GrantByPrincipalResourceTypeSyncUpperBound(syncIDBytes)}, {GrantByNeedsExpansionSyncLowerBound(syncIDBytes), GrantByNeedsExpansionSyncUpperBound(syncIDBytes)}, + {GrantByEntPrincHashSyncLowerBound(syncIDBytes), GrantByEntPrincHashSyncUpperBound(syncIDBytes)}, + {MerkleSyncLowerBound(syncIDBytes), MerkleSyncUpperBound(syncIDBytes)}, {encodeAssetPrefix(syncIDBytes), upperBoundOf(encodeAssetPrefix(syncIDBytes))}, // Stats sidecar — single key per sync; copyRange's [lo, hi) // shape requires a half-open range, so we synthesize one diff --git a/pkg/dotc1z/engine/pebble/cleanup.go b/pkg/dotc1z/engine/pebble/cleanup.go index 0547fafd3..3b8d1f1bc 100644 --- a/pkg/dotc1z/engine/pebble/cleanup.go +++ b/pkg/dotc1z/engine/pebble/cleanup.go @@ -35,6 +35,8 @@ func syncScopedRanges(syncIDBytes []byte) [][2][]byte { {GrantByPrincipalSyncLowerBound(syncIDBytes), GrantByPrincipalSyncUpperBound(syncIDBytes)}, {GrantByPrincipalResourceTypeSyncLowerBound(syncIDBytes), GrantByPrincipalResourceTypeSyncUpperBound(syncIDBytes)}, {GrantByNeedsExpansionSyncLowerBound(syncIDBytes), GrantByNeedsExpansionSyncUpperBound(syncIDBytes)}, + {GrantByEntPrincHashSyncLowerBound(syncIDBytes), GrantByEntPrincHashSyncUpperBound(syncIDBytes)}, + {MerkleSyncLowerBound(syncIDBytes), MerkleSyncUpperBound(syncIDBytes)}, {encodeAssetPrefix(syncIDBytes), upperBoundOf(encodeAssetPrefix(syncIDBytes))}, // Stats sidecar — single key per sync; the half-open range // shape contains exactly the one key for this sync. diff --git a/pkg/dotc1z/engine/pebble/grants.go b/pkg/dotc1z/engine/pebble/grants.go index d72ce2988..3cd228fab 100644 --- a/pkg/dotc1z/engine/pebble/grants.go +++ b/pkg/dotc1z/engine/pebble/grants.go @@ -220,6 +220,8 @@ func (e *Engine) UnsafePutUniqueGrantRecords(ctx context.Context, records ...*v3 priKey []byte priVal []byte idxKeys [][]byte + hashKey []byte + hashVal []byte } enc := make([]encoded, len(records)) @@ -269,6 +271,8 @@ func (e *Engine) UnsafePutUniqueGrantRecords(ctx context.Context, records ...*v3 priKey: encodeGrantKey(idBytes, r.GetExternalId()), priVal: val, idxKeys: grantIndexKeys(idBytes, r), + hashKey: grantHashIndexKey(idBytes, r), + hashVal: grantContentHash(r), } } }(lo, hi) @@ -294,6 +298,11 @@ func (e *Engine) UnsafePutUniqueGrantRecords(ctx context.Context, records ...*v3 return err } } + if enc[i].hashKey != nil { + if err := idxBatch.Set(enc[i].hashKey, enc[i].hashVal, nil); err != nil { + return err + } + } } opts := writeOpts(e.opts.durability) @@ -397,13 +406,21 @@ func grantIndexKeys(syncIDBytes []byte, r *v3.GrantRecord) [][]byte { return keys } -// writeGrantIndexes adds index entries for r to batch. +// writeGrantIndexes adds index entries for r to batch. The nil-valued +// indexes go in via grantIndexKeys; the by_entitlement_principal_hash +// index is the one grant index that carries a VALUE (the grant content +// hash), so it is written separately. func (e *Engine) writeGrantIndexes(batch *pebble.Batch, syncIDBytes []byte, r *v3.GrantRecord) error { for _, k := range grantIndexKeys(syncIDBytes, r) { if err := batch.Set(k, nil, nil); err != nil { return err } } + if hk := grantHashIndexKey(syncIDBytes, r); hk != nil { + if err := batch.Set(hk, grantContentHash(r), nil); err != nil { + return err + } + } return nil } @@ -464,6 +481,22 @@ func (e *Engine) deleteGrantIndexes(batch *pebble.Batch, syncIDBytes []byte, r * if err := batch.Delete(encodeGrantByNeedsExpansionIndexKey(syncIDBytes, ext), nil); err != nil { return err } + // by_entitlement_principal_hash: the bucket hash is derived from the + // principal identity, so deleteGrantIndexes reconstructs the same key + // writeGrantIndexes wrote. Skipped when ent/principal are absent, + // matching grantHashIndexKey's nil-guard. NOTE: this invalidates the + // entitlement's merkle tree — callers that delete grants must rebuild + // it (BuildAllMerkleTrees) before relying on a stored root. + if ent != nil && princ != nil { + bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) + hk := encodeGrantByEntPrincHashIndexKey( + syncIDBytes, ent.GetEntitlementId(), bh, + princ.GetResourceTypeId(), princ.GetResourceId(), ext, + ) + if err := batch.Delete(hk, nil); err != nil { + return err + } + } return nil } diff --git a/pkg/dotc1z/engine/pebble/index_migrations.go b/pkg/dotc1z/engine/pebble/index_migrations.go index cd7098b95..51553c9c6 100644 --- a/pkg/dotc1z/engine/pebble/index_migrations.go +++ b/pkg/dotc1z/engine/pebble/index_migrations.go @@ -8,6 +8,7 @@ import ( "github.com/cockroachdb/pebble/v2" + v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" "github.com/conductorone/baton-sdk/pkg/dotc1z/engine/pebble/codec" ) @@ -63,7 +64,72 @@ type indexMigration struct { // the corresponding index for any existing c1z that doesn't have // it yet. var indexMigrations = []indexMigration{ - // none yet, because we have no existing data. + { + // Backfill the by_entitlement_principal_hash index and the + // per-entitlement merkle trees for files written before either + // existed. Idempotent: re-emitting an index entry is a Set over + // the same key/value, and the trees are rebuilt wholesale. + // New files persist this version at their initial (empty) Open, + // so the inline write path maintains both and the backfill never + // re-runs for them. + Name: "grant_by_entitlement_principal_hash", + Version: 1, + Apply: func(ctx context.Context, e *Engine) error { + return e.backfillGrantHashIndexAndMerkle(ctx) + }, + }, +} + +// backfillGrantHashIndexAndMerkle reconstructs the +// by_entitlement_principal_hash index for every grant in every sync, +// then rebuilds the per-entitlement merkle trees. The index must be +// committed before the trees are built because BuildAllMerkleTrees folds +// over the committed index. +func (e *Engine) backfillGrantHashIndexAndMerkle(ctx context.Context) error { + var syncIDs []string + if err := e.IterateAllSyncRuns(ctx, func(r *v3.SyncRunRecord) bool { + syncIDs = append(syncIDs, r.GetSyncId()) + return true + }); err != nil { + return fmt.Errorf("backfill hash index: list syncs: %w", err) + } + + for _, syncID := range syncIDs { + if err := ctx.Err(); err != nil { + return err + } + idBytes, err := codec.EncodeSyncID(syncID) + if err != nil { + return err + } + // Re-emit the hash index entry for each grant in this sync. + batch := e.db.NewBatch() + err = e.IterateGrantsBySync(ctx, syncID, func(r *v3.GrantRecord) bool { + hk := grantHashIndexKey(idBytes, r) + if hk == nil { + return true + } + if setErr := batch.Set(hk, grantContentHash(r), nil); setErr != nil { + err = setErr + return false + } + return true + }) + if err != nil { + batch.Close() + return fmt.Errorf("backfill hash index: sync %q: %w", syncID, err) + } + if err := batch.Commit(pebble.Sync); err != nil { + batch.Close() + return fmt.Errorf("backfill hash index: commit %q: %w", syncID, err) + } + batch.Close() + + if err := e.BuildAllMerkleTrees(ctx, syncID); err != nil { + return fmt.Errorf("backfill merkle trees: sync %q: %w", syncID, err) + } + } + return nil } // applyIndexMigrations runs on engine Open (writable opens only — diff --git a/pkg/dotc1z/engine/pebble/keys.go b/pkg/dotc1z/engine/pebble/keys.go index 4398a576c..96a1f50b4 100644 --- a/pkg/dotc1z/engine/pebble/keys.go +++ b/pkg/dotc1z/engine/pebble/keys.go @@ -61,6 +61,7 @@ const ( typeIndex byte = 0x07 typeCounter byte = 0x08 typeSession byte = 0x09 + typeMerkle byte = 0x0A typeEngineMeta byte = 0xFF ) @@ -76,6 +77,12 @@ const ( idxGrantByNeedsExpansion byte = 0x05 idxGrantByPrincipalResourceType byte = 0x06 idxGrantByEntitlementResource byte = 0x07 + // idxGrantByEntitlementPrincipalHash sorts grants by + // (entitlement_id, hash(principal)). Unlike every other grant + // index its entries carry a VALUE (the grant content hash). It is + // the substrate the per-entitlement merkle tree (typeMerkle) folds + // over; see merkle.go. + idxGrantByEntitlementPrincipalHash byte = 0x08 ) // --- Grant --- @@ -275,6 +282,135 @@ func encodeGrantByEntitlementResourcePrefix(syncIDBytes []byte, entRT, entRID st return codec.AppendTupleSeparator(buf) } +// --- Grant by (entitlement, principal-hash) + Merkle --- + +// encodeGrantByEntPrincHashIndexKey is the by_entitlement_principal_hash +// secondary index on GrantRecord. Unlike every other grant index it +// interposes a RAW, fixed-width principal-bucket hash between the +// entitlement_id and the principal tuple, so the keyspace sorts by +// (entitlement_id, hash(principal)). That hash-major order is what the +// per-entitlement merkle tree folds over, and a hash PREFIX is a clean +// byte-prefix of the key — which is the property the merkle bucket range +// scans rely on. +// +// v3 | typeIndex | idxGrantByEntitlementPrincipalHash | sync_id | 0x00 | +// entitlement_id | 0x00 | | +// principal_rt | 0x00 | principal_id | 0x00 | external_id +// -> value: grant content hash (sha256, 32 bytes) +// +// Because the bucket hash is raw it can contain 0x00, so the generic +// tuple walkers (lastTupleComponent / decodeTwoTupleComponents) must NOT +// be pointed at a prefix that stops before the hash — their walk would +// derail on those bytes. Use decodeEntPrincHashTail, which accounts for +// the hash's fixed width positionally. +// +// Paired with encodeGrantByEntPrincHashEntPrefix (by-value prefix, with +// trailing sep) and encodeGrantByEntPrincHashBucketPrefix. +func encodeGrantByEntPrincHashIndexKey(syncIDBytes []byte, entitlementID string, bucketHash []byte, principalRT, principalID, externalID string) []byte { + buf := make([]byte, 0, 8+len(syncIDBytes)+len(entitlementID)+len(bucketHash)+len(principalRT)+len(principalID)+len(externalID)) + buf = append(buf, versionV3, typeIndex, idxGrantByEntitlementPrincipalHash) + buf = append(buf, syncIDBytes...) + buf = codec.AppendTupleSeparator(buf) + buf = codec.AppendTupleStrings(buf, entitlementID) + buf = codec.AppendTupleSeparator(buf) + buf = append(buf, bucketHash...) + return codec.AppendTupleStrings(buf, principalRT, principalID, externalID) +} + +// encodeGrantByEntPrincHashEntPrefix is the by-value prefix for "all +// grants in this sync under this entitlement", in hash order. The +// trailing separator is load-bearing (see the keys.go convention doc); +// the raw bucket hash follows it. Its output length is also the offset +// the decoder uses to locate the raw hash region. +func encodeGrantByEntPrincHashEntPrefix(syncIDBytes []byte, entitlementID string) []byte { + buf := make([]byte, 0, 8+len(syncIDBytes)+len(entitlementID)) + buf = append(buf, versionV3, typeIndex, idxGrantByEntitlementPrincipalHash) + buf = append(buf, syncIDBytes...) + buf = codec.AppendTupleSeparator(buf) + buf = codec.AppendTupleStrings(buf, entitlementID) + return codec.AppendTupleSeparator(buf) +} + +// encodeGrantByEntPrincHashBucketPrefix narrows the entitlement prefix to +// a single principal-hash bucket identified by a raw hash prefix. An +// empty hashPrefix yields the whole-entitlement prefix (the depth-0 +// bucket). Used as the LowerBound for a bucket range scan. +func encodeGrantByEntPrincHashBucketPrefix(syncIDBytes []byte, entitlementID string, hashPrefix []byte) []byte { + return append(encodeGrantByEntPrincHashEntPrefix(syncIDBytes, entitlementID), hashPrefix...) +} + +// decodeEntPrincHashTail decodes one index key relative to entPrefix +// (which must be encodeGrantByEntPrincHashEntPrefix output — i.e. it +// ends right before the raw bucket hash). Returns the raw bucket hash +// (a sub-slice of key, valid only as long as key is), the principal +// rt/id and the external_id. ok is false if the key is shorter than +// prefix+hash or the tuple tail is malformed. +func decodeEntPrincHashTail(key, entPrefix []byte) ([]byte, string, string, string, bool) { + if len(key) < len(entPrefix)+merkleBucketHashLen { + return nil, "", "", "", false + } + bucketHash := key[len(entPrefix) : len(entPrefix)+merkleBucketHashLen] + tail := key[len(entPrefix)+merkleBucketHashLen:] + rt, next, err := codec.DecodeTupleStringTo(nil, tail, 0) + if err != nil || next >= len(tail) { + return nil, "", "", "", false + } + id, next2, err := codec.DecodeTupleStringTo(nil, tail, next+1) + if err != nil || next2 >= len(tail) { + return nil, "", "", "", false + } + ext, _, err := codec.DecodeTupleStringTo(nil, tail, next2+1) + if err != nil { + return nil, "", "", "", false + } + return bucketHash, string(rt), string(id), string(ext), true +} + +// GrantByEntPrincHashSyncLowerBound / UpperBound bound the entire +// by_entitlement_principal_hash index for a sync. Exported for the +// cleanup/clone/compaction keyspace plans. +func GrantByEntPrincHashSyncLowerBound(syncIDBytes []byte) []byte { + buf := make([]byte, 0, 3+len(syncIDBytes)) + buf = append(buf, versionV3, typeIndex, idxGrantByEntitlementPrincipalHash) + return append(buf, syncIDBytes...) +} + +func GrantByEntPrincHashSyncUpperBound(syncIDBytes []byte) []byte { + return upperBoundOf(GrantByEntPrincHashSyncLowerBound(syncIDBytes)) +} + +// Merkle node keys. +// +// v3 | typeMerkle | sync_id | 0x00 | entitlement_id | 0x00 | level(1 byte) | bucket_prefix(level raw bytes) +// +// level 0 is the root (bucket_prefix empty); level == tree depth holds +// the leaves, one per non-empty principal-hash bucket. bucket_prefix is +// the first `level` bytes of the principal bucket hash, raw, so it aligns +// byte-for-byte with the index key's bucket-hash region. See merkle.go +// for the node value framing. +func encodeMerkleNodeKey(syncIDBytes []byte, entitlementID string, level byte, bucketPrefix []byte) []byte { + buf := make([]byte, 0, 6+len(syncIDBytes)+len(entitlementID)+len(bucketPrefix)) + buf = append(buf, versionV3, typeMerkle) + buf = append(buf, syncIDBytes...) + buf = codec.AppendTupleSeparator(buf) + buf = codec.AppendTupleStrings(buf, entitlementID) + buf = codec.AppendTupleSeparator(buf) + buf = append(buf, level) + return append(buf, bucketPrefix...) +} + +// MerkleSyncLowerBound / UpperBound bound the entire merkle keyspace for +// a sync. Exported for the cleanup/clone/compaction keyspace plans. +func MerkleSyncLowerBound(syncIDBytes []byte) []byte { + buf := make([]byte, 0, 2+len(syncIDBytes)) + buf = append(buf, versionV3, typeMerkle) + return append(buf, syncIDBytes...) +} + +func MerkleSyncUpperBound(syncIDBytes []byte) []byte { + return upperBoundOf(MerkleSyncLowerBound(syncIDBytes)) +} + // --- ResourceType --- // encodeResourceTypeKey returns the primary key for a resource_type: diff --git a/pkg/dotc1z/engine/pebble/merkle.go b/pkg/dotc1z/engine/pebble/merkle.go new file mode 100644 index 000000000..187913008 --- /dev/null +++ b/pkg/dotc1z/engine/pebble/merkle.go @@ -0,0 +1,632 @@ +package pebble + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/binary" + "errors" + "fmt" + "hash" + "sort" + + "github.com/cockroachdb/pebble/v2" + + v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" +) + +// Per-entitlement grant merkle tree. +// +// Goal: answer "does this entitlement have exactly the same grants as +// some other sync/file?" with a single key read, and when the answer is +// no, identify which principal-hash buckets differ so a caller can load +// only those grants instead of re-reading the whole entitlement. +// +// The tree folds over the by_entitlement_principal_hash index +// (idxGrantByEntitlementPrincipalHash), whose entries are sorted by +// (entitlement_id, hash(principal)) and whose VALUE is the per-grant +// content hash. Two properties make the index the right substrate: +// +// - hash-major order is identical across two files that hold the same +// grants, so a streamed fold produces the same digest; and +// - a hash prefix is a clean byte-prefix of the key, so "all grants in +// bucket P" is a contiguous range scan. +// +// Variable height. The tree depth is chosen from the grant count: +// depth 0 is a single root node (used for empty and small entitlements — +// this is why an entitlement with no grants costs exactly one stored +// node); each additional level consumes one more byte of the bucket +// hash, multiplying the bucket count by 256. Node hashes are +// CONTENT-defined — a node's hash is the fold of every grant content +// hash beneath it, independent of how the subtree is split — so a node +// at one depth is directly comparable to the equivalent prefix-range in +// a tree of a different depth. +// +// On-disk ABI. Both the principal bucket hash and the grant content +// hash are part of the stored format: changing merkleBucketHashLen, the +// content-hash field set (grantContentHash), or the node value framing +// requires an index-migration version bump (see index_migrations.go). + +const ( + // merkleBucketHashLen is the width, in bytes, of the principal + // bucket hash embedded in the index key. 8 bytes (64 bits) bounds + // the maximum tree depth at 8 byte-levels; collisions in the bucket + // address are harmless (they only co-locate principals — the full + // principal identity and external_id still distinguish index rows). + merkleBucketHashLen = 8 + + // merkleTargetBucketSize is the grant count a single leaf bucket + // aims to hold. Depth is grown until buckets are roughly this size. + merkleTargetBucketSize = 256 + + // merkleMaxDepth caps tree depth at the bucket-hash width: one byte + // of hash is consumed per level. + merkleMaxDepth = merkleBucketHashLen +) + +// hashLen is the width of a sha256 digest, used for both the grant +// content hash (index value) and node hashes. +const hashLen = sha256.Size + +// writeLenPrefixed writes an 8-byte big-endian length followed by b. +// Length-prefixing every field makes the canonical encoding injective: +// no concatenation of fields can be confused with a different split. +func writeLenPrefixed(h hash.Hash, b []byte) { + var n [8]byte + binary.BigEndian.PutUint64(n[:], uint64(len(b))) + _, _ = h.Write(n[:]) + _, _ = h.Write(b) +} + +// principalBucketHash is the bucket address for a principal: the first +// merkleBucketHashLen bytes of sha256(rt, id). Identity only — never the +// principal's full object — so the address is stable across syncs even +// when the principal's attributes change. Returns a fresh slice. +func principalBucketHash(rt, id string) []byte { + h := sha256.New() + writeLenPrefixed(h, []byte(rt)) + writeLenPrefixed(h, []byte(id)) + sum := h.Sum(nil) + out := make([]byte, merkleBucketHashLen) + copy(out, sum[:merkleBucketHashLen]) + return out +} + +// grantContentHash is the canonical content hash of a grant — the value +// stored in the hash index and the unit the merkle tree folds. +// +// ABI: the field set below defines what "the same grant" means for the +// diff. It deliberately covers the membership EDGE (entitlement id, +// principal identity, external_id) plus the grant's source-entitlement +// set (expansion provenance), and deliberately EXCLUDES sync-relative +// and transient processing state — sync_id, discovered_at, +// needs_expansion, expansion, and annotations — none of which change +// "which principal holds which entitlement". This is a hand-rolled +// framing, NOT proto marshal: deterministic-proto output is not +// canonical across protobuf library versions, which would make two +// files written by different SDK builds hash identical grants +// differently. Changing this set requires an index-migration bump. +func grantContentHash(r *v3.GrantRecord) []byte { + h := sha256.New() + ent := r.GetEntitlement() + princ := r.GetPrincipal() + writeLenPrefixed(h, []byte(ent.GetEntitlementId())) + writeLenPrefixed(h, []byte(princ.GetResourceTypeId())) + writeLenPrefixed(h, []byte(princ.GetResourceId())) + writeLenPrefixed(h, []byte(r.GetExternalId())) + + // Source-entitlement ids, sorted for order-independence. The map + // values (GrantSourceRecord) are not folded in v1 — only the set of + // source ids, which is the membership-composition signal. + sources := r.GetSources() + ids := make([]string, 0, len(sources)) + for k := range sources { + ids = append(ids, k) + } + sort.Strings(ids) + var nbuf [8]byte + binary.BigEndian.PutUint64(nbuf[:], uint64(len(ids))) + _, _ = h.Write(nbuf[:]) + for _, id := range ids { + writeLenPrefixed(h, []byte(id)) + } + return h.Sum(nil) +} + +// grantHashIndexKey returns the by_entitlement_principal_hash index key +// for r, or nil when the grant lacks the entitlement/principal needed to +// place it (mirrors the by_entitlement index's nil-guard). +func grantHashIndexKey(syncIDBytes []byte, r *v3.GrantRecord) []byte { + ent := r.GetEntitlement() + princ := r.GetPrincipal() + if ent == nil || princ == nil { + return nil + } + bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) + return encodeGrantByEntPrincHashIndexKey( + syncIDBytes, ent.GetEntitlementId(), bh, + princ.GetResourceTypeId(), princ.GetResourceId(), r.GetExternalId(), + ) +} + +// chooseMerkleDepth picks the tree depth for an entitlement with the +// given grant count: 0 when the count fits one target-sized bucket, else +// the smallest depth whose 256^depth buckets bring the average bucket +// under merkleTargetBucketSize, capped at merkleMaxDepth. +func chooseMerkleDepth(count int64) int { + if count <= merkleTargetBucketSize { + return 0 + } + needed := (count + merkleTargetBucketSize - 1) / merkleTargetBucketSize + depth := 0 + capacity := int64(1) + for capacity < needed && depth < merkleMaxDepth { + capacity *= 256 + depth++ + } + return depth +} + +// Node value framing. +// +// root (level 0): depth(1) | count(8 BE) | hash(hashLen) +// leaf (level d): count(8 BE) | hash(hashLen) +// +// The root carries the chosen depth so a reader knows the leaf level +// without scanning. Leaves omit it (their level is in the key). + +func packMerkleRoot(depth int, count int64, h []byte) []byte { + buf := make([]byte, 0, 1+8+len(h)) + buf = append(buf, byte(depth)) + var n [8]byte + binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count + buf = append(buf, n[:]...) + return append(buf, h...) +} + +// unpackMerkleRoot returns (depth, count, hash, ok). ok is false when +// the value is not a well-formed root blob. +func unpackMerkleRoot(val []byte) (int, int64, []byte, bool) { + if len(val) != 1+8+hashLen { + return 0, 0, nil, false + } + depth := int(val[0]) + count := int64(binary.BigEndian.Uint64(val[1:9])) //nolint:gosec // count is a non-negative row count + return depth, count, val[9:], true +} + +func packMerkleLeaf(count int64, h []byte) []byte { + buf := make([]byte, 0, 8+len(h)) + var n [8]byte + binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count + buf = append(buf, n[:]...) + return append(buf, h...) +} + +// BuildAllMerkleTrees rebuilds the per-entitlement merkle tree for every +// entitlement in syncID. Called at seal time (Adapter.EndSync) after all +// grants are written, and by the on-Open migration backfill. Every +// entitlement gets a tree — including those with zero grants, which +// store a single root node — so a reader can always distinguish "empty" +// from "never built". +func (e *Engine) BuildAllMerkleTrees(ctx context.Context, syncID string) error { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return err + } + // Collect entitlement ids first: BuildEntitlementMerkle writes into + // the typeMerkle keyspace while we'd otherwise be mid-iteration over + // typeEntitlement. Different keyspaces, but snapshotting the ids + // keeps the iterator and the writes cleanly separated. + var ents []string + if err := e.IterateEntitlementsBySync(ctx, syncID, func(r *v3.EntitlementRecord) bool { + ents = append(ents, r.GetExternalId()) + return true + }); err != nil { + return fmt.Errorf("BuildAllMerkleTrees: list entitlements: %w", err) + } + for _, ent := range ents { + if err := ctx.Err(); err != nil { + return err + } + if err := e.buildEntitlementMerkle(ctx, idBytes, ent); err != nil { + return fmt.Errorf("BuildAllMerkleTrees: entitlement %q: %w", ent, err) + } + } + return nil +} + +// buildEntitlementMerkle counts an entitlement's grants (pass 1), picks +// the depth from that count, and delegates the fold to +// buildEntitlementMerkleAtDepth (pass 2). +func (e *Engine) buildEntitlementMerkle(ctx context.Context, idBytes []byte, entitlementID string) error { + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) + count, err := e.countHashIndexRange(entPrefix, upperBoundOf(entPrefix)) + if err != nil { + return err + } + return e.buildEntitlementMerkleAtDepth(ctx, idBytes, entitlementID, chooseMerkleDepth(count)) +} + +// buildEntitlementMerkleAtDepth folds the hash index for one entitlement +// into a root (and, when depth > 0, one leaf per non-empty bucket) and +// writes the nodes in a single streaming pass — O(1) memory regardless +// of entitlement size. The depth is taken as a parameter rather than +// derived so the depth-selection seam can be exercised directly: tests +// force a depth that the natural count→depth mapping would only produce +// at a very large grant count, which is how the cross-depth comparison +// path gets covered without seeding tens of thousands of grants. +func (e *Engine) buildEntitlementMerkleAtDepth(ctx context.Context, idBytes []byte, entitlementID string, depth int) error { + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) + upper := upperBoundOf(entPrefix) + + return e.withWrite(func() error { + batch := e.db.NewBatch() + defer batch.Close() + + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upper}) + if err != nil { + return err + } + defer iter.Close() + + rootH := sha256.New() + var ( + leafH hash.Hash + leafCount int64 + curPrefix []byte + haveLeaf bool + total int64 + ) + flushLeaf := func() error { + if depth == 0 || !haveLeaf { + return nil + } + key := encodeMerkleNodeKey(idBytes, entitlementID, byte(depth), curPrefix) + return batch.Set(key, packMerkleLeaf(leafCount, leafH.Sum(nil)), nil) + } + + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return err + } + key := iter.Key() + if len(key) < len(entPrefix)+merkleBucketHashLen { + continue // malformed; skip defensively + } + bucketHash := key[len(entPrefix) : len(entPrefix)+merkleBucketHashLen] + val := iter.Value() // 32-byte content hash + + if depth > 0 { + prefix := bucketHash[:depth] + if !haveLeaf || !bytes.Equal(prefix, curPrefix) { + if err := flushLeaf(); err != nil { + return err + } + curPrefix = append(curPrefix[:0], prefix...) + leafH = sha256.New() + leafCount = 0 + haveLeaf = true + } + _, _ = leafH.Write(val) + leafCount++ + } + _, _ = rootH.Write(val) + total++ + } + if err := iter.Error(); err != nil { + return err + } + if err := flushLeaf(); err != nil { + return err + } + + rootKey := encodeMerkleNodeKey(idBytes, entitlementID, 0, nil) + if err := batch.Set(rootKey, packMerkleRoot(depth, total, rootH.Sum(nil)), nil); err != nil { + return err + } + + opts := writeOpts(e.opts.durability) + if e.IsFreshSync() { + opts = pebble.NoSync + } + return batch.Commit(opts) + }) +} + +// countHashIndexRange counts index entries in [lower, upper) without +// materializing values. +func (e *Engine) countHashIndexRange(lower, upper []byte) (int64, error) { + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return 0, err + } + defer iter.Close() + var n int64 + for iter.First(); iter.Valid(); iter.Next() { + n++ + } + return n, iter.Error() +} + +// MerkleRoot is the result of GetEntitlementMerkleRoot. +type MerkleRoot struct { + Hash []byte + Depth int + Count int64 +} + +// GetEntitlementMerkleRoot returns the stored root for an entitlement. +// ok is false when no tree has been built for it (the caller can fall +// back to ComputeEntitlementRoot, which derives the same digest from the +// index on demand). +func (e *Engine) GetEntitlementMerkleRoot(ctx context.Context, syncID, entitlementID string) (MerkleRoot, bool, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return MerkleRoot{}, false, err + } + val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, 0, nil)) + if err != nil { + if errors.Is(err, pebble.ErrNotFound) { + return MerkleRoot{}, false, nil + } + return MerkleRoot{}, false, err + } + defer closer.Close() + depth, count, h, valid := unpackMerkleRoot(val) + if !valid { + return MerkleRoot{}, false, fmt.Errorf("GetEntitlementMerkleRoot: malformed root for %q", entitlementID) + } + out := make([]byte, len(h)) + copy(out, h) + return MerkleRoot{Hash: out, Depth: depth, Count: count}, true, nil +} + +// ComputeBucketHash folds the hash index over a single principal-hash +// bucket (identified by a raw hash prefix; empty prefix = whole +// entitlement = the root) and returns the content-defined digest plus +// the grant count. This is the authoritative definition of a node hash; +// stored nodes are a cache of it. Depth-independent: the digest depends +// only on the grants in the prefix range, not on any tree's shape. +func (e *Engine) ComputeBucketHash(ctx context.Context, syncID, entitlementID string, hashPrefix []byte) ([]byte, int64, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return nil, 0, err + } + lower := encodeGrantByEntPrincHashBucketPrefix(idBytes, entitlementID, hashPrefix) + upper := upperBoundOf(lower) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return nil, 0, err + } + defer iter.Close() + h := sha256.New() + var count int64 + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return nil, 0, err + } + _, _ = h.Write(iter.Value()) + count++ + } + if err := iter.Error(); err != nil { + return nil, 0, err + } + return h.Sum(nil), count, nil +} + +// IterateGrantsByEntitlementBucket yields the grants in one principal-hash +// bucket of an entitlement (empty hashPrefix = the whole entitlement). +// This is the dirty-bucket loader: after a merkle comparison flags a +// bucket prefix, the caller materializes only those grants. Like the +// other index iterators it does a point Get per entry to fetch the +// primary; orphan index entries are skipped. +func (e *Engine) IterateGrantsByEntitlementBucket(ctx context.Context, syncID, entitlementID string, hashPrefix []byte, yield func(*v3.GrantRecord) bool) error { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return err + } + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) + lower := append(append([]byte(nil), entPrefix...), hashPrefix...) + upper := upperBoundOf(lower) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return err + } + defer iter.Close() + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return err + } + _, _, _, externalID, ok := decodeEntPrincHashTail(iter.Key(), entPrefix) + if !ok { + continue + } + val, closer, getErr := e.db.Get(encodeGrantKey(idBytes, externalID)) + if getErr != nil { + if errors.Is(getErr, pebble.ErrNotFound) { + continue + } + return getErr + } + r := &v3.GrantRecord{} + uErr := unmarshalRecord(val, r) + closer.Close() + if uErr != nil { + return fmt.Errorf("IterateGrantsByEntitlementBucket: unmarshal: %w", uErr) + } + if !yield(r) { + return nil + } + } + return iter.Error() +} + +// DirtyEntitlementBuckets compares this engine's entitlement against +// other's and returns the raw hash-bucket prefixes whose grants differ. +// A single empty prefix means "the whole entitlement differs" (used when +// the comparison granularity is the root, e.g. tiny entitlements). A nil +// (empty) result means the two are identical. +// +// The fast path is a root-hash equality check. On mismatch it descends +// to the shallower of the two trees' depths and compares each bucket at +// that granularity, preferring stored leaf hashes and falling back to an +// on-demand index fold when a side's tree is shallower or absent. +func (e *Engine) DirtyEntitlementBuckets(ctx context.Context, syncID string, other *Engine, otherSyncID, entitlementID string) ([][]byte, error) { + rootA, okA, err := e.GetEntitlementMerkleRoot(ctx, syncID, entitlementID) + if err != nil { + return nil, err + } + rootB, okB, err := other.GetEntitlementMerkleRoot(ctx, otherSyncID, entitlementID) + if err != nil { + return nil, err + } + + // Fast equality via stored roots when both exist. + if okA && okB && bytes.Equal(rootA.Hash, rootB.Hash) { + return nil, nil + } + + // Comparison granularity: the shallower available depth. A missing + // tree is treated as depth 0 (compare at the root → whole-entitlement + // dirty if the computed roots differ). + compareDepth := 0 + if okA && okB { + compareDepth = min(rootA.Depth, rootB.Depth) + } + + if compareDepth == 0 { + // Confirm via the authoritative fold (covers the missing-tree + // case and guards against a stale stored root). + ha, _, err := e.ComputeBucketHash(ctx, syncID, entitlementID, nil) + if err != nil { + return nil, err + } + hb, _, err := other.ComputeBucketHash(ctx, otherSyncID, entitlementID, nil) + if err != nil { + return nil, err + } + if bytes.Equal(ha, hb) { + return nil, nil + } + return [][]byte{{}}, nil + } + + // Union of non-empty bucket prefixes at compareDepth from both sides. + prefixes, err := e.distinctBucketPrefixes(ctx, syncID, entitlementID, compareDepth) + if err != nil { + return nil, err + } + otherPrefixes, err := other.distinctBucketPrefixes(ctx, otherSyncID, entitlementID, compareDepth) + if err != nil { + return nil, err + } + union := mergeSortedPrefixes(prefixes, otherPrefixes) + + var dirty [][]byte + for _, p := range union { + ha, _, err := e.bucketHashPreferStored(ctx, syncID, entitlementID, p, rootA, okA) + if err != nil { + return nil, err + } + hb, _, err := other.bucketHashPreferStored(ctx, otherSyncID, entitlementID, p, rootB, okB) + if err != nil { + return nil, err + } + if !bytes.Equal(ha, hb) { + dirty = append(dirty, p) + } + } + return dirty, nil +} + +// bucketHashPreferStored returns a bucket's hash, using the stored leaf +// node when the tree's depth matches the prefix length (the cheap path), +// otherwise folding the index. root/ok describe the entitlement's stored +// tree for this engine. +func (e *Engine) bucketHashPreferStored(ctx context.Context, syncID, entitlementID string, prefix []byte, root MerkleRoot, ok bool) ([]byte, int64, error) { + if ok && root.Depth == len(prefix) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return nil, 0, err + } + val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, byte(len(prefix)), prefix)) + if err == nil { + defer closer.Close() + if len(val) == 8+hashLen { + count := int64(binary.BigEndian.Uint64(val[:8])) //nolint:gosec // non-negative count + h := make([]byte, hashLen) + copy(h, val[8:]) + return h, count, nil + } + } else if !errors.Is(err, pebble.ErrNotFound) { + return nil, 0, err + } + // fall through to compute on miss/malformed + } + return e.ComputeBucketHash(ctx, syncID, entitlementID, prefix) +} + +// distinctBucketPrefixes returns the sorted, distinct depth-byte hash +// prefixes present in an entitlement's hash index. It seeks past each +// bucket once found, so the cost is O(distinct buckets) seeks rather than +// O(grants). +func (e *Engine) distinctBucketPrefixes(ctx context.Context, syncID, entitlementID string, depth int) ([][]byte, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return nil, err + } + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) + upper := upperBoundOf(entPrefix) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upper}) + if err != nil { + return nil, err + } + defer iter.Close() + + var out [][]byte + for iter.First(); iter.Valid(); { + if err := ctx.Err(); err != nil { + return nil, err + } + key := iter.Key() + if len(key) < len(entPrefix)+depth { + iter.Next() + continue + } + prefix := make([]byte, depth) + copy(prefix, key[len(entPrefix):len(entPrefix)+depth]) + out = append(out, prefix) + // Seek past this whole bucket to the next distinct prefix. + seekTo := upperBoundOf(append(append([]byte(nil), entPrefix...), prefix...)) + if seekTo == nil { + break + } + iter.SeekGE(seekTo) + } + return out, iter.Error() +} + +// mergeSortedPrefixes returns the sorted union of two sorted, de-duped +// prefix lists. +func mergeSortedPrefixes(a, b [][]byte) [][]byte { + out := make([][]byte, 0, len(a)+len(b)) + i, j := 0, 0 + for i < len(a) && j < len(b) { + switch bytes.Compare(a[i], b[j]) { + case 0: + out = append(out, a[i]) + i++ + j++ + case -1: + out = append(out, a[i]) + i++ + default: + out = append(out, b[j]) + j++ + } + } + out = append(out, a[i:]...) + out = append(out, b[j:]...) + return out +} diff --git a/pkg/dotc1z/engine/pebble/merkle_test.go b/pkg/dotc1z/engine/pebble/merkle_test.go new file mode 100644 index 000000000..869e6e9af --- /dev/null +++ b/pkg/dotc1z/engine/pebble/merkle_test.go @@ -0,0 +1,484 @@ +package pebble + +import ( + "bytes" + "context" + "fmt" + "testing" + + "github.com/cockroachdb/pebble/v2" + "github.com/segmentio/ksuid" + + v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" +) + +// putEnt writes an entitlement record whose external_id is entID — the +// same string grants reference via EntitlementRef.EntitlementId, which +// is what BuildAllMerkleTrees keys each tree on. +func putEnt(t testing.TB, e *Engine, ctx context.Context, syncID, entID string) { + t.Helper() + rec := v3.EntitlementRecord_builder{ + SyncId: syncID, + ExternalId: entID, + Resource: v3.ResourceRef_builder{ + ResourceTypeId: "app", + ResourceId: "github", + }.Build(), + }.Build() + if err := e.PutEntitlementRecord(ctx, rec); err != nil { + t.Fatalf("PutEntitlementRecord: %v", err) + } +} + +// makeGrantWithSources is makeGrant plus an optional source-entitlement +// set, which grantContentHash folds in — so two grants with the same +// (entitlement, principal, external_id) but different sources produce +// different content hashes while keeping the SAME index key. +func makeGrantWithSources(syncID, externalID, entID, principalID string, sources ...string) *v3.GrantRecord { + g := makeGrant(syncID, externalID, entID, principalID) + if len(sources) > 0 { + m := make(map[string]*v3.GrantSourceRecord, len(sources)) + for _, s := range sources { + m[s] = v3.GrantSourceRecord_builder{}.Build() + } + g.SetSources(m) + } + return g +} + +// merkleNodeCount counts stored merkle nodes for a sync (across all +// entitlements). +func merkleNodeCount(t testing.TB, e *Engine, syncID string) int { + t.Helper() + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatalf("resolveSyncBytes: %v", err) + } + iter, err := e.db.NewIter(&pebble.IterOptions{ + LowerBound: MerkleSyncLowerBound(idBytes), + UpperBound: MerkleSyncUpperBound(idBytes), + }) + if err != nil { + t.Fatalf("NewIter: %v", err) + } + defer iter.Close() + n := 0 + for iter.First(); iter.Valid(); iter.Next() { + n++ + } + if err := iter.Error(); err != nil { + t.Fatalf("iter: %v", err) + } + return n +} + +// seedEntitlement writes the entitlement record + grants and builds the +// tree, returning the syncID. +func seedEntitlement(t testing.TB, e *Engine, entID string, grants []*v3.GrantRecord) string { + t.Helper() + ctx := context.Background() + syncID := ksuid.New().String() + if err := e.SetCurrentSync(syncID); err != nil { + t.Fatalf("SetCurrentSync: %v", err) + } + putEnt(t, e, ctx, syncID, entID) + for _, g := range grants { + g.SetSyncId(syncID) + if err := e.PutGrantRecord(ctx, g); err != nil { + t.Fatalf("PutGrantRecord: %v", err) + } + } + if err := e.BuildAllMerkleTrees(ctx, syncID); err != nil { + t.Fatalf("BuildAllMerkleTrees: %v", err) + } + return syncID +} + +// seedEntitlementAtDepth is seedEntitlement but forces a specific tree +// depth instead of deriving it from the grant count, so a test can build +// two trees of different heights over a small grant set. +func seedEntitlementAtDepth(t testing.TB, e *Engine, entID string, grants []*v3.GrantRecord, depth int) string { + t.Helper() + ctx := context.Background() + syncID := ksuid.New().String() + if err := e.SetCurrentSync(syncID); err != nil { + t.Fatalf("SetCurrentSync: %v", err) + } + putEnt(t, e, ctx, syncID, entID) + for _, g := range grants { + g.SetSyncId(syncID) + if err := e.PutGrantRecord(ctx, g); err != nil { + t.Fatalf("PutGrantRecord: %v", err) + } + } + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatalf("resolveSyncBytes: %v", err) + } + if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, entID, depth); err != nil { + t.Fatalf("buildEntitlementMerkleAtDepth: %v", err) + } + return syncID +} + +// TestMerkleDifferentDepthsComparison builds two trees of different +// heights (depth 1 vs depth 2) over the SAME entitlement and exercises +// DirtyEntitlementBuckets across them. It validates two things the +// equal-depth tests cannot: +// +// - depth-independence: identical grant content yields the same root +// hash regardless of tree depth, and compares as zero dirty buckets; +// - the mismatched-depth descent: after one principal's grant changes, +// the comparison (at compareDepth = min(1,2) = 1) localizes the +// change to that principal's bucket — the deeper (depth-2) side +// folds its index on demand for the depth-1 prefixes — and leaves a +// known principal in a different bucket clean. +func TestMerkleDifferentDepthsComparison(t *testing.T) { + ctx := context.Background() + + const nPrincipals = 40 + principals := make([]string, nPrincipals) + for i := range principals { + principals[i] = fmt.Sprintf("user-%03d", i) + } + mkGrants := func() []*v3.GrantRecord { + gs := make([]*v3.GrantRecord, 0, nPrincipals) + for i, p := range principals { + gs = append(gs, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", p)) + } + return gs + } + + // depth1Prefix returns the depth-1 bucket prefix (1 raw hash byte) + // for a principal, matching how grants are keyed (principal type + // "user" per makeGrant). + depth1Prefix := func(principalID string) byte { + return principalBucketHash("user", principalID)[0] + } + + ea, _ := newTestEngine(t) + eb, _ := newTestEngine(t) + syncA := seedEntitlementAtDepth(t, ea, "ent-A", mkGrants(), 1) + syncB := seedEntitlementAtDepth(t, eb, "ent-A", mkGrants(), 2) + + ra, okA, err := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") + if err != nil || !okA { + t.Fatalf("root A: ok=%v err=%v", okA, err) + } + rb, okB, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + if err != nil || !okB { + t.Fatalf("root B: ok=%v err=%v", okB, err) + } + if ra.Depth != 1 || rb.Depth != 2 { + t.Fatalf("depths = %d, %d; want 1, 2", ra.Depth, rb.Depth) + } + // Depth-independence: identical content -> identical root despite + // different tree heights. + if !bytes.Equal(ra.Hash, rb.Hash) { + t.Fatalf("different-depth trees over identical content disagree on root:\n A(d1)=%x\n B(d2)=%x", ra.Hash, rb.Hash) + } + dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatalf("DirtyEntitlementBuckets (identical): %v", err) + } + if len(dirty) != 0 { + t.Fatalf("identical content across depths: dirty=%d, want 0", len(dirty)) + } + + // Pick the principal to change and a "clean" principal known to sit + // in a different depth-1 bucket. + changed := principals[0] + changedPrefix := depth1Prefix(changed) + cleanP := "" + for _, p := range principals[1:] { + if depth1Prefix(p) != changedPrefix { + cleanP = p + break + } + } + if cleanP == "" { + t.Skip("no principal landed in a different depth-1 bucket from the changed one; can't assert localization") + } + + // Mutate the changed principal's grant in B (same external_id -> + // same index key, new content hash via an added source) and rebuild + // B's tree at depth 2. + g := makeGrantWithSources(syncB, "g-000", "ent-A", changed, "src-ent") + if err := eb.PutGrantRecord(ctx, g); err != nil { + t.Fatalf("PutGrantRecord (mutate): %v", err) + } + idBytesB, err := eb.resolveSyncBytes(syncB) + if err != nil { + t.Fatal(err) + } + if err := eb.buildEntitlementMerkleAtDepth(ctx, idBytesB, "ent-A", 2); err != nil { + t.Fatalf("rebuild B: %v", err) + } + + rb2, _, _ := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + if bytes.Equal(ra.Hash, rb2.Hash) { + t.Fatal("mutation did not change B's root") + } + + dirty, err = ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatalf("DirtyEntitlementBuckets (changed): %v", err) + } + if len(dirty) == 0 { + t.Fatal("changed principal across depths produced no dirty buckets") + } + // Localization: every dirty entry is a depth-1 prefix (not the + // whole-entitlement empty prefix). + for _, p := range dirty { + if len(p) != 1 { + t.Fatalf("dirty prefix length = %d, want 1 (compareDepth); got whole-entitlement or wrong-depth bucket", len(p)) + } + } + // Loading the dirty buckets in B surfaces the changed principal and + // excludes the known-clean principal. + loaded := map[string]bool{} + for _, prefix := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + loaded[g.GetPrincipal().GetResourceId()] = true + return true + }); err != nil { + t.Fatalf("IterateGrantsByEntitlementBucket: %v", err) + } + } + if !loaded[changed] { + t.Fatalf("dirty buckets did not include the changed principal %q; loaded=%v", changed, loaded) + } + if loaded[cleanP] { + t.Fatalf("dirty buckets wrongly included clean principal %q (different bucket); change was not localized", cleanP) + } +} + +func TestMerkleEmptyEntitlementSingleRoot(t *testing.T) { + ctx := context.Background() + e, _ := newTestEngine(t) + syncID := seedEntitlement(t, e, "ent-empty", nil) + + if got := merkleNodeCount(t, e, syncID); got != 1 { + t.Fatalf("empty entitlement: merkle node count = %d, want 1 (root only)", got) + } + root, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-empty") + if err != nil || !ok { + t.Fatalf("GetEntitlementMerkleRoot: ok=%v err=%v", ok, err) + } + if root.Depth != 0 { + t.Fatalf("empty entitlement depth = %d, want 0", root.Depth) + } + if root.Count != 0 { + t.Fatalf("empty entitlement count = %d, want 0", root.Count) + } +} + +func TestMerkleIdenticalGrantsSameRoot(t *testing.T) { + ctx := context.Background() + mk := func() []*v3.GrantRecord { + return []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrant("", "g2", "ent-A", "bob"), + makeGrant("", "g3", "ent-A", "carol"), + } + } + ea, _ := newTestEngine(t) + eb, _ := newTestEngine(t) + syncA := seedEntitlement(t, ea, "ent-A", mk()) + syncB := seedEntitlement(t, eb, "ent-A", mk()) + + ra, okA, err := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") + if err != nil || !okA { + t.Fatalf("root A: ok=%v err=%v", okA, err) + } + rb, okB, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + if err != nil || !okB { + t.Fatalf("root B: ok=%v err=%v", okB, err) + } + if !bytes.Equal(ra.Hash, rb.Hash) { + t.Fatalf("identical grants produced different roots:\n A=%x\n B=%x", ra.Hash, rb.Hash) + } + dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatalf("DirtyEntitlementBuckets: %v", err) + } + if len(dirty) != 0 { + t.Fatalf("identical grants: dirty buckets = %d, want 0", len(dirty)) + } +} + +func TestMerkleContentChangeDirtyBucket(t *testing.T) { + ctx := context.Background() + // Base set: same in both engines except bob's grant gains a source + // in B. external_id is unchanged, so the index KEY is identical and + // only the content hash (and thus bob's bucket) differs. + baseA := []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrant("", "g2", "ent-A", "bob"), + makeGrant("", "g3", "ent-A", "carol"), + } + baseB := []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrantWithSources("", "g2", "ent-A", "bob", "src-ent"), + makeGrant("", "g3", "ent-A", "carol"), + } + ea, _ := newTestEngine(t) + eb, _ := newTestEngine(t) + syncA := seedEntitlement(t, ea, "ent-A", baseA) + syncB := seedEntitlement(t, eb, "ent-A", baseB) + + ra, _, _ := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") + rb, _, _ := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + if bytes.Equal(ra.Hash, rb.Hash) { + t.Fatal("content change did not change the root hash") + } + + dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatalf("DirtyEntitlementBuckets: %v", err) + } + if len(dirty) == 0 { + t.Fatal("content change produced no dirty buckets") + } + + // Loading the dirty buckets in B must surface bob (the changed + // principal) and must NOT require touching alice/carol's buckets. + found := map[string]bool{} + for _, prefix := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + found[g.GetPrincipal().GetResourceId()] = true + return true + }); err != nil { + t.Fatalf("IterateGrantsByEntitlementBucket: %v", err) + } + } + if !found["bob"] { + t.Fatalf("dirty buckets did not include the changed principal bob; found=%v", found) + } +} + +func TestMerkleAddedGrantDirtyBucket(t *testing.T) { + ctx := context.Background() + baseA := []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrant("", "g2", "ent-A", "bob"), + } + baseB := []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrant("", "g2", "ent-A", "bob"), + makeGrant("", "g3", "ent-A", "dave"), // added in B + } + ea, _ := newTestEngine(t) + eb, _ := newTestEngine(t) + syncA := seedEntitlement(t, ea, "ent-A", baseA) + syncB := seedEntitlement(t, eb, "ent-A", baseB) + + dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatalf("DirtyEntitlementBuckets: %v", err) + } + if len(dirty) == 0 { + t.Fatal("added grant produced no dirty buckets") + } + found := map[string]bool{} + for _, prefix := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + found[g.GetPrincipal().GetResourceId()] = true + return true + }); err != nil { + t.Fatalf("IterateGrantsByEntitlementBucket: %v", err) + } + } + if !found["dave"] { + t.Fatalf("dirty buckets did not include the added principal dave; found=%v", found) + } +} + +func TestMerkleVariableHeight(t *testing.T) { + ctx := context.Background() + + // Small entitlement: under one target bucket -> depth 0, single node. + small := make([]*v3.GrantRecord, 0, 10) + for i := 0; i < 10; i++ { + small = append(small, makeGrant("", ksuid.New().String(), "ent-small", ksuid.New().String())) + } + es, _ := newTestEngine(t) + syncS := seedEntitlement(t, es, "ent-small", small) + rootS, ok, err := es.GetEntitlementMerkleRoot(ctx, syncS, "ent-small") + if err != nil || !ok { + t.Fatalf("small root: ok=%v err=%v", ok, err) + } + if rootS.Depth != 0 { + t.Fatalf("small entitlement depth = %d, want 0", rootS.Depth) + } + if rootS.Count != 10 { + t.Fatalf("small entitlement count = %d, want 10", rootS.Count) + } + if got := merkleNodeCount(t, es, syncS); got != 1 { + t.Fatalf("small entitlement node count = %d, want 1", got) + } + + // Large entitlement: well over the target bucket size -> depth grows, + // and the tree gains leaf nodes beyond the root. + const n = merkleTargetBucketSize*3 + 7 + large := make([]*v3.GrantRecord, 0, n) + for i := 0; i < n; i++ { + large = append(large, makeGrant("", ksuid.New().String(), "ent-large", ksuid.New().String())) + } + el, _ := newTestEngine(t) + syncL := seedEntitlement(t, el, "ent-large", large) + rootL, ok, err := el.GetEntitlementMerkleRoot(ctx, syncL, "ent-large") + if err != nil || !ok { + t.Fatalf("large root: ok=%v err=%v", ok, err) + } + if rootL.Depth < 1 { + t.Fatalf("large entitlement depth = %d, want >= 1", rootL.Depth) + } + if rootL.Count != int64(n) { + t.Fatalf("large entitlement count = %d, want %d", rootL.Count, n) + } + // root + at least 2 leaves (depth>=1 over n grants spreads across + // many buckets). + if got := merkleNodeCount(t, el, syncL); got < 3 { + t.Fatalf("large entitlement node count = %d, want >= 3 (root + leaves)", got) + } +} + +// TestHashIndexIsHashOrdered verifies the new index iterates in +// hash(principal) order: the embedded bucket-hash region is +// non-decreasing across the entitlement's index range. +func TestHashIndexIsHashOrdered(t *testing.T) { + e, _ := newTestEngine(t) + grants := make([]*v3.GrantRecord, 0, 200) + for i := 0; i < 200; i++ { + grants = append(grants, makeGrant("", ksuid.New().String(), "ent-A", ksuid.New().String())) + } + syncID := seedEntitlement(t, e, "ent-A", grants) + + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatal(err) + } + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, "ent-A") + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upperBoundOf(entPrefix)}) + if err != nil { + t.Fatal(err) + } + defer iter.Close() + var prev []byte + count := 0 + for iter.First(); iter.Valid(); iter.Next() { + bh, _, _, _, ok := decodeEntPrincHashTail(iter.Key(), entPrefix) + if !ok { + t.Fatal("failed to decode index tail") + } + if prev != nil && bytes.Compare(bh, prev) < 0 { + t.Fatalf("index not hash-ordered: %x < %x", bh, prev) + } + prev = append(prev[:0], bh...) + count++ + } + if count != 200 { + t.Fatalf("hash index entry count = %d, want 200", count) + } +} diff --git a/pkg/synccompactor/pebble/bucket_plans.go b/pkg/synccompactor/pebble/bucket_plans.go index 0f5c0e239..094eae019 100644 --- a/pkg/synccompactor/pebble/bucket_plans.go +++ b/pkg/synccompactor/pebble/bucket_plans.go @@ -76,6 +76,20 @@ func buildBucketPlans(syncIDBytes []byte) []bucketPlan { lower: enginepkg.GrantByNeedsExpansionSyncLowerBound(syncIDBytes), upper: enginepkg.GrantByNeedsExpansionSyncUpperBound(syncIDBytes), }, + { + name: "grant_by_entitlement_principal_hash", + lower: enginepkg.GrantByEntPrincHashSyncLowerBound(syncIDBytes), + upper: enginepkg.GrantByEntPrincHashSyncUpperBound(syncIDBytes), + }, + { + // Per-entitlement grant merkle nodes. Copied byte-for-byte + // with the sync_id they belong to; because Compact preserves + // the sync_id and the grants move verbatim, the trees stay + // valid in the destination without a rebuild. + name: "grant_merkle", + lower: enginepkg.MerkleSyncLowerBound(syncIDBytes), + upper: enginepkg.MerkleSyncUpperBound(syncIDBytes), + }, { name: "asset", lower: enginepkg.AssetSyncLowerBound(syncIDBytes), From afb78993d42fe081588f9b66faa31e694a314bc3 Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Wed, 10 Jun 2026 11:21:06 -0700 Subject: [PATCH 2/7] betterer --- pkg/dotc1z/engine/pebble/grants.go | 39 +- pkg/dotc1z/engine/pebble/if_newer.go | 13 + pkg/dotc1z/engine/pebble/index_migrations.go | 12 +- pkg/dotc1z/engine/pebble/keys.go | 30 +- pkg/dotc1z/engine/pebble/merkle.go | 628 ++++++++++++++----- pkg/dotc1z/engine/pebble/merkle_test.go | 410 ++++++++++++ 6 files changed, 964 insertions(+), 168 deletions(-) diff --git a/pkg/dotc1z/engine/pebble/grants.go b/pkg/dotc1z/engine/pebble/grants.go index 3cd228fab..bb276e06a 100644 --- a/pkg/dotc1z/engine/pebble/grants.go +++ b/pkg/dotc1z/engine/pebble/grants.go @@ -81,6 +81,14 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord // emitting an external_id on two pages). skipGet := e.takeFreshGrantsEmpty() + // Incremental merkle maintenance applies only on the non-fresh + // path: during a fresh sync the trees don't exist yet (built + // once at seal), so skip even the per-entitlement root probe. + var mm *merkleMutator + if !fresh { + mm = newMerkleMutator(e) + } + // Dedup pre-pass: keep only the LAST occurrence of each // (sync_id, external_id). The map value is the records[] // index — when we re-iterate, we process record i only if @@ -152,6 +160,11 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord return err } closer.Close() + if mm != nil { + if err := mm.remove(idBytes, old); err != nil { + return err + } + } case errors.Is(getErr, pebble.ErrNotFound): // no prior record — write unconditionally default: @@ -164,6 +177,16 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord if err := e.writeGrantIndexes(idxBatch, idBytes, r); err != nil { return err } + if mm != nil { + if err := mm.add(idBytes, r); err != nil { + return err + } + } + } + if mm != nil { + if err := mm.apply(idxBatch); err != nil { + return err + } } opts := writeOpts(e.opts.durability) if fresh { @@ -366,6 +389,14 @@ func (e *Engine) DeleteGrantRecord(ctx context.Context, syncID, externalID strin } closer.Close() + mm := newMerkleMutator(e) + if err := mm.remove(idBytes, old); err != nil { + return err + } + if err := mm.apply(batch); err != nil { + return err + } + if err := batch.Delete(key, nil); err != nil { return err } @@ -484,9 +515,11 @@ func (e *Engine) deleteGrantIndexes(batch *pebble.Batch, syncIDBytes []byte, r * // by_entitlement_principal_hash: the bucket hash is derived from the // principal identity, so deleteGrantIndexes reconstructs the same key // writeGrantIndexes wrote. Skipped when ent/principal are absent, - // matching grantHashIndexKey's nil-guard. NOTE: this invalidates the - // entitlement's merkle tree — callers that delete grants must rebuild - // it (BuildAllMerkleTrees) before relying on a stored root. + // matching grantHashIndexKey's nil-guard. The entitlement's merkle + // tree is kept in step separately: callers on the post-seal mutation + // paths feed the same old record to merkleMutator.remove (and the + // new one to .add), which folds the change into the stored nodes in + // the same batch. if ent != nil && princ != nil { bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) hk := encodeGrantByEntPrincHashIndexKey( diff --git a/pkg/dotc1z/engine/pebble/if_newer.go b/pkg/dotc1z/engine/pebble/if_newer.go index a5d04dcbf..482d8511c 100644 --- a/pkg/dotc1z/engine/pebble/if_newer.go +++ b/pkg/dotc1z/engine/pebble/if_newer.go @@ -37,6 +37,10 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran return e.withWrite(func() error { batch := e.db.NewBatch() defer batch.Close() + // IfNewer is by definition a post-seal mutation path, so the + // merkle trees (cloned along with the sync's keyspace) are kept + // in step incrementally. + mm := newMerkleMutator(e) written := 0 for _, r := range records { if r == nil { @@ -62,6 +66,9 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran if err := e.deleteGrantIndexes(batch, idBytes, old); err != nil { return err } + if err := mm.remove(idBytes, old); err != nil { + return err + } case errors.Is(getErr, pebble.ErrNotFound): // no existing record — write unconditionally default: @@ -77,11 +84,17 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran if err := e.writeGrantIndexes(batch, idBytes, r); err != nil { return err } + if err := mm.add(idBytes, r); err != nil { + return err + } written++ } if written == 0 { return nil } + if err := mm.apply(batch); err != nil { + return err + } return batch.Commit(writeOpts(e.opts.durability)) }) } diff --git a/pkg/dotc1z/engine/pebble/index_migrations.go b/pkg/dotc1z/engine/pebble/index_migrations.go index 51553c9c6..2aaccf092 100644 --- a/pkg/dotc1z/engine/pebble/index_migrations.go +++ b/pkg/dotc1z/engine/pebble/index_migrations.go @@ -68,12 +68,20 @@ var indexMigrations = []indexMigration{ // Backfill the by_entitlement_principal_hash index and the // per-entitlement merkle trees for files written before either // existed. Idempotent: re-emitting an index entry is a Set over - // the same key/value, and the trees are rebuilt wholesale. + // the same key/value, and each tree rebuild range-clears the + // entitlement's typeMerkle keyspace before writing — which is + // also what makes the version bump effective: v1 (sha256-fold, + // root+leaves) nodes are byte-length-identical to v2 (XOR, + // all-levels) nodes, so only the clear removes them. // New files persist this version at their initial (empty) Open, // so the inline write path maintains both and the backfill never // re-runs for them. + // + // v2: XOR combiner, all levels stored sparsely, count on every + // node (RFC 0003). Pre-GA in-place change; no production v3 + // data existed at v1. Name: "grant_by_entitlement_principal_hash", - Version: 1, + Version: 2, Apply: func(ctx context.Context, e *Engine) error { return e.backfillGrantHashIndexAndMerkle(ctx) }, diff --git a/pkg/dotc1z/engine/pebble/keys.go b/pkg/dotc1z/engine/pebble/keys.go index 96a1f50b4..7add2fcab 100644 --- a/pkg/dotc1z/engine/pebble/keys.go +++ b/pkg/dotc1z/engine/pebble/keys.go @@ -383,20 +383,32 @@ func GrantByEntPrincHashSyncUpperBound(syncIDBytes []byte) []byte { // // v3 | typeMerkle | sync_id | 0x00 | entitlement_id | 0x00 | level(1 byte) | bucket_prefix(level raw bytes) // -// level 0 is the root (bucket_prefix empty); level == tree depth holds -// the leaves, one per non-empty principal-hash bucket. bucket_prefix is -// the first `level` bytes of the principal bucket hash, raw, so it aligns -// byte-for-byte with the index key's bucket-hash region. See merkle.go -// for the node value framing. +// level 0 is the root (bucket_prefix empty); every level from 1 to the +// tree's depth holds that level's non-empty nodes, one per non-empty +// principal-hash prefix. bucket_prefix is the first `level` bytes of the +// principal bucket hash, raw, so it aligns byte-for-byte with the index +// key's bucket-hash region — and because the level byte precedes the +// prefix, the children of one node are a contiguous key range one level +// down. See merkle.go for the node value framing. func encodeMerkleNodeKey(syncIDBytes []byte, entitlementID string, level byte, bucketPrefix []byte) []byte { - buf := make([]byte, 0, 6+len(syncIDBytes)+len(entitlementID)+len(bucketPrefix)) + buf := encodeMerkleEntPrefix(syncIDBytes, entitlementID) + buf = append(buf, level) + return append(buf, bucketPrefix...) +} + +// encodeMerkleEntPrefix is the prefix of every merkle node key for one +// entitlement — the range a rebuild clears before writing (the build +// only Sets nodes; without the leading DeleteRange a depth change or an +// emptied bucket would leave stale nodes for the comparison descent to +// read) and the range merkleMutator drops on detecting an inconsistent +// tree. +func encodeMerkleEntPrefix(syncIDBytes []byte, entitlementID string) []byte { + buf := make([]byte, 0, 6+len(syncIDBytes)+len(entitlementID)) buf = append(buf, versionV3, typeMerkle) buf = append(buf, syncIDBytes...) buf = codec.AppendTupleSeparator(buf) buf = codec.AppendTupleStrings(buf, entitlementID) - buf = codec.AppendTupleSeparator(buf) - buf = append(buf, level) - return append(buf, bucketPrefix...) + return codec.AppendTupleSeparator(buf) } // MerkleSyncLowerBound / UpperBound bound the entire merkle keyspace for diff --git a/pkg/dotc1z/engine/pebble/merkle.go b/pkg/dotc1z/engine/pebble/merkle.go index 187913008..13861f018 100644 --- a/pkg/dotc1z/engine/pebble/merkle.go +++ b/pkg/dotc1z/engine/pebble/merkle.go @@ -15,7 +15,7 @@ import ( v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" ) -// Per-entitlement grant merkle tree. +// Per-entitlement grant merkle tree (XOR combiner). // // Goal: answer "does this entitlement have exactly the same grants as // some other sync/file?" with a single key read, and when the answer is @@ -32,20 +32,42 @@ import ( // - a hash prefix is a clean byte-prefix of the key, so "all grants in // bucket P" is a contiguous range scan. // -// Variable height. The tree depth is chosen from the grant count: -// depth 0 is a single root node (used for empty and small entitlements — -// this is why an entitlement with no grants costs exactly one stored -// node); each additional level consumes one more byte of the bucket -// hash, multiplying the bucket count by 256. Node hashes are -// CONTENT-defined — a node's hash is the fold of every grant content -// hash beneath it, independent of how the subtree is split — so a node -// at one depth is directly comparable to the equivalent prefix-range in -// a tree of a different depth. +// Combiner. A node's digest is the XOR of every grant content hash +// beneath it (leaves of the fold stay sha256 — only the combiner is +// XOR). XOR is homomorphic (parent = XOR of children), order-independent, +// and invertible, which buys three things: // -// On-disk ABI. Both the principal bucket hash and the grant content -// hash are part of the stored format: changing merkleBucketHashLen, the -// content-hash field set (grantContentHash), or the node value framing -// requires an index-migration version bump (see index_migrations.go). +// - depth-independence: a node's digest depends only on the grants in +// its prefix range, never on how the subtree is split, so nodes from +// trees of different heights compare directly; +// - O(1) incremental maintenance: post-seal insert/overwrite/delete +// XOR the grant's content hash into/out of the O(depth) nodes on its +// bucket path (see merkleMutator); +// - the empty digest is all-zero (the XOR identity), so an absent node +// reads as {count: 0, digest: 0}. +// +// Every node also stores its grant COUNT. Comparison always checks the +// (count, digest) pair, so a non-empty node whose hashes happened to XOR +// to zero can never be conflated with an empty/absent one. Within one +// tree that cancellation cannot even occur: every key-distinguishing +// field (principal rt/id, external_id) is folded into the content hash, +// so duplicate leaves are impossible by construction. XOR set-hashing is +// not adversarially collision-resistant (Bellare–Micciancio); this tree +// is an optimization, not a trust boundary — see RFC 0003 §9. +// +// Shape. 256-ary radix, variable height: depth 0 is a single root node +// (used for empty and small entitlements — this is why an entitlement +// with no grants costs exactly one stored node); each additional level +// consumes one more byte of the bucket hash. ALL levels are stored, +// sparsely: the root is always materialized (it is the "tree was built" +// marker — absence of the root means "never built", never "empty"), and +// every non-root node is materialized iff its subtree holds ≥1 grant. +// +// On-disk ABI. The principal bucket hash, the grant content hash, the +// combiner, and the node value framing are all part of the stored +// format: changing merkleBucketHashLen, the content-hash field set +// (grantContentHash), the combiner, or the framing requires an +// index-migration version bump (see index_migrations.go). const ( // merkleBucketHashLen is the width, in bytes, of the principal @@ -65,9 +87,19 @@ const ( ) // hashLen is the width of a sha256 digest, used for both the grant -// content hash (index value) and node hashes. +// content hash (index value) and node digests. const hashLen = sha256.Size +// zeroDigest is the XOR identity — the digest of an empty/absent node. +var zeroDigest [hashLen]byte + +// xorInto XORs src into dst in place, over min(len(dst), len(src)). +func xorInto(dst, src []byte) { + for i := range min(len(dst), len(src)) { + dst[i] ^= src[i] + } +} + // writeLenPrefixed writes an 8-byte big-endian length followed by b. // Length-prefixing every field makes the canonical encoding injective: // no concatenation of fields can be confused with a different split. @@ -167,24 +199,28 @@ func chooseMerkleDepth(count int64) int { return depth } -// Node value framing. +// Node value framing. All non-root nodes share one body so interior and +// leaf nodes are read uniformly; the root prepends the chosen depth so a +// reader knows the leaf level without scanning. // -// root (level 0): depth(1) | count(8 BE) | hash(hashLen) -// leaf (level d): count(8 BE) | hash(hashLen) +// node body: count(8 BE) | digest(hashLen) +// root body: depth(1) | count(8 BE) | digest(hashLen) // -// The root carries the chosen depth so a reader knows the leaf level -// without scanning. Leaves omit it (their level is in the key). +// An ABSENT non-root node is {count: 0, digest: 0} by definition — +// readers substitute that for any missing key. An absent ROOT means +// "tree never built" (never "empty"); readers must fall back to the +// on-demand fold, not assume zero. -func packMerkleRoot(depth int, count int64, h []byte) []byte { - buf := make([]byte, 0, 1+8+len(h)) +func packMerkleRoot(depth int, count int64, digest []byte) []byte { + buf := make([]byte, 0, 1+8+len(digest)) buf = append(buf, byte(depth)) var n [8]byte binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count buf = append(buf, n[:]...) - return append(buf, h...) + return append(buf, digest...) } -// unpackMerkleRoot returns (depth, count, hash, ok). ok is false when +// unpackMerkleRoot returns (depth, count, digest, ok). ok is false when // the value is not a well-formed root blob. func unpackMerkleRoot(val []byte) (int, int64, []byte, bool) { if len(val) != 1+8+hashLen { @@ -195,12 +231,21 @@ func unpackMerkleRoot(val []byte) (int, int64, []byte, bool) { return depth, count, val[9:], true } -func packMerkleLeaf(count int64, h []byte) []byte { - buf := make([]byte, 0, 8+len(h)) +func packMerkleNode(count int64, digest []byte) []byte { + buf := make([]byte, 0, 8+len(digest)) var n [8]byte binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count buf = append(buf, n[:]...) - return append(buf, h...) + return append(buf, digest...) +} + +// unpackMerkleNode returns (count, digest, ok) for a non-root node body. +func unpackMerkleNode(val []byte) (int64, []byte, bool) { + if len(val) != 8+hashLen { + return 0, nil, false + } + count := int64(binary.BigEndian.Uint64(val[:8])) //nolint:gosec // non-negative count + return count, val[8:], true } // BuildAllMerkleTrees rebuilds the per-entitlement merkle tree for every @@ -249,43 +294,73 @@ func (e *Engine) buildEntitlementMerkle(ctx context.Context, idBytes []byte, ent } // buildEntitlementMerkleAtDepth folds the hash index for one entitlement -// into a root (and, when depth > 0, one leaf per non-empty bucket) and -// writes the nodes in a single streaming pass — O(1) memory regardless -// of entitlement size. The depth is taken as a parameter rather than -// derived so the depth-selection seam can be exercised directly: tests -// force a depth that the natural count→depth mapping would only produce -// at a very large grant count, which is how the cross-depth comparison -// path gets covered without seeding tens of thousands of grants. +// into a root plus every non-empty node at every level, in a single +// streaming pass — O(depth) memory regardless of entitlement size. +// +// The pass starts by range-deleting the entitlement's whole typeMerkle +// keyspace: the build only ever Sets nodes, so without the clear a +// rebuild that changes depth or empties a bucket would leave stale nodes +// that the comparison descent (which enumerates children from the node +// keyspace) would read — and a stale digest that happens to match the +// peer prunes a real diff. Old and new framings are byte-length +// identical, so stale nodes are not detectable by inspection. +// +// Sorted index order means each node's grants are contiguous, so a +// level's "open" node closes exactly when its prefix changes. Only +// non-empty nodes are ever opened, so sparsity is automatic, not a +// prune pass. +// +// The depth is taken as a parameter rather than derived so the +// depth-selection seam can be exercised directly: tests force a depth +// that the natural count→depth mapping would only produce at a very +// large grant count, which is how the cross-depth comparison path gets +// covered without seeding tens of thousands of grants. func (e *Engine) buildEntitlementMerkleAtDepth(ctx context.Context, idBytes []byte, entitlementID string, depth int) error { entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) upper := upperBoundOf(entPrefix) + nodeLower := encodeMerkleEntPrefix(idBytes, entitlementID) + nodeUpper := upperBoundOf(nodeLower) return e.withWrite(func() error { batch := e.db.NewBatch() defer batch.Close() + // Clear any prior build (see function comment). In-batch + // ordering makes this safe: the Sets below land after the + // tombstone and survive it. + if err := batch.DeleteRange(nodeLower, nodeUpper, nil); err != nil { + return err + } + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upper}) if err != nil { return err } defer iter.Close() - rootH := sha256.New() - var ( - leafH hash.Hash - leafCount int64 - curPrefix []byte - haveLeaf bool - total int64 - ) - flushLeaf := func() error { - if depth == 0 || !haveLeaf { + // One running node per level 1..depth; the root accumulates + // separately (its prefix is always empty, so it never closes + // mid-stream). + type openNode struct { + active bool + prefix []byte + digest [hashLen]byte + count int64 + } + open := make([]openNode, depth+1) + flush := func(level int) error { + n := &open[level] + if !n.active { return nil } - key := encodeMerkleNodeKey(idBytes, entitlementID, byte(depth), curPrefix) - return batch.Set(key, packMerkleLeaf(leafCount, leafH.Sum(nil)), nil) + key := encodeMerkleNodeKey(idBytes, entitlementID, byte(level), n.prefix) + return batch.Set(key, packMerkleNode(n.count, n.digest[:]), nil) } + var ( + rootDigest [hashLen]byte + total int64 + ) for iter.First(); iter.Valid(); iter.Next() { if err := ctx.Err(); err != nil { return err @@ -297,32 +372,37 @@ func (e *Engine) buildEntitlementMerkleAtDepth(ctx context.Context, idBytes []by bucketHash := key[len(entPrefix) : len(entPrefix)+merkleBucketHashLen] val := iter.Value() // 32-byte content hash - if depth > 0 { - prefix := bucketHash[:depth] - if !haveLeaf || !bytes.Equal(prefix, curPrefix) { - if err := flushLeaf(); err != nil { + for level := 1; level <= depth; level++ { + prefix := bucketHash[:level] + n := &open[level] + if !n.active || !bytes.Equal(n.prefix, prefix) { + if err := flush(level); err != nil { return err } - curPrefix = append(curPrefix[:0], prefix...) - leafH = sha256.New() - leafCount = 0 - haveLeaf = true + n.prefix = append(n.prefix[:0], prefix...) + n.digest = [hashLen]byte{} + n.count = 0 + n.active = true } - _, _ = leafH.Write(val) - leafCount++ + xorInto(n.digest[:], val) + n.count++ } - _, _ = rootH.Write(val) + xorInto(rootDigest[:], val) total++ } if err := iter.Error(); err != nil { return err } - if err := flushLeaf(); err != nil { - return err + for level := 1; level <= depth; level++ { + if err := flush(level); err != nil { + return err + } } + // Root is written unconditionally — even at count 0 — as the + // "tree was built" marker. rootKey := encodeMerkleNodeKey(idBytes, entitlementID, 0, nil) - if err := batch.Set(rootKey, packMerkleRoot(depth, total, rootH.Sum(nil)), nil); err != nil { + if err := batch.Set(rootKey, packMerkleRoot(depth, total, rootDigest[:]), nil); err != nil { return err } @@ -358,7 +438,7 @@ type MerkleRoot struct { // GetEntitlementMerkleRoot returns the stored root for an entitlement. // ok is false when no tree has been built for it (the caller can fall -// back to ComputeEntitlementRoot, which derives the same digest from the +// back to ComputeBucketHash, which derives the same digest from the // index on demand). func (e *Engine) GetEntitlementMerkleRoot(ctx context.Context, syncID, entitlementID string) (MerkleRoot, bool, error) { idBytes, err := e.resolveSyncBytes(syncID) @@ -382,10 +462,60 @@ func (e *Engine) GetEntitlementMerkleRoot(ctx context.Context, syncID, entitleme return MerkleRoot{Hash: out, Depth: depth, Count: count}, true, nil } +// getMerkleNode reads one stored non-root node. An absent node returns +// (0, zero digest, present=false, nil) — the XOR identity. +func (e *Engine) getMerkleNode(idBytes []byte, entitlementID string, level int, prefix []byte) (int64, []byte, bool, error) { + val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, byte(level), prefix)) + if err != nil { + if errors.Is(err, pebble.ErrNotFound) { + return 0, zeroDigest[:], false, nil + } + return 0, nil, false, err + } + defer closer.Close() + count, digest, ok := unpackMerkleNode(val) + if !ok { + return 0, nil, false, fmt.Errorf("getMerkleNode: malformed node for %q level %d", entitlementID, level) + } + out := make([]byte, hashLen) + copy(out, digest) + return count, out, true, nil +} + +// merkleChildPrefixes returns the sorted full prefixes (length = level) +// of the stored nodes at `level` under parentPrefix. Because the node +// key embeds the level byte before the prefix bytes, the children of one +// parent are a contiguous key range — cost is O(children present), and +// only non-empty children are ever stored. +func (e *Engine) merkleChildPrefixes(ctx context.Context, idBytes []byte, entitlementID string, level int, parentPrefix []byte) ([][]byte, error) { + stem := encodeMerkleNodeKey(idBytes, entitlementID, byte(level), nil) + lower := append(append([]byte(nil), stem...), parentPrefix...) + upper := upperBoundOf(lower) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return nil, err + } + defer iter.Close() + var out [][]byte + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return nil, err + } + key := iter.Key() + if len(key) != len(stem)+level { + continue // malformed; skip defensively + } + prefix := make([]byte, level) + copy(prefix, key[len(stem):]) + out = append(out, prefix) + } + return out, iter.Error() +} + // ComputeBucketHash folds the hash index over a single principal-hash // bucket (identified by a raw hash prefix; empty prefix = whole -// entitlement = the root) and returns the content-defined digest plus -// the grant count. This is the authoritative definition of a node hash; +// entitlement = the root) and returns the content-defined XOR digest +// plus the grant count. This is the authoritative definition of a node; // stored nodes are a cache of it. Depth-independent: the digest depends // only on the grants in the prefix range, not on any tree's shape. func (e *Engine) ComputeBucketHash(ctx context.Context, syncID, entitlementID string, hashPrefix []byte) ([]byte, int64, error) { @@ -400,19 +530,19 @@ func (e *Engine) ComputeBucketHash(ctx context.Context, syncID, entitlementID st return nil, 0, err } defer iter.Close() - h := sha256.New() + digest := make([]byte, hashLen) var count int64 for iter.First(); iter.Valid(); iter.Next() { if err := ctx.Err(); err != nil { return nil, 0, err } - _, _ = h.Write(iter.Value()) + xorInto(digest, iter.Value()) count++ } if err := iter.Error(); err != nil { return nil, 0, err } - return h.Sum(nil), count, nil + return digest, count, nil } // IterateGrantsByEntitlementBucket yields the grants in one principal-hash @@ -468,10 +598,18 @@ func (e *Engine) IterateGrantsByEntitlementBucket(ctx context.Context, syncID, e // the comparison granularity is the root, e.g. tiny entitlements). A nil // (empty) result means the two are identical. // -// The fast path is a root-hash equality check. On mismatch it descends -// to the shallower of the two trees' depths and compares each bucket at -// that granularity, preferring stored leaf hashes and falling back to an -// on-demand index fold when a side's tree is shallower or absent. +// The fast path is a single root read per side. On mismatch it descends +// level by level, pruning every subtree whose (count, digest) pair +// matches and emitting the differing prefixes at compareDepth — the +// shallower tree's leaf level, where both sides still have directly +// comparable nodes (XOR digests are depth-independent). Children are +// enumerated from the stored node keyspace (union of both sides), so +// descent cost is proportional to the symmetric difference, not the +// fan-out. +// +// A missing root means the tree was never built on that side — NOT that +// the entitlement is empty — so both sides are compared via the +// authoritative on-demand fold instead. func (e *Engine) DirtyEntitlementBuckets(ctx context.Context, syncID string, other *Engine, otherSyncID, entitlementID string) ([][]byte, error) { rootA, okA, err := e.GetEntitlementMerkleRoot(ctx, syncID, entitlementID) if err != nil { @@ -482,129 +620,100 @@ func (e *Engine) DirtyEntitlementBuckets(ctx context.Context, syncID string, oth return nil, err } - // Fast equality via stored roots when both exist. - if okA && okB && bytes.Equal(rootA.Hash, rootB.Hash) { - return nil, nil - } - - // Comparison granularity: the shallower available depth. A missing - // tree is treated as depth 0 (compare at the root → whole-entitlement - // dirty if the computed roots differ). - compareDepth := 0 - if okA && okB { - compareDepth = min(rootA.Depth, rootB.Depth) - } - - if compareDepth == 0 { - // Confirm via the authoritative fold (covers the missing-tree - // case and guards against a stale stored root). - ha, _, err := e.ComputeBucketHash(ctx, syncID, entitlementID, nil) + if !okA || !okB { + ha, ca, err := e.ComputeBucketHash(ctx, syncID, entitlementID, nil) if err != nil { return nil, err } - hb, _, err := other.ComputeBucketHash(ctx, otherSyncID, entitlementID, nil) + hb, cb, err := other.ComputeBucketHash(ctx, otherSyncID, entitlementID, nil) if err != nil { return nil, err } - if bytes.Equal(ha, hb) { + if ca == cb && bytes.Equal(ha, hb) { return nil, nil } return [][]byte{{}}, nil } - // Union of non-empty bucket prefixes at compareDepth from both sides. - prefixes, err := e.distinctBucketPrefixes(ctx, syncID, entitlementID, compareDepth) + if rootA.Count == rootB.Count && bytes.Equal(rootA.Hash, rootB.Hash) { + return nil, nil + } + + // Roots differ. The descent granularity is the shallower tree's + // leaf level; at depth 0 there is nothing below the root. + compareDepth := min(rootA.Depth, rootB.Depth) + if compareDepth == 0 { + return [][]byte{{}}, nil + } + + idBytesA, err := e.resolveSyncBytes(syncID) if err != nil { return nil, err } - otherPrefixes, err := other.distinctBucketPrefixes(ctx, otherSyncID, entitlementID, compareDepth) + idBytesB, err := other.resolveSyncBytes(otherSyncID) if err != nil { return nil, err } - union := mergeSortedPrefixes(prefixes, otherPrefixes) var dirty [][]byte - for _, p := range union { - ha, _, err := e.bucketHashPreferStored(ctx, syncID, entitlementID, p, rootA, okA) + var walk func(prefix []byte, level int) error + walk = func(prefix []byte, level int) error { + if err := ctx.Err(); err != nil { + return err + } + ca, da, _, err := e.getMerkleNode(idBytesA, entitlementID, level, prefix) if err != nil { - return nil, err + return err } - hb, _, err := other.bucketHashPreferStored(ctx, otherSyncID, entitlementID, p, rootB, okB) + cb, db, _, err := other.getMerkleNode(idBytesB, entitlementID, level, prefix) if err != nil { - return nil, err + return err } - if !bytes.Equal(ha, hb) { - dirty = append(dirty, p) + if ca == cb && bytes.Equal(da, db) { + return nil // identical subtree (or absent on both sides) → prune } - } - return dirty, nil -} - -// bucketHashPreferStored returns a bucket's hash, using the stored leaf -// node when the tree's depth matches the prefix length (the cheap path), -// otherwise folding the index. root/ok describe the entitlement's stored -// tree for this engine. -func (e *Engine) bucketHashPreferStored(ctx context.Context, syncID, entitlementID string, prefix []byte, root MerkleRoot, ok bool) ([]byte, int64, error) { - if ok && root.Depth == len(prefix) { - idBytes, err := e.resolveSyncBytes(syncID) + if level == compareDepth { + dirty = append(dirty, prefix) + return nil + } + kidsA, err := e.merkleChildPrefixes(ctx, idBytesA, entitlementID, level+1, prefix) if err != nil { - return nil, 0, err + return err + } + kidsB, err := other.merkleChildPrefixes(ctx, idBytesB, entitlementID, level+1, prefix) + if err != nil { + return err } - val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, byte(len(prefix)), prefix)) - if err == nil { - defer closer.Close() - if len(val) == 8+hashLen { - count := int64(binary.BigEndian.Uint64(val[:8])) //nolint:gosec // non-negative count - h := make([]byte, hashLen) - copy(h, val[8:]) - return h, count, nil + for _, k := range mergeSortedPrefixes(kidsA, kidsB) { + if err := walk(k, level+1); err != nil { + return err } - } else if !errors.Is(err, pebble.ErrNotFound) { - return nil, 0, err } - // fall through to compute on miss/malformed + return nil } - return e.ComputeBucketHash(ctx, syncID, entitlementID, prefix) -} -// distinctBucketPrefixes returns the sorted, distinct depth-byte hash -// prefixes present in an entitlement's hash index. It seeks past each -// bucket once found, so the cost is O(distinct buckets) seeks rather than -// O(grants). -func (e *Engine) distinctBucketPrefixes(ctx context.Context, syncID, entitlementID string, depth int) ([][]byte, error) { - idBytes, err := e.resolveSyncBytes(syncID) + kidsA, err := e.merkleChildPrefixes(ctx, idBytesA, entitlementID, 1, nil) if err != nil { return nil, err } - entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) - upper := upperBoundOf(entPrefix) - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upper}) + kidsB, err := other.merkleChildPrefixes(ctx, idBytesB, entitlementID, 1, nil) if err != nil { return nil, err } - defer iter.Close() - - var out [][]byte - for iter.First(); iter.Valid(); { - if err := ctx.Err(); err != nil { + for _, k := range mergeSortedPrefixes(kidsA, kidsB) { + if err := walk(k, 1); err != nil { return nil, err } - key := iter.Key() - if len(key) < len(entPrefix)+depth { - iter.Next() - continue - } - prefix := make([]byte, depth) - copy(prefix, key[len(entPrefix):len(entPrefix)+depth]) - out = append(out, prefix) - // Seek past this whole bucket to the next distinct prefix. - seekTo := upperBoundOf(append(append([]byte(nil), entPrefix...), prefix...)) - if seekTo == nil { - break - } - iter.SeekGE(seekTo) } - return out, iter.Error() + + // Roots differed but the descent found nothing: with consistent + // trees that's impossible (the root is the XOR of level 1), so a + // stored node is stale or corrupt. Fail safe — whole entitlement + // dirty; the next rebuild heals the tree. + if len(dirty) == 0 { + return [][]byte{{}}, nil + } + return dirty, nil } // mergeSortedPrefixes returns the sorted union of two sorted, de-duped @@ -630,3 +739,214 @@ func mergeSortedPrefixes(a, b [][]byte) [][]byte { out = append(out, b[j:]...) return out } + +// --- Incremental maintenance (post-seal) --- + +// merkleMutator accumulates per-node (XOR, count) deltas for a batch of +// post-seal grant mutations and applies each touched node exactly once. +// +// Why an accumulator instead of read-modify-write per mutation: the +// updates target a plain pebble.Batch, which does NOT read through its +// own writes — and every mutation in a batch touches the root node, so +// naive per-mutation RMW would lose deltas. Accumulating also collapses +// N writes per node into one. All grant writers run under withWrite's +// mutex, so reading current node values from the DB inside apply is +// race-free. +// +// Lifecycle: one mutator per write batch. Callers feed remove(old) / +// add(new) as they process records (an overwrite that moves the grant to +// a different bucket — principal changed — is exactly remove+add), then +// call apply(batch) once before commit. +// +// Entitlements whose tree was never built (no stored root) are skipped: +// the seal-time build or the on-Open backfill will construct them from +// the index. This also makes the mutator free during a fresh sync — but +// callers on the fresh-sync bulk path should skip constructing one +// anyway to avoid the per-entitlement root probe. +type merkleMutator struct { + e *Engine + ents map[string]*mutatorEnt // keyed by string(root node key) +} + +type mutatorEnt struct { + idBytes []byte + entID string + rootKey []byte + present bool // stored root exists; if false all deltas are dropped + depth int + rootCount int64 // count read from the stored root + rootDigest [hashLen]byte // digest read from the stored root + xor [hashLen]byte // accumulated root delta + countDelta int64 + nodes map[string]*mutatorNode // levels 1..depth, keyed by string(node key) +} + +type mutatorNode struct { + key []byte + xor [hashLen]byte + countDelta int64 +} + +func newMerkleMutator(e *Engine) *merkleMutator { + return &merkleMutator{e: e, ents: make(map[string]*mutatorEnt)} +} + +// entFor returns the (cached) per-entitlement state, probing the stored +// root on first touch. The cached root snapshot stays valid for the +// mutator's lifetime because all writers serialize through withWrite. +func (m *merkleMutator) entFor(idBytes []byte, entitlementID string) (*mutatorEnt, error) { + rootKey := encodeMerkleNodeKey(idBytes, entitlementID, 0, nil) + k := string(rootKey) + if me, ok := m.ents[k]; ok { + return me, nil + } + me := &mutatorEnt{idBytes: idBytes, entID: entitlementID, rootKey: rootKey, nodes: make(map[string]*mutatorNode)} + val, closer, err := m.e.db.Get(rootKey) + switch { + case err == nil: + depth, count, digest, ok := unpackMerkleRoot(val) + closer.Close() + if ok { + me.present = true + me.depth = depth + me.rootCount = count + copy(me.rootDigest[:], digest) + } + // Malformed root: leave present=false so mutations are dropped; + // the stale root heals at the next rebuild. + case errors.Is(err, pebble.ErrNotFound): + // No tree — deltas for this entitlement are no-ops. + default: + return nil, err + } + m.ents[k] = me + return me, nil +} + +// add records r's insertion into its entitlement's tree. +func (m *merkleMutator) add(idBytes []byte, r *v3.GrantRecord) error { + return m.delta(idBytes, r, 1) +} + +// remove records r's removal from its entitlement's tree. +func (m *merkleMutator) remove(idBytes []byte, r *v3.GrantRecord) error { + return m.delta(idBytes, r, -1) +} + +func (m *merkleMutator) delta(idBytes []byte, r *v3.GrantRecord, sign int64) error { + ent := r.GetEntitlement() + princ := r.GetPrincipal() + if ent == nil || princ == nil { + return nil // not in the hash index → not in the tree + } + me, err := m.entFor(idBytes, ent.GetEntitlementId()) + if err != nil { + return err + } + if !me.present { + return nil + } + h := grantContentHash(r) + xorInto(me.xor[:], h) + me.countDelta += sign + if me.depth == 0 { + return nil + } + bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) + for level := 1; level <= me.depth; level++ { + key := encodeMerkleNodeKey(idBytes, ent.GetEntitlementId(), byte(level), bh[:level]) + nk := string(key) + n, ok := me.nodes[nk] + if !ok { + n = &mutatorNode{key: key} + me.nodes[nk] = n + } + xorInto(n.xor[:], h) + n.countDelta += sign + } + return nil +} + +// apply folds the accumulated deltas into the stored nodes via batch. +// Nodes whose delta cancelled to zero (e.g. an overwrite that changed +// only excluded fields) are skipped; a non-root node whose count reaches +// zero is deleted (restoring sparsity); the root is rewritten in place. +// A count that would go negative means the stored tree disagrees with +// the mutation stream — the tree is dropped wholesale (DeleteRange), so +// readers fall back to the on-demand fold until the next rebuild. +func (m *merkleMutator) apply(batch *pebble.Batch) error { + for _, me := range m.ents { + if !me.present { + continue + } + if err := m.applyEnt(batch, me); err != nil { + return err + } + } + return nil +} + +func (m *merkleMutator) applyEnt(batch *pebble.Batch, me *mutatorEnt) error { + if me.countDelta == 0 && me.xor == zeroDigest && len(me.nodes) == 0 { + return nil + } + dropTree := func() error { + // In-batch ordering: this tombstone lands after any node Sets + // already staged for this entitlement and removes them too. + lo := encodeMerkleEntPrefix(me.idBytes, me.entID) + return batch.DeleteRange(lo, upperBoundOf(lo), nil) + } + if me.rootCount+me.countDelta < 0 { + return dropTree() + } + for _, n := range me.nodes { + if n.countDelta == 0 && n.xor == zeroDigest { + continue + } + var ( + curCount int64 + curDigest [hashLen]byte + ) + val, closer, err := m.e.db.Get(n.key) + switch { + case err == nil: + c, d, ok := unpackMerkleNode(val) + closer.Close() + if !ok { + return dropTree() + } + curCount = c + copy(curDigest[:], d) + case errors.Is(err, pebble.ErrNotFound): + // absent node = {0, zero} + default: + return err + } + newCount := curCount + n.countDelta + if newCount < 0 { + return dropTree() + } + xorInto(curDigest[:], n.xor[:]) + if newCount == 0 { + // An emptied node's digest must cancel to exactly zero + // (count 0 ⇒ digest 0); anything else means the stored + // tree disagrees with the mutation stream. + if curDigest != zeroDigest { + return dropTree() + } + if err := batch.Delete(n.key, nil); err != nil { + return err + } + continue + } + if err := batch.Set(n.key, packMerkleNode(newCount, curDigest[:]), nil); err != nil { + return err + } + } + if me.countDelta == 0 && me.xor == zeroDigest { + return nil + } + newDigest := me.rootDigest + xorInto(newDigest[:], me.xor[:]) + return batch.Set(me.rootKey, packMerkleRoot(me.depth, me.rootCount+me.countDelta, newDigest[:]), nil) +} diff --git a/pkg/dotc1z/engine/pebble/merkle_test.go b/pkg/dotc1z/engine/pebble/merkle_test.go index 869e6e9af..1af4a9797 100644 --- a/pkg/dotc1z/engine/pebble/merkle_test.go +++ b/pkg/dotc1z/engine/pebble/merkle_test.go @@ -482,3 +482,413 @@ func TestHashIndexIsHashOrdered(t *testing.T) { t.Fatalf("hash index entry count = %d, want 200", count) } } + +// dumpMerkleNodes snapshots every merkle node key/value for a sync. +// Used to byte-compare an incrementally-maintained tree against a +// from-scratch rebuild. +func dumpMerkleNodes(t testing.TB, e *Engine, syncID string) map[string][]byte { + t.Helper() + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatalf("resolveSyncBytes: %v", err) + } + iter, err := e.db.NewIter(&pebble.IterOptions{ + LowerBound: MerkleSyncLowerBound(idBytes), + UpperBound: MerkleSyncUpperBound(idBytes), + }) + if err != nil { + t.Fatalf("NewIter: %v", err) + } + defer iter.Close() + out := map[string][]byte{} + for iter.First(); iter.Valid(); iter.Next() { + out[string(iter.Key())] = append([]byte(nil), iter.Value()...) + } + if err := iter.Error(); err != nil { + t.Fatalf("iter: %v", err) + } + return out +} + +// requireSameMerkleNodes fails with a per-key diff when two node +// snapshots differ. +func requireSameMerkleNodes(t *testing.T, got, want map[string][]byte) { + t.Helper() + for k, wv := range want { + gv, ok := got[k] + if !ok { + t.Errorf("missing node %x (want %x)", k, wv) + continue + } + if !bytes.Equal(gv, wv) { + t.Errorf("node %x differs:\n got %x\nwant %x", k, gv, wv) + } + } + for k, gv := range got { + if _, ok := want[k]; !ok { + t.Errorf("extra node %x = %x", k, gv) + } + } +} + +// TestMerkleAllLevelsSparseConsistent verifies the all-levels build: +// every stored node is non-empty, each interior node is exactly the XOR +// (and count-sum) of its children, the root is the fold of level 1, no +// nodes exist beyond the chosen depth, and a stored node byte-matches +// the authoritative on-demand fold of its bucket. +func TestMerkleAllLevelsSparseConsistent(t *testing.T) { + ctx := context.Background() + e, _ := newTestEngine(t) + const n = 60 + grants := make([]*v3.GrantRecord, 0, n) + for i := 0; i < n; i++ { + grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) + } + syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatal(err) + } + + root, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + if err != nil || !ok { + t.Fatalf("root: ok=%v err=%v", ok, err) + } + if root.Depth != 2 || root.Count != n { + t.Fatalf("root depth=%d count=%d, want 2, %d", root.Depth, root.Count, n) + } + + level1, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 1, nil) + if err != nil { + t.Fatal(err) + } + if len(level1) == 0 { + t.Fatal("no level-1 nodes stored") + } + var ( + rootXor [hashLen]byte + rootCount int64 + level2N int + ) + for _, p1 := range level1 { + c1, d1, present, err := e.getMerkleNode(idBytes, "ent-A", 1, p1) + if err != nil || !present { + t.Fatalf("level-1 node %x: present=%v err=%v", p1, present, err) + } + if c1 < 1 { + t.Fatalf("level-1 node %x stored with count %d; empty nodes must not be materialized", p1, c1) + } + children, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, p1) + if err != nil { + t.Fatal(err) + } + if len(children) == 0 { + t.Fatalf("level-1 node %x has no stored children", p1) + } + var ( + childXor [hashLen]byte + childCount int64 + ) + for _, p2 := range children { + c2, d2, present2, err := e.getMerkleNode(idBytes, "ent-A", 2, p2) + if err != nil || !present2 { + t.Fatalf("level-2 node %x: present=%v err=%v", p2, present2, err) + } + if c2 < 1 { + t.Fatalf("level-2 node %x stored with count %d", p2, c2) + } + xorInto(childXor[:], d2) + childCount += c2 + } + if childCount != c1 || !bytes.Equal(childXor[:], d1) { + t.Fatalf("interior node %x != fold of children: count %d vs %d", p1, c1, childCount) + } + level2N += len(children) + xorInto(rootXor[:], d1) + rootCount += c1 + } + if rootCount != root.Count || !bytes.Equal(rootXor[:], root.Hash) { + t.Fatalf("root != fold of level 1: count %d vs %d", root.Count, rootCount) + } + + // Exactly root + level1 + level2 nodes — nothing beyond the depth. + if got, want := merkleNodeCount(t, e, syncID), 1+len(level1)+level2N; got != want { + t.Fatalf("total node count = %d, want %d (root + L1 + L2 only)", got, want) + } + + // A stored node is a cache of the authoritative fold. + h, c, err := e.ComputeBucketHash(ctx, syncID, "ent-A", level1[0]) + if err != nil { + t.Fatal(err) + } + c1, d1, _, err := e.getMerkleNode(idBytes, "ent-A", 1, level1[0]) + if err != nil { + t.Fatal(err) + } + if c != c1 || !bytes.Equal(h, d1) { + t.Fatalf("stored node disagrees with ComputeBucketHash: count %d vs %d", c1, c) + } +} + +// TestMerkleRebuildClearsStaleNodes verifies the leading DeleteRange in +// the build: a rebuild at a shallower depth must remove the deeper +// levels of the prior build, or the comparison descent would read them. +func TestMerkleRebuildClearsStaleNodes(t *testing.T) { + ctx := context.Background() + e, _ := newTestEngine(t) + grants := make([]*v3.GrantRecord, 0, 40) + for i := 0; i < 40; i++ { + grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) + } + syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatal(err) + } + + level2, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, nil) + if err != nil { + t.Fatal(err) + } + if len(level2) == 0 { + t.Fatal("depth-2 build produced no level-2 nodes") + } + rootBefore, _, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + if err != nil { + t.Fatal(err) + } + + if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 1); err != nil { + t.Fatalf("rebuild at depth 1: %v", err) + } + + rootAfter, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + if err != nil || !ok { + t.Fatalf("root after rebuild: ok=%v err=%v", ok, err) + } + if rootAfter.Depth != 1 { + t.Fatalf("root depth after rebuild = %d, want 1", rootAfter.Depth) + } + // Depth-independence: same content, same root digest. + if !bytes.Equal(rootBefore.Hash, rootAfter.Hash) || rootBefore.Count != rootAfter.Count { + t.Fatal("rebuild at different depth changed the root digest/count over identical content") + } + level2After, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, nil) + if err != nil { + t.Fatal(err) + } + if len(level2After) != 0 { + t.Fatalf("%d stale level-2 nodes survived the depth-1 rebuild", len(level2After)) + } +} + +// TestMerkleIncrementalEqualsRebuild is the §7 keystone invariant: after +// a sequence of post-seal inserts, content overwrites, a bucket-moving +// (principal-changing) overwrite, an excluded-field no-op overwrite, +// deletes, and a multi-record batch, the incrementally-maintained tree +// byte-equals a from-scratch rebuild. +func TestMerkleIncrementalEqualsRebuild(t *testing.T) { + ctx := context.Background() + e, _ := newTestEngine(t) + grants := make([]*v3.GrantRecord, 0, 30) + for i := 0; i < 30; i++ { + grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) + } + syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatal(err) + } + + put := func(g *v3.GrantRecord) { + t.Helper() + g.SetSyncId(syncID) + if err := e.PutGrantRecord(ctx, g); err != nil { + t.Fatalf("PutGrantRecord: %v", err) + } + } + + // Post-seal inserts. + put(makeGrant(syncID, "g-100", "ent-A", "new-user-1")) + put(makeGrant(syncID, "g-101", "ent-A", "new-user-2")) + // Content overwrite: same index key, sources changed. + put(makeGrantWithSources(syncID, "g-005", "ent-A", "user-005", "src-ent")) + // Bucket-moving overwrite: same external_id, principal changed — + // must apply as remove(old path) + add(new path). + put(makeGrant(syncID, "g-006", "ent-A", "user-moved")) + // Excluded-field overwrite: needs_expansion is not part of the + // content hash, so this must leave the tree untouched. + noop := makeGrant(syncID, "g-007", "ent-A", "user-007") + noop.SetNeedsExpansion(true) + put(noop) + // Deletes. + for _, ext := range []string{"g-008", "g-009"} { + if err := e.DeleteGrantRecord(ctx, syncID, ext); err != nil { + t.Fatalf("DeleteGrantRecord(%s): %v", ext, err) + } + } + // Multi-record batch: inserts + an overwrite in one PutGrantRecords + // call, exercising the per-node delta accumulator (all of them + // share the root). + batch := []*v3.GrantRecord{ + makeGrant(syncID, "g-110", "ent-A", "batch-user-1"), + makeGrant(syncID, "g-111", "ent-A", "batch-user-2"), + makeGrant(syncID, "g-112", "ent-A", "batch-user-3"), + makeGrantWithSources(syncID, "g-010", "ent-A", "user-010", "src-2"), + } + if err := e.PutGrantRecords(ctx, batch...); err != nil { + t.Fatalf("PutGrantRecords: %v", err) + } + + // Sparsity restored on delete: unless another remaining principal + // shares user-008's depth-2 prefix, its leaf must be gone. + remaining := []string{"user-moved", "new-user-1", "new-user-2", "batch-user-1", "batch-user-2", "batch-user-3"} + for i := 0; i < 30; i++ { + if i == 6 || i == 8 || i == 9 { + continue // moved or deleted + } + remaining = append(remaining, fmt.Sprintf("user-%03d", i)) + } + deletedPrefix := principalBucketHash("user", "user-008")[:2] + shared := false + for _, p := range remaining { + if bytes.Equal(principalBucketHash("user", p)[:2], deletedPrefix) { + shared = true + break + } + } + if !shared { + _, _, present, err := e.getMerkleNode(idBytes, "ent-A", 2, deletedPrefix) + if err != nil { + t.Fatal(err) + } + if present { + t.Fatal("emptied leaf node survived an incremental delete; sparsity not restored") + } + } + + incremental := dumpMerkleNodes(t, e, syncID) + if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 2); err != nil { + t.Fatalf("rebuild: %v", err) + } + rebuilt := dumpMerkleNodes(t, e, syncID) + requireSameMerkleNodes(t, incremental, rebuilt) +} + +// TestMerkleSameBatchSameBucketRMW pins the accumulator behavior: one +// PutGrantRecords batch adds several grants that land in the SAME +// depth-1 bucket. A naive read-modify-write against the batch would +// lose all but one delta (a plain pebble.Batch doesn't read through its +// own writes); the accumulator must fold all of them into one node +// write. +func TestMerkleSameBatchSameBucketRMW(t *testing.T) { + ctx := context.Background() + e, _ := newTestEngine(t) + + // Find three principals whose bucket hashes share a first byte. + collide := []string{"seed-principal"} + target := principalBucketHash("user", collide[0])[0] + for i := 0; len(collide) < 3; i++ { + p := fmt.Sprintf("cand-%d", i) + if principalBucketHash("user", p)[0] == target { + collide = append(collide, p) + } + } + + base := make([]*v3.GrantRecord, 0, 5) + for i := 0; i < 5; i++ { + base = append(base, makeGrant("", fmt.Sprintf("b-%d", i), "ent-A", fmt.Sprintf("base-%d", i))) + } + syncID := seedEntitlementAtDepth(t, e, "ent-A", base, 1) + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + t.Fatal(err) + } + + gs := make([]*v3.GrantRecord, 0, len(collide)) + for i, p := range collide { + gs = append(gs, makeGrant(syncID, fmt.Sprintf("x-%d", i), "ent-A", p)) + } + if err := e.PutGrantRecords(ctx, gs...); err != nil { + t.Fatalf("PutGrantRecords: %v", err) + } + + want := int64(len(collide)) + for i := 0; i < 5; i++ { + if principalBucketHash("user", fmt.Sprintf("base-%d", i))[0] == target { + want++ + } + } + count, _, present, err := e.getMerkleNode(idBytes, "ent-A", 1, []byte{target}) + if err != nil || !present { + t.Fatalf("bucket node %x: present=%v err=%v", target, present, err) + } + if count != want { + t.Fatalf("same-batch deltas lost: bucket count = %d, want %d", count, want) + } + + incremental := dumpMerkleNodes(t, e, syncID) + if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 1); err != nil { + t.Fatalf("rebuild: %v", err) + } + requireSameMerkleNodes(t, incremental, dumpMerkleNodes(t, e, syncID)) +} + +// TestMerkleMissingRootFallback: a missing root means "tree never +// built", not "no grants". Comparison against a populated-but-unbuilt +// side must fall back to the authoritative fold — clean when content is +// identical, whole-entitlement dirty when it differs. +func TestMerkleMissingRootFallback(t *testing.T) { + ctx := context.Background() + mk := func() []*v3.GrantRecord { + return []*v3.GrantRecord{ + makeGrant("", "g1", "ent-A", "alice"), + makeGrant("", "g2", "ent-A", "bob"), + makeGrant("", "g3", "ent-A", "carol"), + } + } + ea, _ := newTestEngine(t) + syncA := seedEntitlement(t, ea, "ent-A", mk()) + + // B holds the same grants but never builds a tree. + eb, _ := newTestEngine(t) + syncB := ksuid.New().String() + if err := eb.SetCurrentSync(syncB); err != nil { + t.Fatal(err) + } + putEnt(t, eb, ctx, syncB, "ent-A") + for _, g := range mk() { + g.SetSyncId(syncB) + if err := eb.PutGrantRecord(ctx, g); err != nil { + t.Fatal(err) + } + } + if _, ok, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A"); err != nil || ok { + t.Fatalf("B unexpectedly has a root: ok=%v err=%v", ok, err) + } + + for name, dirtyFn := range map[string]func() ([][]byte, error){ + "A vs B": func() ([][]byte, error) { return ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") }, + "B vs A": func() ([][]byte, error) { return eb.DirtyEntitlementBuckets(ctx, syncB, ea, syncA, "ent-A") }, + } { + dirty, err := dirtyFn() + if err != nil { + t.Fatalf("%s: %v", name, err) + } + if len(dirty) != 0 { + t.Fatalf("%s: identical content with one tree unbuilt: dirty=%d, want 0 (fold fallback)", name, len(dirty)) + } + } + + // Diverge B; the fallback must now flag the whole entitlement. + if err := eb.PutGrantRecord(ctx, makeGrant(syncB, "g9", "ent-A", "zed")); err != nil { + t.Fatal(err) + } + dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") + if err != nil { + t.Fatal(err) + } + if len(dirty) != 1 || len(dirty[0]) != 0 { + t.Fatalf("diverged content with one tree unbuilt: dirty=%v, want one empty prefix", dirty) + } +} From f7c82f5471a54d9fb57b3e76d4d82eb80ed71a0d Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Wed, 10 Jun 2026 17:54:54 -0700 Subject: [PATCH 3/7] use trie for diffing --- pkg/dotc1z/engine/pebble/adapter_diff.go | 99 +++++++- .../engine/pebble/adapter_diff_trie_test.go | 202 ++++++++++++++++ .../engine/pebble/diff_grants_bench_test.go | 224 ++++++++++++++++++ pkg/dotc1z/engine/pebble/merkle.go | 56 ++++- 4 files changed, 568 insertions(+), 13 deletions(-) create mode 100644 pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go create mode 100644 pkg/dotc1z/engine/pebble/diff_grants_bench_test.go diff --git a/pkg/dotc1z/engine/pebble/adapter_diff.go b/pkg/dotc1z/engine/pebble/adapter_diff.go index 8b6a6530a..ee8bc7922 100644 --- a/pkg/dotc1z/engine/pebble/adapter_diff.go +++ b/pkg/dotc1z/engine/pebble/adapter_diff.go @@ -21,13 +21,17 @@ import ( // deletions are NOT captured — that matches the SQLite behavior // in pkg/dotc1z/diff.go (additions-only diff). // -// Strategy: for each record-type-bearing keyspace under -// appliedSyncID, iterate the source keys, recompute the same -// record's primary key under baseSyncID, Get from base; if base -// returns ErrNotFound, write the value under diffSyncID's -// keyspace in the same record type. Index entries are -// recomputed on write so the diff sync has its own (fresh) -// indexes that match the records that landed. +// Strategy: for the small record types (resource_types, resources, +// entitlements, assets), iterate the source keys under appliedSyncID, +// recompute the same record's primary key under baseSyncID, Get from +// base; if base returns ErrNotFound, write the value under +// diffSyncID's keyspace in the same record type. GRANTS — the type +// that dominates every real file — are diffed via the per-entitlement +// hash tries instead (see diffGrants): unchanged entitlements are +// skipped with a single root comparison, and only the principal-hash +// buckets that actually differ are scanned. Index entries are +// recomputed on write so the diff sync has its own (fresh) indexes +// that match the records that landed. // // Returns the diff sync's ID. func generateSyncDiff(ctx context.Context, a *Adapter, baseSyncID, appliedSyncID string) (string, error) { @@ -95,7 +99,7 @@ func generateSyncDiff(ctx context.Context, a *Adapter, baseSyncID, appliedSyncID if err := diffEntitlements(ctx, a, baseBytes, appliedBytes, diffSyncID); err != nil { return "", fmt.Errorf("generate-diff: entitlements: %w", err) } - if err := diffGrants(ctx, a, baseBytes, appliedBytes, diffSyncID); err != nil { + if err := diffGrants(ctx, a, baseBytes, appliedBytes, baseSyncID, appliedSyncID, diffSyncID); err != nil { return "", fmt.Errorf("generate-diff: grants: %w", err) } if err := diffAssets(ctx, a, baseBytes, appliedBytes, diffSyncID); err != nil { @@ -182,7 +186,84 @@ func diffEntitlements(ctx context.Context, a *Adapter, baseBytes, appliedBytes [ }) } -func diffGrants(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, diffSyncID string) error { +// diffGrants computes the grants set difference using the +// per-entitlement hash tries instead of scanning every applied grant. +// +// Walk: enumerate the distinct entitlement_ids in the applied sync's +// hash index (one seek each), compare each entitlement's trie between +// base and applied — one root read per side when nothing changed, the +// overwhelmingly common case — and materialize only the principal-hash +// buckets the comparison flags as dirty. Grants in dirty buckets are +// probed against base's PRIMARY keyspace by external_id, which +// preserves the additions-only contract exactly: a grant whose +// external_id exists in base under a different entitlement/principal +// is still "present in base" (not emitted), which is why the probe +// targets the primary key rather than comparing index keys. +// +// Why pruning is sound: equal (count, digest) node pairs mean the two +// sides hold identical grant content-hash sets, and the content hash +// folds external_id — so a clean entitlement/bucket cannot contain an +// applied external_id that base lacks. Dirtiness over-approximates +// additions (it also fires for removals and source-set changes, which +// the base probe then filters out), never under-approximates them. +// +// NOTE: grants without an entitlement or principal ref have no +// hash-index entry and are invisible to the trie. They are silently +// skipped. A future O(1) coverage check (e.g. a stored grant count in +// the sync run record) will restore detection of this case. +func diffGrants(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, baseSyncID, appliedSyncID, diffSyncID string) error { + eng := a.engine + + ents, err := eng.distinctHashIndexEntitlements(ctx, appliedBytes) + if err != nil { + return err + } + for _, ent := range ents { + if err := ctx.Err(); err != nil { + return err + } + // Entitlements only present in base (fully removed) are not in + // `ents` and are correctly skipped: they cannot contain + // additions. Tries missing on either side (e.g. a ghost + // entitlement with grants but no entitlement record) degrade + // to an on-demand index fold inside DirtyEntitlementBuckets. + dirty, err := eng.DirtyEntitlementBuckets(ctx, baseSyncID, eng, appliedSyncID, ent) + if err != nil { + return err + } + for _, prefix := range dirty { + var innerErr error + err := eng.IterateGrantsByEntitlementBucket(ctx, appliedSyncID, ent, prefix, func(rec *v3.GrantRecord) bool { + exists, probeErr := existsAt(eng.DB(), encodeGrantKey(baseBytes, rec.GetExternalId())) + if probeErr != nil { + innerErr = probeErr + return false + } + if exists { + return true + } + rec.SetSyncId(diffSyncID) + if putErr := eng.PutGrantRecord(ctx, rec); putErr != nil { + innerErr = putErr + return false + } + return true + }) + if err != nil { + return err + } + if innerErr != nil { + return innerErr + } + } + } + return nil +} + +// diffGrantsFullScan is the O(applied grants) path: iterate every +// applied grant, probe base by external_id, emit on miss. Used +// directly by benchmarks and available as a correctness reference. +func diffGrantsFullScan(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, diffSyncID string) error { srcPrefix := encodeGrantPrefix(appliedBytes) return iterDiff(ctx, a.engine.DB(), srcPrefix, upperBoundOf(srcPrefix), func(_ []byte, val []byte) error { var rec v3.GrantRecord diff --git a/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go b/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go new file mode 100644 index 000000000..cd1c59c5c --- /dev/null +++ b/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go @@ -0,0 +1,202 @@ +package pebble + +import ( + "context" + "testing" + + "github.com/cockroachdb/pebble/v2" + + v2 "github.com/conductorone/baton-sdk/pb/c1/connector/v2" + v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" + "github.com/conductorone/baton-sdk/pkg/connectorstore" +) + +func mkV2Ent(id string) *v2.Entitlement { + return v2.Entitlement_builder{ + Id: id, + Resource: v2.Resource_builder{ + Id: v2.ResourceId_builder{ + ResourceType: "app", + Resource: "github", + }.Build(), + }.Build(), + }.Build() +} + +// runSync starts a sync, writes the given entitlements + grants, and +// seals it (EndSync builds the per-entitlement tries). Returns the +// sync id. +func runSync(t *testing.T, a *Adapter, ents []*v2.Entitlement, grants []*v2.Grant) string { + t.Helper() + ctx := context.Background() + syncID, err := a.StartNewSync(ctx, connectorstore.SyncTypeFull, "") + if err != nil { + t.Fatalf("StartNewSync: %v", err) + } + if len(ents) > 0 { + if err := a.PutEntitlements(ctx, ents...); err != nil { + t.Fatalf("PutEntitlements: %v", err) + } + } + if len(grants) > 0 { + if err := a.PutGrants(ctx, grants...); err != nil { + t.Fatalf("PutGrants: %v", err) + } + } + if err := a.EndSync(ctx); err != nil { + t.Fatalf("EndSync: %v", err) + } + return syncID +} + +// diffGrantIDs runs generateSyncDiff and returns the set of grant +// external_ids that landed in the diff sync. +func diffGrantIDs(t *testing.T, a *Adapter, baseSyncID, appliedSyncID string) map[string]bool { + t.Helper() + ctx := context.Background() + diffID, err := generateSyncDiff(ctx, a, baseSyncID, appliedSyncID) + if err != nil { + t.Fatalf("generateSyncDiff: %v", err) + } + got := map[string]bool{} + if err := a.engine.IterateGrantsBySync(ctx, diffID, func(r *v3.GrantRecord) bool { + got[r.GetExternalId()] = true + return true + }); err != nil { + t.Fatalf("IterateGrantsBySync(diff): %v", err) + } + return got +} + +func requireDiffIDs(t *testing.T, got map[string]bool, want ...string) { + t.Helper() + wantSet := map[string]bool{} + for _, id := range want { + wantSet[id] = true + if !got[id] { + t.Errorf("diff missing expected grant %q", id) + } + } + for id := range got { + if !wantSet[id] { + t.Errorf("diff contains unexpected grant %q", id) + } + } +} + +// TestGenerateSyncDiffTrieMultiEntitlement exercises the trie-driven +// grants diff across the full semantic matrix in one pair of syncs: +// unchanged entitlements (pruned at the root), an addition inside an +// existing entitlement, an addition under a brand-new entitlement, a +// removal (must not be emitted), and a grant that MOVED to a different +// entitlement while keeping its external_id (dirty bucket on both +// entitlements, but per the additions-only-by-external_id contract it +// must not be emitted). +func TestGenerateSyncDiffTrieMultiEntitlement(t *testing.T) { + a := newAdapter(t) + + base := runSync(t, a, + []*v2.Entitlement{mkV2Ent("ent-A"), mkV2Ent("ent-B"), mkV2Ent("ent-C")}, + []*v2.Grant{ + mkV2Grant("g1", "ent-A", "user", "alice"), + mkV2Grant("g2", "ent-A", "user", "bob"), + mkV2Grant("g3", "ent-B", "user", "carol"), + mkV2Grant("g4", "ent-B", "user", "dave"), + mkV2Grant("g5", "ent-C", "user", "eve"), + mkV2Grant("g6", "ent-A", "user", "frank"), + }) + applied := runSync(t, a, + []*v2.Entitlement{mkV2Ent("ent-A"), mkV2Ent("ent-B"), mkV2Ent("ent-C"), mkV2Ent("ent-D")}, + []*v2.Grant{ + mkV2Grant("g1", "ent-A", "user", "alice"), // unchanged + mkV2Grant("g2", "ent-A", "user", "bob"), // unchanged + mkV2Grant("g3", "ent-B", "user", "carol"), // unchanged + mkV2Grant("g4", "ent-B", "user", "dave"), // unchanged + // g5 removed (ent-C empty in applied) — removals not emitted. + mkV2Grant("g6", "ent-B", "user", "frank"), // moved ent-A→ent-B, same external_id — NOT an addition + mkV2Grant("g7", "ent-B", "user", "grace"), // addition in an existing entitlement + mkV2Grant("g8", "ent-D", "user", "heidi"), // addition under a new entitlement + }) + + got := diffGrantIDs(t, a, base, applied) + requireDiffIDs(t, got, "g7", "g8") +} + +// TestGenerateSyncDiffOrphanGrantSkipped documents that a grant +// without entitlement/principal refs has no hash-index entry and is +// invisible to the trie-driven diff path. It is silently skipped. +// Normal grants in the same sync (g2) are still emitted correctly. +// TODO: restore detection via an O(1) coverage check (e.g. a stored +// grant count in the sync run record). +func TestGenerateSyncDiffOrphanGrantSkipped(t *testing.T) { + ctx := context.Background() + a := newAdapter(t) + + base := runSync(t, a, + []*v2.Entitlement{mkV2Ent("ent-A")}, + []*v2.Grant{mkV2Grant("g1", "ent-A", "user", "alice")}) + + applied, err := a.StartNewSync(ctx, connectorstore.SyncTypeFull, "") + if err != nil { + t.Fatalf("StartNewSync: %v", err) + } + if err := a.PutEntitlements(ctx, mkV2Ent("ent-A")); err != nil { + t.Fatalf("PutEntitlements: %v", err) + } + if err := a.PutGrants(ctx, + mkV2Grant("g1", "ent-A", "user", "alice"), + mkV2Grant("g2", "ent-A", "user", "bob"), + ); err != nil { + t.Fatalf("PutGrants: %v", err) + } + orphan := v3.GrantRecord_builder{ + SyncId: applied, + ExternalId: "g-orphan", + }.Build() + if err := a.engine.PutGrantRecord(ctx, orphan); err != nil { + t.Fatalf("PutGrantRecord(orphan): %v", err) + } + if err := a.EndSync(ctx); err != nil { + t.Fatalf("EndSync: %v", err) + } + + got := diffGrantIDs(t, a, base, applied) + // g-orphan has no hash-index entry — silently skipped by trie diff. + requireDiffIDs(t, got, "g2") +} + +// TestGenerateSyncDiffTrieWithoutTrees simulates a file whose tries +// are missing (e.g. written before they existed and not yet +// backfilled): the hash index is intact, so the trie path runs, and +// DirtyEntitlementBuckets degrades to the on-demand index fold per +// entitlement. The diff must still be exact. +func TestGenerateSyncDiffTrieWithoutTrees(t *testing.T) { + a := newAdapter(t) + + base := runSync(t, a, + []*v2.Entitlement{mkV2Ent("ent-A"), mkV2Ent("ent-B")}, + []*v2.Grant{ + mkV2Grant("g1", "ent-A", "user", "alice"), + mkV2Grant("g2", "ent-B", "user", "bob"), + }) + applied := runSync(t, a, + []*v2.Entitlement{mkV2Ent("ent-A"), mkV2Ent("ent-B")}, + []*v2.Grant{ + mkV2Grant("g1", "ent-A", "user", "alice"), + mkV2Grant("g2", "ent-B", "user", "bob"), + mkV2Grant("g3", "ent-B", "user", "carol"), + }) + + for _, sid := range []string{base, applied} { + idBytes, err := a.engine.resolveSyncBytes(sid) + if err != nil { + t.Fatal(err) + } + if err := a.engine.db.DeleteRange(MerkleSyncLowerBound(idBytes), MerkleSyncUpperBound(idBytes), pebble.Sync); err != nil { + t.Fatalf("DeleteRange(merkle %s): %v", sid, err) + } + } + + got := diffGrantIDs(t, a, base, applied) + requireDiffIDs(t, got, "g3") +} diff --git a/pkg/dotc1z/engine/pebble/diff_grants_bench_test.go b/pkg/dotc1z/engine/pebble/diff_grants_bench_test.go new file mode 100644 index 000000000..6820bf465 --- /dev/null +++ b/pkg/dotc1z/engine/pebble/diff_grants_bench_test.go @@ -0,0 +1,224 @@ +package pebble + +import ( + "context" + "fmt" + "os" + "testing" + + "github.com/segmentio/ksuid" + + v2 "github.com/conductorone/baton-sdk/pb/c1/connector/v2" + "github.com/conductorone/baton-sdk/pkg/connectorstore" + "github.com/conductorone/baton-sdk/pkg/dotc1z/engine/pebble/codec" +) + +// Diff-strategy benchmark — trie (diffGrants) vs full-scan (diffGrantsFullScan). +// +// Fixture shape: N entitlements × M grants in base, plus a small addition set in +// applied (diffBenchDirtyEnts existing ents each gain diffBenchAddPerDirty grants, +// plus diffBenchNewEnts brand-new entitlements). The fixture is scale-independent: +// only N×M changes between scale levels; the addition count stays fixed. +// +// Scales (grant counts in base): +// +// 10K — 50 ents × 200 grants, depth-0 trees. Always runs (also with -short). +// 1M — 1 000 ents × 1 000 grants, depth-1 trees. Default. +// 50M — 5 000 ents × 10 000 grants, depth-1 trees. Set BATONSDK_BENCH_DIFF_LONG=1. +// WARNING: seeding 2×50M grants takes ~20 min. Run overnight or on CI. +// +// Run examples: +// +// # default (1M) +// go test -run=^$ -bench=BenchmarkDiffGrants -benchtime=1x ./pkg/dotc1z/engine/pebble +// +// # quick smoke (10K) +// go test -run=^$ -bench=BenchmarkDiffGrants -benchtime=1x -short ./pkg/dotc1z/engine/pebble +// +// # full suite including 50M +// BATONSDK_BENCH_DIFF_LONG=1 go test -run=^$ -bench=BenchmarkDiffGrants -benchtime=1x \ +// ./pkg/dotc1z/engine/pebble +const ( + // additions injected into applied — same count at every scale. + diffBenchDirtyEnts = 5 // existing entitlements that gain new grants in applied + diffBenchAddPerDirty = 20 // new grants per dirty entitlement + diffBenchNewEnts = 5 // brand-new entitlements present only in applied + diffBenchGrantsNewEnt = 20 // grants per new entitlement + + diffBenchTotalAdded = diffBenchDirtyEnts*diffBenchAddPerDirty + + diffBenchNewEnts*diffBenchGrantsNewEnt // = 200 +) + +type diffScale struct { + tag string + ents int + grantsPerEnt int +} + +// diffBenchScales returns the scale levels to run. +// - -short → 10K only (fast smoke). +// - default → 1M. +// - BATONSDK_BENCH_DIFF_LONG=1 → 1M + 50M. +func diffBenchScales() []diffScale { + if testing.Short() { + return []diffScale{{tag: "10K", ents: 50, grantsPerEnt: 200}} + } + scales := []diffScale{{tag: "1M", ents: 1_000, grantsPerEnt: 1_000}} + if os.Getenv("BATONSDK_BENCH_DIFF_LONG") != "" { + scales = append(scales, diffScale{tag: "50M", ents: 5_000, grantsPerEnt: 10_000}) + } + return scales +} + +// seedGrantDiffBench builds the base and applied syncs. Seeding is the +// expensive part; call this once per benchmark sub-function before the loop. +// +// Grants are PUT entitlement-by-entitlement to keep peak memory per call at +// O(grantsPerEnt) rather than O(ents×grantsPerEnt). +// newAdapterNoSync opens an adapter with DurabilityNoSync so the 200 +// per-diff grant writes don't each pay a WAL fsync. +func newAdapterNoSync(t testing.TB) *Adapter { + t.Helper() + e, _ := newTestEngine(t, WithDurability(DurabilityNoSync)) + return NewAdapter(e) +} + +func seedGrantDiffBench(b *testing.B, a *Adapter, ents, grantsPerEnt int) (baseSyncID, appliedSyncID string) { + b.Helper() + ctx := context.Background() + + baseEnts := make([]*v2.Entitlement, ents) + for i := range baseEnts { + baseEnts[i] = mkV2Ent(fmt.Sprintf("ent-%04d", i)) + } + + // Base sync ---------------------------------------------------------------- + var err error + baseSyncID, err = a.StartNewSync(ctx, connectorstore.SyncTypeFull, "") + if err != nil { + b.Fatalf("StartNewSync(base): %v", err) + } + if err := a.PutEntitlements(ctx, baseEnts...); err != nil { + b.Fatalf("PutEntitlements(base): %v", err) + } + for e := range ents { + entID := fmt.Sprintf("ent-%04d", e) + batch := make([]*v2.Grant, grantsPerEnt) + for g := range batch { + batch[g] = mkV2Grant( + fmt.Sprintf("%s:g%06d", entID, g), + entID, "user", fmt.Sprintf("u%08d", g), + ) + } + if err := a.PutGrants(ctx, batch...); err != nil { + b.Fatalf("PutGrants(base ent %d): %v", e, err) + } + } + if err := a.EndSync(ctx); err != nil { + b.Fatalf("EndSync(base): %v", err) + } + + // Applied sync ------------------------------------------------------------- + appliedSyncID, err = a.StartNewSync(ctx, connectorstore.SyncTypeFull, "") + if err != nil { + b.Fatalf("StartNewSync(applied): %v", err) + } + appliedEnts := make([]*v2.Entitlement, 0, ents+diffBenchNewEnts) + appliedEnts = append(appliedEnts, baseEnts...) + for i := range diffBenchNewEnts { + appliedEnts = append(appliedEnts, mkV2Ent(fmt.Sprintf("new-ent-%04d", i))) + } + if err := a.PutEntitlements(ctx, appliedEnts...); err != nil { + b.Fatalf("PutEntitlements(applied): %v", err) + } + + // Existing entitlements: identical to base, plus additions in the first few. + for e := range ents { + entID := fmt.Sprintf("ent-%04d", e) + cap := grantsPerEnt + if e < diffBenchDirtyEnts { + cap += diffBenchAddPerDirty + } + batch := make([]*v2.Grant, 0, cap) + for g := range grantsPerEnt { + batch = append(batch, mkV2Grant( + fmt.Sprintf("%s:g%06d", entID, g), + entID, "user", fmt.Sprintf("u%08d", g), + )) + } + if e < diffBenchDirtyEnts { + for g := range diffBenchAddPerDirty { + batch = append(batch, mkV2Grant( + fmt.Sprintf("%s:added%04d", entID, g), + entID, "user", fmt.Sprintf("added-u%d-%08d", e, g), + )) + } + } + if err := a.PutGrants(ctx, batch...); err != nil { + b.Fatalf("PutGrants(applied ent %d): %v", e, err) + } + } + + // Brand-new entitlements present only in applied. + for i := range diffBenchNewEnts { + entID := fmt.Sprintf("new-ent-%04d", i) + batch := make([]*v2.Grant, diffBenchGrantsNewEnt) + for g := range batch { + batch[g] = mkV2Grant( + fmt.Sprintf("%s:g%06d", entID, g), + entID, "user", fmt.Sprintf("new-ent-u%d-%08d", i, g), + ) + } + if err := a.PutGrants(ctx, batch...); err != nil { + b.Fatalf("PutGrants(applied new-ent %d): %v", i, err) + } + } + + if err := a.EndSync(ctx); err != nil { + b.Fatalf("EndSync(applied): %v", err) + } + return baseSyncID, appliedSyncID +} + +// runDiffGrantsBench is the shared body for the two diff benchmarks. +func runDiffGrantsBench(b *testing.B, diffFn func(context.Context, *Adapter, []byte, []byte, string, string, string) error) { + b.Helper() + for _, sc := range diffBenchScales() { + b.Run(sc.tag, func(b *testing.B) { + a := newAdapterNoSync(b) + ctx := context.Background() + baseSyncID, appliedSyncID := seedGrantDiffBench(b, a, sc.ents, sc.grantsPerEnt) + + baseBytes, err := codec.EncodeSyncID(baseSyncID) + if err != nil { + b.Fatalf("encode base: %v", err) + } + appliedBytes, err := codec.EncodeSyncID(appliedSyncID) + if err != nil { + b.Fatalf("encode applied: %v", err) + } + + b.ReportMetric(float64(sc.ents*sc.grantsPerEnt), "base_grants") + b.ReportMetric(float64(diffBenchTotalAdded), "expected_additions") + + for b.Loop() { + diffID := ksuid.New().String() + if err := diffFn(ctx, a, baseBytes, appliedBytes, baseSyncID, appliedSyncID, diffID); err != nil { + b.Fatalf("diff: %v", err) + } + } + }) + } +} + +// fullScanAdapter wraps diffGrantsFullScan to match the diffGrants signature. +func fullScanAdapter(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, _, _, diffID string) error { + return diffGrantsFullScan(ctx, a, baseBytes, appliedBytes, diffID) +} + +// BenchmarkDiffGrants_Trie / BenchmarkDiffGrants_FullScan use DurabilityNoSync +// so that the 200 per-diff grant writes don't each pay a WAL fsync. The writes +// still happen (churn is present), but sync latency is excluded from the +// measurement. This isolates the read/comparison cost of each strategy. +func BenchmarkDiffGrants_Trie(b *testing.B) { runDiffGrantsBench(b, diffGrants) } +func BenchmarkDiffGrants_FullScan(b *testing.B) { runDiffGrantsBench(b, fullScanAdapter) } diff --git a/pkg/dotc1z/engine/pebble/merkle.go b/pkg/dotc1z/engine/pebble/merkle.go index 13861f018..9bc075426 100644 --- a/pkg/dotc1z/engine/pebble/merkle.go +++ b/pkg/dotc1z/engine/pebble/merkle.go @@ -13,6 +13,7 @@ import ( "github.com/cockroachdb/pebble/v2" v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" + "github.com/conductorone/baton-sdk/pkg/dotc1z/engine/pebble/codec" ) // Per-entitlement grant merkle tree (XOR combiner). @@ -286,7 +287,7 @@ func (e *Engine) BuildAllMerkleTrees(ctx context.Context, syncID string) error { // buildEntitlementMerkleAtDepth (pass 2). func (e *Engine) buildEntitlementMerkle(ctx context.Context, idBytes []byte, entitlementID string) error { entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) - count, err := e.countHashIndexRange(entPrefix, upperBoundOf(entPrefix)) + count, err := e.countKeysInRange(entPrefix, upperBoundOf(entPrefix)) if err != nil { return err } @@ -414,9 +415,10 @@ func (e *Engine) buildEntitlementMerkleAtDepth(ctx context.Context, idBytes []by }) } -// countHashIndexRange counts index entries in [lower, upper) without -// materializing values. -func (e *Engine) countHashIndexRange(lower, upper []byte) (int64, error) { +// countKeysInRange counts keys in [lower, upper) without materializing +// values. Used by the build's count→depth pass and by the diff driver's +// primary-vs-index coverage guard. +func (e *Engine) countKeysInRange(lower, upper []byte) (int64, error) { iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) if err != nil { return 0, err @@ -429,6 +431,52 @@ func (e *Engine) countHashIndexRange(lower, upper []byte) (int64, error) { return n, iter.Error() } +// distinctHashIndexEntitlements returns the distinct entitlement_ids +// present in a sync's by_entitlement_principal_hash index, in index +// order. It seeks past each entitlement's whole range once its id is +// captured, so the cost is O(entitlements) seeks, not O(grants). This +// is the diff driver's work list, and it deliberately comes from the +// INDEX rather than the entitlement records: a grant whose entitlement +// has no record still has index entries (and a fold), so it still gets +// compared — only a tree was never built for it. +func (e *Engine) distinctHashIndexEntitlements(ctx context.Context, syncIDBytes []byte) ([]string, error) { + lower := GrantByEntPrincHashSyncLowerBound(syncIDBytes) + upper := GrantByEntPrincHashSyncUpperBound(syncIDBytes) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return nil, err + } + defer iter.Close() + + // Keys are lower | 0x00 | tuple(entitlement_id) | 0x00 | bucketHash | … + entStart := len(lower) + 1 + var out []string + for iter.First(); iter.Valid(); { + if err := ctx.Err(); err != nil { + return nil, err + } + key := iter.Key() + if len(key) <= entStart { + iter.Next() + continue // malformed; skip defensively + } + entBytes, _, decErr := codec.DecodeTupleStringTo(nil, key[entStart:], 0) + if decErr != nil { + iter.Next() + continue + } + ent := string(entBytes) + out = append(out, ent) + // Seek past this entitlement's whole index range. + seekTo := upperBoundOf(encodeGrantByEntPrincHashEntPrefix(syncIDBytes, ent)) + if seekTo == nil { + break + } + iter.SeekGE(seekTo) + } + return out, iter.Error() +} + // MerkleRoot is the result of GetEntitlementMerkleRoot. type MerkleRoot struct { Hash []byte From 24d67991ac96d7708ea684ccc180e513e4c66d00 Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Thu, 11 Jun 2026 15:04:17 -0700 Subject: [PATCH 4/7] migration can chill, this isn't in yeta --- pkg/dotc1z/engine/pebble/index_migrations.go | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/pkg/dotc1z/engine/pebble/index_migrations.go b/pkg/dotc1z/engine/pebble/index_migrations.go index 2aaccf092..f6395878b 100644 --- a/pkg/dotc1z/engine/pebble/index_migrations.go +++ b/pkg/dotc1z/engine/pebble/index_migrations.go @@ -69,19 +69,16 @@ var indexMigrations = []indexMigration{ // per-entitlement merkle trees for files written before either // existed. Idempotent: re-emitting an index entry is a Set over // the same key/value, and each tree rebuild range-clears the - // entitlement's typeMerkle keyspace before writing — which is - // also what makes the version bump effective: v1 (sha256-fold, - // root+leaves) nodes are byte-length-identical to v2 (XOR, - // all-levels) nodes, so only the clear removes them. + // entitlement's typeMerkle keyspace before writing. // New files persist this version at their initial (empty) Open, // so the inline write path maintains both and the backfill never // re-runs for them. // - // v2: XOR combiner, all levels stored sparsely, count on every - // node (RFC 0003). Pre-GA in-place change; no production v3 - // data existed at v1. + // v1: XOR combiner, all levels stored sparsely, count on every + // node (RFC 0003). Uses xxHash64 (8-byte digests) for both + // principalBucketHash and grantContentHash. Name: "grant_by_entitlement_principal_hash", - Version: 2, + Version: 1, Apply: func(ctx context.Context, e *Engine) error { return e.backfillGrantHashIndexAndMerkle(ctx) }, From a7277c56b302fc124bcfca2a7698f7208ff3d92e Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Thu, 11 Jun 2026 15:04:52 -0700 Subject: [PATCH 5/7] use xxhash64 --- pkg/dotc1z/engine/pebble/merkle.go | 63 +++++++++++++----------------- 1 file changed, 28 insertions(+), 35 deletions(-) diff --git a/pkg/dotc1z/engine/pebble/merkle.go b/pkg/dotc1z/engine/pebble/merkle.go index 9bc075426..5946dfdca 100644 --- a/pkg/dotc1z/engine/pebble/merkle.go +++ b/pkg/dotc1z/engine/pebble/merkle.go @@ -3,13 +3,12 @@ package pebble import ( "bytes" "context" - "crypto/sha256" "encoding/binary" "errors" "fmt" - "hash" "sort" + "github.com/cespare/xxhash/v2" "github.com/cockroachdb/pebble/v2" v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" @@ -34,8 +33,8 @@ import ( // bucket P" is a contiguous range scan. // // Combiner. A node's digest is the XOR of every grant content hash -// beneath it (leaves of the fold stay sha256 — only the combiner is -// XOR). XOR is homomorphic (parent = XOR of children), order-independent, +// beneath it (leaves are xxHash64 digests — only the combiner is XOR). +// XOR is homomorphic (parent = XOR of children), order-independent, // and invertible, which buys three things: // // - depth-independence: a node's digest depends only on the grants in @@ -87,9 +86,9 @@ const ( merkleMaxDepth = merkleBucketHashLen ) -// hashLen is the width of a sha256 digest, used for both the grant +// hashLen is the width of an xxHash64 digest, used for both the grant // content hash (index value) and node digests. -const hashLen = sha256.Size +const hashLen = 8 // zeroDigest is the XOR identity — the digest of an empty/absent node. var zeroDigest [hashLen]byte @@ -101,27 +100,17 @@ func xorInto(dst, src []byte) { } } -// writeLenPrefixed writes an 8-byte big-endian length followed by b. -// Length-prefixing every field makes the canonical encoding injective: -// no concatenation of fields can be confused with a different split. -func writeLenPrefixed(h hash.Hash, b []byte) { - var n [8]byte - binary.BigEndian.PutUint64(n[:], uint64(len(b))) - _, _ = h.Write(n[:]) - _, _ = h.Write(b) -} - -// principalBucketHash is the bucket address for a principal: the first -// merkleBucketHashLen bytes of sha256(rt, id). Identity only — never the -// principal's full object — so the address is stable across syncs even -// when the principal's attributes change. Returns a fresh slice. +// principalBucketHash is the bucket address for a principal: the 8-byte +// xxHash64 of (rt + "\x00" + id). Identity only — never the principal's +// full object — so the address is stable across syncs even when the +// principal's attributes change. Returns a fresh slice. func principalBucketHash(rt, id string) []byte { - h := sha256.New() - writeLenPrefixed(h, []byte(rt)) - writeLenPrefixed(h, []byte(id)) - sum := h.Sum(nil) + h := xxhash.New() + _, _ = h.WriteString(rt) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(id) out := make([]byte, merkleBucketHashLen) - copy(out, sum[:merkleBucketHashLen]) + binary.BigEndian.PutUint64(out, h.Sum64()) return out } @@ -140,13 +129,17 @@ func principalBucketHash(rt, id string) []byte { // files written by different SDK builds hash identical grants // differently. Changing this set requires an index-migration bump. func grantContentHash(r *v3.GrantRecord) []byte { - h := sha256.New() + h := xxhash.New() ent := r.GetEntitlement() princ := r.GetPrincipal() - writeLenPrefixed(h, []byte(ent.GetEntitlementId())) - writeLenPrefixed(h, []byte(princ.GetResourceTypeId())) - writeLenPrefixed(h, []byte(princ.GetResourceId())) - writeLenPrefixed(h, []byte(r.GetExternalId())) + _, _ = h.WriteString(ent.GetEntitlementId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(princ.GetResourceTypeId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(princ.GetResourceId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(r.GetExternalId()) + _, _ = h.Write([]byte{0}) // Source-entitlement ids, sorted for order-independence. The map // values (GrantSourceRecord) are not folded in v1 — only the set of @@ -157,13 +150,13 @@ func grantContentHash(r *v3.GrantRecord) []byte { ids = append(ids, k) } sort.Strings(ids) - var nbuf [8]byte - binary.BigEndian.PutUint64(nbuf[:], uint64(len(ids))) - _, _ = h.Write(nbuf[:]) for _, id := range ids { - writeLenPrefixed(h, []byte(id)) + _, _ = h.WriteString(id) + _, _ = h.Write([]byte{0}) } - return h.Sum(nil) + out := make([]byte, hashLen) + binary.BigEndian.PutUint64(out, h.Sum64()) + return out } // grantHashIndexKey returns the by_entitlement_principal_hash index key From a88f19110e3294b20697c853f1d1f3aa213a7da0 Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Thu, 11 Jun 2026 19:07:11 -0700 Subject: [PATCH 6/7] use bit count leaf only buckets, rename to grant_digest --- pkg/dotc1z/engine/pebble/adapter.go | 16 +- .../engine/pebble/adapter_clone_sync.go | 2 +- pkg/dotc1z/engine/pebble/adapter_diff.go | 17 +- .../engine/pebble/adapter_diff_trie_test.go | 4 +- pkg/dotc1z/engine/pebble/cleanup.go | 2 +- pkg/dotc1z/engine/pebble/digest.go | 900 ++++++++++++++++ .../pebble/{merkle_test.go => digest_test.go} | 479 +++++---- pkg/dotc1z/engine/pebble/grant_digest.go | 260 +++++ pkg/dotc1z/engine/pebble/grants.go | 24 +- pkg/dotc1z/engine/pebble/if_newer.go | 10 +- pkg/dotc1z/engine/pebble/index_migrations.go | 38 +- pkg/dotc1z/engine/pebble/keys.go | 94 +- pkg/dotc1z/engine/pebble/merkle.go | 993 ------------------ pkg/synccompactor/pebble/bucket_plans.go | 15 +- 14 files changed, 1535 insertions(+), 1319 deletions(-) create mode 100644 pkg/dotc1z/engine/pebble/digest.go rename pkg/dotc1z/engine/pebble/{merkle_test.go => digest_test.go} (57%) create mode 100644 pkg/dotc1z/engine/pebble/grant_digest.go delete mode 100644 pkg/dotc1z/engine/pebble/merkle.go diff --git a/pkg/dotc1z/engine/pebble/adapter.go b/pkg/dotc1z/engine/pebble/adapter.go index ea822eec7..9f5dc751a 100644 --- a/pkg/dotc1z/engine/pebble/adapter.go +++ b/pkg/dotc1z/engine/pebble/adapter.go @@ -228,14 +228,14 @@ func (a *Adapter) EndSync(ctx context.Context) error { zap.Error(err), ) } - // Build the per-entitlement grant merkle trees at seal time, after - // all grants are written and while still on the fresh-sync NoSync - // path (the EndFreshSync flush below hardens the nodes). Non-fatal, - // like the stats sidecar: a missing/stale tree only forces a diff - // consumer onto the on-demand ComputeBucketHash fold, and the - // on-Open migration backfills it next time the file opens writable. - if err := a.engine.BuildAllMerkleTrees(ctx, existing.GetSyncId()); err != nil { - ctxzap.Extract(ctx).Warn("pebble: build grant merkle trees failed; grant-diff callers will fall back to on-demand index folds until the next Open backfills them", + // Build the per-entitlement grant digests at seal time, after all + // grants are written and while still on the fresh-sync NoSync path + // (the EndFreshSync flush below hardens the nodes). Non-fatal, like + // the stats sidecar: a missing/stale digest only forces a diff + // consumer onto the on-demand index fold, and the on-Open migration + // backfills it next time the file opens writable. + if err := a.engine.BuildAllGrantDigests(ctx, existing.GetSyncId()); err != nil { + ctxzap.Extract(ctx).Warn("pebble: build grant digests failed; grant-diff callers will fall back to on-demand index folds until the next Open backfills them", zap.String("sync_id", existing.GetSyncId()), zap.Error(err), ) diff --git a/pkg/dotc1z/engine/pebble/adapter_clone_sync.go b/pkg/dotc1z/engine/pebble/adapter_clone_sync.go index 9698f62d3..28aeb20b4 100644 --- a/pkg/dotc1z/engine/pebble/adapter_clone_sync.go +++ b/pkg/dotc1z/engine/pebble/adapter_clone_sync.go @@ -109,7 +109,7 @@ func cloneSync( {GrantByPrincipalResourceTypeSyncLowerBound(syncIDBytes), GrantByPrincipalResourceTypeSyncUpperBound(syncIDBytes)}, {GrantByNeedsExpansionSyncLowerBound(syncIDBytes), GrantByNeedsExpansionSyncUpperBound(syncIDBytes)}, {GrantByEntPrincHashSyncLowerBound(syncIDBytes), GrantByEntPrincHashSyncUpperBound(syncIDBytes)}, - {MerkleSyncLowerBound(syncIDBytes), MerkleSyncUpperBound(syncIDBytes)}, + {DigestSyncLowerBound(syncIDBytes), DigestSyncUpperBound(syncIDBytes)}, {encodeAssetPrefix(syncIDBytes), upperBoundOf(encodeAssetPrefix(syncIDBytes))}, // Stats sidecar — single key per sync; copyRange's [lo, hi) // shape requires a half-open range, so we synthesize one diff --git a/pkg/dotc1z/engine/pebble/adapter_diff.go b/pkg/dotc1z/engine/pebble/adapter_diff.go index ee8bc7922..af7fcdcd1 100644 --- a/pkg/dotc1z/engine/pebble/adapter_diff.go +++ b/pkg/dotc1z/engine/pebble/adapter_diff.go @@ -27,7 +27,7 @@ import ( // base; if base returns ErrNotFound, write the value under // diffSyncID's keyspace in the same record type. GRANTS — the type // that dominates every real file — are diffed via the per-entitlement -// hash tries instead (see diffGrants): unchanged entitlements are +// grant digests instead (see diffGrants): unchanged entitlements are // skipped with a single root comparison, and only the principal-hash // buckets that actually differ are scanned. Index entries are // recomputed on write so the diff sync has its own (fresh) indexes @@ -187,10 +187,11 @@ func diffEntitlements(ctx context.Context, a *Adapter, baseBytes, appliedBytes [ } // diffGrants computes the grants set difference using the -// per-entitlement hash tries instead of scanning every applied grant. +// per-entitlement grant digests instead of scanning every applied +// grant. // // Walk: enumerate the distinct entitlement_ids in the applied sync's -// hash index (one seek each), compare each entitlement's trie between +// hash index (one seek each), compare each entitlement's digest between // base and applied — one root read per side when nothing changed, the // overwhelmingly common case — and materialize only the principal-hash // buckets the comparison flags as dirty. Grants in dirty buckets are @@ -208,13 +209,13 @@ func diffEntitlements(ctx context.Context, a *Adapter, baseBytes, appliedBytes [ // the base probe then filters out), never under-approximates them. // // NOTE: grants without an entitlement or principal ref have no -// hash-index entry and are invisible to the trie. They are silently +// hash-index entry and are invisible to the digest. They are silently // skipped. A future O(1) coverage check (e.g. a stored grant count in // the sync run record) will restore detection of this case. func diffGrants(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, baseSyncID, appliedSyncID, diffSyncID string) error { eng := a.engine - ents, err := eng.distinctHashIndexEntitlements(ctx, appliedBytes) + ents, err := eng.distinctDigestPartitions(ctx, grantDigestSpec, appliedBytes) if err != nil { return err } @@ -224,16 +225,16 @@ func diffGrants(ctx context.Context, a *Adapter, baseBytes, appliedBytes []byte, } // Entitlements only present in base (fully removed) are not in // `ents` and are correctly skipped: they cannot contain - // additions. Tries missing on either side (e.g. a ghost + // additions. Digests missing on either side (e.g. a ghost // entitlement with grants but no entitlement record) degrade // to an on-demand index fold inside DirtyEntitlementBuckets. dirty, err := eng.DirtyEntitlementBuckets(ctx, baseSyncID, eng, appliedSyncID, ent) if err != nil { return err } - for _, prefix := range dirty { + for _, bucket := range dirty { var innerErr error - err := eng.IterateGrantsByEntitlementBucket(ctx, appliedSyncID, ent, prefix, func(rec *v3.GrantRecord) bool { + err := eng.IterateGrantsByEntitlementBucket(ctx, appliedSyncID, ent, bucket, func(rec *v3.GrantRecord) bool { exists, probeErr := existsAt(eng.DB(), encodeGrantKey(baseBytes, rec.GetExternalId())) if probeErr != nil { innerErr = probeErr diff --git a/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go b/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go index cd1c59c5c..291739690 100644 --- a/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go +++ b/pkg/dotc1z/engine/pebble/adapter_diff_trie_test.go @@ -192,8 +192,8 @@ func TestGenerateSyncDiffTrieWithoutTrees(t *testing.T) { if err != nil { t.Fatal(err) } - if err := a.engine.db.DeleteRange(MerkleSyncLowerBound(idBytes), MerkleSyncUpperBound(idBytes), pebble.Sync); err != nil { - t.Fatalf("DeleteRange(merkle %s): %v", sid, err) + if err := a.engine.db.DeleteRange(DigestSyncLowerBound(idBytes), DigestSyncUpperBound(idBytes), pebble.Sync); err != nil { + t.Fatalf("DeleteRange(digest %s): %v", sid, err) } } diff --git a/pkg/dotc1z/engine/pebble/cleanup.go b/pkg/dotc1z/engine/pebble/cleanup.go index 3b8d1f1bc..eeb390ca0 100644 --- a/pkg/dotc1z/engine/pebble/cleanup.go +++ b/pkg/dotc1z/engine/pebble/cleanup.go @@ -36,7 +36,7 @@ func syncScopedRanges(syncIDBytes []byte) [][2][]byte { {GrantByPrincipalResourceTypeSyncLowerBound(syncIDBytes), GrantByPrincipalResourceTypeSyncUpperBound(syncIDBytes)}, {GrantByNeedsExpansionSyncLowerBound(syncIDBytes), GrantByNeedsExpansionSyncUpperBound(syncIDBytes)}, {GrantByEntPrincHashSyncLowerBound(syncIDBytes), GrantByEntPrincHashSyncUpperBound(syncIDBytes)}, - {MerkleSyncLowerBound(syncIDBytes), MerkleSyncUpperBound(syncIDBytes)}, + {DigestSyncLowerBound(syncIDBytes), DigestSyncUpperBound(syncIDBytes)}, {encodeAssetPrefix(syncIDBytes), upperBoundOf(encodeAssetPrefix(syncIDBytes))}, // Stats sidecar — single key per sync; the half-open range // shape contains exactly the one key for this sync. diff --git a/pkg/dotc1z/engine/pebble/digest.go b/pkg/dotc1z/engine/pebble/digest.go new file mode 100644 index 000000000..919d471ac --- /dev/null +++ b/pkg/dotc1z/engine/pebble/digest.go @@ -0,0 +1,900 @@ +package pebble + +import ( + "bytes" + "context" + "encoding/binary" + "errors" + "fmt" + + "github.com/cockroachdb/pebble/v2" + + "github.com/conductorone/baton-sdk/pkg/dotc1z/engine/pebble/codec" +) + +// Bucketed XOR set digests over bucket-hash indexes. +// +// Goal: answer "does this partition hold exactly the same records as +// some other sync/file?" with a single key read, and when the answer is +// no, identify which hash buckets differ so a caller can load only +// those records instead of re-reading the whole partition. +// +// The digest is generic over any secondary index matching the shape +// described on digestIndexSpec — a partition prefix followed by a raw +// fixed-width bucket hash, with a per-record content hash as the +// value. Two properties make such an index the right substrate: +// +// - hash-major order is identical across two files that hold the same +// records, so a streamed fold produces the same digest; and +// - a bucket is a contiguous bit-range of the hash, which is a +// contiguous range of the index key, so "all records in bucket P" +// is a single range scan. +// +// The grant instantiation (partition = entitlement, bucket hash = +// hash(principal identity)) lives in grant_digest.go. +// +// Combiner. A node's digest is the XOR of every content hash beneath +// it (the content hashes themselves are whatever the index writer +// chose — only the combiner is XOR). XOR is homomorphic (parent = XOR +// of children), order-independent, and invertible, which buys three +// things: +// +// - split-independence: a bucket's digest depends only on the records +// in its hash range, never on how the range is subdivided, so +// buckets from digests of different widths compare directly (a +// width-w bucket is the XOR of its two width-(w+1) halves); +// - O(1) incremental maintenance: post-seal insert/overwrite/delete +// XOR the record's content hash into/out of the root and the one +// leaf on its bucket path (see digestMutator); +// - the empty digest is all-zero (the XOR identity), so an absent +// leaf reads as {count: 0, digest: 0}. +// +// Every node also stores its record COUNT. Comparison always checks the +// (count, digest) pair, so a non-empty node whose hashes happened to XOR +// to zero can never be conflated with an empty/absent one. XOR +// set-hashing is not adversarially collision-resistant +// (Bellare–Micciancio); a digest is an optimization, not a trust +// boundary — see RFC 0003 §9. Index writers should fold every +// key-distinguishing field into the content hash so duplicate leaves +// (and thus in-partition self-cancellation) are impossible by +// construction. +// +// Shape. A flat hash table, not a multi-level radix tree: one root plus +// a single leaf level of 2^width buckets, where width — in BITS of the +// bucket hash, 0..digestMaxWidthBits — is chosen per partition so the +// average bucket holds at most digestTargetBucketSize records +// (chooseDigestWidth). Width 0 means root-only (small and empty +// partitions cost exactly one stored node). Growing capacity one bit at +// a time keeps realized bucket occupancy within a 2x band of the +// target; a byte-per-level radix could only pick capacities of 256^k, +// so a partition just past a boundary would store up to ~256x more +// nodes than needed. +// +// Interior levels are deliberately absent: hierarchical pruning only +// pays when the leaf level is too large to scan, and at <= 2^16 leaves +// a contiguous range scan beats a level-by-level descent. Comparison is +// a single merge scan of both sides' leaf levels (see +// dirtyPartitionBuckets); cross-width comparison folds the finer side's +// leaves down to the coarser width on the fly, which split-independence +// makes exact. For the same reason digestTargetBucketSize is a soft +// tuning knob, not an ABI constant: digests built with different +// targets still compare correctly. +// +// Leaves are stored sparsely — a leaf is materialized iff its bucket +// holds >=1 record. The root is always materialized (it is the "digest +// was built" marker — absence of the root means "never built", never +// "empty"). +// +// On-disk ABI. The bucket-hash width, each index's content-hash +// definition, the combiner, the leaf-prefix encoding, and the node +// value framing are all part of the stored format: changing +// digestBucketHashLen, an index's content-hash field set, the combiner, +// or the framing requires an index-migration version bump (see +// index_migrations.go). digestTargetBucketSize is NOT part of the ABI +// (see above). + +const ( + // digestBucketHashLen is the width, in bytes, of the raw bucket + // hash embedded in a digested index's key. Collisions in the bucket + // address are harmless (they only co-locate records — the index + // key's tail still distinguishes rows). + digestBucketHashLen = 8 + + // digestTargetBucketSize is the record count a single leaf bucket + // aims to hold. The width is grown until 2^width buckets bring the + // average bucket under this. Tunable without a migration: the width + // is read from the stored root and cross-width comparison folds to + // the coarser side. + digestTargetBucketSize = 512 + + // digestMaxWidthBits caps the leaf-level width. 2^16 buckets keeps + // the comparison's full leaf scan trivially cheap; past the cap + // (digestTargetBucketSize << 16 records) buckets simply grow beyond + // the target. + digestMaxWidthBits = 16 + + // digestLeafPrefixLen is the stored byte width of a leaf key's + // bucket prefix: the bucket index LEFT-ALIGNED in 16 bits, so leaf + // keys sort in bucket-hash order at every width and folding to a + // coarser width is "take the top bits". ABI. + digestLeafPrefixLen = 2 +) + +// Node-key levels: the root is level 0 (empty prefix); the single leaf +// level is 1 (digestLeafPrefixLen-byte prefix). See encodeDigestNodeKey. +const ( + digestLevelRoot byte = 0 + digestLevelLeaf byte = 1 +) + +// hashLen is the width of a content hash (index value) and of a node +// digest. +const hashLen = 8 + +// zeroDigest is the XOR identity — the digest of an empty/absent node. +var zeroDigest [hashLen]byte + +// xorInto XORs src into dst in place, over min(len(dst), len(src)). +func xorInto(dst, src []byte) { + for i := range min(len(dst), len(src)) { + dst[i] ^= src[i] + } +} + +// digestIndexSpec describes one bucket-hash index the digest core can +// fold. The index must have the shape +// +// index key = partitionPrefix(sync, partition) | | +// index value = content hash (hashLen bytes) +// +// and, for partition enumeration, +// +// index key = syncBounds(sync).lower | 0x00 | tuple(partition) | 0x00 | | … +// +// (i.e. the partition is the first tuple component after the sync-wide +// lower bound — the standard by-value index layout). +// +// The content hash defines record identity for the diff and must fold +// every key-distinguishing field (see the package comment); the bucket +// hash must be derived from fields that are stable across syncs. +type digestIndexSpec struct { + // indexID discriminates this index's nodes inside the typeDigest + // keyspace. Conventionally the digested index's own idx* byte. ABI. + indexID byte + + // partitionPrefix returns the index-key prefix covering one + // partition's entries, ending immediately before the raw bucket + // hash (trailing separator included). + partitionPrefix func(syncIDBytes []byte, partition string) []byte + + // syncBounds bounds the index's entire keyspace for one sync. + syncBounds func(syncIDBytes []byte) (lower, upper []byte) +} + +// DigestBucket addresses one hash bucket of a partition: the records +// whose bucket hash starts with the top Bits bits of Index. The zero +// value (Bits 0) addresses the whole partition. +type DigestBucket struct { + Index uint32 + Bits int +} + +// leafKeyPrefix returns the stored digestLeafPrefixLen-byte node-key +// prefix for a leaf bucket: the index left-aligned in 16 bits. +// Requires 1 <= Bits <= digestMaxWidthBits. +func (b DigestBucket) leafKeyPrefix() []byte { + out := make([]byte, digestLeafPrefixLen) + binary.BigEndian.PutUint16(out, uint16(b.Index)<<(16-b.Bits)) //nolint:gosec // Index < 2^Bits <= 2^16 by construction + return out +} + +// bucketOfHash returns the width-`bits` bucket holding bucket hash bh. +// Requires 1 <= bits <= digestMaxWidthBits. +func bucketOfHash(bh []byte, bits int) DigestBucket { + u := binary.BigEndian.Uint16(bh[:digestLeafPrefixLen]) + return DigestBucket{Index: uint32(u >> (16 - bits)), Bits: bits} +} + +// bucketBounds returns the index key range [lower, upper) covering a +// bucket's records in one partition. The raw bucket hash is a clean +// byte region of the index key, so a bit-granular bucket maps to plain +// uint64 arithmetic on that region. +func (s digestIndexSpec) bucketBounds(idBytes []byte, partition string, b DigestBucket) ([]byte, []byte) { + prefix := s.partitionPrefix(idBytes, partition) + if b.Bits == 0 { + return prefix, upperBoundOf(prefix) + } + boundAt := func(hash uint64) []byte { + out := append(append(make([]byte, 0, len(prefix)+digestBucketHashLen), prefix...), 0, 0, 0, 0, 0, 0, 0, 0) + binary.BigEndian.PutUint64(out[len(prefix):], hash) + return out + } + shift := uint(64 - b.Bits) //nolint:gosec // Bits in [1, digestMaxWidthBits] + lower := boundAt(uint64(b.Index) << shift) + if uint64(b.Index)+1 == uint64(1)< digestMaxWidthBits { + return 0, 0, nil, false + } + widthBits := int(val[0]) + count := int64(binary.BigEndian.Uint64(val[1:9])) //nolint:gosec // count is a non-negative row count + return widthBits, count, val[9:], true +} + +func packDigestLeaf(count int64, digest []byte) []byte { + buf := make([]byte, 0, 8+len(digest)) + var n [8]byte + binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count + buf = append(buf, n[:]...) + return append(buf, digest...) +} + +// unpackDigestLeaf returns (count, digest, ok) for a leaf node body. +func unpackDigestLeaf(val []byte) (int64, []byte, bool) { + if len(val) != 8+hashLen { + return 0, nil, false + } + count := int64(binary.BigEndian.Uint64(val[:8])) //nolint:gosec // non-negative count + return count, val[8:], true +} + +// buildPartitionDigest counts a partition's index entries (pass 1), +// picks the width from that count, and delegates the fold to +// buildPartitionDigestAtWidth (pass 2). +func (e *Engine) buildPartitionDigest(ctx context.Context, spec digestIndexSpec, idBytes []byte, partition string) error { + prefix := spec.partitionPrefix(idBytes, partition) + count, err := e.countKeysInRange(prefix, upperBoundOf(prefix)) + if err != nil { + return err + } + return e.buildPartitionDigestAtWidth(ctx, spec, idBytes, partition, chooseDigestWidth(count)) +} + +// buildPartitionDigestAtWidth folds the index for one partition into a +// root plus every non-empty leaf, in a single streaming pass — O(1) +// memory regardless of partition size. +// +// The pass starts by range-deleting the partition's whole node +// keyspace: the build only ever Sets nodes, so without the clear a +// rebuild that changes width or empties a bucket would leave stale +// leaves that the comparison merge scan (which enumerates leaves from +// the node keyspace) would read — and a stale digest that happens to +// match the peer prunes a real diff. Old and new framings are +// byte-length identical, so stale nodes are not detectable by +// inspection. +// +// Sorted index order means each bucket's entries are contiguous, so the +// single "open" leaf closes exactly when its prefix changes. Only +// non-empty leaves are ever opened, so sparsity is automatic, not a +// prune pass. +// +// The width is taken as a parameter rather than derived so the +// width-selection seam can be exercised directly: tests force a width +// that the natural count→width mapping would only produce at a very +// large record count, which is how the cross-width comparison path gets +// covered without seeding hundreds of thousands of records. +func (e *Engine) buildPartitionDigestAtWidth(ctx context.Context, spec digestIndexSpec, idBytes []byte, partition string, widthBits int) error { + prefix := spec.partitionPrefix(idBytes, partition) + upper := upperBoundOf(prefix) + nodeLower := encodeDigestPartitionPrefix(idBytes, spec.indexID, partition) + nodeUpper := upperBoundOf(nodeLower) + + return e.withWrite(func() error { + batch := e.db.NewBatch() + defer batch.Close() + + // Clear any prior build (see function comment). In-batch + // ordering makes this safe: the Sets below land after the + // tombstone and survive it. + if err := batch.DeleteRange(nodeLower, nodeUpper, nil); err != nil { + return err + } + + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: prefix, UpperBound: upper}) + if err != nil { + return err + } + defer iter.Close() + + // lowMask clears the sub-bucket bits of a left-aligned 16-bit + // prefix, leaving the leaf's stored key prefix value. + var lowMask uint16 + if widthBits > 0 { + lowMask = ^uint16(0) >> widthBits + } + + var ( + leafOpen bool + leafLV uint16 // left-aligned stored prefix of the open leaf + leafDigest [hashLen]byte + leafCount int64 + rootDigest [hashLen]byte + total int64 + ) + flushLeaf := func() error { + if !leafOpen { + return nil + } + var lp [digestLeafPrefixLen]byte + binary.BigEndian.PutUint16(lp[:], leafLV) + key := encodeDigestNodeKey(idBytes, spec.indexID, partition, digestLevelLeaf, lp[:]) + return batch.Set(key, packDigestLeaf(leafCount, leafDigest[:]), nil) + } + + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return err + } + key := iter.Key() + if len(key) < len(prefix)+digestBucketHashLen { + continue // malformed; skip defensively + } + val := iter.Value() // per-record content hash + if widthBits > 0 { + lv := binary.BigEndian.Uint16(key[len(prefix):]) &^ lowMask + if !leafOpen || lv != leafLV { + if err := flushLeaf(); err != nil { + return err + } + leafLV = lv + leafDigest = [hashLen]byte{} + leafCount = 0 + leafOpen = true + } + xorInto(leafDigest[:], val) + leafCount++ + } + xorInto(rootDigest[:], val) + total++ + } + if err := iter.Error(); err != nil { + return err + } + if err := flushLeaf(); err != nil { + return err + } + + // Root is written unconditionally — even at count 0 — as the + // "digest was built" marker. + rootKey := encodeDigestNodeKey(idBytes, spec.indexID, partition, digestLevelRoot, nil) + if err := batch.Set(rootKey, packDigestRoot(widthBits, total, rootDigest[:]), nil); err != nil { + return err + } + + opts := writeOpts(e.opts.durability) + if e.IsFreshSync() { + opts = pebble.NoSync + } + return batch.Commit(opts) + }) +} + +// countKeysInRange counts keys in [lower, upper) without materializing +// values. Used by the build's count→width pass and by the diff driver's +// primary-vs-index coverage guard. +func (e *Engine) countKeysInRange(lower, upper []byte) (int64, error) { + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return 0, err + } + defer iter.Close() + var n int64 + for iter.First(); iter.Valid(); iter.Next() { + n++ + } + return n, iter.Error() +} + +// distinctDigestPartitions returns the distinct partitions present in a +// sync's digested index, in index order. It seeks past each partition's +// whole range once its id is captured, so the cost is O(partitions) +// seeks, not O(entries). This is the diff driver's work list, and it +// deliberately comes from the INDEX rather than the partition-owning +// records: an entry whose partition has no owning record still has +// index entries (and a fold), so it still gets compared — only a digest +// was never built for it. +func (e *Engine) distinctDigestPartitions(ctx context.Context, spec digestIndexSpec, syncIDBytes []byte) ([]string, error) { + lower, upper := spec.syncBounds(syncIDBytes) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return nil, err + } + defer iter.Close() + + // Keys are lower | 0x00 | tuple(partition) | 0x00 | bucketHash | … + // (the spec's shape contract). + partStart := len(lower) + 1 + var out []string + for iter.First(); iter.Valid(); { + if err := ctx.Err(); err != nil { + return nil, err + } + key := iter.Key() + if len(key) <= partStart { + iter.Next() + continue // malformed; skip defensively + } + partBytes, _, decErr := codec.DecodeTupleStringTo(nil, key[partStart:], 0) + if decErr != nil { + iter.Next() + continue + } + partition := string(partBytes) + out = append(out, partition) + // Seek past this partition's whole index range. + seekTo := upperBoundOf(spec.partitionPrefix(syncIDBytes, partition)) + if seekTo == nil { + break + } + iter.SeekGE(seekTo) + } + return out, iter.Error() +} + +// DigestRoot is a partition's stored root digest. +type DigestRoot struct { + Hash []byte + Bits int // leaf-level width in bits; 0 = root-only digest + Count int64 +} + +// getPartitionDigestRoot returns the stored root for a partition. ok is +// false when no digest has been built for it (the caller can fall back +// to computeBucketDigest, which derives the same digest from the index +// on demand). +func (e *Engine) getPartitionDigestRoot(spec digestIndexSpec, idBytes []byte, partition string) (DigestRoot, bool, error) { + val, closer, err := e.db.Get(encodeDigestNodeKey(idBytes, spec.indexID, partition, digestLevelRoot, nil)) + if err != nil { + if errors.Is(err, pebble.ErrNotFound) { + return DigestRoot{}, false, nil + } + return DigestRoot{}, false, err + } + defer closer.Close() + widthBits, count, h, valid := unpackDigestRoot(val) + if !valid { + return DigestRoot{}, false, fmt.Errorf("getPartitionDigestRoot: malformed root for %q", partition) + } + out := make([]byte, len(h)) + copy(out, h) + return DigestRoot{Hash: out, Bits: widthBits, Count: count}, true, nil +} + +// getDigestLeaf reads one stored leaf by its key prefix. An absent leaf +// returns (0, zero digest, present=false, nil) — the XOR identity. +func (e *Engine) getDigestLeaf(spec digestIndexSpec, idBytes []byte, partition string, leafPrefix []byte) (int64, []byte, bool, error) { + val, closer, err := e.db.Get(encodeDigestNodeKey(idBytes, spec.indexID, partition, digestLevelLeaf, leafPrefix)) + if err != nil { + if errors.Is(err, pebble.ErrNotFound) { + return 0, zeroDigest[:], false, nil + } + return 0, nil, false, err + } + defer closer.Close() + count, digest, ok := unpackDigestLeaf(val) + if !ok { + return 0, nil, false, fmt.Errorf("getDigestLeaf: malformed leaf for %q", partition) + } + out := make([]byte, hashLen) + copy(out, digest) + return count, out, true, nil +} + +// foldedBucket is one entry of a folded leaf scan: the (XOR, count) +// aggregate of the consecutive stored leaves sharing the top `bits` +// bits of their bucket index. +type foldedBucket struct { + idx uint32 + count int64 + digest [hashLen]byte +} + +// foldedLeafBuckets scans a partition's stored leaf level and folds it +// to width foldBits (which must be <= the width the digest was built +// at), returning the non-empty buckets in index order. One contiguous +// range scan; folding is exact because leaf prefixes are left-aligned +// (so keys sort in bucket-hash order at any width) and XOR digests are +// split-independent. +func (e *Engine) foldedLeafBuckets(ctx context.Context, spec digestIndexSpec, idBytes []byte, partition string, foldBits int) ([]foldedBucket, error) { + stem := encodeDigestNodeKey(idBytes, spec.indexID, partition, digestLevelLeaf, nil) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: stem, UpperBound: upperBoundOf(stem)}) + if err != nil { + return nil, err + } + defer iter.Close() + var out []foldedBucket + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return nil, err + } + key := iter.Key() + if len(key) != len(stem)+digestLeafPrefixLen { + continue // malformed; skip defensively + } + lv := binary.BigEndian.Uint16(key[len(stem):]) + idx := uint32(lv >> (16 - foldBits)) + count, digest, ok := unpackDigestLeaf(iter.Value()) + if !ok { + return nil, fmt.Errorf("foldedLeafBuckets: malformed leaf for %q", partition) + } + if n := len(out); n > 0 && out[n-1].idx == idx { + out[n-1].count += count + xorInto(out[n-1].digest[:], digest) + continue + } + fb := foldedBucket{idx: idx, count: count} + copy(fb.digest[:], digest) + out = append(out, fb) + } + return out, iter.Error() +} + +// computeBucketDigest folds the index over a single bucket (the zero +// bucket = whole partition = the root) and returns the content-defined +// XOR digest plus the record count. This is the authoritative +// definition of a node; stored nodes are a cache of it. +// Split-independent: the digest depends only on the records in the +// bucket's hash range, not on any digest's width. +func (e *Engine) computeBucketDigest(ctx context.Context, spec digestIndexSpec, idBytes []byte, partition string, bucket DigestBucket) ([]byte, int64, error) { + lower, upper := spec.bucketBounds(idBytes, partition, bucket) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return nil, 0, err + } + defer iter.Close() + digest := make([]byte, hashLen) + var count int64 + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return nil, 0, err + } + xorInto(digest, iter.Value()) + count++ + } + if err := iter.Error(); err != nil { + return nil, 0, err + } + return digest, count, nil +} + +// dirtyPartitionBuckets compares this engine's partition against +// other's and returns the buckets whose records differ. A single zero +// bucket (Bits 0) means "the whole partition differs" (used when the +// comparison granularity is the root, e.g. small partitions). A nil +// (empty) result means the two are identical. +// +// The fast path is a single root read per side. On mismatch both sides' +// leaf levels are folded to compareBits — the narrower digest's width, +// where both sides have directly comparable buckets (XOR digests are +// split-independent) — and merge-compared in one pass. Each side's fold +// is a single contiguous range scan of its stored leaves, so cost is +// O(leaves), bounded by count/digestTargetBucketSize per side. +// +// A missing root means the digest was never built on that side — NOT +// that the partition is empty — so both sides are compared via the +// authoritative on-demand fold instead. +func (e *Engine) dirtyPartitionBuckets(ctx context.Context, spec digestIndexSpec, idBytes []byte, other *Engine, otherIDBytes []byte, partition string) ([]DigestBucket, error) { + rootA, okA, err := e.getPartitionDigestRoot(spec, idBytes, partition) + if err != nil { + return nil, err + } + rootB, okB, err := other.getPartitionDigestRoot(spec, otherIDBytes, partition) + if err != nil { + return nil, err + } + + if !okA || !okB { + ha, ca, err := e.computeBucketDigest(ctx, spec, idBytes, partition, DigestBucket{}) + if err != nil { + return nil, err + } + hb, cb, err := other.computeBucketDigest(ctx, spec, otherIDBytes, partition, DigestBucket{}) + if err != nil { + return nil, err + } + if ca == cb && bytes.Equal(ha, hb) { + return nil, nil + } + return []DigestBucket{{}}, nil + } + + if rootA.Count == rootB.Count && bytes.Equal(rootA.Hash, rootB.Hash) { + return nil, nil + } + + // Roots differ. The comparison granularity is the narrower digest's + // width; at width 0 there is nothing below the root. + compareBits := min(rootA.Bits, rootB.Bits) + if compareBits == 0 { + return []DigestBucket{{}}, nil + } + + fa, err := e.foldedLeafBuckets(ctx, spec, idBytes, partition, compareBits) + if err != nil { + return nil, err + } + fb, err := other.foldedLeafBuckets(ctx, spec, otherIDBytes, partition, compareBits) + if err != nil { + return nil, err + } + + // Merge the two sorted folded-bucket streams. A bucket present on + // only one side is dirty by construction (stored leaves are never + // empty); a shared bucket is dirty iff its (count, digest) differs. + var dirty []DigestBucket + i, j := 0, 0 + for i < len(fa) || j < len(fb) { + switch { + case j == len(fb) || (i < len(fa) && fa[i].idx < fb[j].idx): + dirty = append(dirty, DigestBucket{Index: fa[i].idx, Bits: compareBits}) + i++ + case i == len(fa) || fb[j].idx < fa[i].idx: + dirty = append(dirty, DigestBucket{Index: fb[j].idx, Bits: compareBits}) + j++ + default: + if fa[i].count != fb[j].count || fa[i].digest != fb[j].digest { + dirty = append(dirty, DigestBucket{Index: fa[i].idx, Bits: compareBits}) + } + i++ + j++ + } + } + + // Roots differed but the merge found nothing: with consistent + // digests that's impossible (the root is the XOR of the leaves), so + // a stored node is stale or corrupt. Fail safe — whole partition + // dirty; the next rebuild heals the digest. + if len(dirty) == 0 { + return []DigestBucket{{}}, nil + } + return dirty, nil +} + +// --- Incremental maintenance (post-seal) --- + +// digestMutator accumulates per-node (XOR, count) deltas for a batch of +// post-seal record mutations against one digested index, and applies +// each touched node exactly once. +// +// Why an accumulator instead of read-modify-write per mutation: the +// updates target a plain pebble.Batch, which does NOT read through its +// own writes — and every mutation in a batch touches the root node, so +// naive per-mutation RMW would lose deltas. Accumulating also collapses +// N writes per node into one. All record writers run under withWrite's +// mutex, so reading current node values from the DB inside apply is +// race-free. +// +// Lifecycle: one mutator per write batch. Callers feed removeHash(old) +// / addHash(new) as they process records (an overwrite that moves a +// record to a different bucket is exactly remove+add), then call +// apply(batch) once before commit. +// +// Partitions whose digest was never built (no stored root) are skipped: +// the seal-time build or the on-Open backfill will construct them from +// the index. This also makes the mutator free during a fresh sync — but +// callers on the fresh-sync bulk path should skip constructing one +// anyway to avoid the per-partition root probe. +type digestMutator struct { + e *Engine + spec digestIndexSpec + parts map[string]*mutatorPartition // keyed by string(root node key) +} + +type mutatorPartition struct { + idBytes []byte + partition string + rootKey []byte + present bool // stored root exists; if false all deltas are dropped + bits int // leaf-level width read from the stored root + rootCount int64 // count read from the stored root + rootDigest [hashLen]byte // digest read from the stored root + xor [hashLen]byte // accumulated root delta + countDelta int64 + nodes map[string]*mutatorNode // leaves, keyed by string(node key) +} + +type mutatorNode struct { + key []byte + xor [hashLen]byte + countDelta int64 +} + +func newDigestMutator(e *Engine, spec digestIndexSpec) *digestMutator { + return &digestMutator{e: e, spec: spec, parts: make(map[string]*mutatorPartition)} +} + +// partFor returns the (cached) per-partition state, probing the stored +// root on first touch. The cached root snapshot stays valid for the +// mutator's lifetime because all writers serialize through withWrite. +func (m *digestMutator) partFor(idBytes []byte, partition string) (*mutatorPartition, error) { + rootKey := encodeDigestNodeKey(idBytes, m.spec.indexID, partition, digestLevelRoot, nil) + k := string(rootKey) + if mp, ok := m.parts[k]; ok { + return mp, nil + } + mp := &mutatorPartition{idBytes: idBytes, partition: partition, rootKey: rootKey, nodes: make(map[string]*mutatorNode)} + val, closer, err := m.e.db.Get(rootKey) + switch { + case err == nil: + widthBits, count, digest, ok := unpackDigestRoot(val) + closer.Close() + if ok { + mp.present = true + mp.bits = widthBits + mp.rootCount = count + copy(mp.rootDigest[:], digest) + } + // Malformed root: leave present=false so mutations are dropped; + // the stale root heals at the next rebuild. + case errors.Is(err, pebble.ErrNotFound): + // No digest — deltas for this partition are no-ops. + default: + return nil, err + } + m.parts[k] = mp + return mp, nil +} + +// addHash records the insertion of a record with the given bucket and +// content hashes into its partition's digest. +func (m *digestMutator) addHash(idBytes []byte, partition string, bucketHash, contentHash []byte) error { + return m.delta(idBytes, partition, bucketHash, contentHash, 1) +} + +// removeHash records the removal of a record from its partition's +// digest. +func (m *digestMutator) removeHash(idBytes []byte, partition string, bucketHash, contentHash []byte) error { + return m.delta(idBytes, partition, bucketHash, contentHash, -1) +} + +func (m *digestMutator) delta(idBytes []byte, partition string, bucketHash, contentHash []byte, sign int64) error { + mp, err := m.partFor(idBytes, partition) + if err != nil { + return err + } + if !mp.present { + return nil + } + xorInto(mp.xor[:], contentHash) + mp.countDelta += sign + if mp.bits == 0 { + return nil + } + key := encodeDigestNodeKey(idBytes, m.spec.indexID, partition, digestLevelLeaf, bucketOfHash(bucketHash, mp.bits).leafKeyPrefix()) + nk := string(key) + n, ok := mp.nodes[nk] + if !ok { + n = &mutatorNode{key: key} + mp.nodes[nk] = n + } + xorInto(n.xor[:], contentHash) + n.countDelta += sign + return nil +} + +// apply folds the accumulated deltas into the stored nodes via batch. +// Nodes whose delta cancelled to zero (e.g. an overwrite that changed +// only excluded fields) are skipped; a leaf whose count reaches zero is +// deleted (restoring sparsity); the root is rewritten in place. A count +// that would go negative means the stored digest disagrees with the +// mutation stream — the digest is dropped wholesale (DeleteRange), so +// readers fall back to the on-demand fold until the next rebuild. +func (m *digestMutator) apply(batch *pebble.Batch) error { + for _, mp := range m.parts { + if !mp.present { + continue + } + if err := m.applyPartition(batch, mp); err != nil { + return err + } + } + return nil +} + +func (m *digestMutator) applyPartition(batch *pebble.Batch, mp *mutatorPartition) error { + if mp.countDelta == 0 && mp.xor == zeroDigest && len(mp.nodes) == 0 { + return nil + } + dropDigest := func() error { + // In-batch ordering: this tombstone lands after any node Sets + // already staged for this partition and removes them too. + lo := encodeDigestPartitionPrefix(mp.idBytes, m.spec.indexID, mp.partition) + return batch.DeleteRange(lo, upperBoundOf(lo), nil) + } + if mp.rootCount+mp.countDelta < 0 { + return dropDigest() + } + for _, n := range mp.nodes { + if n.countDelta == 0 && n.xor == zeroDigest { + continue + } + var ( + curCount int64 + curDigest [hashLen]byte + ) + val, closer, err := m.e.db.Get(n.key) + switch { + case err == nil: + c, d, ok := unpackDigestLeaf(val) + closer.Close() + if !ok { + return dropDigest() + } + curCount = c + copy(curDigest[:], d) + case errors.Is(err, pebble.ErrNotFound): + // absent leaf = {0, zero} + default: + return err + } + newCount := curCount + n.countDelta + if newCount < 0 { + return dropDigest() + } + xorInto(curDigest[:], n.xor[:]) + if newCount == 0 { + // An emptied leaf's digest must cancel to exactly zero + // (count 0 ⇒ digest 0); anything else means the stored + // digest disagrees with the mutation stream. + if curDigest != zeroDigest { + return dropDigest() + } + if err := batch.Delete(n.key, nil); err != nil { + return err + } + continue + } + if err := batch.Set(n.key, packDigestLeaf(newCount, curDigest[:]), nil); err != nil { + return err + } + } + if mp.countDelta == 0 && mp.xor == zeroDigest { + return nil + } + newDigest := mp.rootDigest + xorInto(newDigest[:], mp.xor[:]) + return batch.Set(mp.rootKey, packDigestRoot(mp.bits, mp.rootCount+mp.countDelta, newDigest[:]), nil) +} diff --git a/pkg/dotc1z/engine/pebble/merkle_test.go b/pkg/dotc1z/engine/pebble/digest_test.go similarity index 57% rename from pkg/dotc1z/engine/pebble/merkle_test.go rename to pkg/dotc1z/engine/pebble/digest_test.go index 1af4a9797..4784bc09c 100644 --- a/pkg/dotc1z/engine/pebble/merkle_test.go +++ b/pkg/dotc1z/engine/pebble/digest_test.go @@ -3,6 +3,7 @@ package pebble import ( "bytes" "context" + "encoding/binary" "fmt" "testing" @@ -14,7 +15,7 @@ import ( // putEnt writes an entitlement record whose external_id is entID — the // same string grants reference via EntitlementRef.EntitlementId, which -// is what BuildAllMerkleTrees keys each tree on. +// is what BuildAllGrantDigests keys each digest on. func putEnt(t testing.TB, e *Engine, ctx context.Context, syncID, entID string) { t.Helper() rec := v3.EntitlementRecord_builder{ @@ -46,17 +47,17 @@ func makeGrantWithSources(syncID, externalID, entID, principalID string, sources return g } -// merkleNodeCount counts stored merkle nodes for a sync (across all -// entitlements). -func merkleNodeCount(t testing.TB, e *Engine, syncID string) int { +// digestNodeCount counts stored digest nodes for a sync (across all +// partitions and digested indexes). +func digestNodeCount(t testing.TB, e *Engine, syncID string) int { t.Helper() idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatalf("resolveSyncBytes: %v", err) } iter, err := e.db.NewIter(&pebble.IterOptions{ - LowerBound: MerkleSyncLowerBound(idBytes), - UpperBound: MerkleSyncUpperBound(idBytes), + LowerBound: DigestSyncLowerBound(idBytes), + UpperBound: DigestSyncUpperBound(idBytes), }) if err != nil { t.Fatalf("NewIter: %v", err) @@ -72,8 +73,32 @@ func merkleNodeCount(t testing.TB, e *Engine, syncID string) int { return n } +// rawLeafPrefixes returns the stored 2-byte leaf key prefixes for one +// entitlement's grant digest, in key order. +func rawLeafPrefixes(t testing.TB, e *Engine, idBytes []byte, entID string) [][]byte { + t.Helper() + stem := encodeDigestNodeKey(idBytes, grantDigestSpec.indexID, entID, digestLevelLeaf, nil) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: stem, UpperBound: upperBoundOf(stem)}) + if err != nil { + t.Fatalf("NewIter: %v", err) + } + defer iter.Close() + var out [][]byte + for iter.First(); iter.Valid(); iter.Next() { + key := iter.Key() + if len(key) != len(stem)+digestLeafPrefixLen { + t.Fatalf("leaf key with prefix length %d, want %d", len(key)-len(stem), digestLeafPrefixLen) + } + out = append(out, append([]byte(nil), key[len(stem):]...)) + } + if err := iter.Error(); err != nil { + t.Fatalf("iter: %v", err) + } + return out +} + // seedEntitlement writes the entitlement record + grants and builds the -// tree, returning the syncID. +// digest, returning the syncID. func seedEntitlement(t testing.TB, e *Engine, entID string, grants []*v3.GrantRecord) string { t.Helper() ctx := context.Background() @@ -84,20 +109,21 @@ func seedEntitlement(t testing.TB, e *Engine, entID string, grants []*v3.GrantRe putEnt(t, e, ctx, syncID, entID) for _, g := range grants { g.SetSyncId(syncID) - if err := e.PutGrantRecord(ctx, g); err != nil { - t.Fatalf("PutGrantRecord: %v", err) - } } - if err := e.BuildAllMerkleTrees(ctx, syncID); err != nil { - t.Fatalf("BuildAllMerkleTrees: %v", err) + if err := e.PutGrantRecords(ctx, grants...); err != nil { + t.Fatalf("PutGrantRecords: %v", err) + } + if err := e.BuildAllGrantDigests(ctx, syncID); err != nil { + t.Fatalf("BuildAllGrantDigests: %v", err) } return syncID } -// seedEntitlementAtDepth is seedEntitlement but forces a specific tree -// depth instead of deriving it from the grant count, so a test can build -// two trees of different heights over a small grant set. -func seedEntitlementAtDepth(t testing.TB, e *Engine, entID string, grants []*v3.GrantRecord, depth int) string { +// seedEntitlementAtWidth is seedEntitlement but forces a specific +// leaf-level width instead of deriving it from the grant count, so a +// test can build two digests of different widths over a small grant +// set. +func seedEntitlementAtWidth(t testing.TB, e *Engine, entID string, grants []*v3.GrantRecord, widthBits int) string { t.Helper() ctx := context.Background() syncID := ksuid.New().String() @@ -107,33 +133,34 @@ func seedEntitlementAtDepth(t testing.TB, e *Engine, entID string, grants []*v3. putEnt(t, e, ctx, syncID, entID) for _, g := range grants { g.SetSyncId(syncID) - if err := e.PutGrantRecord(ctx, g); err != nil { - t.Fatalf("PutGrantRecord: %v", err) - } + } + if err := e.PutGrantRecords(ctx, grants...); err != nil { + t.Fatalf("PutGrantRecords: %v", err) } idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatalf("resolveSyncBytes: %v", err) } - if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, entID, depth); err != nil { - t.Fatalf("buildEntitlementMerkleAtDepth: %v", err) + if err := e.buildPartitionDigestAtWidth(ctx, grantDigestSpec, idBytes, entID, widthBits); err != nil { + t.Fatalf("buildPartitionDigestAtWidth: %v", err) } return syncID } -// TestMerkleDifferentDepthsComparison builds two trees of different -// heights (depth 1 vs depth 2) over the SAME entitlement and exercises +// TestDigestDifferentWidthsComparison builds two digests of different +// widths (4 vs 8 bits) over the SAME entitlement and exercises // DirtyEntitlementBuckets across them. It validates two things the -// equal-depth tests cannot: +// equal-width tests cannot: // -// - depth-independence: identical grant content yields the same root -// hash regardless of tree depth, and compares as zero dirty buckets; -// - the mismatched-depth descent: after one principal's grant changes, -// the comparison (at compareDepth = min(1,2) = 1) localizes the -// change to that principal's bucket — the deeper (depth-2) side -// folds its index on demand for the depth-1 prefixes — and leaves a -// known principal in a different bucket clean. -func TestMerkleDifferentDepthsComparison(t *testing.T) { +// - split-independence: identical grant content yields the same root +// hash regardless of digest width, and compares as zero dirty +// buckets; +// - the cross-width merge: after one principal's grant changes, the +// comparison (at compareBits = min(4,8) = 4) localizes the change to +// that principal's width-4 bucket — the finer (width-8) side's +// leaves fold down to width 4 during the scan — and leaves a known +// principal in a different bucket clean. +func TestDigestDifferentWidthsComparison(t *testing.T) { ctx := context.Background() const nPrincipals = 40 @@ -149,60 +176,59 @@ func TestMerkleDifferentDepthsComparison(t *testing.T) { return gs } - // depth1Prefix returns the depth-1 bucket prefix (1 raw hash byte) - // for a principal, matching how grants are keyed (principal type - // "user" per makeGrant). - depth1Prefix := func(principalID string) byte { - return principalBucketHash("user", principalID)[0] + // bucket4 returns the width-4 bucket index for a principal, matching + // how grants are keyed (principal type "user" per makeGrant). + bucket4 := func(principalID string) uint32 { + return bucketOfHash(principalBucketHash("user", principalID), 4).Index } ea, _ := newTestEngine(t) eb, _ := newTestEngine(t) - syncA := seedEntitlementAtDepth(t, ea, "ent-A", mkGrants(), 1) - syncB := seedEntitlementAtDepth(t, eb, "ent-A", mkGrants(), 2) + syncA := seedEntitlementAtWidth(t, ea, "ent-A", mkGrants(), 4) + syncB := seedEntitlementAtWidth(t, eb, "ent-A", mkGrants(), 8) - ra, okA, err := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") + ra, okA, err := ea.GetEntitlementDigestRoot(ctx, syncA, "ent-A") if err != nil || !okA { t.Fatalf("root A: ok=%v err=%v", okA, err) } - rb, okB, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + rb, okB, err := eb.GetEntitlementDigestRoot(ctx, syncB, "ent-A") if err != nil || !okB { t.Fatalf("root B: ok=%v err=%v", okB, err) } - if ra.Depth != 1 || rb.Depth != 2 { - t.Fatalf("depths = %d, %d; want 1, 2", ra.Depth, rb.Depth) + if ra.Bits != 4 || rb.Bits != 8 { + t.Fatalf("widths = %d, %d; want 4, 8", ra.Bits, rb.Bits) } - // Depth-independence: identical content -> identical root despite - // different tree heights. + // Split-independence: identical content -> identical root despite + // different digest widths. if !bytes.Equal(ra.Hash, rb.Hash) { - t.Fatalf("different-depth trees over identical content disagree on root:\n A(d1)=%x\n B(d2)=%x", ra.Hash, rb.Hash) + t.Fatalf("different-width digests over identical content disagree on root:\n A(w4)=%x\n B(w8)=%x", ra.Hash, rb.Hash) } dirty, err := ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") if err != nil { t.Fatalf("DirtyEntitlementBuckets (identical): %v", err) } if len(dirty) != 0 { - t.Fatalf("identical content across depths: dirty=%d, want 0", len(dirty)) + t.Fatalf("identical content across widths: dirty=%d, want 0", len(dirty)) } // Pick the principal to change and a "clean" principal known to sit - // in a different depth-1 bucket. + // in a different width-4 bucket. changed := principals[0] - changedPrefix := depth1Prefix(changed) + changedBucket := bucket4(changed) cleanP := "" for _, p := range principals[1:] { - if depth1Prefix(p) != changedPrefix { + if bucket4(p) != changedBucket { cleanP = p break } } if cleanP == "" { - t.Skip("no principal landed in a different depth-1 bucket from the changed one; can't assert localization") + t.Skip("no principal landed in a different width-4 bucket from the changed one; can't assert localization") } // Mutate the changed principal's grant in B (same external_id -> // same index key, new content hash via an added source) and rebuild - // B's tree at depth 2. + // B's digest at width 8. g := makeGrantWithSources(syncB, "g-000", "ent-A", changed, "src-ent") if err := eb.PutGrantRecord(ctx, g); err != nil { t.Fatalf("PutGrantRecord (mutate): %v", err) @@ -211,11 +237,11 @@ func TestMerkleDifferentDepthsComparison(t *testing.T) { if err != nil { t.Fatal(err) } - if err := eb.buildEntitlementMerkleAtDepth(ctx, idBytesB, "ent-A", 2); err != nil { + if err := eb.buildPartitionDigestAtWidth(ctx, grantDigestSpec, idBytesB, "ent-A", 8); err != nil { t.Fatalf("rebuild B: %v", err) } - rb2, _, _ := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + rb2, _, _ := eb.GetEntitlementDigestRoot(ctx, syncB, "ent-A") if bytes.Equal(ra.Hash, rb2.Hash) { t.Fatal("mutation did not change B's root") } @@ -225,20 +251,20 @@ func TestMerkleDifferentDepthsComparison(t *testing.T) { t.Fatalf("DirtyEntitlementBuckets (changed): %v", err) } if len(dirty) == 0 { - t.Fatal("changed principal across depths produced no dirty buckets") + t.Fatal("changed principal across widths produced no dirty buckets") } - // Localization: every dirty entry is a depth-1 prefix (not the - // whole-entitlement empty prefix). - for _, p := range dirty { - if len(p) != 1 { - t.Fatalf("dirty prefix length = %d, want 1 (compareDepth); got whole-entitlement or wrong-depth bucket", len(p)) + // Localization: every dirty entry is a width-4 bucket (not the + // whole-entitlement zero bucket). + for _, b := range dirty { + if b.Bits != 4 { + t.Fatalf("dirty bucket bits = %d, want 4 (compareBits); got whole-entitlement or wrong-width bucket", b.Bits) } } // Loading the dirty buckets in B surfaces the changed principal and // excludes the known-clean principal. loaded := map[string]bool{} - for _, prefix := range dirty { - if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + for _, b := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", b, func(g *v3.GrantRecord) bool { loaded[g.GetPrincipal().GetResourceId()] = true return true }); err != nil { @@ -253,27 +279,27 @@ func TestMerkleDifferentDepthsComparison(t *testing.T) { } } -func TestMerkleEmptyEntitlementSingleRoot(t *testing.T) { +func TestDigestEmptyEntitlementSingleRoot(t *testing.T) { ctx := context.Background() e, _ := newTestEngine(t) syncID := seedEntitlement(t, e, "ent-empty", nil) - if got := merkleNodeCount(t, e, syncID); got != 1 { - t.Fatalf("empty entitlement: merkle node count = %d, want 1 (root only)", got) + if got := digestNodeCount(t, e, syncID); got != 1 { + t.Fatalf("empty entitlement: digest node count = %d, want 1 (root only)", got) } - root, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-empty") + root, ok, err := e.GetEntitlementDigestRoot(ctx, syncID, "ent-empty") if err != nil || !ok { - t.Fatalf("GetEntitlementMerkleRoot: ok=%v err=%v", ok, err) + t.Fatalf("GetEntitlementDigestRoot: ok=%v err=%v", ok, err) } - if root.Depth != 0 { - t.Fatalf("empty entitlement depth = %d, want 0", root.Depth) + if root.Bits != 0 { + t.Fatalf("empty entitlement width = %d, want 0", root.Bits) } if root.Count != 0 { t.Fatalf("empty entitlement count = %d, want 0", root.Count) } } -func TestMerkleIdenticalGrantsSameRoot(t *testing.T) { +func TestDigestIdenticalGrantsSameRoot(t *testing.T) { ctx := context.Background() mk := func() []*v3.GrantRecord { return []*v3.GrantRecord{ @@ -287,11 +313,11 @@ func TestMerkleIdenticalGrantsSameRoot(t *testing.T) { syncA := seedEntitlement(t, ea, "ent-A", mk()) syncB := seedEntitlement(t, eb, "ent-A", mk()) - ra, okA, err := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") + ra, okA, err := ea.GetEntitlementDigestRoot(ctx, syncA, "ent-A") if err != nil || !okA { t.Fatalf("root A: ok=%v err=%v", okA, err) } - rb, okB, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + rb, okB, err := eb.GetEntitlementDigestRoot(ctx, syncB, "ent-A") if err != nil || !okB { t.Fatalf("root B: ok=%v err=%v", okB, err) } @@ -307,7 +333,7 @@ func TestMerkleIdenticalGrantsSameRoot(t *testing.T) { } } -func TestMerkleContentChangeDirtyBucket(t *testing.T) { +func TestDigestContentChangeDirtyBucket(t *testing.T) { ctx := context.Background() // Base set: same in both engines except bob's grant gains a source // in B. external_id is unchanged, so the index KEY is identical and @@ -327,8 +353,8 @@ func TestMerkleContentChangeDirtyBucket(t *testing.T) { syncA := seedEntitlement(t, ea, "ent-A", baseA) syncB := seedEntitlement(t, eb, "ent-A", baseB) - ra, _, _ := ea.GetEntitlementMerkleRoot(ctx, syncA, "ent-A") - rb, _, _ := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A") + ra, _, _ := ea.GetEntitlementDigestRoot(ctx, syncA, "ent-A") + rb, _, _ := eb.GetEntitlementDigestRoot(ctx, syncB, "ent-A") if bytes.Equal(ra.Hash, rb.Hash) { t.Fatal("content change did not change the root hash") } @@ -344,8 +370,8 @@ func TestMerkleContentChangeDirtyBucket(t *testing.T) { // Loading the dirty buckets in B must surface bob (the changed // principal) and must NOT require touching alice/carol's buckets. found := map[string]bool{} - for _, prefix := range dirty { - if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + for _, b := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", b, func(g *v3.GrantRecord) bool { found[g.GetPrincipal().GetResourceId()] = true return true }); err != nil { @@ -357,7 +383,7 @@ func TestMerkleContentChangeDirtyBucket(t *testing.T) { } } -func TestMerkleAddedGrantDirtyBucket(t *testing.T) { +func TestDigestAddedGrantDirtyBucket(t *testing.T) { ctx := context.Background() baseA := []*v3.GrantRecord{ makeGrant("", "g1", "ent-A", "alice"), @@ -381,8 +407,8 @@ func TestMerkleAddedGrantDirtyBucket(t *testing.T) { t.Fatal("added grant produced no dirty buckets") } found := map[string]bool{} - for _, prefix := range dirty { - if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", prefix, func(g *v3.GrantRecord) bool { + for _, b := range dirty { + if err := eb.IterateGrantsByEntitlementBucket(ctx, syncB, "ent-A", b, func(g *v3.GrantRecord) bool { found[g.GetPrincipal().GetResourceId()] = true return true }); err != nil { @@ -394,57 +420,66 @@ func TestMerkleAddedGrantDirtyBucket(t *testing.T) { } } -func TestMerkleVariableHeight(t *testing.T) { +func TestDigestVariableWidth(t *testing.T) { ctx := context.Background() - // Small entitlement: under one target bucket -> depth 0, single node. + // Small entitlement: under one target bucket -> width 0, single node. small := make([]*v3.GrantRecord, 0, 10) for i := 0; i < 10; i++ { small = append(small, makeGrant("", ksuid.New().String(), "ent-small", ksuid.New().String())) } es, _ := newTestEngine(t) syncS := seedEntitlement(t, es, "ent-small", small) - rootS, ok, err := es.GetEntitlementMerkleRoot(ctx, syncS, "ent-small") + rootS, ok, err := es.GetEntitlementDigestRoot(ctx, syncS, "ent-small") if err != nil || !ok { t.Fatalf("small root: ok=%v err=%v", ok, err) } - if rootS.Depth != 0 { - t.Fatalf("small entitlement depth = %d, want 0", rootS.Depth) + if rootS.Bits != 0 { + t.Fatalf("small entitlement width = %d, want 0", rootS.Bits) } if rootS.Count != 10 { t.Fatalf("small entitlement count = %d, want 10", rootS.Count) } - if got := merkleNodeCount(t, es, syncS); got != 1 { + if got := digestNodeCount(t, es, syncS); got != 1 { t.Fatalf("small entitlement node count = %d, want 1", got) } - // Large entitlement: well over the target bucket size -> depth grows, - // and the tree gains leaf nodes beyond the root. - const n = merkleTargetBucketSize*3 + 7 + // Large entitlement: well over the target bucket size -> the width + // grows one bit at a time, and the digest gains leaf nodes beyond + // the root. + const n = digestTargetBucketSize*3 + 7 large := make([]*v3.GrantRecord, 0, n) for i := 0; i < n; i++ { large = append(large, makeGrant("", ksuid.New().String(), "ent-large", ksuid.New().String())) } el, _ := newTestEngine(t) syncL := seedEntitlement(t, el, "ent-large", large) - rootL, ok, err := el.GetEntitlementMerkleRoot(ctx, syncL, "ent-large") + rootL, ok, err := el.GetEntitlementDigestRoot(ctx, syncL, "ent-large") if err != nil || !ok { t.Fatalf("large root: ok=%v err=%v", ok, err) } - if rootL.Depth < 1 { - t.Fatalf("large entitlement depth = %d, want >= 1", rootL.Depth) + if want := chooseDigestWidth(n); rootL.Bits != want { + t.Fatalf("large entitlement width = %d, want %d", rootL.Bits, want) } if rootL.Count != int64(n) { t.Fatalf("large entitlement count = %d, want %d", rootL.Count, n) } - // root + at least 2 leaves (depth>=1 over n grants spreads across - // many buckets). - if got := merkleNodeCount(t, el, syncL); got < 3 { + // root + at least 2 leaves (width>=1 over n grants spreads across + // multiple buckets). + if got := digestNodeCount(t, el, syncL); got < 3 { t.Fatalf("large entitlement node count = %d, want >= 3 (root + leaves)", got) } + // Capacity invariant: 2^width buckets at the target size must cover + // the count, and width-1 must not (else the width is too large). + if int64(1)< 0 && int64(1)<<(rootL.Bits-1)*digestTargetBucketSize >= n { + t.Fatalf("width %d is one bit wider than the count %d needs", rootL.Bits, n) + } } -// TestHashIndexIsHashOrdered verifies the new index iterates in +// TestHashIndexIsHashOrdered verifies the index iterates in // hash(principal) order: the embedded bucket-hash region is // non-decreasing across the entitlement's index range. func TestHashIndexIsHashOrdered(t *testing.T) { @@ -483,18 +518,18 @@ func TestHashIndexIsHashOrdered(t *testing.T) { } } -// dumpMerkleNodes snapshots every merkle node key/value for a sync. -// Used to byte-compare an incrementally-maintained tree against a +// dumpDigestNodes snapshots every digest node key/value for a sync. +// Used to byte-compare an incrementally-maintained digest against a // from-scratch rebuild. -func dumpMerkleNodes(t testing.TB, e *Engine, syncID string) map[string][]byte { +func dumpDigestNodes(t testing.TB, e *Engine, syncID string) map[string][]byte { t.Helper() idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatalf("resolveSyncBytes: %v", err) } iter, err := e.db.NewIter(&pebble.IterOptions{ - LowerBound: MerkleSyncLowerBound(idBytes), - UpperBound: MerkleSyncUpperBound(idBytes), + LowerBound: DigestSyncLowerBound(idBytes), + UpperBound: DigestSyncUpperBound(idBytes), }) if err != nil { t.Fatalf("NewIter: %v", err) @@ -510,9 +545,9 @@ func dumpMerkleNodes(t testing.TB, e *Engine, syncID string) map[string][]byte { return out } -// requireSameMerkleNodes fails with a per-key diff when two node +// requireSameDigestNodes fails with a per-key diff when two node // snapshots differ. -func requireSameMerkleNodes(t *testing.T, got, want map[string][]byte) { +func requireSameDigestNodes(t *testing.T, got, want map[string][]byte) { t.Helper() for k, wv := range want { gv, ok := got[k] @@ -531,12 +566,13 @@ func requireSameMerkleNodes(t *testing.T, got, want map[string][]byte) { } } -// TestMerkleAllLevelsSparseConsistent verifies the all-levels build: -// every stored node is non-empty, each interior node is exactly the XOR -// (and count-sum) of its children, the root is the fold of level 1, no -// nodes exist beyond the chosen depth, and a stored node byte-matches -// the authoritative on-demand fold of its bucket. -func TestMerkleAllLevelsSparseConsistent(t *testing.T) { +// TestDigestLeafFoldConsistent verifies the leaf-level build and the +// fold machinery the comparison rests on: every stored leaf is +// non-empty, the root is exactly the XOR (and count-sum) of the leaves, +// no nodes exist beyond root + leaves, folding the leaf level to a +// coarser width matches a manual regrouping, and a stored leaf +// byte-matches the authoritative on-demand fold of its bucket. +func TestDigestLeafFoldConsistent(t *testing.T) { ctx := context.Background() e, _ := newTestEngine(t) const n = 60 @@ -544,157 +580,161 @@ func TestMerkleAllLevelsSparseConsistent(t *testing.T) { for i := 0; i < n; i++ { grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) } - syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + syncID := seedEntitlementAtWidth(t, e, "ent-A", grants, 8) idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatal(err) } - root, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + root, ok, err := e.GetEntitlementDigestRoot(ctx, syncID, "ent-A") if err != nil || !ok { t.Fatalf("root: ok=%v err=%v", ok, err) } - if root.Depth != 2 || root.Count != n { - t.Fatalf("root depth=%d count=%d, want 2, %d", root.Depth, root.Count, n) + if root.Bits != 8 || root.Count != n { + t.Fatalf("root width=%d count=%d, want 8, %d", root.Bits, root.Count, n) } - level1, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 1, nil) + // Folding at the build width returns the stored leaves one-to-one. + leaves, err := e.foldedLeafBuckets(ctx, grantDigestSpec, idBytes, "ent-A", 8) if err != nil { t.Fatal(err) } - if len(level1) == 0 { - t.Fatal("no level-1 nodes stored") + if len(leaves) == 0 { + t.Fatal("no leaf nodes stored") } var ( rootXor [hashLen]byte rootCount int64 - level2N int ) - for _, p1 := range level1 { - c1, d1, present, err := e.getMerkleNode(idBytes, "ent-A", 1, p1) - if err != nil || !present { - t.Fatalf("level-1 node %x: present=%v err=%v", p1, present, err) - } - if c1 < 1 { - t.Fatalf("level-1 node %x stored with count %d; empty nodes must not be materialized", p1, c1) - } - children, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, p1) - if err != nil { - t.Fatal(err) + for _, l := range leaves { + if l.count < 1 { + t.Fatalf("leaf %d stored with count %d; empty leaves must not be materialized", l.idx, l.count) } - if len(children) == 0 { - t.Fatalf("level-1 node %x has no stored children", p1) - } - var ( - childXor [hashLen]byte - childCount int64 - ) - for _, p2 := range children { - c2, d2, present2, err := e.getMerkleNode(idBytes, "ent-A", 2, p2) - if err != nil || !present2 { - t.Fatalf("level-2 node %x: present=%v err=%v", p2, present2, err) - } - if c2 < 1 { - t.Fatalf("level-2 node %x stored with count %d", p2, c2) - } - xorInto(childXor[:], d2) - childCount += c2 - } - if childCount != c1 || !bytes.Equal(childXor[:], d1) { - t.Fatalf("interior node %x != fold of children: count %d vs %d", p1, c1, childCount) - } - level2N += len(children) - xorInto(rootXor[:], d1) - rootCount += c1 + xorInto(rootXor[:], l.digest[:]) + rootCount += l.count } if rootCount != root.Count || !bytes.Equal(rootXor[:], root.Hash) { - t.Fatalf("root != fold of level 1: count %d vs %d", root.Count, rootCount) + t.Fatalf("root != fold of leaves: count %d vs %d", root.Count, rootCount) } - // Exactly root + level1 + level2 nodes — nothing beyond the depth. - if got, want := merkleNodeCount(t, e, syncID), 1+len(level1)+level2N; got != want { - t.Fatalf("total node count = %d, want %d (root + L1 + L2 only)", got, want) + // Exactly root + leaves — nothing else in the keyspace. + if got, want := digestNodeCount(t, e, syncID), 1+len(leaves); got != want { + t.Fatalf("total node count = %d, want %d (root + leaves only)", got, want) } - // A stored node is a cache of the authoritative fold. - h, c, err := e.ComputeBucketHash(ctx, syncID, "ent-A", level1[0]) + // Folding to a coarser width matches a manual regroup of the + // build-width leaves. + leaves4, err := e.foldedLeafBuckets(ctx, grantDigestSpec, idBytes, "ent-A", 4) if err != nil { t.Fatal(err) } - c1, d1, _, err := e.getMerkleNode(idBytes, "ent-A", 1, level1[0]) + manual := map[uint32]*foldedBucket{} + var order []uint32 + for _, l := range leaves { + idx := l.idx >> 4 + fb, ok := manual[idx] + if !ok { + fb = &foldedBucket{idx: idx} + manual[idx] = fb + order = append(order, idx) + } + fb.count += l.count + xorInto(fb.digest[:], l.digest[:]) + } + if len(leaves4) != len(order) { + t.Fatalf("fold to width 4: %d buckets, want %d", len(leaves4), len(order)) + } + for i, idx := range order { + got, want := leaves4[i], manual[idx] + if got.idx != want.idx || got.count != want.count || got.digest != want.digest { + t.Fatalf("folded bucket %d mismatch: got {%d %d %x}, want {%d %d %x}", + i, got.idx, got.count, got.digest, want.idx, want.count, want.digest) + } + } + + // A stored leaf is a cache of the authoritative fold. + b := DigestBucket{Index: leaves[0].idx, Bits: 8} + h, c, err := e.ComputeEntitlementBucketDigest(ctx, syncID, "ent-A", b) if err != nil { t.Fatal(err) } - if c != c1 || !bytes.Equal(h, d1) { - t.Fatalf("stored node disagrees with ComputeBucketHash: count %d vs %d", c1, c) + lc, ld, present, err := e.getDigestLeaf(grantDigestSpec, idBytes, "ent-A", b.leafKeyPrefix()) + if err != nil || !present { + t.Fatalf("leaf %d: present=%v err=%v", b.Index, present, err) + } + if c != lc || !bytes.Equal(h, ld) { + t.Fatalf("stored leaf disagrees with ComputeEntitlementBucketDigest: count %d vs %d", lc, c) } } -// TestMerkleRebuildClearsStaleNodes verifies the leading DeleteRange in -// the build: a rebuild at a shallower depth must remove the deeper -// levels of the prior build, or the comparison descent would read them. -func TestMerkleRebuildClearsStaleNodes(t *testing.T) { +// TestDigestRebuildClearsStaleNodes verifies the leading DeleteRange in +// the build: a rebuild at a narrower width must remove the prior +// build's finer-grained leaves, or the comparison merge scan would read +// them. +func TestDigestRebuildClearsStaleNodes(t *testing.T) { ctx := context.Background() e, _ := newTestEngine(t) grants := make([]*v3.GrantRecord, 0, 40) for i := 0; i < 40; i++ { grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) } - syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + syncID := seedEntitlementAtWidth(t, e, "ent-A", grants, 8) idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatal(err) } - level2, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, nil) - if err != nil { - t.Fatal(err) - } - if len(level2) == 0 { - t.Fatal("depth-2 build produced no level-2 nodes") + before := rawLeafPrefixes(t, e, idBytes, "ent-A") + if len(before) == 0 { + t.Fatal("width-8 build produced no leaves") } - rootBefore, _, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + rootBefore, _, err := e.GetEntitlementDigestRoot(ctx, syncID, "ent-A") if err != nil { t.Fatal(err) } - if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 1); err != nil { - t.Fatalf("rebuild at depth 1: %v", err) + if err := e.buildPartitionDigestAtWidth(ctx, grantDigestSpec, idBytes, "ent-A", 4); err != nil { + t.Fatalf("rebuild at width 4: %v", err) } - rootAfter, ok, err := e.GetEntitlementMerkleRoot(ctx, syncID, "ent-A") + rootAfter, ok, err := e.GetEntitlementDigestRoot(ctx, syncID, "ent-A") if err != nil || !ok { t.Fatalf("root after rebuild: ok=%v err=%v", ok, err) } - if rootAfter.Depth != 1 { - t.Fatalf("root depth after rebuild = %d, want 1", rootAfter.Depth) + if rootAfter.Bits != 4 { + t.Fatalf("root width after rebuild = %d, want 4", rootAfter.Bits) } - // Depth-independence: same content, same root digest. + // Split-independence: same content, same root digest. if !bytes.Equal(rootBefore.Hash, rootAfter.Hash) || rootBefore.Count != rootAfter.Count { - t.Fatal("rebuild at different depth changed the root digest/count over identical content") + t.Fatal("rebuild at different width changed the root digest/count over identical content") } - level2After, err := e.merkleChildPrefixes(ctx, idBytes, "ent-A", 2, nil) - if err != nil { - t.Fatal(err) + // Every surviving leaf prefix must be width-4 aligned (low 12 bits + // of the left-aligned prefix zero) — a width-8 leaf that escaped the + // range-clear would fail this. + after := rawLeafPrefixes(t, e, idBytes, "ent-A") + if len(after) == 0 || len(after) > 16 { + t.Fatalf("width-4 rebuild stored %d leaves, want 1..16", len(after)) } - if len(level2After) != 0 { - t.Fatalf("%d stale level-2 nodes survived the depth-1 rebuild", len(level2After)) + for _, p := range after { + if lv := binary.BigEndian.Uint16(p); lv&0x0FFF != 0 { + t.Fatalf("stale leaf prefix %x survived the width-4 rebuild", p) + } } } -// TestMerkleIncrementalEqualsRebuild is the §7 keystone invariant: after +// TestDigestIncrementalEqualsRebuild is the §7 keystone invariant: after // a sequence of post-seal inserts, content overwrites, a bucket-moving // (principal-changing) overwrite, an excluded-field no-op overwrite, -// deletes, and a multi-record batch, the incrementally-maintained tree -// byte-equals a from-scratch rebuild. -func TestMerkleIncrementalEqualsRebuild(t *testing.T) { +// deletes, and a multi-record batch, the incrementally-maintained +// digest byte-equals a from-scratch rebuild. +func TestDigestIncrementalEqualsRebuild(t *testing.T) { ctx := context.Background() e, _ := newTestEngine(t) grants := make([]*v3.GrantRecord, 0, 30) for i := 0; i < 30; i++ { grants = append(grants, makeGrant("", fmt.Sprintf("g-%03d", i), "ent-A", fmt.Sprintf("user-%03d", i))) } - syncID := seedEntitlementAtDepth(t, e, "ent-A", grants, 2) + syncID := seedEntitlementAtWidth(t, e, "ent-A", grants, 8) idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatal(err) @@ -717,7 +757,7 @@ func TestMerkleIncrementalEqualsRebuild(t *testing.T) { // must apply as remove(old path) + add(new path). put(makeGrant(syncID, "g-006", "ent-A", "user-moved")) // Excluded-field overwrite: needs_expansion is not part of the - // content hash, so this must leave the tree untouched. + // content hash, so this must leave the digest untouched. noop := makeGrant(syncID, "g-007", "ent-A", "user-007") noop.SetNeedsExpansion(true) put(noop) @@ -741,7 +781,7 @@ func TestMerkleIncrementalEqualsRebuild(t *testing.T) { } // Sparsity restored on delete: unless another remaining principal - // shares user-008's depth-2 prefix, its leaf must be gone. + // shares user-008's width-8 bucket, its leaf must be gone. remaining := []string{"user-moved", "new-user-1", "new-user-2", "batch-user-1", "batch-user-2", "batch-user-3"} for i := 0; i < 30; i++ { if i == 6 || i == 8 || i == 9 { @@ -749,16 +789,16 @@ func TestMerkleIncrementalEqualsRebuild(t *testing.T) { } remaining = append(remaining, fmt.Sprintf("user-%03d", i)) } - deletedPrefix := principalBucketHash("user", "user-008")[:2] + deletedLeaf := bucketOfHash(principalBucketHash("user", "user-008"), 8).leafKeyPrefix() shared := false for _, p := range remaining { - if bytes.Equal(principalBucketHash("user", p)[:2], deletedPrefix) { + if bytes.Equal(bucketOfHash(principalBucketHash("user", p), 8).leafKeyPrefix(), deletedLeaf) { shared = true break } } if !shared { - _, _, present, err := e.getMerkleNode(idBytes, "ent-A", 2, deletedPrefix) + _, _, present, err := e.getDigestLeaf(grantDigestSpec, idBytes, "ent-A", deletedLeaf) if err != nil { t.Fatal(err) } @@ -767,25 +807,26 @@ func TestMerkleIncrementalEqualsRebuild(t *testing.T) { } } - incremental := dumpMerkleNodes(t, e, syncID) - if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 2); err != nil { + incremental := dumpDigestNodes(t, e, syncID) + if err := e.buildPartitionDigestAtWidth(ctx, grantDigestSpec, idBytes, "ent-A", 8); err != nil { t.Fatalf("rebuild: %v", err) } - rebuilt := dumpMerkleNodes(t, e, syncID) - requireSameMerkleNodes(t, incremental, rebuilt) + rebuilt := dumpDigestNodes(t, e, syncID) + requireSameDigestNodes(t, incremental, rebuilt) } -// TestMerkleSameBatchSameBucketRMW pins the accumulator behavior: one +// TestDigestSameBatchSameBucketRMW pins the accumulator behavior: one // PutGrantRecords batch adds several grants that land in the SAME -// depth-1 bucket. A naive read-modify-write against the batch would +// width-8 bucket. A naive read-modify-write against the batch would // lose all but one delta (a plain pebble.Batch doesn't read through its // own writes); the accumulator must fold all of them into one node // write. -func TestMerkleSameBatchSameBucketRMW(t *testing.T) { +func TestDigestSameBatchSameBucketRMW(t *testing.T) { ctx := context.Background() e, _ := newTestEngine(t) - // Find three principals whose bucket hashes share a first byte. + // Find three principals whose bucket hashes share a first byte — + // the top 8 bits, i.e. the same width-8 bucket. collide := []string{"seed-principal"} target := principalBucketHash("user", collide[0])[0] for i := 0; len(collide) < 3; i++ { @@ -799,7 +840,7 @@ func TestMerkleSameBatchSameBucketRMW(t *testing.T) { for i := 0; i < 5; i++ { base = append(base, makeGrant("", fmt.Sprintf("b-%d", i), "ent-A", fmt.Sprintf("base-%d", i))) } - syncID := seedEntitlementAtDepth(t, e, "ent-A", base, 1) + syncID := seedEntitlementAtWidth(t, e, "ent-A", base, 8) idBytes, err := e.resolveSyncBytes(syncID) if err != nil { t.Fatal(err) @@ -819,26 +860,26 @@ func TestMerkleSameBatchSameBucketRMW(t *testing.T) { want++ } } - count, _, present, err := e.getMerkleNode(idBytes, "ent-A", 1, []byte{target}) + count, _, present, err := e.getDigestLeaf(grantDigestSpec, idBytes, "ent-A", []byte{target, 0}) if err != nil || !present { - t.Fatalf("bucket node %x: present=%v err=%v", target, present, err) + t.Fatalf("bucket leaf %x: present=%v err=%v", target, present, err) } if count != want { t.Fatalf("same-batch deltas lost: bucket count = %d, want %d", count, want) } - incremental := dumpMerkleNodes(t, e, syncID) - if err := e.buildEntitlementMerkleAtDepth(ctx, idBytes, "ent-A", 1); err != nil { + incremental := dumpDigestNodes(t, e, syncID) + if err := e.buildPartitionDigestAtWidth(ctx, grantDigestSpec, idBytes, "ent-A", 8); err != nil { t.Fatalf("rebuild: %v", err) } - requireSameMerkleNodes(t, incremental, dumpMerkleNodes(t, e, syncID)) + requireSameDigestNodes(t, incremental, dumpDigestNodes(t, e, syncID)) } -// TestMerkleMissingRootFallback: a missing root means "tree never +// TestDigestMissingRootFallback: a missing root means "digest never // built", not "no grants". Comparison against a populated-but-unbuilt // side must fall back to the authoritative fold — clean when content is // identical, whole-entitlement dirty when it differs. -func TestMerkleMissingRootFallback(t *testing.T) { +func TestDigestMissingRootFallback(t *testing.T) { ctx := context.Background() mk := func() []*v3.GrantRecord { return []*v3.GrantRecord{ @@ -850,7 +891,7 @@ func TestMerkleMissingRootFallback(t *testing.T) { ea, _ := newTestEngine(t) syncA := seedEntitlement(t, ea, "ent-A", mk()) - // B holds the same grants but never builds a tree. + // B holds the same grants but never builds a digest. eb, _ := newTestEngine(t) syncB := ksuid.New().String() if err := eb.SetCurrentSync(syncB); err != nil { @@ -863,20 +904,20 @@ func TestMerkleMissingRootFallback(t *testing.T) { t.Fatal(err) } } - if _, ok, err := eb.GetEntitlementMerkleRoot(ctx, syncB, "ent-A"); err != nil || ok { + if _, ok, err := eb.GetEntitlementDigestRoot(ctx, syncB, "ent-A"); err != nil || ok { t.Fatalf("B unexpectedly has a root: ok=%v err=%v", ok, err) } - for name, dirtyFn := range map[string]func() ([][]byte, error){ - "A vs B": func() ([][]byte, error) { return ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") }, - "B vs A": func() ([][]byte, error) { return eb.DirtyEntitlementBuckets(ctx, syncB, ea, syncA, "ent-A") }, + for name, dirtyFn := range map[string]func() ([]DigestBucket, error){ + "A vs B": func() ([]DigestBucket, error) { return ea.DirtyEntitlementBuckets(ctx, syncA, eb, syncB, "ent-A") }, + "B vs A": func() ([]DigestBucket, error) { return eb.DirtyEntitlementBuckets(ctx, syncB, ea, syncA, "ent-A") }, } { dirty, err := dirtyFn() if err != nil { t.Fatalf("%s: %v", name, err) } if len(dirty) != 0 { - t.Fatalf("%s: identical content with one tree unbuilt: dirty=%d, want 0 (fold fallback)", name, len(dirty)) + t.Fatalf("%s: identical content with one digest unbuilt: dirty=%d, want 0 (fold fallback)", name, len(dirty)) } } @@ -888,7 +929,7 @@ func TestMerkleMissingRootFallback(t *testing.T) { if err != nil { t.Fatal(err) } - if len(dirty) != 1 || len(dirty[0]) != 0 { - t.Fatalf("diverged content with one tree unbuilt: dirty=%v, want one empty prefix", dirty) + if len(dirty) != 1 || dirty[0].Bits != 0 { + t.Fatalf("diverged content with one digest unbuilt: dirty=%v, want one whole-entitlement bucket", dirty) } } diff --git a/pkg/dotc1z/engine/pebble/grant_digest.go b/pkg/dotc1z/engine/pebble/grant_digest.go new file mode 100644 index 000000000..5416b57eb --- /dev/null +++ b/pkg/dotc1z/engine/pebble/grant_digest.go @@ -0,0 +1,260 @@ +package pebble + +import ( + "context" + "encoding/binary" + "errors" + "fmt" + "sort" + + "github.com/cespare/xxhash/v2" + "github.com/cockroachdb/pebble/v2" + + v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" +) + +// Grant instantiation of the digest core (digest.go): the per- +// entitlement grant digest, folded over the +// by_entitlement_principal_hash index. +// +// partition = entitlement_id +// bucket hash = principalBucketHash (identity of the principal) +// content hash = grantContentHash (the membership edge) +// +// This answers "does this entitlement have exactly the same grants as +// some other sync/file?" with a single root read, and localizes any +// difference to principal-hash buckets so the diff driver loads only +// those grants (see adapter_diff.go). + +// grantDigestSpec wires the grant hash index into the digest core. The +// index's key layout (encodeGrantByEntPrincHashIndexKey) satisfies the +// digestIndexSpec shape contract: partition prefix, then the raw 8-byte +// bucket hash, then the principal/external_id tail; the value is the +// grant content hash. +var grantDigestSpec = digestIndexSpec{ + indexID: idxGrantByEntitlementPrincipalHash, + partitionPrefix: encodeGrantByEntPrincHashEntPrefix, + syncBounds: func(syncIDBytes []byte) ([]byte, []byte) { + return GrantByEntPrincHashSyncLowerBound(syncIDBytes), GrantByEntPrincHashSyncUpperBound(syncIDBytes) + }, +} + +// principalBucketHash is the bucket address for a principal: the 8-byte +// xxHash64 of (rt + "\x00" + id). Identity only — never the principal's +// full object — so the address is stable across syncs even when the +// principal's attributes change. Returns a fresh slice. +func principalBucketHash(rt, id string) []byte { + h := xxhash.New() + _, _ = h.WriteString(rt) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(id) + out := make([]byte, digestBucketHashLen) + binary.BigEndian.PutUint64(out, h.Sum64()) + return out +} + +// grantContentHash is the canonical content hash of a grant — the value +// stored in the hash index and the unit the grant digest folds. +// +// ABI: the field set below defines what "the same grant" means for the +// diff. It deliberately covers the membership EDGE (entitlement id, +// principal identity, external_id) plus the grant's source-entitlement +// set (expansion provenance), and deliberately EXCLUDES sync-relative +// and transient processing state — sync_id, discovered_at, +// needs_expansion, expansion, and annotations — none of which change +// "which principal holds which entitlement". This is a hand-rolled +// framing, NOT proto marshal: deterministic-proto output is not +// canonical across protobuf library versions, which would make two +// files written by different SDK builds hash identical grants +// differently. Changing this set requires an index-migration bump. +func grantContentHash(r *v3.GrantRecord) []byte { + h := xxhash.New() + ent := r.GetEntitlement() + princ := r.GetPrincipal() + _, _ = h.WriteString(ent.GetEntitlementId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(princ.GetResourceTypeId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(princ.GetResourceId()) + _, _ = h.Write([]byte{0}) + _, _ = h.WriteString(r.GetExternalId()) + _, _ = h.Write([]byte{0}) + + // Source-entitlement ids, sorted for order-independence. The map + // values (GrantSourceRecord) are not folded in v1 — only the set of + // source ids, which is the membership-composition signal. + sources := r.GetSources() + ids := make([]string, 0, len(sources)) + for k := range sources { + ids = append(ids, k) + } + sort.Strings(ids) + for _, id := range ids { + _, _ = h.WriteString(id) + _, _ = h.Write([]byte{0}) + } + out := make([]byte, hashLen) + binary.BigEndian.PutUint64(out, h.Sum64()) + return out +} + +// grantHashIndexKey returns the by_entitlement_principal_hash index key +// for r, or nil when the grant lacks the entitlement/principal needed to +// place it (mirrors the by_entitlement index's nil-guard). +func grantHashIndexKey(syncIDBytes []byte, r *v3.GrantRecord) []byte { + ent := r.GetEntitlement() + princ := r.GetPrincipal() + if ent == nil || princ == nil { + return nil + } + bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) + return encodeGrantByEntPrincHashIndexKey( + syncIDBytes, ent.GetEntitlementId(), bh, + princ.GetResourceTypeId(), princ.GetResourceId(), r.GetExternalId(), + ) +} + +// BuildAllGrantDigests rebuilds the grant digest for every entitlement +// in syncID. Called at seal time (Adapter.EndSync) after all grants are +// written, and by the on-Open migration backfill. Every entitlement +// gets a digest — including those with zero grants, which store a +// single root node — so a reader can always distinguish "empty" from +// "never built". +func (e *Engine) BuildAllGrantDigests(ctx context.Context, syncID string) error { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return err + } + // Collect entitlement ids first: the build writes into the + // typeDigest keyspace while we'd otherwise be mid-iteration over + // typeEntitlement. Different keyspaces, but snapshotting the ids + // keeps the iterator and the writes cleanly separated. + var ents []string + if err := e.IterateEntitlementsBySync(ctx, syncID, func(r *v3.EntitlementRecord) bool { + ents = append(ents, r.GetExternalId()) + return true + }); err != nil { + return fmt.Errorf("BuildAllGrantDigests: list entitlements: %w", err) + } + for _, ent := range ents { + if err := ctx.Err(); err != nil { + return err + } + if err := e.buildPartitionDigest(ctx, grantDigestSpec, idBytes, ent); err != nil { + return fmt.Errorf("BuildAllGrantDigests: entitlement %q: %w", ent, err) + } + } + return nil +} + +// GetEntitlementDigestRoot returns the stored grant-digest root for an +// entitlement. ok is false when no digest has been built for it (the +// caller can fall back to ComputeEntitlementBucketDigest, which derives +// the same digest from the index on demand). +func (e *Engine) GetEntitlementDigestRoot(ctx context.Context, syncID, entitlementID string) (DigestRoot, bool, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return DigestRoot{}, false, err + } + return e.getPartitionDigestRoot(grantDigestSpec, idBytes, entitlementID) +} + +// ComputeEntitlementBucketDigest folds the grant hash index over a +// single bucket of an entitlement (the zero bucket = the whole +// entitlement) — the authoritative on-demand counterpart of the stored +// digest nodes. +func (e *Engine) ComputeEntitlementBucketDigest(ctx context.Context, syncID, entitlementID string, bucket DigestBucket) ([]byte, int64, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return nil, 0, err + } + return e.computeBucketDigest(ctx, grantDigestSpec, idBytes, entitlementID, bucket) +} + +// DirtyEntitlementBuckets compares this engine's entitlement against +// other's and returns the buckets whose grants differ — see +// dirtyPartitionBuckets for the comparison contract (zero bucket = +// whole entitlement; nil = identical). +func (e *Engine) DirtyEntitlementBuckets(ctx context.Context, syncID string, other *Engine, otherSyncID, entitlementID string) ([]DigestBucket, error) { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return nil, err + } + otherIDBytes, err := other.resolveSyncBytes(otherSyncID) + if err != nil { + return nil, err + } + return e.dirtyPartitionBuckets(ctx, grantDigestSpec, idBytes, other, otherIDBytes, entitlementID) +} + +// IterateGrantsByEntitlementBucket yields the grants in one +// principal-hash bucket of an entitlement (the zero bucket = the whole +// entitlement). This is the dirty-bucket loader: after a digest +// comparison flags a bucket, the caller materializes only those grants. +// Like the other index iterators it does a point Get per entry to fetch +// the primary; orphan index entries are skipped. +func (e *Engine) IterateGrantsByEntitlementBucket(ctx context.Context, syncID, entitlementID string, bucket DigestBucket, yield func(*v3.GrantRecord) bool) error { + idBytes, err := e.resolveSyncBytes(syncID) + if err != nil { + return err + } + entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) + lower, upper := grantDigestSpec.bucketBounds(idBytes, entitlementID, bucket) + iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) + if err != nil { + return err + } + defer iter.Close() + for iter.First(); iter.Valid(); iter.Next() { + if err := ctx.Err(); err != nil { + return err + } + _, _, _, externalID, ok := decodeEntPrincHashTail(iter.Key(), entPrefix) + if !ok { + continue + } + val, closer, getErr := e.db.Get(encodeGrantKey(idBytes, externalID)) + if getErr != nil { + if errors.Is(getErr, pebble.ErrNotFound) { + continue + } + return getErr + } + r := &v3.GrantRecord{} + uErr := unmarshalRecord(val, r) + closer.Close() + if uErr != nil { + return fmt.Errorf("IterateGrantsByEntitlementBucket: unmarshal: %w", uErr) + } + if !yield(r) { + return nil + } + } + return iter.Error() +} + +// newGrantDigestMutator returns a digestMutator bound to the grant +// digest; feed it with addGrant/removeGrant. +func newGrantDigestMutator(e *Engine) *digestMutator { + return newDigestMutator(e, grantDigestSpec) +} + +// addGrant records r's insertion into its entitlement's grant digest. +func (m *digestMutator) addGrant(idBytes []byte, r *v3.GrantRecord) error { + return m.grantDelta(idBytes, r, m.addHash) +} + +// removeGrant records r's removal from its entitlement's grant digest. +func (m *digestMutator) removeGrant(idBytes []byte, r *v3.GrantRecord) error { + return m.grantDelta(idBytes, r, m.removeHash) +} + +func (m *digestMutator) grantDelta(idBytes []byte, r *v3.GrantRecord, apply func(idBytes []byte, partition string, bucketHash, contentHash []byte) error) error { + ent := r.GetEntitlement() + princ := r.GetPrincipal() + if ent == nil || princ == nil { + return nil // not in the hash index → not in the digest + } + bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) + return apply(idBytes, ent.GetEntitlementId(), bh, grantContentHash(r)) +} diff --git a/pkg/dotc1z/engine/pebble/grants.go b/pkg/dotc1z/engine/pebble/grants.go index bb276e06a..1c99f9596 100644 --- a/pkg/dotc1z/engine/pebble/grants.go +++ b/pkg/dotc1z/engine/pebble/grants.go @@ -81,12 +81,12 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord // emitting an external_id on two pages). skipGet := e.takeFreshGrantsEmpty() - // Incremental merkle maintenance applies only on the non-fresh - // path: during a fresh sync the trees don't exist yet (built + // Incremental digest maintenance applies only on the non-fresh + // path: during a fresh sync the digests don't exist yet (built // once at seal), so skip even the per-entitlement root probe. - var mm *merkleMutator + var mm *digestMutator if !fresh { - mm = newMerkleMutator(e) + mm = newGrantDigestMutator(e) } // Dedup pre-pass: keep only the LAST occurrence of each @@ -161,7 +161,7 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord } closer.Close() if mm != nil { - if err := mm.remove(idBytes, old); err != nil { + if err := mm.removeGrant(idBytes, old); err != nil { return err } } @@ -178,7 +178,7 @@ func (e *Engine) PutGrantRecords(ctx context.Context, records ...*v3.GrantRecord return err } if mm != nil { - if err := mm.add(idBytes, r); err != nil { + if err := mm.addGrant(idBytes, r); err != nil { return err } } @@ -389,8 +389,8 @@ func (e *Engine) DeleteGrantRecord(ctx context.Context, syncID, externalID strin } closer.Close() - mm := newMerkleMutator(e) - if err := mm.remove(idBytes, old); err != nil { + mm := newGrantDigestMutator(e) + if err := mm.removeGrant(idBytes, old); err != nil { return err } if err := mm.apply(batch); err != nil { @@ -515,10 +515,10 @@ func (e *Engine) deleteGrantIndexes(batch *pebble.Batch, syncIDBytes []byte, r * // by_entitlement_principal_hash: the bucket hash is derived from the // principal identity, so deleteGrantIndexes reconstructs the same key // writeGrantIndexes wrote. Skipped when ent/principal are absent, - // matching grantHashIndexKey's nil-guard. The entitlement's merkle - // tree is kept in step separately: callers on the post-seal mutation - // paths feed the same old record to merkleMutator.remove (and the - // new one to .add), which folds the change into the stored nodes in + // matching grantHashIndexKey's nil-guard. The entitlement's digest is + // kept in step separately: callers on the post-seal mutation paths + // feed the same old record to digestMutator.removeGrant (and the new + // one to .addGrant), which folds the change into the stored nodes in // the same batch. if ent != nil && princ != nil { bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) diff --git a/pkg/dotc1z/engine/pebble/if_newer.go b/pkg/dotc1z/engine/pebble/if_newer.go index 482d8511c..678854d17 100644 --- a/pkg/dotc1z/engine/pebble/if_newer.go +++ b/pkg/dotc1z/engine/pebble/if_newer.go @@ -38,9 +38,9 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran batch := e.db.NewBatch() defer batch.Close() // IfNewer is by definition a post-seal mutation path, so the - // merkle trees (cloned along with the sync's keyspace) are kept - // in step incrementally. - mm := newMerkleMutator(e) + // grant digests (cloned along with the sync's keyspace) are + // kept in step incrementally. + mm := newGrantDigestMutator(e) written := 0 for _, r := range records { if r == nil { @@ -66,7 +66,7 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran if err := e.deleteGrantIndexes(batch, idBytes, old); err != nil { return err } - if err := mm.remove(idBytes, old); err != nil { + if err := mm.removeGrant(idBytes, old); err != nil { return err } case errors.Is(getErr, pebble.ErrNotFound): @@ -84,7 +84,7 @@ func (e *Engine) PutGrantRecordsIfNewer(ctx context.Context, records ...*v3.Gran if err := e.writeGrantIndexes(batch, idBytes, r); err != nil { return err } - if err := mm.add(idBytes, r); err != nil { + if err := mm.addGrant(idBytes, r); err != nil { return err } written++ diff --git a/pkg/dotc1z/engine/pebble/index_migrations.go b/pkg/dotc1z/engine/pebble/index_migrations.go index f6395878b..9f94c09e6 100644 --- a/pkg/dotc1z/engine/pebble/index_migrations.go +++ b/pkg/dotc1z/engine/pebble/index_migrations.go @@ -66,31 +66,39 @@ type indexMigration struct { var indexMigrations = []indexMigration{ { // Backfill the by_entitlement_principal_hash index and the - // per-entitlement merkle trees for files written before either + // per-entitlement grant digests for files written before either // existed. Idempotent: re-emitting an index entry is a Set over - // the same key/value, and each tree rebuild range-clears the - // entitlement's typeMerkle keyspace before writing. + // the same key/value, and each digest rebuild range-clears the + // partition's typeDigest keyspace before writing. // New files persist this version at their initial (empty) Open, // so the inline write path maintains both and the backfill never // re-runs for them. // - // v1: XOR combiner, all levels stored sparsely, count on every - // node (RFC 0003). Uses xxHash64 (8-byte digests) for both - // principalBucketHash and grantContentHash. + // v1: XOR combiner, 256-ary radix with all levels stored + // sparsely, count on every node (RFC 0003). Uses xxHash64 + // (8-byte digests) for both principalBucketHash and + // grantContentHash. + // v2: flat shape — root + a single leaf level of 2^width + // buckets (width in bits, chosen per entitlement), leaf keys + // carrying a 2-byte left-aligned bucket index; node keys carry + // the digested index's id after the sync_id (see + // encodeDigestNodeKey). Index entries are unchanged; the bump + // forces a digest rebuild, whose per-partition range-clear + // removes the v1 nodes. Name: "grant_by_entitlement_principal_hash", - Version: 1, + Version: 2, Apply: func(ctx context.Context, e *Engine) error { - return e.backfillGrantHashIndexAndMerkle(ctx) + return e.backfillGrantHashIndexAndDigests(ctx) }, }, } -// backfillGrantHashIndexAndMerkle reconstructs the +// backfillGrantHashIndexAndDigests reconstructs the // by_entitlement_principal_hash index for every grant in every sync, -// then rebuilds the per-entitlement merkle trees. The index must be -// committed before the trees are built because BuildAllMerkleTrees folds -// over the committed index. -func (e *Engine) backfillGrantHashIndexAndMerkle(ctx context.Context) error { +// then rebuilds the per-entitlement grant digests. The index must be +// committed before the digests are built because BuildAllGrantDigests +// folds over the committed index. +func (e *Engine) backfillGrantHashIndexAndDigests(ctx context.Context) error { var syncIDs []string if err := e.IterateAllSyncRuns(ctx, func(r *v3.SyncRunRecord) bool { syncIDs = append(syncIDs, r.GetSyncId()) @@ -130,8 +138,8 @@ func (e *Engine) backfillGrantHashIndexAndMerkle(ctx context.Context) error { } batch.Close() - if err := e.BuildAllMerkleTrees(ctx, syncID); err != nil { - return fmt.Errorf("backfill merkle trees: sync %q: %w", syncID, err) + if err := e.BuildAllGrantDigests(ctx, syncID); err != nil { + return fmt.Errorf("backfill grant digests: sync %q: %w", syncID, err) } } return nil diff --git a/pkg/dotc1z/engine/pebble/keys.go b/pkg/dotc1z/engine/pebble/keys.go index 7add2fcab..341c53610 100644 --- a/pkg/dotc1z/engine/pebble/keys.go +++ b/pkg/dotc1z/engine/pebble/keys.go @@ -61,7 +61,7 @@ const ( typeIndex byte = 0x07 typeCounter byte = 0x08 typeSession byte = 0x09 - typeMerkle byte = 0x0A + typeDigest byte = 0x0A typeEngineMeta byte = 0xFF ) @@ -80,8 +80,8 @@ const ( // idxGrantByEntitlementPrincipalHash sorts grants by // (entitlement_id, hash(principal)). Unlike every other grant // index its entries carry a VALUE (the grant content hash). It is - // the substrate the per-entitlement merkle tree (typeMerkle) folds - // over; see merkle.go. + // the substrate the per-entitlement grant digest (typeDigest) + // folds over; see digest.go and grant_digest.go. idxGrantByEntitlementPrincipalHash byte = 0x08 ) @@ -282,21 +282,21 @@ func encodeGrantByEntitlementResourcePrefix(syncIDBytes []byte, entRT, entRID st return codec.AppendTupleSeparator(buf) } -// --- Grant by (entitlement, principal-hash) + Merkle --- +// --- Grant by (entitlement, principal-hash) + digest nodes --- // encodeGrantByEntPrincHashIndexKey is the by_entitlement_principal_hash // secondary index on GrantRecord. Unlike every other grant index it // interposes a RAW, fixed-width principal-bucket hash between the // entitlement_id and the principal tuple, so the keyspace sorts by // (entitlement_id, hash(principal)). That hash-major order is what the -// per-entitlement merkle tree folds over, and a hash PREFIX is a clean -// byte-prefix of the key — which is the property the merkle bucket range -// scans rely on. +// per-entitlement grant digest folds over, and a bucket — a bit-range +// of the hash — is a contiguous key range, which is the property the +// digest bucket range scans rely on (see digestIndexSpec.bucketBounds). // // v3 | typeIndex | idxGrantByEntitlementPrincipalHash | sync_id | 0x00 | -// entitlement_id | 0x00 | | +// entitlement_id | 0x00 | | // principal_rt | 0x00 | principal_id | 0x00 | external_id -// -> value: grant content hash (sha256, 32 bytes) +// -> value: grant content hash (xxHash64, 8 bytes) // // Because the bucket hash is raw it can contain 0x00, so the generic // tuple walkers (lastTupleComponent / decodeTwoTupleComponents) must NOT @@ -305,7 +305,7 @@ func encodeGrantByEntitlementResourcePrefix(syncIDBytes []byte, entRT, entRID st // the hash's fixed width positionally. // // Paired with encodeGrantByEntPrincHashEntPrefix (by-value prefix, with -// trailing sep) and encodeGrantByEntPrincHashBucketPrefix. +// trailing sep) and digestIndexSpec.bucketBounds (digest.go). func encodeGrantByEntPrincHashIndexKey(syncIDBytes []byte, entitlementID string, bucketHash []byte, principalRT, principalID, externalID string) []byte { buf := make([]byte, 0, 8+len(syncIDBytes)+len(entitlementID)+len(bucketHash)+len(principalRT)+len(principalID)+len(externalID)) buf = append(buf, versionV3, typeIndex, idxGrantByEntitlementPrincipalHash) @@ -331,14 +331,6 @@ func encodeGrantByEntPrincHashEntPrefix(syncIDBytes []byte, entitlementID string return codec.AppendTupleSeparator(buf) } -// encodeGrantByEntPrincHashBucketPrefix narrows the entitlement prefix to -// a single principal-hash bucket identified by a raw hash prefix. An -// empty hashPrefix yields the whole-entitlement prefix (the depth-0 -// bucket). Used as the LowerBound for a bucket range scan. -func encodeGrantByEntPrincHashBucketPrefix(syncIDBytes []byte, entitlementID string, hashPrefix []byte) []byte { - return append(encodeGrantByEntPrincHashEntPrefix(syncIDBytes, entitlementID), hashPrefix...) -} - // decodeEntPrincHashTail decodes one index key relative to entPrefix // (which must be encodeGrantByEntPrincHashEntPrefix output — i.e. it // ends right before the raw bucket hash). Returns the raw bucket hash @@ -346,11 +338,11 @@ func encodeGrantByEntPrincHashBucketPrefix(syncIDBytes []byte, entitlementID str // rt/id and the external_id. ok is false if the key is shorter than // prefix+hash or the tuple tail is malformed. func decodeEntPrincHashTail(key, entPrefix []byte) ([]byte, string, string, string, bool) { - if len(key) < len(entPrefix)+merkleBucketHashLen { + if len(key) < len(entPrefix)+digestBucketHashLen { return nil, "", "", "", false } - bucketHash := key[len(entPrefix) : len(entPrefix)+merkleBucketHashLen] - tail := key[len(entPrefix)+merkleBucketHashLen:] + bucketHash := key[len(entPrefix) : len(entPrefix)+digestBucketHashLen] + tail := key[len(entPrefix)+digestBucketHashLen:] rt, next, err := codec.DecodeTupleStringTo(nil, tail, 0) if err != nil || next >= len(tail) { return nil, "", "", "", false @@ -379,48 +371,54 @@ func GrantByEntPrincHashSyncUpperBound(syncIDBytes []byte) []byte { return upperBoundOf(GrantByEntPrincHashSyncLowerBound(syncIDBytes)) } -// Merkle node keys. +// Digest node keys. // -// v3 | typeMerkle | sync_id | 0x00 | entitlement_id | 0x00 | level(1 byte) | bucket_prefix(level raw bytes) +// v3 | typeDigest | sync_id | index_id(1 byte) | 0x00 | partition | 0x00 | level(1 byte) | bucket_prefix // -// level 0 is the root (bucket_prefix empty); every level from 1 to the -// tree's depth holds that level's non-empty nodes, one per non-empty -// principal-hash prefix. bucket_prefix is the first `level` bytes of the -// principal bucket hash, raw, so it aligns byte-for-byte with the index -// key's bucket-hash region — and because the level byte precedes the -// prefix, the children of one node are a contiguous key range one level -// down. See merkle.go for the node value framing. -func encodeMerkleNodeKey(syncIDBytes []byte, entitlementID string, level byte, bucketPrefix []byte) []byte { - buf := encodeMerkleEntPrefix(syncIDBytes, entitlementID) +// index_id discriminates WHICH digested index the node belongs to (the +// digested index's own idx* byte, see digestIndexSpec) — it sits right +// after the fixed-width sync_id so one sync's digests for all indexes +// are still a single contiguous range for the cleanup/clone plans. +// level 0 is the root (bucket_prefix empty); level 1 is the single leaf +// level, one node per non-empty bucket, whose bucket_prefix is the +// bucket index LEFT-ALIGNED in 2 raw bytes (digestLeafPrefixLen). The +// left alignment makes leaf keys sort in bucket-hash order at every +// digest width, so the comparison's fold-to-coarser-width merge is a +// single contiguous scan of this range. See digest.go for the node +// value framing. +func encodeDigestNodeKey(syncIDBytes []byte, indexID byte, partition string, level byte, bucketPrefix []byte) []byte { + buf := encodeDigestPartitionPrefix(syncIDBytes, indexID, partition) buf = append(buf, level) return append(buf, bucketPrefix...) } -// encodeMerkleEntPrefix is the prefix of every merkle node key for one -// entitlement — the range a rebuild clears before writing (the build -// only Sets nodes; without the leading DeleteRange a depth change or an -// emptied bucket would leave stale nodes for the comparison descent to -// read) and the range merkleMutator drops on detecting an inconsistent -// tree. -func encodeMerkleEntPrefix(syncIDBytes []byte, entitlementID string) []byte { - buf := make([]byte, 0, 6+len(syncIDBytes)+len(entitlementID)) - buf = append(buf, versionV3, typeMerkle) +// encodeDigestPartitionPrefix is the prefix of every digest node key +// for one (index, partition) — the range a rebuild clears before +// writing (the build only Sets nodes; without the leading DeleteRange a +// width change or an emptied bucket would leave stale nodes for the +// comparison merge scan to read) and the range digestMutator drops on +// detecting an inconsistent digest. +func encodeDigestPartitionPrefix(syncIDBytes []byte, indexID byte, partition string) []byte { + buf := make([]byte, 0, 7+len(syncIDBytes)+len(partition)) + buf = append(buf, versionV3, typeDigest) buf = append(buf, syncIDBytes...) + buf = append(buf, indexID) buf = codec.AppendTupleSeparator(buf) - buf = codec.AppendTupleStrings(buf, entitlementID) + buf = codec.AppendTupleStrings(buf, partition) return codec.AppendTupleSeparator(buf) } -// MerkleSyncLowerBound / UpperBound bound the entire merkle keyspace for -// a sync. Exported for the cleanup/clone/compaction keyspace plans. -func MerkleSyncLowerBound(syncIDBytes []byte) []byte { +// DigestSyncLowerBound / UpperBound bound the entire digest keyspace +// (all digested indexes) for a sync. Exported for the +// cleanup/clone/compaction keyspace plans. +func DigestSyncLowerBound(syncIDBytes []byte) []byte { buf := make([]byte, 0, 2+len(syncIDBytes)) - buf = append(buf, versionV3, typeMerkle) + buf = append(buf, versionV3, typeDigest) return append(buf, syncIDBytes...) } -func MerkleSyncUpperBound(syncIDBytes []byte) []byte { - return upperBoundOf(MerkleSyncLowerBound(syncIDBytes)) +func DigestSyncUpperBound(syncIDBytes []byte) []byte { + return upperBoundOf(DigestSyncLowerBound(syncIDBytes)) } // --- ResourceType --- diff --git a/pkg/dotc1z/engine/pebble/merkle.go b/pkg/dotc1z/engine/pebble/merkle.go deleted file mode 100644 index 5946dfdca..000000000 --- a/pkg/dotc1z/engine/pebble/merkle.go +++ /dev/null @@ -1,993 +0,0 @@ -package pebble - -import ( - "bytes" - "context" - "encoding/binary" - "errors" - "fmt" - "sort" - - "github.com/cespare/xxhash/v2" - "github.com/cockroachdb/pebble/v2" - - v3 "github.com/conductorone/baton-sdk/pb/c1/storage/v3" - "github.com/conductorone/baton-sdk/pkg/dotc1z/engine/pebble/codec" -) - -// Per-entitlement grant merkle tree (XOR combiner). -// -// Goal: answer "does this entitlement have exactly the same grants as -// some other sync/file?" with a single key read, and when the answer is -// no, identify which principal-hash buckets differ so a caller can load -// only those grants instead of re-reading the whole entitlement. -// -// The tree folds over the by_entitlement_principal_hash index -// (idxGrantByEntitlementPrincipalHash), whose entries are sorted by -// (entitlement_id, hash(principal)) and whose VALUE is the per-grant -// content hash. Two properties make the index the right substrate: -// -// - hash-major order is identical across two files that hold the same -// grants, so a streamed fold produces the same digest; and -// - a hash prefix is a clean byte-prefix of the key, so "all grants in -// bucket P" is a contiguous range scan. -// -// Combiner. A node's digest is the XOR of every grant content hash -// beneath it (leaves are xxHash64 digests — only the combiner is XOR). -// XOR is homomorphic (parent = XOR of children), order-independent, -// and invertible, which buys three things: -// -// - depth-independence: a node's digest depends only on the grants in -// its prefix range, never on how the subtree is split, so nodes from -// trees of different heights compare directly; -// - O(1) incremental maintenance: post-seal insert/overwrite/delete -// XOR the grant's content hash into/out of the O(depth) nodes on its -// bucket path (see merkleMutator); -// - the empty digest is all-zero (the XOR identity), so an absent node -// reads as {count: 0, digest: 0}. -// -// Every node also stores its grant COUNT. Comparison always checks the -// (count, digest) pair, so a non-empty node whose hashes happened to XOR -// to zero can never be conflated with an empty/absent one. Within one -// tree that cancellation cannot even occur: every key-distinguishing -// field (principal rt/id, external_id) is folded into the content hash, -// so duplicate leaves are impossible by construction. XOR set-hashing is -// not adversarially collision-resistant (Bellare–Micciancio); this tree -// is an optimization, not a trust boundary — see RFC 0003 §9. -// -// Shape. 256-ary radix, variable height: depth 0 is a single root node -// (used for empty and small entitlements — this is why an entitlement -// with no grants costs exactly one stored node); each additional level -// consumes one more byte of the bucket hash. ALL levels are stored, -// sparsely: the root is always materialized (it is the "tree was built" -// marker — absence of the root means "never built", never "empty"), and -// every non-root node is materialized iff its subtree holds ≥1 grant. -// -// On-disk ABI. The principal bucket hash, the grant content hash, the -// combiner, and the node value framing are all part of the stored -// format: changing merkleBucketHashLen, the content-hash field set -// (grantContentHash), the combiner, or the framing requires an -// index-migration version bump (see index_migrations.go). - -const ( - // merkleBucketHashLen is the width, in bytes, of the principal - // bucket hash embedded in the index key. 8 bytes (64 bits) bounds - // the maximum tree depth at 8 byte-levels; collisions in the bucket - // address are harmless (they only co-locate principals — the full - // principal identity and external_id still distinguish index rows). - merkleBucketHashLen = 8 - - // merkleTargetBucketSize is the grant count a single leaf bucket - // aims to hold. Depth is grown until buckets are roughly this size. - merkleTargetBucketSize = 256 - - // merkleMaxDepth caps tree depth at the bucket-hash width: one byte - // of hash is consumed per level. - merkleMaxDepth = merkleBucketHashLen -) - -// hashLen is the width of an xxHash64 digest, used for both the grant -// content hash (index value) and node digests. -const hashLen = 8 - -// zeroDigest is the XOR identity — the digest of an empty/absent node. -var zeroDigest [hashLen]byte - -// xorInto XORs src into dst in place, over min(len(dst), len(src)). -func xorInto(dst, src []byte) { - for i := range min(len(dst), len(src)) { - dst[i] ^= src[i] - } -} - -// principalBucketHash is the bucket address for a principal: the 8-byte -// xxHash64 of (rt + "\x00" + id). Identity only — never the principal's -// full object — so the address is stable across syncs even when the -// principal's attributes change. Returns a fresh slice. -func principalBucketHash(rt, id string) []byte { - h := xxhash.New() - _, _ = h.WriteString(rt) - _, _ = h.Write([]byte{0}) - _, _ = h.WriteString(id) - out := make([]byte, merkleBucketHashLen) - binary.BigEndian.PutUint64(out, h.Sum64()) - return out -} - -// grantContentHash is the canonical content hash of a grant — the value -// stored in the hash index and the unit the merkle tree folds. -// -// ABI: the field set below defines what "the same grant" means for the -// diff. It deliberately covers the membership EDGE (entitlement id, -// principal identity, external_id) plus the grant's source-entitlement -// set (expansion provenance), and deliberately EXCLUDES sync-relative -// and transient processing state — sync_id, discovered_at, -// needs_expansion, expansion, and annotations — none of which change -// "which principal holds which entitlement". This is a hand-rolled -// framing, NOT proto marshal: deterministic-proto output is not -// canonical across protobuf library versions, which would make two -// files written by different SDK builds hash identical grants -// differently. Changing this set requires an index-migration bump. -func grantContentHash(r *v3.GrantRecord) []byte { - h := xxhash.New() - ent := r.GetEntitlement() - princ := r.GetPrincipal() - _, _ = h.WriteString(ent.GetEntitlementId()) - _, _ = h.Write([]byte{0}) - _, _ = h.WriteString(princ.GetResourceTypeId()) - _, _ = h.Write([]byte{0}) - _, _ = h.WriteString(princ.GetResourceId()) - _, _ = h.Write([]byte{0}) - _, _ = h.WriteString(r.GetExternalId()) - _, _ = h.Write([]byte{0}) - - // Source-entitlement ids, sorted for order-independence. The map - // values (GrantSourceRecord) are not folded in v1 — only the set of - // source ids, which is the membership-composition signal. - sources := r.GetSources() - ids := make([]string, 0, len(sources)) - for k := range sources { - ids = append(ids, k) - } - sort.Strings(ids) - for _, id := range ids { - _, _ = h.WriteString(id) - _, _ = h.Write([]byte{0}) - } - out := make([]byte, hashLen) - binary.BigEndian.PutUint64(out, h.Sum64()) - return out -} - -// grantHashIndexKey returns the by_entitlement_principal_hash index key -// for r, or nil when the grant lacks the entitlement/principal needed to -// place it (mirrors the by_entitlement index's nil-guard). -func grantHashIndexKey(syncIDBytes []byte, r *v3.GrantRecord) []byte { - ent := r.GetEntitlement() - princ := r.GetPrincipal() - if ent == nil || princ == nil { - return nil - } - bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) - return encodeGrantByEntPrincHashIndexKey( - syncIDBytes, ent.GetEntitlementId(), bh, - princ.GetResourceTypeId(), princ.GetResourceId(), r.GetExternalId(), - ) -} - -// chooseMerkleDepth picks the tree depth for an entitlement with the -// given grant count: 0 when the count fits one target-sized bucket, else -// the smallest depth whose 256^depth buckets bring the average bucket -// under merkleTargetBucketSize, capped at merkleMaxDepth. -func chooseMerkleDepth(count int64) int { - if count <= merkleTargetBucketSize { - return 0 - } - needed := (count + merkleTargetBucketSize - 1) / merkleTargetBucketSize - depth := 0 - capacity := int64(1) - for capacity < needed && depth < merkleMaxDepth { - capacity *= 256 - depth++ - } - return depth -} - -// Node value framing. All non-root nodes share one body so interior and -// leaf nodes are read uniformly; the root prepends the chosen depth so a -// reader knows the leaf level without scanning. -// -// node body: count(8 BE) | digest(hashLen) -// root body: depth(1) | count(8 BE) | digest(hashLen) -// -// An ABSENT non-root node is {count: 0, digest: 0} by definition — -// readers substitute that for any missing key. An absent ROOT means -// "tree never built" (never "empty"); readers must fall back to the -// on-demand fold, not assume zero. - -func packMerkleRoot(depth int, count int64, digest []byte) []byte { - buf := make([]byte, 0, 1+8+len(digest)) - buf = append(buf, byte(depth)) - var n [8]byte - binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count - buf = append(buf, n[:]...) - return append(buf, digest...) -} - -// unpackMerkleRoot returns (depth, count, digest, ok). ok is false when -// the value is not a well-formed root blob. -func unpackMerkleRoot(val []byte) (int, int64, []byte, bool) { - if len(val) != 1+8+hashLen { - return 0, 0, nil, false - } - depth := int(val[0]) - count := int64(binary.BigEndian.Uint64(val[1:9])) //nolint:gosec // count is a non-negative row count - return depth, count, val[9:], true -} - -func packMerkleNode(count int64, digest []byte) []byte { - buf := make([]byte, 0, 8+len(digest)) - var n [8]byte - binary.BigEndian.PutUint64(n[:], uint64(count)) //nolint:gosec // non-negative row count - buf = append(buf, n[:]...) - return append(buf, digest...) -} - -// unpackMerkleNode returns (count, digest, ok) for a non-root node body. -func unpackMerkleNode(val []byte) (int64, []byte, bool) { - if len(val) != 8+hashLen { - return 0, nil, false - } - count := int64(binary.BigEndian.Uint64(val[:8])) //nolint:gosec // non-negative count - return count, val[8:], true -} - -// BuildAllMerkleTrees rebuilds the per-entitlement merkle tree for every -// entitlement in syncID. Called at seal time (Adapter.EndSync) after all -// grants are written, and by the on-Open migration backfill. Every -// entitlement gets a tree — including those with zero grants, which -// store a single root node — so a reader can always distinguish "empty" -// from "never built". -func (e *Engine) BuildAllMerkleTrees(ctx context.Context, syncID string) error { - idBytes, err := e.resolveSyncBytes(syncID) - if err != nil { - return err - } - // Collect entitlement ids first: BuildEntitlementMerkle writes into - // the typeMerkle keyspace while we'd otherwise be mid-iteration over - // typeEntitlement. Different keyspaces, but snapshotting the ids - // keeps the iterator and the writes cleanly separated. - var ents []string - if err := e.IterateEntitlementsBySync(ctx, syncID, func(r *v3.EntitlementRecord) bool { - ents = append(ents, r.GetExternalId()) - return true - }); err != nil { - return fmt.Errorf("BuildAllMerkleTrees: list entitlements: %w", err) - } - for _, ent := range ents { - if err := ctx.Err(); err != nil { - return err - } - if err := e.buildEntitlementMerkle(ctx, idBytes, ent); err != nil { - return fmt.Errorf("BuildAllMerkleTrees: entitlement %q: %w", ent, err) - } - } - return nil -} - -// buildEntitlementMerkle counts an entitlement's grants (pass 1), picks -// the depth from that count, and delegates the fold to -// buildEntitlementMerkleAtDepth (pass 2). -func (e *Engine) buildEntitlementMerkle(ctx context.Context, idBytes []byte, entitlementID string) error { - entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) - count, err := e.countKeysInRange(entPrefix, upperBoundOf(entPrefix)) - if err != nil { - return err - } - return e.buildEntitlementMerkleAtDepth(ctx, idBytes, entitlementID, chooseMerkleDepth(count)) -} - -// buildEntitlementMerkleAtDepth folds the hash index for one entitlement -// into a root plus every non-empty node at every level, in a single -// streaming pass — O(depth) memory regardless of entitlement size. -// -// The pass starts by range-deleting the entitlement's whole typeMerkle -// keyspace: the build only ever Sets nodes, so without the clear a -// rebuild that changes depth or empties a bucket would leave stale nodes -// that the comparison descent (which enumerates children from the node -// keyspace) would read — and a stale digest that happens to match the -// peer prunes a real diff. Old and new framings are byte-length -// identical, so stale nodes are not detectable by inspection. -// -// Sorted index order means each node's grants are contiguous, so a -// level's "open" node closes exactly when its prefix changes. Only -// non-empty nodes are ever opened, so sparsity is automatic, not a -// prune pass. -// -// The depth is taken as a parameter rather than derived so the -// depth-selection seam can be exercised directly: tests force a depth -// that the natural count→depth mapping would only produce at a very -// large grant count, which is how the cross-depth comparison path gets -// covered without seeding tens of thousands of grants. -func (e *Engine) buildEntitlementMerkleAtDepth(ctx context.Context, idBytes []byte, entitlementID string, depth int) error { - entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) - upper := upperBoundOf(entPrefix) - nodeLower := encodeMerkleEntPrefix(idBytes, entitlementID) - nodeUpper := upperBoundOf(nodeLower) - - return e.withWrite(func() error { - batch := e.db.NewBatch() - defer batch.Close() - - // Clear any prior build (see function comment). In-batch - // ordering makes this safe: the Sets below land after the - // tombstone and survive it. - if err := batch.DeleteRange(nodeLower, nodeUpper, nil); err != nil { - return err - } - - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: entPrefix, UpperBound: upper}) - if err != nil { - return err - } - defer iter.Close() - - // One running node per level 1..depth; the root accumulates - // separately (its prefix is always empty, so it never closes - // mid-stream). - type openNode struct { - active bool - prefix []byte - digest [hashLen]byte - count int64 - } - open := make([]openNode, depth+1) - flush := func(level int) error { - n := &open[level] - if !n.active { - return nil - } - key := encodeMerkleNodeKey(idBytes, entitlementID, byte(level), n.prefix) - return batch.Set(key, packMerkleNode(n.count, n.digest[:]), nil) - } - - var ( - rootDigest [hashLen]byte - total int64 - ) - for iter.First(); iter.Valid(); iter.Next() { - if err := ctx.Err(); err != nil { - return err - } - key := iter.Key() - if len(key) < len(entPrefix)+merkleBucketHashLen { - continue // malformed; skip defensively - } - bucketHash := key[len(entPrefix) : len(entPrefix)+merkleBucketHashLen] - val := iter.Value() // 32-byte content hash - - for level := 1; level <= depth; level++ { - prefix := bucketHash[:level] - n := &open[level] - if !n.active || !bytes.Equal(n.prefix, prefix) { - if err := flush(level); err != nil { - return err - } - n.prefix = append(n.prefix[:0], prefix...) - n.digest = [hashLen]byte{} - n.count = 0 - n.active = true - } - xorInto(n.digest[:], val) - n.count++ - } - xorInto(rootDigest[:], val) - total++ - } - if err := iter.Error(); err != nil { - return err - } - for level := 1; level <= depth; level++ { - if err := flush(level); err != nil { - return err - } - } - - // Root is written unconditionally — even at count 0 — as the - // "tree was built" marker. - rootKey := encodeMerkleNodeKey(idBytes, entitlementID, 0, nil) - if err := batch.Set(rootKey, packMerkleRoot(depth, total, rootDigest[:]), nil); err != nil { - return err - } - - opts := writeOpts(e.opts.durability) - if e.IsFreshSync() { - opts = pebble.NoSync - } - return batch.Commit(opts) - }) -} - -// countKeysInRange counts keys in [lower, upper) without materializing -// values. Used by the build's count→depth pass and by the diff driver's -// primary-vs-index coverage guard. -func (e *Engine) countKeysInRange(lower, upper []byte) (int64, error) { - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) - if err != nil { - return 0, err - } - defer iter.Close() - var n int64 - for iter.First(); iter.Valid(); iter.Next() { - n++ - } - return n, iter.Error() -} - -// distinctHashIndexEntitlements returns the distinct entitlement_ids -// present in a sync's by_entitlement_principal_hash index, in index -// order. It seeks past each entitlement's whole range once its id is -// captured, so the cost is O(entitlements) seeks, not O(grants). This -// is the diff driver's work list, and it deliberately comes from the -// INDEX rather than the entitlement records: a grant whose entitlement -// has no record still has index entries (and a fold), so it still gets -// compared — only a tree was never built for it. -func (e *Engine) distinctHashIndexEntitlements(ctx context.Context, syncIDBytes []byte) ([]string, error) { - lower := GrantByEntPrincHashSyncLowerBound(syncIDBytes) - upper := GrantByEntPrincHashSyncUpperBound(syncIDBytes) - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) - if err != nil { - return nil, err - } - defer iter.Close() - - // Keys are lower | 0x00 | tuple(entitlement_id) | 0x00 | bucketHash | … - entStart := len(lower) + 1 - var out []string - for iter.First(); iter.Valid(); { - if err := ctx.Err(); err != nil { - return nil, err - } - key := iter.Key() - if len(key) <= entStart { - iter.Next() - continue // malformed; skip defensively - } - entBytes, _, decErr := codec.DecodeTupleStringTo(nil, key[entStart:], 0) - if decErr != nil { - iter.Next() - continue - } - ent := string(entBytes) - out = append(out, ent) - // Seek past this entitlement's whole index range. - seekTo := upperBoundOf(encodeGrantByEntPrincHashEntPrefix(syncIDBytes, ent)) - if seekTo == nil { - break - } - iter.SeekGE(seekTo) - } - return out, iter.Error() -} - -// MerkleRoot is the result of GetEntitlementMerkleRoot. -type MerkleRoot struct { - Hash []byte - Depth int - Count int64 -} - -// GetEntitlementMerkleRoot returns the stored root for an entitlement. -// ok is false when no tree has been built for it (the caller can fall -// back to ComputeBucketHash, which derives the same digest from the -// index on demand). -func (e *Engine) GetEntitlementMerkleRoot(ctx context.Context, syncID, entitlementID string) (MerkleRoot, bool, error) { - idBytes, err := e.resolveSyncBytes(syncID) - if err != nil { - return MerkleRoot{}, false, err - } - val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, 0, nil)) - if err != nil { - if errors.Is(err, pebble.ErrNotFound) { - return MerkleRoot{}, false, nil - } - return MerkleRoot{}, false, err - } - defer closer.Close() - depth, count, h, valid := unpackMerkleRoot(val) - if !valid { - return MerkleRoot{}, false, fmt.Errorf("GetEntitlementMerkleRoot: malformed root for %q", entitlementID) - } - out := make([]byte, len(h)) - copy(out, h) - return MerkleRoot{Hash: out, Depth: depth, Count: count}, true, nil -} - -// getMerkleNode reads one stored non-root node. An absent node returns -// (0, zero digest, present=false, nil) — the XOR identity. -func (e *Engine) getMerkleNode(idBytes []byte, entitlementID string, level int, prefix []byte) (int64, []byte, bool, error) { - val, closer, err := e.db.Get(encodeMerkleNodeKey(idBytes, entitlementID, byte(level), prefix)) - if err != nil { - if errors.Is(err, pebble.ErrNotFound) { - return 0, zeroDigest[:], false, nil - } - return 0, nil, false, err - } - defer closer.Close() - count, digest, ok := unpackMerkleNode(val) - if !ok { - return 0, nil, false, fmt.Errorf("getMerkleNode: malformed node for %q level %d", entitlementID, level) - } - out := make([]byte, hashLen) - copy(out, digest) - return count, out, true, nil -} - -// merkleChildPrefixes returns the sorted full prefixes (length = level) -// of the stored nodes at `level` under parentPrefix. Because the node -// key embeds the level byte before the prefix bytes, the children of one -// parent are a contiguous key range — cost is O(children present), and -// only non-empty children are ever stored. -func (e *Engine) merkleChildPrefixes(ctx context.Context, idBytes []byte, entitlementID string, level int, parentPrefix []byte) ([][]byte, error) { - stem := encodeMerkleNodeKey(idBytes, entitlementID, byte(level), nil) - lower := append(append([]byte(nil), stem...), parentPrefix...) - upper := upperBoundOf(lower) - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) - if err != nil { - return nil, err - } - defer iter.Close() - var out [][]byte - for iter.First(); iter.Valid(); iter.Next() { - if err := ctx.Err(); err != nil { - return nil, err - } - key := iter.Key() - if len(key) != len(stem)+level { - continue // malformed; skip defensively - } - prefix := make([]byte, level) - copy(prefix, key[len(stem):]) - out = append(out, prefix) - } - return out, iter.Error() -} - -// ComputeBucketHash folds the hash index over a single principal-hash -// bucket (identified by a raw hash prefix; empty prefix = whole -// entitlement = the root) and returns the content-defined XOR digest -// plus the grant count. This is the authoritative definition of a node; -// stored nodes are a cache of it. Depth-independent: the digest depends -// only on the grants in the prefix range, not on any tree's shape. -func (e *Engine) ComputeBucketHash(ctx context.Context, syncID, entitlementID string, hashPrefix []byte) ([]byte, int64, error) { - idBytes, err := e.resolveSyncBytes(syncID) - if err != nil { - return nil, 0, err - } - lower := encodeGrantByEntPrincHashBucketPrefix(idBytes, entitlementID, hashPrefix) - upper := upperBoundOf(lower) - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) - if err != nil { - return nil, 0, err - } - defer iter.Close() - digest := make([]byte, hashLen) - var count int64 - for iter.First(); iter.Valid(); iter.Next() { - if err := ctx.Err(); err != nil { - return nil, 0, err - } - xorInto(digest, iter.Value()) - count++ - } - if err := iter.Error(); err != nil { - return nil, 0, err - } - return digest, count, nil -} - -// IterateGrantsByEntitlementBucket yields the grants in one principal-hash -// bucket of an entitlement (empty hashPrefix = the whole entitlement). -// This is the dirty-bucket loader: after a merkle comparison flags a -// bucket prefix, the caller materializes only those grants. Like the -// other index iterators it does a point Get per entry to fetch the -// primary; orphan index entries are skipped. -func (e *Engine) IterateGrantsByEntitlementBucket(ctx context.Context, syncID, entitlementID string, hashPrefix []byte, yield func(*v3.GrantRecord) bool) error { - idBytes, err := e.resolveSyncBytes(syncID) - if err != nil { - return err - } - entPrefix := encodeGrantByEntPrincHashEntPrefix(idBytes, entitlementID) - lower := append(append([]byte(nil), entPrefix...), hashPrefix...) - upper := upperBoundOf(lower) - iter, err := e.db.NewIter(&pebble.IterOptions{LowerBound: lower, UpperBound: upper}) - if err != nil { - return err - } - defer iter.Close() - for iter.First(); iter.Valid(); iter.Next() { - if err := ctx.Err(); err != nil { - return err - } - _, _, _, externalID, ok := decodeEntPrincHashTail(iter.Key(), entPrefix) - if !ok { - continue - } - val, closer, getErr := e.db.Get(encodeGrantKey(idBytes, externalID)) - if getErr != nil { - if errors.Is(getErr, pebble.ErrNotFound) { - continue - } - return getErr - } - r := &v3.GrantRecord{} - uErr := unmarshalRecord(val, r) - closer.Close() - if uErr != nil { - return fmt.Errorf("IterateGrantsByEntitlementBucket: unmarshal: %w", uErr) - } - if !yield(r) { - return nil - } - } - return iter.Error() -} - -// DirtyEntitlementBuckets compares this engine's entitlement against -// other's and returns the raw hash-bucket prefixes whose grants differ. -// A single empty prefix means "the whole entitlement differs" (used when -// the comparison granularity is the root, e.g. tiny entitlements). A nil -// (empty) result means the two are identical. -// -// The fast path is a single root read per side. On mismatch it descends -// level by level, pruning every subtree whose (count, digest) pair -// matches and emitting the differing prefixes at compareDepth — the -// shallower tree's leaf level, where both sides still have directly -// comparable nodes (XOR digests are depth-independent). Children are -// enumerated from the stored node keyspace (union of both sides), so -// descent cost is proportional to the symmetric difference, not the -// fan-out. -// -// A missing root means the tree was never built on that side — NOT that -// the entitlement is empty — so both sides are compared via the -// authoritative on-demand fold instead. -func (e *Engine) DirtyEntitlementBuckets(ctx context.Context, syncID string, other *Engine, otherSyncID, entitlementID string) ([][]byte, error) { - rootA, okA, err := e.GetEntitlementMerkleRoot(ctx, syncID, entitlementID) - if err != nil { - return nil, err - } - rootB, okB, err := other.GetEntitlementMerkleRoot(ctx, otherSyncID, entitlementID) - if err != nil { - return nil, err - } - - if !okA || !okB { - ha, ca, err := e.ComputeBucketHash(ctx, syncID, entitlementID, nil) - if err != nil { - return nil, err - } - hb, cb, err := other.ComputeBucketHash(ctx, otherSyncID, entitlementID, nil) - if err != nil { - return nil, err - } - if ca == cb && bytes.Equal(ha, hb) { - return nil, nil - } - return [][]byte{{}}, nil - } - - if rootA.Count == rootB.Count && bytes.Equal(rootA.Hash, rootB.Hash) { - return nil, nil - } - - // Roots differ. The descent granularity is the shallower tree's - // leaf level; at depth 0 there is nothing below the root. - compareDepth := min(rootA.Depth, rootB.Depth) - if compareDepth == 0 { - return [][]byte{{}}, nil - } - - idBytesA, err := e.resolveSyncBytes(syncID) - if err != nil { - return nil, err - } - idBytesB, err := other.resolveSyncBytes(otherSyncID) - if err != nil { - return nil, err - } - - var dirty [][]byte - var walk func(prefix []byte, level int) error - walk = func(prefix []byte, level int) error { - if err := ctx.Err(); err != nil { - return err - } - ca, da, _, err := e.getMerkleNode(idBytesA, entitlementID, level, prefix) - if err != nil { - return err - } - cb, db, _, err := other.getMerkleNode(idBytesB, entitlementID, level, prefix) - if err != nil { - return err - } - if ca == cb && bytes.Equal(da, db) { - return nil // identical subtree (or absent on both sides) → prune - } - if level == compareDepth { - dirty = append(dirty, prefix) - return nil - } - kidsA, err := e.merkleChildPrefixes(ctx, idBytesA, entitlementID, level+1, prefix) - if err != nil { - return err - } - kidsB, err := other.merkleChildPrefixes(ctx, idBytesB, entitlementID, level+1, prefix) - if err != nil { - return err - } - for _, k := range mergeSortedPrefixes(kidsA, kidsB) { - if err := walk(k, level+1); err != nil { - return err - } - } - return nil - } - - kidsA, err := e.merkleChildPrefixes(ctx, idBytesA, entitlementID, 1, nil) - if err != nil { - return nil, err - } - kidsB, err := other.merkleChildPrefixes(ctx, idBytesB, entitlementID, 1, nil) - if err != nil { - return nil, err - } - for _, k := range mergeSortedPrefixes(kidsA, kidsB) { - if err := walk(k, 1); err != nil { - return nil, err - } - } - - // Roots differed but the descent found nothing: with consistent - // trees that's impossible (the root is the XOR of level 1), so a - // stored node is stale or corrupt. Fail safe — whole entitlement - // dirty; the next rebuild heals the tree. - if len(dirty) == 0 { - return [][]byte{{}}, nil - } - return dirty, nil -} - -// mergeSortedPrefixes returns the sorted union of two sorted, de-duped -// prefix lists. -func mergeSortedPrefixes(a, b [][]byte) [][]byte { - out := make([][]byte, 0, len(a)+len(b)) - i, j := 0, 0 - for i < len(a) && j < len(b) { - switch bytes.Compare(a[i], b[j]) { - case 0: - out = append(out, a[i]) - i++ - j++ - case -1: - out = append(out, a[i]) - i++ - default: - out = append(out, b[j]) - j++ - } - } - out = append(out, a[i:]...) - out = append(out, b[j:]...) - return out -} - -// --- Incremental maintenance (post-seal) --- - -// merkleMutator accumulates per-node (XOR, count) deltas for a batch of -// post-seal grant mutations and applies each touched node exactly once. -// -// Why an accumulator instead of read-modify-write per mutation: the -// updates target a plain pebble.Batch, which does NOT read through its -// own writes — and every mutation in a batch touches the root node, so -// naive per-mutation RMW would lose deltas. Accumulating also collapses -// N writes per node into one. All grant writers run under withWrite's -// mutex, so reading current node values from the DB inside apply is -// race-free. -// -// Lifecycle: one mutator per write batch. Callers feed remove(old) / -// add(new) as they process records (an overwrite that moves the grant to -// a different bucket — principal changed — is exactly remove+add), then -// call apply(batch) once before commit. -// -// Entitlements whose tree was never built (no stored root) are skipped: -// the seal-time build or the on-Open backfill will construct them from -// the index. This also makes the mutator free during a fresh sync — but -// callers on the fresh-sync bulk path should skip constructing one -// anyway to avoid the per-entitlement root probe. -type merkleMutator struct { - e *Engine - ents map[string]*mutatorEnt // keyed by string(root node key) -} - -type mutatorEnt struct { - idBytes []byte - entID string - rootKey []byte - present bool // stored root exists; if false all deltas are dropped - depth int - rootCount int64 // count read from the stored root - rootDigest [hashLen]byte // digest read from the stored root - xor [hashLen]byte // accumulated root delta - countDelta int64 - nodes map[string]*mutatorNode // levels 1..depth, keyed by string(node key) -} - -type mutatorNode struct { - key []byte - xor [hashLen]byte - countDelta int64 -} - -func newMerkleMutator(e *Engine) *merkleMutator { - return &merkleMutator{e: e, ents: make(map[string]*mutatorEnt)} -} - -// entFor returns the (cached) per-entitlement state, probing the stored -// root on first touch. The cached root snapshot stays valid for the -// mutator's lifetime because all writers serialize through withWrite. -func (m *merkleMutator) entFor(idBytes []byte, entitlementID string) (*mutatorEnt, error) { - rootKey := encodeMerkleNodeKey(idBytes, entitlementID, 0, nil) - k := string(rootKey) - if me, ok := m.ents[k]; ok { - return me, nil - } - me := &mutatorEnt{idBytes: idBytes, entID: entitlementID, rootKey: rootKey, nodes: make(map[string]*mutatorNode)} - val, closer, err := m.e.db.Get(rootKey) - switch { - case err == nil: - depth, count, digest, ok := unpackMerkleRoot(val) - closer.Close() - if ok { - me.present = true - me.depth = depth - me.rootCount = count - copy(me.rootDigest[:], digest) - } - // Malformed root: leave present=false so mutations are dropped; - // the stale root heals at the next rebuild. - case errors.Is(err, pebble.ErrNotFound): - // No tree — deltas for this entitlement are no-ops. - default: - return nil, err - } - m.ents[k] = me - return me, nil -} - -// add records r's insertion into its entitlement's tree. -func (m *merkleMutator) add(idBytes []byte, r *v3.GrantRecord) error { - return m.delta(idBytes, r, 1) -} - -// remove records r's removal from its entitlement's tree. -func (m *merkleMutator) remove(idBytes []byte, r *v3.GrantRecord) error { - return m.delta(idBytes, r, -1) -} - -func (m *merkleMutator) delta(idBytes []byte, r *v3.GrantRecord, sign int64) error { - ent := r.GetEntitlement() - princ := r.GetPrincipal() - if ent == nil || princ == nil { - return nil // not in the hash index → not in the tree - } - me, err := m.entFor(idBytes, ent.GetEntitlementId()) - if err != nil { - return err - } - if !me.present { - return nil - } - h := grantContentHash(r) - xorInto(me.xor[:], h) - me.countDelta += sign - if me.depth == 0 { - return nil - } - bh := principalBucketHash(princ.GetResourceTypeId(), princ.GetResourceId()) - for level := 1; level <= me.depth; level++ { - key := encodeMerkleNodeKey(idBytes, ent.GetEntitlementId(), byte(level), bh[:level]) - nk := string(key) - n, ok := me.nodes[nk] - if !ok { - n = &mutatorNode{key: key} - me.nodes[nk] = n - } - xorInto(n.xor[:], h) - n.countDelta += sign - } - return nil -} - -// apply folds the accumulated deltas into the stored nodes via batch. -// Nodes whose delta cancelled to zero (e.g. an overwrite that changed -// only excluded fields) are skipped; a non-root node whose count reaches -// zero is deleted (restoring sparsity); the root is rewritten in place. -// A count that would go negative means the stored tree disagrees with -// the mutation stream — the tree is dropped wholesale (DeleteRange), so -// readers fall back to the on-demand fold until the next rebuild. -func (m *merkleMutator) apply(batch *pebble.Batch) error { - for _, me := range m.ents { - if !me.present { - continue - } - if err := m.applyEnt(batch, me); err != nil { - return err - } - } - return nil -} - -func (m *merkleMutator) applyEnt(batch *pebble.Batch, me *mutatorEnt) error { - if me.countDelta == 0 && me.xor == zeroDigest && len(me.nodes) == 0 { - return nil - } - dropTree := func() error { - // In-batch ordering: this tombstone lands after any node Sets - // already staged for this entitlement and removes them too. - lo := encodeMerkleEntPrefix(me.idBytes, me.entID) - return batch.DeleteRange(lo, upperBoundOf(lo), nil) - } - if me.rootCount+me.countDelta < 0 { - return dropTree() - } - for _, n := range me.nodes { - if n.countDelta == 0 && n.xor == zeroDigest { - continue - } - var ( - curCount int64 - curDigest [hashLen]byte - ) - val, closer, err := m.e.db.Get(n.key) - switch { - case err == nil: - c, d, ok := unpackMerkleNode(val) - closer.Close() - if !ok { - return dropTree() - } - curCount = c - copy(curDigest[:], d) - case errors.Is(err, pebble.ErrNotFound): - // absent node = {0, zero} - default: - return err - } - newCount := curCount + n.countDelta - if newCount < 0 { - return dropTree() - } - xorInto(curDigest[:], n.xor[:]) - if newCount == 0 { - // An emptied node's digest must cancel to exactly zero - // (count 0 ⇒ digest 0); anything else means the stored - // tree disagrees with the mutation stream. - if curDigest != zeroDigest { - return dropTree() - } - if err := batch.Delete(n.key, nil); err != nil { - return err - } - continue - } - if err := batch.Set(n.key, packMerkleNode(newCount, curDigest[:]), nil); err != nil { - return err - } - } - if me.countDelta == 0 && me.xor == zeroDigest { - return nil - } - newDigest := me.rootDigest - xorInto(newDigest[:], me.xor[:]) - return batch.Set(me.rootKey, packMerkleRoot(me.depth, me.rootCount+me.countDelta, newDigest[:]), nil) -} diff --git a/pkg/synccompactor/pebble/bucket_plans.go b/pkg/synccompactor/pebble/bucket_plans.go index 094eae019..204bbfe17 100644 --- a/pkg/synccompactor/pebble/bucket_plans.go +++ b/pkg/synccompactor/pebble/bucket_plans.go @@ -82,13 +82,14 @@ func buildBucketPlans(syncIDBytes []byte) []bucketPlan { upper: enginepkg.GrantByEntPrincHashSyncUpperBound(syncIDBytes), }, { - // Per-entitlement grant merkle nodes. Copied byte-for-byte - // with the sync_id they belong to; because Compact preserves - // the sync_id and the grants move verbatim, the trees stay - // valid in the destination without a rebuild. - name: "grant_merkle", - lower: enginepkg.MerkleSyncLowerBound(syncIDBytes), - upper: enginepkg.MerkleSyncUpperBound(syncIDBytes), + // Digest nodes (all digested indexes, e.g. the per-entitlement + // grant digest). Copied byte-for-byte with the sync_id they + // belong to; because Compact preserves the sync_id and the + // records move verbatim, the digests stay valid in the + // destination without a rebuild. + name: "digest", + lower: enginepkg.DigestSyncLowerBound(syncIDBytes), + upper: enginepkg.DigestSyncUpperBound(syncIDBytes), }, { name: "asset", From 118ecb60bf713edb7df8dcab6c3b7ff5679b407a Mon Sep 17 00:00:00 2001 From: MJ Palanker Date: Thu, 18 Jun 2026 09:04:28 -0700 Subject: [PATCH 7/7] check the length of hash vals to avoid corruption --- pkg/dotc1z/engine/pebble/digest.go | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/pkg/dotc1z/engine/pebble/digest.go b/pkg/dotc1z/engine/pebble/digest.go index 919d471ac..5d216a900 100644 --- a/pkg/dotc1z/engine/pebble/digest.go +++ b/pkg/dotc1z/engine/pebble/digest.go @@ -373,6 +373,15 @@ func (e *Engine) buildPartitionDigestAtWidth(ctx context.Context, spec digestInd continue // malformed; skip defensively } val := iter.Value() // per-record content hash + if len(val) != hashLen { + // Index writers always emit exactly hashLen bytes + // (grantContentHash); a wrong length is a corrupt or + // mis-encoded entry. xorInto would fold only a prefix and + // quietly corrupt the digest, so reject it. At seal the + // caller downgrades a build error to "no digest" (readers + // fall back to the on-demand fold), so this fails safe. + return fmt.Errorf("buildPartitionDigestAtWidth: content hash for %q is %d bytes, want %d", partition, len(val), hashLen) + } if widthBits > 0 { lv := binary.BigEndian.Uint16(key[len(prefix):]) &^ lowMask if !leafOpen || lv != leafLV { @@ -591,7 +600,13 @@ func (e *Engine) computeBucketDigest(ctx context.Context, spec digestIndexSpec, if err := ctx.Err(); err != nil { return nil, 0, err } - xorInto(digest, iter.Value()) + val := iter.Value() + if len(val) != hashLen { + // See buildPartitionDigestAtWidth: a mis-length index value is + // corruption; reject it rather than silently fold a prefix. + return nil, 0, fmt.Errorf("computeBucketDigest: content hash for %q is %d bytes, want %d", partition, len(val), hashLen) + } + xorInto(digest, val) count++ } if err := iter.Error(); err != nil {