From 7565b82e21b35f8339dcce2fb3dc6688b3da27b2 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 12:23:16 +0000 Subject: [PATCH 01/32] feat(audit): pair every request entry with a response Add AuditWriter::begin, which appends the request entry and returns an AuditRequest handle that owes its response. The handle responds in the request's own schema and correlation, and a response appended through AuditWriter::append under the same schema and correlation answers it too. A handle dropped unanswered writes the product's unfinished record as the response, so an early return, a panic, or a canceled future still pairs the request entry. For the file destination that line is queued into the group commit and, if the runtime shuts down first, written when the last reference to the file is dropped. Signed-off-by: Jeremi Joslin --- crates/registry-platform-audit/src/lib.rs | 5 +- crates/registry-platform-audit/src/writer.rs | 665 ++++++++++++++++++- 2 files changed, 662 insertions(+), 8 deletions(-) diff --git a/crates/registry-platform-audit/src/lib.rs b/crates/registry-platform-audit/src/lib.rs index 6b6c87708..a5f4a04c3 100644 --- a/crates/registry-platform-audit/src/lib.rs +++ b/crates/registry-platform-audit/src/lib.rs @@ -5,6 +5,9 @@ //! file or to stdout: a `request` entry before protected I/O and a //! `response` entry with the outcome, sharing one correlation. It fails //! closed: an entry the destination does not accept is an error. +//! [`AuditWriter::begin`] writes the `request` entry and returns an +//! [`AuditRequest`] that owes the `response`: dropped unanswered, it writes +//! the product's `unfinished` record, so no request entry stays unpaired. //! - [`AuditProfile`] and [`AuditKeyHasher`] derive keyed, domain-separated //! references so audit records never carry raw identifiers. //! - [`redact`] minimizes query strings, email addresses, and phone numbers. @@ -25,7 +28,7 @@ mod writer; #[cfg(unix)] pub use writer::{ AuditDestination, AuditDestinationError, AuditDestinationKind, AuditEntry, AuditPhase, - AuditUnavailable, AuditUnavailableReason, AuditWriter, FileDestination, + AuditRequest, AuditUnavailable, AuditUnavailableReason, AuditWriter, FileDestination, DEFAULT_AUDIT_RETAIN_DAYS, DEFAULT_AUDIT_ROTATE_BYTES, MAX_AUDIT_RETAIN_DAYS, MIN_AUDIT_ROTATE_BYTES, }; diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index 6f984aa6a..cf393a067 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -17,7 +17,7 @@ use std::{ os::unix::fs::{DirBuilderExt, FileExt, MetadataExt, OpenOptionsExt, PermissionsExt}, path::{Path, PathBuf}, sync::{ - atomic::{AtomicBool, AtomicU64, Ordering}, + atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}, Arc, Mutex as StdMutex, }, time::{Duration, SystemTime}, @@ -44,6 +44,7 @@ const MAX_ENTRY_BYTES: usize = 1024 * 1024; const MAX_SCHEMA_BYTES: usize = 128; const MAX_CORRELATION_BYTES: usize = 256; const SEGMENT_SEQUENCE_DIGITS: usize = 8; +const DETACHED_LOCK_ATTEMPTS: usize = 1024; const SECONDS_PER_DAY: u64 = 24 * 60 * 60; const TIME_FORMAT: &[FormatItem<'static>] = format_description!("[year]-[month]-[day]T[hour]:[minute]:[second].[subsecond digits:3]Z"); @@ -525,6 +526,59 @@ impl AuditDestination { #[derive(Clone)] pub struct AuditWriter { inner: Arc, + open: Arc, +} + +/// The schema and correlation a request entry and its responses share. +type RequestKey = (String, String); + +/// The request entries whose [`AuditRequest`] still owes a response, by +/// schema and correlation, oldest first. +#[derive(Default)] +struct OpenRequests(StdMutex>>>); + +impl OpenRequests { + fn open(&self, key: RequestKey) -> Arc { + let answered = Arc::new(AtomicBool::new(false)); + if let Ok(mut open) = self.0.lock() { + open.entry(key).or_default().push(Arc::clone(&answered)); + } + answered + } + + /// Mark the oldest open request under `key` answered. + fn answer(&self, key: &RequestKey) { + let Ok(mut open) = self.0.lock() else { + return; + }; + if let Some(waiting) = open.get_mut(key) { + if !waiting.is_empty() { + waiting.remove(0).store(true, Ordering::Release); + } + if waiting.is_empty() { + open.remove(key); + } + } + } + + /// Close `answered` under `key`, reporting whether it is still owed a + /// response. Checked and removed under one lock, so a response cannot be + /// counted after the owner decided it had none. + fn close(&self, key: &RequestKey, answered: &Arc) -> bool { + let Ok(mut open) = self.0.lock() else { + return !answered.load(Ordering::Acquire); + }; + if answered.load(Ordering::Acquire) { + return false; + } + if let Some(waiting) = open.get_mut(key) { + waiting.retain(|candidate| !Arc::ptr_eq(candidate, answered)); + if waiting.is_empty() { + open.remove(key); + } + } + true + } } enum WriterInner { @@ -565,6 +619,7 @@ impl AuditWriter { }; Ok(Self { inner: Arc::new(inner), + open: Arc::default(), }) } @@ -574,13 +629,25 @@ impl AuditWriter { pub fn from_line_sink(sink: Box) -> Self { Self { inner: Arc::new(WriterInner::Stream(Arc::new(LineStream::new(sink)))), + open: Arc::default(), } } /// Append one entry. For the file destination this returns only after the /// entry's bytes are durable. Canceling the caller does not cancel an /// enqueued file write or the other entries in its group commit. + /// + /// An accepted `response` entry answers the oldest [`AuditRequest`] still + /// open under the same schema and correlation. pub async fn append(&self, entry: AuditEntry) -> Result<(), AuditUnavailable> { + self.write(&entry).await?; + if entry.phase == AuditPhase::Response { + self.open.answer(&(entry.schema, entry.correlation)); + } + Ok(()) + } + + async fn write(&self, entry: &AuditEntry) -> Result<(), AuditUnavailable> { let line = entry.to_line()?; match self.inner.as_ref() { WriterInner::File(file) => { @@ -598,6 +665,67 @@ impl AuditWriter { } } + /// Append the `request` entry of one audited operation and return the + /// [`AuditRequest`] that owes its `response` entry. + /// + /// `request` and `unfinished` are the product's minimized records. The + /// handle writes `unfinished` as the `response` entry if it is dropped + /// before any response is accepted, so an early return, an error, a + /// panic, or a canceled future still pairs the request entry. Every + /// `response` the handle writes carries the request's schema and + /// correlation. A response appended through [`Self::append`] under the + /// same schema and correlation answers it too, so an operation whose + /// outcome is written elsewhere only holds the handle until it returns. + pub async fn begin( + &self, + schema: impl Into, + correlation: impl Into, + request: Value, + unfinished: Value, + ) -> Result { + let schema = schema.into(); + let correlation = correlation.into(); + if !unfinished.is_object() { + return Err(AuditUnavailable::new(AuditUnavailableReason::InvalidEntry)); + } + self.write(&AuditEntry::request( + schema.clone(), + correlation.clone(), + request, + )) + .await?; + let answered = self.open.open((schema.clone(), correlation.clone())); + Ok(AuditRequest { + writer: self.clone(), + schema, + correlation, + answered, + unfinished, + }) + } + + /// Write `entry` without waiting for the destination to accept it. + fn append_detached(&self, entry: &AuditEntry) { + let line = match entry.to_line() { + Ok(line) => line, + Err(_) => { + tracing::error!("an unfinished response entry is malformed and was not written"); + return; + } + }; + match self.inner.as_ref() { + WriterInner::File(file) => file.enqueue_detached(line), + WriterInner::Stream(stream) => { + // Written inline, unlike `write`: a blocking task queued from + // a drop during runtime shutdown may never run, and the + // unfinished response is the entry that must not be lost. + if stream.append(&line).is_err() { + tracing::error!("an unfinished response entry was not accepted"); + } + } + } + } + /// Report whether the writer can still accept entries. For the file /// destination this also confirms the writer still owns the active file. pub async fn ready(&self) -> bool { @@ -635,6 +763,87 @@ impl AuditWriter { } } +/// An accepted `request` entry that still owes its `response` entry. +/// +/// [`AuditRequest::respond`] appends a `response` entry and waits for the +/// destination to accept it; an operation may respond more than once. A +/// handle dropped before any response was accepted writes the `unfinished` +/// record given to [`AuditWriter::begin`] as its `response` entry. That write +/// cannot be awaited, so an operation with a known outcome, a refusal +/// included, responds with it instead of relying on the drop. +#[must_use = "an audit request writes its unfinished response entry when dropped"] +pub struct AuditRequest { + writer: AuditWriter, + schema: String, + correlation: String, + /// Set once a response entry under this schema and correlation was + /// accepted. + answered: Arc, + /// The record written if the handle is dropped unanswered. + unfinished: Value, +} + +impl std::fmt::Debug for AuditRequest { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter + .debug_struct("AuditRequest") + .field("schema", &self.schema) + .field("answered", &self.is_answered()) + .finish_non_exhaustive() + } +} + +impl AuditRequest { + /// The correlation its request and response entries share. + #[must_use] + pub fn correlation(&self) -> &str { + &self.correlation + } + + /// Whether a response entry was accepted. + #[must_use] + pub fn is_answered(&self) -> bool { + self.answered.load(Ordering::Acquire) + } + + /// Append one `response` entry. A refused entry leaves the request + /// unanswered, so a later drop still writes the unfinished record. + pub async fn respond(&mut self, record: Value) -> Result<(), AuditUnavailable> { + self.writer + .write(&AuditEntry::response( + self.schema.clone(), + self.correlation.clone(), + record, + )) + .await?; + // Answer this request, not an older one open under its correlation. + self.writer.open.close( + &(self.schema.clone(), self.correlation.clone()), + &self.answered, + ); + self.answered.store(true, Ordering::Release); + Ok(()) + } + + /// Append `record` as the only `response` entry and release the handle. + pub async fn finish(mut self, record: Value) -> Result<(), AuditUnavailable> { + self.respond(record).await + } +} + +impl Drop for AuditRequest { + fn drop(&mut self) { + let key = ( + std::mem::take(&mut self.schema), + std::mem::take(&mut self.correlation), + ); + if self.writer.open.close(&key, &self.answered) { + let entry = AuditEntry::response(key.0, key.1, std::mem::take(&mut self.unfinished)); + self.writer.append_detached(&entry); + } + } +} + struct LineStream { out: StdMutex>, healthy: AtomicBool, @@ -767,6 +976,65 @@ impl GroupCommitFile { } self.file.ready().await } + + /// Queue `line` without waiting for it to be durable, then flush it on + /// the current runtime. A line queued when no runtime can flush it, or + /// whose flush is canceled at shutdown, is written by the next append or + /// when the last reference to the file is dropped. + fn enqueue_detached(self: &Arc, line: String) { + let runtime = tokio::runtime::Handle::try_current().ok(); + let mut line = Some(line); + // Every holder of the state lock releases it without awaiting, so a + // short wait is enough unless the state is poisoned by a stop. + for _ in 0..DETACHED_LOCK_ATTEMPTS { + if let Ok(mut state) = self.state.try_lock() { + if state.stopped { + tracing::error!( + "audit writer stopped; an unfinished response entry was not written" + ); + return; + } + state.pending.extend(line.take()); + state.enqueued = state.enqueued.saturating_add(1); + break; + } + std::thread::yield_now(); + } + let Some(runtime) = runtime else { + if line.is_some() { + tracing::error!("audit file state is busy outside a runtime; an unfinished response entry was not written"); + } + return; + }; + let file = Arc::clone(self); + runtime.spawn(async move { + let result = match line { + Some(line) => file.append(line).await, + None => { + let _writer = file.flush.lock().await; + file.flush_once().await + } + }; + if result.is_err() { + tracing::error!("an unfinished response entry was not accepted"); + } + }); + } +} + +impl Drop for GroupCommitFile { + /// Write the lines still queued, such as an unfinished response whose + /// flush was canceled when the runtime shut down. + fn drop(&mut self) { + let state = self.state.get_mut(); + if state.stopped || state.pending.is_empty() { + return; + } + let lines = std::mem::take(&mut state.pending); + if let Err(error) = self.file.write_lines_blocking(lines) { + tracing::error!(%error, "queued audit entries were not written at shutdown"); + } + } } /// A single-writer JSON Lines file with online size rotation and age-based @@ -783,6 +1051,7 @@ struct SegmentedFile { retain: Duration, state: tokio::sync::Mutex, healthy: AtomicBool, + in_flight: Arc, lock_fingerprint: FileFingerprint, writer_lock: File, #[cfg(test)] @@ -864,6 +1133,7 @@ impl SegmentedFile { next_sequence, }), healthy: AtomicBool::new(true), + in_flight: Arc::new(AtomicUsize::new(0)), lock_fingerprint, writer_lock, #[cfg(test)] @@ -906,7 +1176,49 @@ impl SegmentedFile { ))); } let mut state = self.state.lock().await; - let request = AppendRequest { + let request = self.append_request(&state, lines)?; + let in_flight = InFlightWrite::start(&self.in_flight); + let outcome = tokio::task::spawn_blocking(move || { + let _in_flight = in_flight; + request.run() + }) + .await + .map_err(|error| AuditError::Io(io::Error::other(error))) + .and_then(|result| result); + self.settle(&mut state, outcome) + } + + /// Write `lines` on the calling thread, outside any runtime. The owner of + /// the last reference calls it, so no other write can hold the state. + fn write_lines_blocking(&self, lines: Vec) -> Result<(), AuditError> { + if !self.healthy.load(Ordering::Acquire) { + return Err(AuditError::Io(io::Error::other( + "audit writer stopped after a failed write", + ))); + } + // A canceled caller can leave its blocking write running; writing + // beside it could interleave two runs of lines in the active file. + if self.in_flight.load(Ordering::Acquire) != 0 { + return Err(AuditError::Io(io::Error::other( + "an earlier audit write is still in flight", + ))); + } + let mut state = self + .state + .try_lock() + .map_err(|_| AuditError::Io(io::Error::other("audit file state is held")))?; + let outcome = self + .append_request(&state, lines) + .and_then(AppendRequest::run); + self.settle(&mut state, outcome) + } + + fn append_request( + &self, + state: &FileState, + lines: Vec, + ) -> Result { + Ok(AppendRequest { path: self.path.clone(), rotate_bytes: self.rotate_bytes, retain: self.retain, @@ -919,11 +1231,14 @@ impl SegmentedFile { next_sequence: state.next_sequence, #[cfg(test)] sync_hook: self.sync_hook.clone(), - }; - let outcome = tokio::task::spawn_blocking(move || request.run()) - .await - .map_err(|error| AuditError::Io(io::Error::other(error))) - .and_then(|result| result); + }) + } + + fn settle( + &self, + state: &mut FileState, + outcome: Result, + ) -> Result<(), AuditError> { match outcome { Ok(result) => { if let Some(active) = result.replacement { @@ -941,6 +1256,23 @@ impl SegmentedFile { } } +/// Counts one blocking write from its start until its thread finishes it, +/// even when the caller that awaited it was canceled. +struct InFlightWrite(Arc); + +impl InFlightWrite { + fn start(counter: &Arc) -> Self { + counter.fetch_add(1, Ordering::AcqRel); + Self(Arc::clone(counter)) + } +} + +impl Drop for InFlightWrite { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::AcqRel); + } +} + struct AppendRequest { path: PathBuf, rotate_bytes: u64, @@ -1399,6 +1731,7 @@ mod tests { file.sync_hook = Some(hook); AuditWriter { inner: Arc::new(WriterInner::File(Arc::new(GroupCommitFile::new(file)))), + open: Arc::default(), } } @@ -2711,4 +3044,322 @@ mod tests { .await .expect_err("leftover hash-chained journal"); } + + const SCHEMA: &str = "registry.test.audit/v2"; + + fn unfinished() -> Value { + json!({"operationId": "read", "outcome": "unfinished"}) + } + + fn buffered() -> (AuditWriter, SharedBuffer) { + let buffer = SharedBuffer::default(); + ( + AuditWriter::from_line_sink(Box::new(buffer.clone())), + buffer, + ) + } + + fn buffered_lines(buffer: &SharedBuffer) -> Vec { + String::from_utf8(buffer.0.lock().expect("buffer").clone()) + .expect("utf-8") + .lines() + .map(|line| serde_json::from_str(line).expect("json line")) + .collect() + } + + fn assert_paired(entries: &[Value], outcome: &str) { + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["schema"], entries[0]["schema"]); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!(entries[1]["record"]["outcome"], outcome); + } + + #[tokio::test] + async fn an_answered_request_writes_only_its_responses() { + let (writer, buffer) = buffered(); + let mut request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + assert_eq!(request.correlation(), "req-1"); + assert!(!request.is_answered()); + request + .respond(json!({"outcome": "returned"})) + .await + .expect("response"); + assert!(request.is_answered()); + request + .respond(json!({"outcome": "returned"})) + .await + .expect("second response"); + drop(request); + let entries = buffered_lines(&buffer); + assert_eq!(entries.len(), 3); + assert!(entries[1..] + .iter() + .all(|entry| entry["record"]["outcome"] == "returned" + && entry["correlation"] == "req-1" + && entry["schema"] == SCHEMA)); + } + + #[tokio::test] + async fn a_response_appended_elsewhere_answers_the_open_request() { + let (writer, buffer) = buffered(); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + .expect("response"); + assert!(request.is_answered()); + drop(request); + assert_paired(&buffered_lines(&buffer), "returned"); + } + + #[tokio::test] + async fn a_response_in_another_schema_does_not_answer_the_request() { + let (writer, buffer) = buffered(); + let request = writer + .begin(SCHEMA, "req-1", json!({"operationId": "run"}), unfinished()) + .await + .expect("request"); + writer + .append(AuditEntry::response( + "registry.test.other/v1", + "req-1", + json!({"outcome": "refused"}), + )) + .await + .expect("other schema"); + assert!(!request.is_answered()); + drop(request); + let entries = buffered_lines(&buffer); + assert_eq!(entries.len(), 3); + assert_eq!(entries[2]["schema"], SCHEMA); + assert_eq!(entries[2]["correlation"], "req-1"); + assert_eq!(entries[2]["record"]["outcome"], "unfinished"); + } + + #[tokio::test] + async fn one_response_answers_one_of_two_requests_sharing_a_correlation() { + let (writer, buffer) = buffered(); + let first = writer + .begin(SCHEMA, "shared", json!({"n": 1}), unfinished()) + .await + .expect("first"); + let mut second = writer + .begin(SCHEMA, "shared", json!({"n": 2}), unfinished()) + .await + .expect("second"); + second + .respond(json!({"outcome": "returned"})) + .await + .expect("second answers itself"); + assert!(!first.is_answered(), "the second's response is its own"); + drop(second); + drop(first); + let outcomes: Vec<_> = buffered_lines(&buffer) + .iter() + .filter(|entry| entry["phase"] == "response") + .map(|entry| entry["record"]["outcome"].clone()) + .collect(); + assert_eq!(outcomes, [json!("returned"), json!("unfinished")]); + } + + #[tokio::test] + async fn a_request_dropped_unanswered_writes_its_unfinished_response() { + let (writer, buffer) = buffered(); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + drop(request); + assert_paired(&buffered_lines(&buffer), "unfinished"); + } + + #[tokio::test] + async fn an_early_error_return_pairs_the_request() { + async fn refused(writer: &AuditWriter) -> Result<(), &'static str> { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .map_err(|_| "audit")?; + Err("not found")?; + Ok(()) + } + let (writer, buffer) = buffered(); + assert_eq!(refused(&writer).await, Err("not found")); + assert_paired(&buffered_lines(&buffer), "unfinished"); + } + + #[tokio::test] + async fn a_refused_response_leaves_the_request_owed() { + let (writer, buffer) = buffered(); + let mut request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + let refused = request.respond(json!(["not", "an", "object"])).await; + assert_eq!( + refused.map_err(|error| error.reason()), + Err(AuditUnavailableReason::InvalidEntry) + ); + assert!(!request.is_answered()); + drop(request); + assert_paired(&buffered_lines(&buffer), "unfinished"); + } + + #[tokio::test] + async fn an_unfinished_record_that_is_not_an_object_writes_no_request() { + let (writer, buffer) = buffered(); + let refused = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + json!("gone"), + ) + .await; + assert!(refused.is_err()); + assert!(buffered_lines(&buffer).is_empty()); + } + + #[tokio::test] + async fn a_canceled_operation_pairs_its_request_in_the_file() { + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let (started, begun) = tokio::sync::oneshot::channel(); + let operation = tokio::spawn({ + let writer = writer.clone(); + async move { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + started.send(()).expect("signal"); + std::future::pending::<()>().await; + } + }); + begun.await.expect("begun"); + operation.abort(); + assert!(operation.await.expect_err("aborted").is_cancelled()); + // A later append is ordered after the queued unfinished response. + writer + .append(AuditEntry::response(SCHEMA, "other", json!({}))) + .await + .expect("later entry"); + let entries = lines(&path); + assert_paired(&entries[..2], "unfinished"); + assert_eq!(entries[2]["correlation"], "other"); + } + + #[tokio::test] + async fn a_panicking_operation_pairs_its_request() { + let (writer, buffer) = buffered(); + let operation = tokio::spawn({ + let writer = writer.clone(); + async move { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + panic!("handler failed"); + } + }); + assert!(operation.await.expect_err("panicked").is_panic()); + assert_paired(&buffered_lines(&buffer), "unfinished"); + } + + #[test] + fn a_command_that_exits_after_an_early_return_pairs_its_request() { + // An operator command runs on a current-thread runtime and exits as + // soon as its future returns, before a spawned flush can run. + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let result: Result<(), &str> = runtime.block_on(async move { + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "erase"}), + unfinished(), + ) + .await + .expect("request"); + Err("database unavailable") + }); + assert_eq!(result, Err("database unavailable")); + drop(runtime); + assert_paired(&lines(&path), "unfinished"); + } + + #[tokio::test] + async fn a_stopped_writer_refuses_the_request_and_writes_nothing() { + let writer = AuditWriter::from_line_sink(Box::new(FailingSink)); + assert!(writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished() + ) + .await + .is_err()); + assert!(!writer.ready().await); + } } From c66be30da460a2845c42b1eaabaf482f665d91ef Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 12:35:46 +0000 Subject: [PATCH 02/32] fix(casework): pair every audited request with a response Hold the shared AuditRequest for each caller-requested operation, so a refusal, a failed commit, or a canceled request that returns after the request entry still writes {event, outcome: "unfinished"} as its response under the same correlation. Audit add_review_note, which wrote no entries: it now records who added a note and its history event, never the note's text or audience. Refs #1576 Signed-off-by: Jeremi Joslin --- crates/registry-casework/src/audit.rs | 93 ++++++++++++------- crates/registry-casework/src/review.rs | 27 +++++- .../tests/postgres_transactions.rs | 61 +++++++++++- .../tests/review_postgres.rs | 79 ++++++++++++++++ 4 files changed, 222 insertions(+), 38 deletions(-) diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index 57aeb7aed..286573c58 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -7,10 +7,13 @@ //! operation that recorded no domain event, such as an idempotent replay, //! appends one `response` entry naming its terminal outcome instead, so no //! caller-requested result is released without an accepted `response` entry. +//! One that returns without committing, a refusal, a failure, or a canceled +//! request, appends `{event, outcome: "unfinished"}` as its `response` entry +//! when its operation is dropped, so no `request` entry stays unpaired. //! Entries carry only event metadata and keyed references, never source //! selectors, free-text reasons, receipts, or issuer and subject identities. -use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditWriter}; +use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditRequest, AuditWriter}; use serde_json::{json, Map, Value}; use uuid::Uuid; @@ -62,19 +65,20 @@ impl CaseworkAudit { .and_then(Value::as_str) .ok_or(StoreError::Corrupt)? .to_owned(); - let correlation = Uuid::new_v4().to_string(); - self.writer - .append(AuditEntry::request( + let unfinished = self.minimized(json!({"event": event, "outcome": "unfinished"}))?; + let request = self + .writer + .begin( CASEWORK_AUDIT_SCHEMA, - correlation.clone(), + Uuid::new_v4().to_string(), record, - )) + unfinished, + ) .await .map_err(|_| StoreError::AuditUnavailable)?; Ok(AuditOperation { audit: self.clone(), - correlation, - request_event: Some(event), + pairing: Pairing::Requested { request, event }, outcome: None, responses: Vec::new(), }) @@ -90,8 +94,9 @@ impl CaseworkAudit { } Ok(AuditOperation { audit: self.clone(), - correlation: Uuid::new_v4().to_string(), - request_event: None, + pairing: Pairing::Background { + correlation: Uuid::new_v4().to_string(), + }, outcome: None, responses: Vec::new(), }) @@ -161,18 +166,28 @@ impl AuditOutcome { /// One audited operation whose `request` entry was accepted. It collects the /// minimized `response` records its transaction produces and appends them -/// once the transaction commits. +/// once the transaction commits. Dropped before then, a caller-requested +/// operation appends its `unfinished` response entry. #[must_use = "an audited operation appends its response entries only when completed"] pub(crate) struct AuditOperation { audit: CaseworkAudit, - correlation: String, - /// The event of the accepted `request` entry; absent for background work, - /// which writes no `request` entry. - request_event: Option, + pairing: Pairing, outcome: Option, responses: Vec<(Uuid, Value)>, } +/// How an operation's `response` entries are correlated. +enum Pairing { + /// A caller-requested operation: its accepted `request` entry, which owes + /// the `response` entries, and the event that entry names. + Requested { + request: AuditRequest, + event: String, + }, + /// Background work no caller requested, which writes no `request` entry. + Background { correlation: String }, +} + impl AuditOperation { /// Name the terminal outcome of a caller-requested operation that may /// record no domain event. It is appended as the operation's `response` @@ -186,7 +201,10 @@ impl AuditOperation { /// nor a terminal outcome to append, since its result would leave without /// an accepted `response` entry. fn ensure_terminal(&self) -> Result<(), StoreError> { - if self.request_event.is_some() && self.responses.is_empty() && self.outcome.is_none() { + if matches!(self.pairing, Pairing::Requested { .. }) + && self.responses.is_empty() + && self.outcome.is_none() + { tracing::error!( "a Casework audited operation has no response entry to append; its result is withheld" ); @@ -265,29 +283,34 @@ impl AuditOperation { /// destination unavailable while the committed change stays in place. pub(crate) async fn complete(self) -> Result<(), StoreError> { self.ensure_terminal()?; - let mut records: Vec = self - .responses - .into_iter() - .map(|(_, record)| record) - .collect(); + let Self { + audit, + mut pairing, + outcome, + responses, + } = self; + let mut records: Vec = responses.into_iter().map(|(_, record)| record).collect(); if records.is_empty() { - if let (Some(event), Some(outcome)) = (&self.request_event, self.outcome) { - records.push( - self.audit - .minimized(json!({"event": event, "outcome": outcome.as_str()}))?, - ); + if let (Pairing::Requested { event, .. }, Some(outcome)) = (&pairing, outcome) { + records + .push(audit.minimized(json!({"event": event, "outcome": outcome.as_str()}))?); } } for record in records { - self.audit - .writer - .append(AuditEntry::response( - CASEWORK_AUDIT_SCHEMA, - self.correlation.clone(), - record, - )) - .await - .map_err(|_| StoreError::AuditUnavailable)?; + match &mut pairing { + Pairing::Requested { request, .. } => request.respond(record).await, + Pairing::Background { correlation } => { + audit + .writer + .append(AuditEntry::response( + CASEWORK_AUDIT_SCHEMA, + correlation.clone(), + record, + )) + .await + } + } + .map_err(|_| StoreError::AuditUnavailable)?; } Ok(()) } diff --git a/crates/registry-casework/src/review.rs b/crates/registry-casework/src/review.rs index 7850557d4..17b833ba9 100644 --- a/crates/registry-casework/src/review.rs +++ b/crates/registry-casework/src/review.rs @@ -3742,6 +3742,14 @@ impl PostgresStore { request: ReviewNoteRequest, idempotency_key: &str, ) -> Result { + let mut audit = self + .begin_audit(crate::audit::request_record( + "review_note_added", + Some(actor), + &actor.profile_id, + json!({}), + )) + .await?; if request.note.trim().is_empty() || request.note.len() > 2_000 || request.note.chars().any(char::is_control) @@ -3775,7 +3783,8 @@ impl PostgresStore { ) .await? { - transaction.commit().await?; + audit.record_outcome(crate::audit::AuditOutcome::Replayed); + audit.commit(transaction).await?; return serde_json::from_value(response).map_err(ReviewRuntimeError::from); } let entry = ReviewHistoryEntry { @@ -3819,7 +3828,21 @@ impl PostgresStore { &serde_json::to_value(&entry)?, ) .await?; - transaction.commit().await?; + // The note's text and audience stay in the review history; the audit + // record names only who added a note and which history event it is. + audit.record( + entry.event_id, + json!({ + "event": "casework.review_note_added", + "eventId": entry.event_id, + "actor": { + "issuer": actor.principal.issuer, + "subject": actor.principal.subject, + }, + "profileId": actor.profile_id, + }), + )?; + audit.commit(transaction).await?; Ok(entry) } diff --git a/crates/registry-casework/tests/postgres_transactions.rs b/crates/registry-casework/tests/postgres_transactions.rs index 3b2383fb8..0b918d7ce 100644 --- a/crates/registry-casework/tests/postgres_transactions.rs +++ b/crates/registry-casework/tests/postgres_transactions.rs @@ -1743,6 +1743,27 @@ async fn a_resubmitted_proposal_supersedes_the_earlier_application_item() { assert_eq!(current.state, OccurrenceState::WaitingApplication); } +/// The correlations of request entries no response entry answers. +fn unpaired_requests(entries: &[serde_json::Value]) -> Vec { + entries + .iter() + .filter(|entry| entry["phase"] == "request") + .filter(|request| { + !entries.iter().any(|entry| { + entry["phase"] == "response" + && entry["schema"] == request["schema"] + && entry["correlation"] == request["correlation"] + }) + }) + .map(|request| { + request["correlation"] + .as_str() + .unwrap_or_default() + .to_owned() + }) + .collect() +} + const SETTLEMENT_REASON: &str = "The source refused the saved evidence version; the registrar confirmed no change was made."; const SETTLEMENT_DECIDED_BY: &str = "Registrar duty officer, ticket OPS-4411"; @@ -1890,12 +1911,17 @@ impl SettlementFixture { .await .expect("settlement snapshot") .get(0); + // A refusal pairs its request entry with an `unfinished` response; + // only a response recording a committed change counts as a write. snapshot["auditResponses"] = serde_json::json!(self .audit .entries() .iter() - .filter(|entry| entry["phase"] == "response") + .filter(|entry| { + entry["phase"] == "response" && entry["record"]["outcome"] != "unfinished" + }) .count()); + snapshot["unpairedRequests"] = serde_json::json!(unpaired_requests(&self.audit.entries())); snapshot } @@ -2155,6 +2181,39 @@ async fn a_refused_audit_response_reports_unavailable_after_the_settlement_commi ); } +#[tokio::test] +async fn a_refusal_after_the_request_entry_pairs_it_with_an_unfinished_response() { + let fixture = settlement_fixture("refusal_pairs_request", true).await; + let written = fixture.audit.entries().len(); + let missing = fixture + .store + .claim(&fixture.holder, uuid::Uuid::new_v4(), 1, "claim-missing") + .await; + assert!(matches!(missing, Err(StoreError::NotFound)), "{missing:?}"); + let item = fixture.store.item(fixture.item_id).await.expect("item"); + let again = fixture + .store + .claim(&fixture.holder, item.item_id, item.revision, "claim-again") + .await; + assert!( + matches!(again, Err(StoreError::AlreadyClaimed)), + "{again:?}" + ); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 4, "{entries:#?}"); + for pair in entries.chunks(2) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[1]["schema"], pair[0]["schema"]); + assert_eq!(pair[1]["correlation"], pair[0]["correlation"]); + assert_eq!( + pair[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); + } +} + #[tokio::test] async fn a_live_execution_lease_refuses_settlement_and_writes_nothing() { let fixture = settlement_fixture("settle_live_lease", true).await; diff --git a/crates/registry-casework/tests/review_postgres.rs b/crates/registry-casework/tests/review_postgres.rs index 9840bcd04..5b7036c9d 100644 --- a/crates/registry-casework/tests/review_postgres.rs +++ b/crates/registry-casework/tests/review_postgres.rs @@ -3318,6 +3318,85 @@ async fn review_cancellation_is_audited_after_the_cancel_commits() { assert!(!audit_text.contains(&created.accepted.request_id.to_string())); } +#[tokio::test] +async fn review_notes_are_audited_without_their_text() { + let fixture = fixture().await; + let (service, audit) = service_with_audit(&fixture, project("1")); + let created = service + .create_review_request( + &fixture.producer, + request("note-audit", "producer-ref-note-audit"), + "create-note-audit", + ) + .await + .expect("create note audit review"); + let request_id = created.accepted.request_id; + let note = |text: &str| ReviewNoteRequest { + audience: ReviewHistoryAudience::Requester, + note: text.to_owned(), + }; + let added = service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note("a note that stays out of the audit"), + "note-audit", + ) + .await + .expect("add note"); + let record = audited_response( + &audit, + "review_note_added", + "eventId", + &added.event_id.to_string(), + ); + assert!(!serde_json::to_string(&audit.entries()) + .expect("audit JSON") + .contains("stays out of the audit")); + assert!(record.get("principalPseudonym").is_some()); + + let (replay_service, replay_audit) = service_with_audit(&fixture, project("1")); + replay_service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note("a note that stays out of the audit"), + "note-audit", + ) + .await + .expect("replay note"); + assert_replay_audited(&replay_audit, "review_note_added"); + + // A refused note still pairs the request entry it wrote. + let (refused_service, refused_audit) = service_with_audit(&fixture, project("1")); + assert!(matches!( + refused_service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note(" "), + "note-audit-refused", + ) + .await, + Err(ReviewRuntimeError::Invalid) + )); + let entries = refused_audit.entries(); + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + json!({"event": "casework.review_note_added", "outcome": "unfinished"}) + ); +} + #[tokio::test] async fn review_task_ownership_transitions_are_audited() { let fixture = fixture().await; From e4a9cd8f103d9c43d2c43c47f2b046af45645600 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 12:46:38 +0000 Subject: [PATCH 03/32] test(casework): expect a withheld result to pair its request entry An operation that has no response record to append still withholds its result, and now writes its unfinished outcome as the response, so its request entry is no longer left unpaired. Refs #1576 Signed-off-by: Jeremi Joslin --- crates/registry-casework/src/audit.rs | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index 286573c58..22300c3df 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -613,9 +613,17 @@ mod tests { operation.complete().await, Err(StoreError::AuditUnavailable) )); + // The result is withheld, and the request entry is still paired: the + // operation writes its unfinished outcome as the response. let entries = capture.entries(); - assert_eq!(entries.len(), 1, "only the request entry was written"); + assert_eq!(entries.len(), 2, "{entries:?}"); assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + json!({"event": "casework.task_claimed", "outcome": "unfinished"}) + ); } #[tokio::test] From f17ed102ab22a1bbade084e5d4d4518c357409bb Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 13:04:40 +0000 Subject: [PATCH 04/32] fix(scheduling): answer every commitment request entry A failed capacity transaction, records replaced under a commitment, and a reused or expired idempotency key wrote a request entry and no response. They now write a response with the outcome unfinished and a closed reason (commitment.failed, commitment.facts-stale, idempotency.key-reused, idempotency.expired), never an authorization verdict. The request is held through the shared AuditRequest handle, so a commitment that returns or is canceled before it answers writes commitment.unfinished. A refusal, from the ledger or the permission check, is now answered only once its response entry is accepted, and service.unavailable otherwise, instead of logging the write failure and answering the refusal anyway. SCHEDULING-SEC-14 and the runtime configuration reference say so. Refs #1582 Signed-off-by: Jeremi Joslin --- crates/registry-scheduling/src/audit.rs | 38 ++- crates/registry-scheduling/src/service.rs | 250 ++++++++++-------- .../tests/postgres_commitments.rs | 210 +++++++++++++++ products/scheduling/CHANGELOG.md | 8 + products/scheduling/RUNTIME-CONFIG.md | 17 +- .../contracts/security-invariant-matrix.yaml | 26 +- .../contracts/security-test-traceability.yaml | 5 + 7 files changed, 428 insertions(+), 126 deletions(-) diff --git a/crates/registry-scheduling/src/audit.rs b/crates/registry-scheduling/src/audit.rs index cc50f46a7..e65d6f224 100644 --- a/crates/registry-scheduling/src/audit.rs +++ b/crates/registry-scheduling/src/audit.rs @@ -5,10 +5,15 @@ //! opens and one `response` entry once the decision is known: after commit //! for an allowed commitment, after rollback for a refused one. Both share a //! correlation, which is also the `eventId` the response record carries. +//! A commitment nothing decided, because its transaction failed, the +//! environment records moved under it, or its idempotency key was refused, +//! still answers its request entry with an `unfinished` response naming the +//! reason; one that returns or is canceled without answering writes the +//! `commitment.unfinished` response when its request handle is dropped. //! Entries carry only pseudonymized references and closed codes, never a raw //! principal, grant, claim identifier, or free-text reason. -use registry_platform_audit::{AuditEntry, AuditUnavailable, AuditWriter}; +use registry_platform_audit::{AuditEntry, AuditRequest, AuditUnavailable, AuditWriter}; use serde_json::Value; use uuid::Uuid; @@ -50,6 +55,25 @@ impl SchedulingAudit { .await } + /// Append the `request` entry of one audited operation and return the + /// handle that owes its `response`. Dropped unanswered, the handle writes + /// `unfinished` as the response under the same correlation. + pub async fn begin( + &self, + correlation: Uuid, + record: Value, + unfinished: Value, + ) -> Result { + self.writer + .begin( + SCHEDULING_AUDIT_SCHEMA, + correlation.to_string(), + record, + unfinished, + ) + .await + } + /// Append the `response` entry of one audited operation. pub async fn response(&self, correlation: Uuid, record: Value) -> Result<(), AuditUnavailable> { self.writer @@ -73,6 +97,18 @@ pub fn request_record(mut record: Value) -> Value { record } +/// The `response` form of a commitment nothing decided: the request's fields +/// with the `unfinished` outcome and the closed `reason` it did not finish. +#[must_use] +pub fn unfinished_record(record: Value, reason: &str) -> Value { + let mut record = request_record(record); + if let Some(fields) = record.as_object_mut() { + fields.insert("outcome".to_owned(), Value::String("unfinished".to_owned())); + fields.insert("reason".to_owned(), Value::String(reason.to_owned())); + } + record +} + /// Stamp a `response` record with its audit identity, refusing a record that /// already carries a different one. #[must_use] diff --git a/crates/registry-scheduling/src/service.rs b/crates/registry-scheduling/src/service.rs index 80e9f852f..1aaca3f40 100644 --- a/crates/registry-scheduling/src/service.rs +++ b/crates/registry-scheduling/src/service.rs @@ -16,7 +16,9 @@ //! a replayed idempotency key answers as it first did. use chrono::{DateTime, TimeDelta, Utc}; -use registry_platform_audit::{AuditKeyHasher, AuthorizationAuditEvent, AuthorizationOutcome}; +use registry_platform_audit::{ + AuditKeyHasher, AuditRequest, AuthorizationAuditEvent, AuthorizationOutcome, +}; use registry_platform_calendar::CalendarInterval; use registry_platform_canonical_json::canonicalize_json; use registry_platform_oidc::GrantClaims; @@ -36,7 +38,7 @@ use sha2::{Digest as _, Sha256}; use std::borrow::Cow; use uuid::Uuid; -use crate::audit::{request_record, with_event_id, SchedulingAudit}; +use crate::audit::{request_record, unfinished_record, with_event_id, SchedulingAudit}; use crate::cursors::{ bind_stored, cursor_expiry, decode_cursor, encode_cursor, CursorError, ListingPosition, StoredCursor, @@ -595,7 +597,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, HOLD_CREATE_ACTION) .await?; let outcome = self @@ -655,7 +657,7 @@ impl SchedulingService { // closing a hold cannot name a resource. let commitment = self.commitment(caller, &actor, &grant, &release_key, &request_hash, now, 0)?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, HOLD_RELEASE_ACTION) .await?; let outcome = self.store.release_hold(hold_id, commitment).await; @@ -736,7 +738,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CREATE_ACTION) .await?; let outcome = self @@ -782,7 +784,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CREATE_ACTION) .await?; let outcome = self @@ -862,7 +864,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_RESCHEDULE_ACTION) .await?; let outcome = self @@ -928,7 +930,7 @@ impl SchedulingService { now, 0, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CANCEL_ACTION) .await?; let outcome = self @@ -1141,7 +1143,8 @@ impl SchedulingService { /// the refusal. Readable availability is not authority to book: the /// permission must name all three. /// - /// Both refusals are audited. A refusal decided here never opens the + /// Both refusals are audited, and answered only once their entry is + /// accepted. A refusal decided here never opens the /// capacity transaction, so nothing further in the request would record /// that it happened, and a caller probing which services and locations its /// grant reaches would leave no audit entry. Each refusal is one `response` @@ -1159,7 +1162,7 @@ impl SchedulingService { Uuid::new_v4(), grantless_refusal_record(&self.hasher, &self.scheduling_id, caller, action), ) - .await; + .await?; return Err(ServiceError::Problem(ProblemCode::OperationNotAuthorized)); }; let allowed = grant @@ -1190,43 +1193,47 @@ impl SchedulingService { "authorization.refused", ), ) - .await; + .await?; Err(ServiceError::Problem(ProblemCode::OperationNotAuthorized)) } } - /// Write one authorization refusal as the `response` entry of - /// `correlation`, whose identity it also carries as `eventId`. A - /// destination that refuses it must not change the caller's answer: the - /// decision is already made and the caller is refused either way, so the - /// failure is logged loudly and the refusal stands. - async fn record_refusal(&self, correlation: Uuid, record: Result) { - match record.map(|record| with_event_id(correlation, record)) { - Ok(Some(record)) => { - if let Err(failure) = self.audit.response(correlation, record).await { - tracing::error!(%failure, "the refusal audit entry could not be recorded"); - } - } - Ok(None) => { - tracing::error!("the refusal audit entry carries another identity"); - } - Err(refused) => { - tracing::error!(error = %refused, "the refusal audit entry could not be built"); - } - } + /// Write one refusal, or one commitment nothing decided, as the + /// `response` entry of `correlation`, whose identity it also carries as + /// `eventId`. It fails closed like an allowed response: a refusal whose + /// entry the destination does not accept is answered + /// `service.unavailable`, so no refusal is answered without its audit. + async fn record_refusal( + &self, + correlation: Uuid, + record: Result, + ) -> Result<(), ServiceError> { + let record = with_event_id(correlation, record?).ok_or_else(|| { + ServiceError::internal("the refusal audit record carries another identity") + })?; + self.audit + .response(correlation, record) + .await + .map_err(|failure| { + tracing::error!(%failure, "the refusal audit entry was refused"); + ServiceError::Problem(ProblemCode::ServiceUnavailable) + }) } /// Append the `request` entry of one commitment and return the - /// correlation its `response` entry will carry. It names the caller, the - /// grant, and the operation the capacity transaction will decide, and no - /// outcome. A destination that refuses it refuses the commitment: the - /// capacity transaction does not open. + /// correlation its `response` entry will carry, with the handle that owes + /// it. It names the caller, the grant, and the operation the capacity + /// transaction will decide, and no outcome. A destination that refuses it + /// refuses the commitment: the capacity transaction does not open. The + /// caller holds the handle until it answers: the response written under + /// the correlation answers it, and a commitment that returns or is + /// canceled before then writes the `commitment.unfinished` response. async fn audit_request( &self, caller: &Caller, grant: &GrantClaims, operation: &str, - ) -> Result { + ) -> Result<(Uuid, AuditRequest), ServiceError> { let record = audit_record( &self.hasher, &self.scheduling_id, @@ -1237,14 +1244,20 @@ impl SchedulingService { "authorization.allowed", )?; let correlation = Uuid::new_v4(); - self.audit - .request(correlation, request_record(record)) + let unfinished = with_event_id( + correlation, + unfinished_record(record.clone(), "commitment.unfinished"), + ) + .ok_or_else(|| ServiceError::internal("the commitment audit record carries an identity"))?; + let request = self + .audit + .begin(correlation, request_record(record), unfinished) .await .map_err(|failure| { tracing::error!(%failure, "the commitment request audit entry was refused"); ServiceError::Problem(ProblemCode::ServiceUnavailable) })?; - Ok(correlation) + Ok((correlation, request)) } /// Append the `response` entry that gates an answer: the allowed entry @@ -1398,9 +1411,11 @@ impl SchedulingService { /// accepted, and a replayed receipt, including one a concurrent identical /// request won, only once the `response` entry recording its decision /// is. A refusal writes its receipt under the caller's idempotency key so - /// a replay of that key answers the same, and writes its denied - /// `response` entry when the refusal was an authorization decision. Both - /// entries carry the `correlation` of the commitment's `request` entry. + /// a replay of that key answers the same, and is answered only once its + /// `response` entry is accepted: denied for a decision the ledger took, + /// `unfinished` with its reason for a failed transaction, a replaced + /// environment, or a refused idempotency key. Every entry carries the + /// `correlation` of the commitment's `request` entry. #[allow(clippy::too_many_arguments)] async fn commitment_outcome( &self, @@ -1439,32 +1454,54 @@ impl SchedulingService { Ok(CommitmentAnswer::Minted(T::from_minted(minted))) } Err(error) => { - if matches!( + let mut problem = problem_of(&error); + // Every commitment answers its request entry, refused as much + // as allowed, and one nothing decided as well. The match is + // exhaustive on purpose: a new variant must state which side + // it falls on rather than inherit silence from a wildcard. + let mut response = match &error { + // The transaction failed and the caller sees no detail. + CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) => { + tracing::error!(%error, "the Scheduling store failed mid-commitment"); + Unanswered::Unfinished("commitment.failed") + } + // A records replacement moved under this request, and the + // caller retries; the swap is an expected operator act, so + // this is a warning, not a failure. + CommitError::FactsStale => { + tracing::warn!( + %error, + "the environment records were replaced while a commitment was in flight" + ); + Unanswered::Unfinished("commitment.facts-stale") + } + // The idempotency layer refused the key, not the + // commitment, and the key says nothing about what a grant + // reaches, so this is no authorization decision. + CommitError::KeyReused | CommitError::KeyExpired => { + Unanswered::Unfinished(key_refusal_reason(&error)) + } + CommitError::Unauthorized => Unanswered::Denied("authorization.refused"), + CommitError::Refused(_) + | CommitError::HoldCeiling + | CommitError::RevisionMismatch + | CommitError::CutoffPassed => Unanswered::Denied("authorization.profile"), + }; + // A refusal writes its receipt so a replay of the key answers + // the same. A failed transaction and a replaced environment + // decided nothing, and an expired receipt cannot be + // recreated. A reused key still passes through the + // insert-or-replay path: a concurrent identical winner is + // replayed, while a different request hash remains key-reused. + let receipted = !matches!( error, - CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) - ) { - // Nothing was decided: no receipt, no response entry, and - // the caller sees no detail. - tracing::error!(%error, "the Scheduling store failed mid-commitment"); - return Err(ServiceError::Problem(ProblemCode::ServiceUnavailable)); - } - if matches!(error, CommitError::FactsStale) { - // A records replacement moved under this request. Nothing - // was decided and the caller retries; the swap is an - // expected operator act, so this is a warning, not a - // failure. - tracing::warn!( - %error, - "the environment records were replaced while a commitment was in flight" - ); - return Err(ServiceError::Problem(ProblemCode::ServiceUnavailable)); - } - let problem = problem_of(&error); - // An expired receipt cannot be recreated. A reused key still - // passes through the insert-or-replay path: a concurrent - // identical winner is replayed, while a different request - // hash remains key-reused. - if !matches!(error, CommitError::KeyExpired) { + CommitError::Store(_) + | CommitError::Query(_) + | CommitError::Hooks(_) + | CommitError::FactsStale + | CommitError::KeyExpired + ); + if receipted { let receipt = self.commitment( caller, actor, @@ -1506,7 +1543,8 @@ impl SchedulingService { Err( failure @ (CommitError::KeyReused | CommitError::KeyExpired), ) => { - return Err(ServiceError::Problem(problem_of(&failure))); + problem = problem_of(&failure); + response = Unanswered::Unfinished(key_refusal_reason(&failure)); } Err(failure) => { tracing::error!(%failure, "the refused attempt receipt could not be recorded"); @@ -1518,43 +1556,28 @@ impl SchedulingService { } } } - // Every commitment the ledger decides is attributable, refused - // as much as allowed. The match is exhaustive on purpose: a - // new variant must state which side it falls on rather than - // inherit silence from a wildcard. - let audited = match error { - // Nothing was decided. The transaction failed, or the - // environment moved and the caller retries against the - // current records, so there is no verdict to attribute. - CommitError::Store(_) - | CommitError::Query(_) - | CommitError::Hooks(_) - | CommitError::FactsStale => None, - // The idempotency layer refused the key, not the - // commitment. The attempt receipt above already records - // it, and the key says nothing about what a grant reaches. - CommitError::KeyReused | CommitError::KeyExpired => None, - CommitError::Unauthorized => Some("authorization.refused"), - CommitError::Refused(_) - | CommitError::HoldCeiling - | CommitError::RevisionMismatch - | CommitError::CutoffPassed => Some("authorization.profile"), - }; - if let Some(reason) = audited { - self.record_refusal( - correlation, - audit_record( - &self.hasher, - &self.scheduling_id, - caller, - grant, - operation, - AuthorizationOutcome::Denied, - reason, - ), + let record = match response { + Unanswered::Denied(reason) => audit_record( + &self.hasher, + &self.scheduling_id, + caller, + grant, + operation, + AuthorizationOutcome::Denied, + reason, + ), + Unanswered::Unfinished(reason) => audit_record( + &self.hasher, + &self.scheduling_id, + caller, + grant, + operation, + AuthorizationOutcome::Allowed, + "authorization.allowed", ) - .await; - } + .map(|record| unfinished_record(record, reason)), + }; + self.record_refusal(correlation, record).await?; Err(ServiceError::Problem(problem)) } } @@ -2110,6 +2133,24 @@ impl ClaimRow { } } +/// The `response` a commitment that is not answered by a success or a replay +/// writes: a refusal the ledger or the permission check decided, or the +/// closed reason a commitment nothing decided did not finish. +#[derive(Clone, Copy)] +enum Unanswered { + Denied(&'static str), + Unfinished(&'static str), +} + +/// The closed reason an idempotency key refusal records. +fn key_refusal_reason(error: &CommitError) -> &'static str { + if matches!(error, CommitError::KeyExpired) { + "idempotency.expired" + } else { + "idempotency.key-reused" + } +} + fn problem_of(error: &CommitError) -> ProblemCode { match error { CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) => { @@ -2122,8 +2163,7 @@ fn problem_of(error: &CommitError) -> ProblemCode { CommitError::Unauthorized => ProblemCode::OperationNotAuthorized, CommitError::RevisionMismatch => ProblemCode::RevisionMismatch, CommitError::CutoffPassed => ProblemCode::CancellationCutoffPassed, - // Never reached: the outcome handler intercepts a stale-facts - // refusal before it projects, because nothing was decided. + // Nothing was decided: the caller retries against the current records. CommitError::FactsStale => ProblemCode::ServiceUnavailable, } } diff --git a/crates/registry-scheduling/tests/postgres_commitments.rs b/crates/registry-scheduling/tests/postgres_commitments.rs index 133319c83..3f552b241 100644 --- a/crates/registry-scheduling/tests/postgres_commitments.rs +++ b/crates/registry-scheduling/tests/postgres_commitments.rs @@ -6550,3 +6550,213 @@ async fn migration_refuses_to_drop_unpublished_audit_and_drops_a_drained_outbox( .await .expect("the schema is at this release"); } + +/// The single `response` entry that answers the commitment's `request` +/// entry among `appended`, after checking the pair shares one correlation +/// and records a commitment nothing decided. +fn one_unfinished_response(appended: &[Value], reason: &str) -> Value { + let response = one_request_and_one_response(appended); + assert_eq!(response["outcome"], "unfinished", "{response}"); + assert_eq!(response["reason"], reason, "{response}"); + response +} + +/// SCHEDULING-SEC-14: a capacity transaction that fails decided nothing, and +/// its request entry is still answered: one response entry records that the +/// commitment did not finish. +#[tokio::test] +async fn a_failed_capacity_transaction_pairs_its_request_entry() { + let fx = hook_fixture().await; + fx.admin + .batch_execute( + "ALTER TABLE registry_outbox + ADD CONSTRAINT test_refuse_hook_capture + CHECK (event_type <> 'confirmed-observer')", + ) + .await + .expect("install the transaction failure seam"); + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "failed-transaction", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + let response = + one_unfinished_response(&fx.capture.entries().split_off(before), "commitment.failed"); + assert_eq!(response["operation"], "appointment.create"); +} + +/// SCHEDULING-SEC-14: an idempotency key refused as reused, and one refused +/// as expired, each answer the request entry the commitment wrote, without +/// recording the key refusal as an authorization decision. +#[tokio::test] +async fn an_idempotency_key_refusal_pairs_its_request_entry() { + let fx = fixture().await; + let (first, second) = first_overlapping_pair(&fx, OFFERING, 90, 260).await; + let (status, _) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, first)}), + ) + .await; + assert_eq!(status, StatusCode::CREATED); + + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, second)}), + ) + .await; + assert_eq!(status, StatusCode::CONFLICT, "{problem}"); + one_unfinished_response( + &fx.capture.entries().split_off(before), + "idempotency.key-reused", + ); + + fx.store + .erase_expired_attempts(Utc::now() + TimeDelta::days(8)) + .await + .expect("the retention sweep runs"); + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, first)}), + ) + .await; + assert_eq!(status, StatusCode::GONE, "{problem}"); + one_unfinished_response( + &fx.capture.entries().split_off(before), + "idempotency.expired", + ); +} + +/// SCHEDULING-SEC-14: a records replacement that lands while a commitment +/// waits on its revision guard leaves the commitment undecided, and its +/// request entry is still answered. +#[tokio::test] +async fn a_records_swap_under_a_commitment_pairs_its_request_entry() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let (http, agent, capture) = (fx.http.clone(), fx.agent.clone(), fx.capture.clone()); + let before = capture.entries().len(); + let mut admin = fx.admin; + let swap = admin + .transaction() + .await + .expect("the stand-in records swap opens"); + swap.execute( + "UPDATE scheduling_meta SET facts_revision = facts_revision + 1 WHERE singleton", + &[], + ) + .await + .expect("the stand-in swap moves the records revision"); + + let commitment = tokio::spawn(send( + http, + "POST".to_owned(), + "/v1/appointments".to_owned(), + agent, + Some("swapped-records".to_owned()), + Some(body), + )); + // Wait until the commitment is blocked on the revision guard, which it + // reaches only after reading the records it was evaluated against. + loop { + swap.batch_execute("SELECT pg_stat_clear_snapshot()") + .await + .expect("refresh the activity snapshot"); + let waiting: i64 = swap + .query_one( + "SELECT count(*) FROM pg_stat_activity + WHERE wait_event_type = 'Lock' AND query LIKE '%FROM scheduling_meta%FOR SHARE%'", + &[], + ) + .await + .expect("read the waiting commitment") + .get(0); + if waiting > 0 { + break; + } + assert!( + !commitment.is_finished(), + "the commitment finished before reaching its revision guard" + ); + tokio::task::yield_now().await; + } + swap.commit().await.expect("the stand-in swap commits"); + + let (status, problem) = commitment.await.expect("the commitment answers"); + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + one_unfinished_response( + &capture.entries().split_off(before), + "commitment.facts-stale", + ); +} + +/// SCHEDULING-SEC-14: a refusal is answered only once its response entry is +/// accepted, like an allowed commitment. When the destination refuses it, the +/// caller is told the service is unavailable, for a permission mismatch +/// decided before the transaction and for a refusal the ledger decided. +#[tokio::test] +async fn a_refusal_whose_response_entry_is_refused_answers_service_unavailable() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let (status, appointment) = fx + .post( + "/v1/appointments", + &fx.agent, + "refusal-audit-create", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::CREATED); + let appointment_id = appointment["appointmentId"].as_str().unwrap().to_owned(); + let stale = appointment["revision"].as_u64().unwrap() + 1; + + // A ledger refusal: the request entry is accepted, its response is not. + fx.capture.refuse_after(fx.capture.entries().len() + 1); + let other = first_slot(&fx, OFFERING, 480, 620).await; + let (status, problem) = fx + .post( + &format!("/v1/appointments/{appointment_id}/reschedule"), + &fx.agent, + "refusal-audit-stale", + json!({"observedRevision": stale, "admission": admission(&fx, OFFERING, other)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); +} + +/// SCHEDULING-SEC-14: a permission mismatch refused before the transaction +/// is answered only once its response entry is accepted. +#[tokio::test] +async fn a_permission_refusal_the_destination_refuses_answers_service_unavailable() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + fx.capture.refuse_after(fx.capture.entries().len()); + let (status, problem) = fx + .post( + "/v1/holds", + &agent_token_outside_its_bounds(), + "outside-bounds-unaudited", + admission(&fx, OFFERING, slot), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); +} diff --git a/products/scheduling/CHANGELOG.md b/products/scheduling/CHANGELOG.md index 766783b1d..f00af1eb7 100644 --- a/products/scheduling/CHANGELOG.md +++ b/products/scheduling/CHANGELOG.md @@ -21,6 +21,14 @@ response entry, recording the decision the receipt carries, is accepted; a refused one answers `service.unavailable` without the receipt. A permission refused before the transaction is one response entry. + - Every commitment request entry is answered. A refusal, decided by the + ledger or by the permission check, is answered only once its response + entry is accepted, and `service.unavailable` otherwise, where it was + previously answered with the write failure only logged. A failed + transaction, records replaced under a commitment, and a reused or expired + idempotency key now write a response with the outcome `unfinished` and a + closed reason, and a commitment that returns or is canceled before + answering writes `commitment.unfinished`. - Hook delivery writes an attempt's request entry before egress and its terminal response entry with the same correlation; a refused entry leaves the delivery pending and sends nothing. diff --git a/products/scheduling/RUNTIME-CONFIG.md b/products/scheduling/RUNTIME-CONFIG.md index ea73009c0..88a5f27ee 100644 --- a/products/scheduling/RUNTIME-CONFIG.md +++ b/products/scheduling/RUNTIME-CONFIG.md @@ -99,13 +99,16 @@ and its `response` entry after the transaction commits or rolls back; a permission refused before the transaction is one `response` entry. Audit fails closed: a refused `request` entry opens no transaction, and a refused `response` entry for a committed change answers `service.unavailable` with the change -committed. One case is best effort instead: a refusal the ledger itself -decides (an admission refusal, the hold ceiling, a lapsed grant, a stale -observed revision, or a cancellation past its cutoff), and a permission -mismatch refused before the transaction opens, still reach the caller when -their `response` entry cannot be written; the write failure is only logged, -and the journal is left holding a `request` entry with no paired `response`, -or, for a permission mismatch, no entry at all. `/readyz` reports unavailable +committed. A refusal, whether the ledger decided it (an admission refusal, +the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation +past its cutoff) or the permission check refused it before the transaction +opened, reaches the caller only once its `response` entry is accepted, and +answers `service.unavailable` otherwise. A commitment nothing decided still +answers its `request` entry: a failed transaction, records replaced under it, +or a reused or expired idempotency key writes a `response` with the outcome +`unfinished` and the reason `commitment.failed`, `commitment.facts-stale`, +`idempotency.key-reused`, or `idempotency.expired`, and one that returns or is +canceled before answering writes `commitment.unfinished`. `/readyz` reports unavailable while the destination refuses writes. An expired hold writes its history entry as `system` and no audit entry. diff --git a/products/scheduling/contracts/security-invariant-matrix.yaml b/products/scheduling/contracts/security-invariant-matrix.yaml index ded7f4ece..78107bb9b 100644 --- a/products/scheduling/contracts/security-invariant-matrix.yaml +++ b/products/scheduling/contracts/security-invariant-matrix.yaml @@ -254,25 +254,25 @@ invariants: transaction opens is one response entry. A refused commitment is one the ledger decided: an admission refusal, the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation past its cutoff. - Both that refusal and a permission mismatch write their response entry - best effort: the failure is only logged, the caller still receives the - refusal already decided, and the journal is left holding a request - entry with no paired response, or, for a permission mismatch, no entry - at all. A failed transaction and a replaced environment decided nothing - and write no response, and an idempotency key refusal is carried by the - attempt receipt instead. The match that selects between them is - exhaustive, so a new outcome states its side. A replayed receipt, - including one a concurrent identical request won, is answered only - after its own response entry is accepted, and that entry records the + Every request entry is answered. A commitment nothing decided, because + its transaction failed, the environment records were replaced under it, + or its idempotency key was refused as reused or expired, writes one + response entry with the outcome unfinished and a closed reason, and + never an authorization verdict. One that returns or is canceled before + it answers writes the commitment.unfinished response when its request + handle is dropped. The match that selects between them is exhaustive, + so a new outcome states its side. A replayed receipt, including one a + concurrent identical request won, is answered only after its own + response entry is accepted, and that entry records the decision the receipt carries: allowed for a success, denied with the original reason for a refusal. refusal: >- Record authorization.allowed or authorization.refused as an audit reason, never as a problem code, and never carry the caller's raw identifier. A refused request entry opens no transaction, and a refused response entry - for a committed commitment or a replayed receipt answers - service.unavailable without releasing the receipt, so no commitment or - replay is answered without its audit. + for a committed commitment, a replayed receipt, a refusal, or a + commitment nothing decided answers service.unavailable without releasing + the receipt, so no answer leaves without its audit. negativeTest: path: crates/registry-scheduling/tests/postgres_commitments.rs name: a_replayed_receipt_is_released_only_after_its_response_entry_is_accepted diff --git a/products/scheduling/contracts/security-test-traceability.yaml b/products/scheduling/contracts/security-test-traceability.yaml index ae796f9de..6b92cfe38 100644 --- a/products/scheduling/contracts/security-test-traceability.yaml +++ b/products/scheduling/contracts/security-test-traceability.yaml @@ -125,6 +125,11 @@ entries: - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_replayed_receipt_is_released_only_after_its_response_entry_is_accepted} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_replayed_refusal_is_recorded_as_the_refusal_it_replays} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_racing_replay_is_not_released_when_its_response_entry_is_refused} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_failed_capacity_transaction_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: an_idempotency_key_refusal_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_records_swap_under_a_commitment_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_refusal_whose_response_entry_is_refused_answers_service_unavailable} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_permission_refusal_the_destination_refuses_answers_service_unavailable} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_concurrent_identical_request_replays_the_winning_receipt} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: concurrent_identical_admissible_requests_replay_one_winning_success} - {path: crates/registry-scheduling/src/service.rs, name: a_replayed_receipt_records_the_decision_it_carries} From ca0c54b2b0ca19e6d2821457cc15fc66a0ed2295 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 13:20:59 +0000 Subject: [PATCH 05/32] fix(render): pair each call's audit entries under a server-drawn id The audit correlation was the caller's Idempotency-Key, which two calls may share, so their request and response entries could not be told apart. Every call now draws its own correlation and the caller's key is recorded only as correlationId. The request entry is held through the shared AuditRequest handle, so a call dropped before its outcome, such as a disconnected caller, writes an unfinished response. The answer is built before the response entry is written, so a rendered entry is never recorded for a document that could not be sent. Closes #1575 Signed-off-by: Jeremi Joslin --- crates/registry-render/src/audit.rs | 48 ++++-- crates/registry-render/src/server.rs | 151 ++++++++++++------ crates/registry-render/tests/serve.rs | 5 +- .../content/docs/operate/registry-render.mdx | 6 +- products/render/ACCEPTANCE.md | 2 +- products/render/README.md | 10 +- products/render/SECURITY-MATRIX.md | 2 +- .../render/integrations/openfn/JOURNEY.md | 5 +- 8 files changed, 151 insertions(+), 78 deletions(-) diff --git a/crates/registry-render/src/audit.rs b/crates/registry-render/src/audit.rs index 7e594e002..c7b7b6d05 100644 --- a/crates/registry-render/src/audit.rs +++ b/crates/registry-render/src/audit.rs @@ -3,14 +3,18 @@ //! and a `response` entry carrying the outcome before the document leaves; //! both fail closed. A refusal decided before any render (validation, 401, //! 413) is one `response` entry, so the log distinguishes "no attempt" from -//! a refused one. The log carries no hash chain or signature. +//! a refused one. A render whose call ends before its outcome is written, a +//! caller that disconnects included, writes an `unfinished` response, so no +//! request entry stays unpaired. The pairing correlation is drawn by the +//! server for every call; the caller's `Idempotency-Key` is only echoed in +//! the record as `correlationId`. The log carries no hash chain or signature. //! //! Events carry no data values and no asset bytes: identifiers, versions, //! hashes, outcomes, caller, trace and correlation ids only. use serde::Serialize; -use registry_platform_audit::{AuditDestination, AuditEntry, AuditWriter}; +use registry_platform_audit::{AuditDestination, AuditEntry, AuditRequest, AuditWriter}; use crate::problem::{ProblemKind, RenderProblem}; @@ -31,8 +35,9 @@ pub struct RenderAuditEvent { pub document_version: u32, pub bundle_version: u32, pub bundle_hash: String, - /// "rendered" or "refused"; absent on the request entry written before - /// the render starts. + /// "rendered", "refused", or "unfinished" for a call that ended before + /// its outcome; absent on the request entry written before the render + /// starts. #[serde(skip_serializing_if = "Option::is_none")] pub outcome: Option<&'static str>, /// Problem slug for refusals. @@ -116,15 +121,11 @@ impl RenderAuditEvent { } } -/// The envelope correlation for one request: the caller's idempotency key -/// when it is a usable correlation, otherwise a fresh random id. The key is -/// already bounded to 128 characters; one carrying a control character is -/// not a valid correlation, so it gets a drawn id instead. -pub fn correlation(correlation_id: Option<&str>) -> String { - match correlation_id { - Some(id) if !id.is_empty() && !id.chars().any(char::is_control) => id.to_owned(), - _ => uuid::Uuid::new_v4().to_string(), - } +/// The envelope correlation for one call: a fresh random id the server +/// draws. The caller's `Idempotency-Key` is not unique to one call, so it is +/// never the value that pairs a call's entries. +pub fn correlation() -> String { + uuid::Uuid::new_v4().to_string() } impl RenderAudit { @@ -144,16 +145,29 @@ impl RenderAudit { Self { writer } } - /// Append the request entry. It must be accepted before the render - /// starts. + /// Append the request entry and return the handle that owes its + /// response. It must be accepted before the render starts. A call that + /// ends before it responds writes `unfinished` as the response. pub async fn request( &self, correlation: &str, event: RenderAuditEvent, - ) -> Result<(), RenderProblem> { + ) -> Result { + let mut unfinished = record(RenderAuditEvent { + outcome: Some("unfinished"), + ..event.clone() + })?; + if let Some(fields) = unfinished.as_object_mut() { + fields.remove("pdfSha256"); + fields.remove("dataSha256"); + } let record = record(event)?; - self.append(AuditEntry::request(AUDIT_SCHEMA, correlation, record)) + self.writer + .begin(AUDIT_SCHEMA, correlation, record, unfinished) .await + .map_err(|err| { + RenderProblem::new(ProblemKind::AuditFailed, format!("audit append: {err}")) + }) } /// Append the response entry. It must be accepted before the caller diff --git a/crates/registry-render/src/server.rs b/crates/registry-render/src/server.rs index 51d0d36d8..fd7b8380a 100644 --- a/crates/registry-render/src/server.rs +++ b/crates/registry-render/src/server.rs @@ -265,7 +265,7 @@ async fn require_bearer( correlation.as_deref(), trace_id(request.headers()).as_deref(), ); - let envelope = crate::audit::correlation(correlation.as_deref()); + let envelope = crate::audit::correlation(); if let Err(audit_problem) = service.audit.response(&envelope, event).await { tracing::error!(problem = %audit_problem, "401 audit append failed"); } @@ -322,7 +322,7 @@ async fn refuse_oversized_bodies( correlation.as_deref(), trace_id(request.headers()).as_deref(), ); - let envelope = crate::audit::correlation(correlation.as_deref()); + let envelope = crate::audit::correlation(); if let Err(audit_problem) = service.audit.response(&envelope, event).await { tracing::error!(problem = %audit_problem, "413 audit append failed"); } @@ -582,8 +582,9 @@ async fn render_route( let document_type = sanitize_document_type(&document_type); let trace = trace_id(&headers); let correlation = correlation_id(&headers); - // One envelope correlation joins this call's request and response entries. - let envelope = crate::audit::correlation(correlation.as_deref()); + // One server-drawn envelope correlation joins this call's request and + // response entries; the caller's key is only echoed in the record. + let envelope = crate::audit::correlation(); let caller = service.caller_fingerprint.clone(); if method != Method::POST { return refuse( @@ -641,28 +642,37 @@ async fn render_route( trace.as_deref(), ); // The request entry is accepted before the render starts; a refused one - // means the render never starts and no response entry follows. - if let Err(audit_problem) = service.audit.request(&envelope, started.clone()).await { - return problem_response(&audit_problem); - } - let outcome = run_render(&service, worker_request).await; - let event = match &outcome { - Ok(rendered) => RenderAuditEvent { - document_id: document_type.clone(), - document_version: rendered.document_version, - bundle_version: rendered.bundle_version, - bundle_hash: rendered.bundle_hash.clone(), - outcome: Some("rendered"), - problem: None, - pdf_sha256: Some(rendered.pdf_sha256.clone()), - data_sha256: Some(rendered.data_sha256.clone()), - caller, - correlation_id: correlation.clone(), - trace_id: trace.clone(), - renderer_version: crate::display_version(), - typst_pin: crate::TYPST_PIN.to_owned(), - }, - Err(problem) => started.refused_after_start(problem), + // means the render never starts and no response entry follows. A call + // dropped from here on writes its unfinished response. + let _request = match service.audit.request(&envelope, started.clone()).await { + Ok(request) => request, + Err(audit_problem) => return problem_response(&audit_problem), + }; + // The answer is built before its response entry, so the entry records + // the outcome the caller actually receives. + let outcome = run_render(&service, worker_request) + .await + .and_then(|rendered| { + let event = RenderAuditEvent { + document_id: document_type.clone(), + document_version: rendered.document_version, + bundle_version: rendered.bundle_version, + bundle_hash: rendered.bundle_hash.clone(), + outcome: Some("rendered"), + problem: None, + pdf_sha256: Some(rendered.pdf_sha256.clone()), + data_sha256: Some(rendered.data_sha256.clone()), + caller, + correlation_id: correlation.clone(), + trace_id: trace.clone(), + renderer_version: crate::display_version(), + typst_pin: crate::TYPST_PIN.to_owned(), + }; + Ok((event, respond_rendered(&headers, rendered)?)) + }); + let (event, outcome) = match outcome { + Ok((event, response)) => (event, Ok(response)), + Err(problem) => (started.refused_after_start(&problem), Err(problem)), }; // The response entry is accepted before anything leaves: audit failure // fails closed and withholds the document. @@ -670,20 +680,17 @@ async fn render_route( return problem_response(&audit_problem); } match outcome { - Ok(rendered) => match respond_rendered(&headers, rendered) { - Ok(mut response) => { - if let Some(correlation) = correlation - .as_ref() - .and_then(|c| header::HeaderValue::from_str(c).ok()) - { - response - .headers_mut() - .insert("idempotency-key", correlation); - } + Ok(mut response) => { + if let Some(correlation) = correlation + .as_ref() + .and_then(|c| header::HeaderValue::from_str(c).ok()) + { response + .headers_mut() + .insert("idempotency-key", correlation); } - Err(problem) => problem_response(&problem), - }, + response + } Err(problem) => problem_response(&problem), } } @@ -1028,11 +1035,18 @@ mod tests { "with no render slot free the request must be waiting at the render step" ); let accepted = lines.accepted(); - assert_eq!(accepted.len(), 1, "only the request entry: {accepted:?}"); + // The call dropped while it waited for a render slot, as a canceled + // request is: its request entry is paired with an unfinished + // response, and nothing else was written. + assert_eq!(accepted.len(), 2, "{accepted:?}"); let entry = &accepted[0]; assert_eq!(entry["schema"], crate::audit::AUDIT_SCHEMA); assert_eq!(entry["phase"], "request"); - assert_eq!(entry["correlation"], "effect-1234"); + let correlation = entry["correlation"].as_str().expect("correlation"); + assert_eq!(correlation.len(), 36, "a drawn correlation: {correlation}"); + assert_eq!(accepted[1]["phase"], "response"); + assert_eq!(accepted[1]["correlation"], correlation); + assert_eq!(accepted[1]["record"]["outcome"], "unfinished"); let record = &entry["record"]; assert!( record.get("outcome").is_none(), @@ -1095,7 +1109,8 @@ mod tests { let accepted = lines.accepted(); assert_eq!(accepted.len(), 1, "{accepted:?}"); assert_eq!(accepted[0]["phase"], "response"); - assert_eq!(accepted[0]["correlation"], "effect-5678"); + assert_ne!(accepted[0]["correlation"], "effect-5678"); + assert_eq!(accepted[0]["record"]["correlationId"], "effect-5678"); assert_eq!(accepted[0]["record"]["outcome"], "refused"); assert_eq!(accepted[0]["record"]["problem"], "issued-at-missing"); } @@ -1115,8 +1130,9 @@ mod tests { assert_eq!(accepted.len(), 2, "{accepted:?}"); let entry = &accepted[1]; assert_eq!(entry["phase"], "response"); - assert_eq!(entry["correlation"], "effect-9012"); + assert_eq!(entry["correlation"], accepted[0]["correlation"]); let record = &entry["record"]; + assert_eq!(record["correlationId"], "effect-9012"); assert_eq!(record["outcome"], "refused"); assert_eq!(record["problem"], "internal"); assert_eq!(record["documentId"], "receipt"); @@ -1129,16 +1145,49 @@ mod tests { } #[test] - fn the_envelope_correlation_reuses_the_idempotency_key_or_draws_one() { + fn the_envelope_correlation_is_always_drawn_by_the_server() { + let drawn = crate::audit::correlation(); + assert_eq!(drawn.len(), 36, "a random UUID: {drawn}"); + assert_ne!(drawn, crate::audit::correlation()); + } + + /// Two calls that carry the same caller key still pair unambiguously: + /// the key is the caller's own reference, never the journal's join. + #[tokio::test] + async fn concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries() { + let lines = AuditLines::new(None); + let service = service(&lines, 2); + let (first, second) = tokio::join!( + router(Arc::clone(&service)).oneshot(render_request(Some("shared-key"))), + router(Arc::clone(&service)).oneshot(render_request(Some("shared-key"))), + ); + // Whatever each call's outcome, both reach the render step. assert_eq!( - crate::audit::correlation(Some("effect-1234")), - "effect-1234" + first.expect("router").status(), + second.expect("router").status() ); - let drawn = crate::audit::correlation(None); - assert_eq!(drawn.len(), 36, "a random UUID: {drawn}"); - assert_ne!(drawn, crate::audit::correlation(None)); - // A key the envelope cannot carry verbatim gets a drawn value too; - // the record keeps the caller's key as before. - assert_eq!(crate::audit::correlation(Some("tab\there")).len(), 36); + let accepted = lines.accepted(); + assert_eq!(accepted.len(), 4, "{accepted:?}"); + let mut correlations = std::collections::BTreeMap::>::new(); + for entry in &accepted { + assert_eq!(entry["record"]["correlationId"], "shared-key"); + correlations + .entry( + entry["correlation"] + .as_str() + .expect("correlation") + .to_owned(), + ) + .or_default() + .push(entry["phase"].as_str().expect("phase").to_owned()); + } + assert_eq!( + correlations.len(), + 2, + "one correlation per call: {correlations:?}" + ); + for phases in correlations.values() { + assert_eq!(phases, &["request", "response"]); + } } } diff --git a/crates/registry-render/tests/serve.rs b/crates/registry-render/tests/serve.rs index 84b5d4af3..5ab10bf3c 100644 --- a/crates/registry-render/tests/serve.rs +++ b/crates/registry-render/tests/serve.rs @@ -747,7 +747,10 @@ fn a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation() { ); } } - assert_eq!(entries[0]["correlation"], "effect-1234"); + // The pairing correlation is always drawn by the server; the caller's + // key stays in the record for the caller's own cross-referencing. + let drawn = entries[0]["correlation"].as_str().expect("correlation"); + assert_eq!(drawn.len(), 36, "a drawn random id: {drawn}"); assert_eq!(entries[0]["record"]["correlationId"], "effect-1234"); let drawn = entries[2]["correlation"].as_str().expect("correlation"); assert_eq!(drawn.len(), 36, "a drawn random id: {drawn}"); diff --git a/docs/site/src/content/docs/operate/registry-render.mdx b/docs/site/src/content/docs/operate/registry-render.mdx index b25431017..f01db47cc 100644 --- a/docs/site/src/content/docs/operate/registry-render.mdx +++ b/docs/site/src/content/docs/operate/registry-render.mdx @@ -172,8 +172,10 @@ Each line is one entry: `schema` (`render.registrystack.org/audit/v1`), `eventId `phase` (`request` or `response`), `correlation`, and the value-free `record`: hashes, versions, the caller's key fingerprint, the outcome, and the trace and correlation ids, never data values or image bytes. The `request` entry carries no outcome. Both entries of one render share their -`correlation`: the caller's `Idempotency-Key`, or a random identifier when there is none. For -example, to list outcomes by correlation: +`correlation`, a random identifier the server draws for every call; the caller's `Idempotency-Key` +is recorded only as the record's `correlationId`, because two calls may carry the same key. A call +that ends before its outcome is written, such as one whose caller disconnected, has the outcome +`unfinished`. For example, to list outcomes by correlation: ```sh jq -c '{correlation, phase, outcome: .record.outcome}' /var/lib/registry-render/audit/render.jsonl diff --git a/products/render/ACCEPTANCE.md b/products/render/ACCEPTANCE.md index 20acde544..8cb9e1733 100644 --- a/products/render/ACCEPTANCE.md +++ b/products/render/ACCEPTANCE.md @@ -18,7 +18,7 @@ hardening (2026-09-19); see EVIDENCE.md for the change list. | Resource enforcement (kill at timeout, recycle, recover) | `serve.rs`: `pathological_renders_are_bounded_and_the_service_recovers` (504 problem, service healthy afterwards, both kills audited); the CLI shares the wall: `scaffold::compile_is_bounded_by_a_timeout` | | Auth (constant-time, indistinguishable 401s, audited, key-file normalization) | `serve.rs`: `unauthorized_requests_are_refused_and_audited`, `api_key_file_with_one_trailing_newline_is_trimmed`, `api_key_with_stray_whitespace_is_refused_at_startup` | | HTTP contract (headers, JSON variant, correlation, problems, 413, versions in /health) | `serve.rs`: `render_returns_pdf_with_hash_headers_matching_golden`, `json_variant_serves_openfn_clients`, `missing_issued_at_and_bad_data_are_named_problems`, `oversized_bodies_are_refused_after_auth_as_problems`, `serve_health_and_ready` | -| Audit (request entry before the render, response entry before the response, failures audited, value-free) | `serve.rs`: `a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation`, `a_refused_request_entry_prevents_the_render`, `a_refused_response_entry_withholds_the_pdf`, `audit_events_are_value_free` (canary + API-key scans), `unauthorized_requests_are_refused_and_audited`, `pathological_renders_are_bounded_and_the_service_recovers`; unit: `server::tests::the_request_entry_is_accepted_before_the_render_starts`, `server::tests::a_refused_request_entry_prevents_the_render`, `server::tests::a_refusal_before_the_render_is_one_response_entry`, `server::tests::a_refusal_after_the_render_starts_keeps_the_render_identity`, `runtime::tests::the_audit_block_takes_the_shared_destination_shape`, `runtime::tests::an_audit_block_outside_the_shared_shape_is_refused`; `scaffold.rs`: `check_proves_the_audit_file_resolves_under_the_root`, `check_refuses_to_prove_a_stdout_audit_destination` | +| Audit (request entry before the render, response entry before the response, failures audited, value-free) | `serve.rs`: `a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation`, `a_refused_request_entry_prevents_the_render`, `a_refused_response_entry_withholds_the_pdf`, `audit_events_are_value_free` (canary + API-key scans), `unauthorized_requests_are_refused_and_audited`, `pathological_renders_are_bounded_and_the_service_recovers`; unit: `server::tests::the_request_entry_is_accepted_before_the_render_starts`, `server::tests::a_refused_request_entry_prevents_the_render`, `server::tests::a_refusal_before_the_render_is_one_response_entry`, `server::tests::a_refusal_after_the_render_starts_keeps_the_render_identity`, `server::tests::concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries`, `runtime::tests::the_audit_block_takes_the_shared_destination_shape`, `runtime::tests::an_audit_block_outside_the_shared_shape_is_refused`; `scaffold.rs`: `check_proves_the_audit_file_resolves_under_the_root`, `check_refuses_to_prove_a_stdout_audit_destination` | | Sealed-bundle-only serving (startup and per render) | `serve.rs`: `tampered_bundle_refuses_to_serve` (exit code = BundleTampered), `bundle_drift_after_serve_starts_is_refused_per_render` (tamper and unseal after startup) | | Cross-machine byte stability; closure drift | `.github/workflows/render-golden.yml` runs the golden, serve, and scaffold suites on ubuntu-24.04 and macos-14 against the same `golden.json`; `golden_hashes_match` also pins each bundle's file closure and requires every non-virtual dep to be manifest-governed | | DX (scaffold compiles offline, validate dry-run, errors) | Manual smoke 2026-09-17 recorded in EVIDENCE-style: `registry-render init` + first compile offline, warning-clean after the font fix; `registry-render validate` refuses bad data with exit 9 and pointers | diff --git a/products/render/README.md b/products/render/README.md index 26bbbaa0c..5f852679e 100644 --- a/products/render/README.md +++ b/products/render/README.md @@ -99,9 +99,13 @@ exits 101): Stack audit writer. A render writes a `request` entry before the worker starts and a `response` entry with the outcome before the document leaves, both failing closed; a refusal before any render is one `response` entry. - Each line is the envelope `{schema, eventId, time, phase, correlation, - record}` with schema `render.registrystack.org/audit/v1`; `correlation` is - the caller's `Idempotency-Key`, or a random id when there is none. The + A call that ends before its outcome is written, such as a disconnected + caller, writes an `unfinished` response, so every `request` entry is + paired. Each line is the envelope `{schema, eventId, time, phase, + correlation, record}` with schema `render.registrystack.org/audit/v1`; + `correlation` is a random id the server draws for every call, and the + caller's `Idempotency-Key` is echoed only as the record's `correlationId`, + since two calls may carry the same key. The runtime's `audit` block names a `file` (the default, with `path` and optional `rotateBytes` and `retainDays`) or `stdout` destination. The log carries no hash chain or signature, so it is not tamper-evident on the diff --git a/products/render/SECURITY-MATRIX.md b/products/render/SECURITY-MATRIX.md index 51bdb5754..4db6c2cbc 100644 --- a/products/render/SECURITY-MATRIX.md +++ b/products/render/SECURITY-MATRIX.md @@ -15,7 +15,7 @@ fixed or explicitly documented as residuals below), extended by the PR | 5 | Caller without the API key reads documents or renders | Bearer authentication as a layer on `/v1/*` **before body buffering**; constant-time compare; ≥32-byte ASCII key; identical 401 bodies for missing/wrong/short keys; the key file gets exactly one trailing line ending trimmed and any other whitespace refuses startup (no silently mis-armed key with `/health` green) | `serve::unauthorized_requests_are_refused_and_audited` (two indistinguishable 401s); `serve::api_key_file_with_one_trailing_newline_is_trimmed`; `serve::api_key_with_stray_whitespace_is_refused_at_startup`; `/v1/documents` behind the same layer | | 6 | The append-only audit log is used as an unauthenticated write oracle (huge or hostile route params) | Route parameters are bounded (64 chars) and kebab-validated (`sanitize_document_type`) before any audit append, including pre-auth 401 events | code-reviewed; audit shape asserted in `serve::unauthorized_requests_are_refused_and_audited` | | 7 | Request data values or the API key leak into the audit log or logs | Audit events carry a closed, value-free field set; lifecycle logs use fixed dimensions (method class, route template, status, latency, trace id) | `serve::audit_events_are_value_free` (canary + key scans over the audit file) | -| 8 | A render starts, or a document leaves, with no accepted audit record of it | The shared audit writer accepts a `file` entry only after `fsync` (owner-only file, single-writer lock) and a `stdout` entry only after it is written and flushed, then stops accepting after any failure. The `request` entry is accepted before the worker starts, else `503 audit-failed` and no render; the `response` entry is accepted before the document leaves, else `503 audit-failed` and no PDF or hash headers; `/ready` reports a stopped writer. Host-level tampering with the log is not defended here: the log carries no hash chain or signature (residual below) | `server::tests::the_request_entry_is_accepted_before_the_render_starts`; `server::tests::a_refused_request_entry_prevents_the_render`; `serve::a_refused_request_entry_prevents_the_render`; `serve::a_refused_response_entry_withholds_the_pdf`; `serve::a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation` (file mode 0600) | +| 8 | A render starts, or a document leaves, with no accepted audit record of it | The shared audit writer accepts a `file` entry only after `fsync` (owner-only file, single-writer lock) and a `stdout` entry only after it is written and flushed, then stops accepting after any failure. The `request` entry is accepted before the worker starts, else `503 audit-failed` and no render; the `response` entry is accepted before the document leaves, else `503 audit-failed` and no PDF or hash headers; `/ready` reports a stopped writer. Each call pairs its entries under a correlation the server draws, never the caller's `Idempotency-Key` (echoed only as `correlationId`), so calls sharing a key cannot be confused, and a call that ends before its outcome is written writes an `unfinished` response. Host-level tampering with the log is not defended here: the log carries no hash chain or signature (residual below) | `server::tests::the_request_entry_is_accepted_before_the_render_starts`; `server::tests::a_refused_request_entry_prevents_the_render`; `serve::a_refused_request_entry_prevents_the_render`; `serve::a_refused_response_entry_withholds_the_pdf`; `serve::a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation` (file mode 0600); `server::tests::concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries` | | 9 | A render succeeds but is wrong on paper (missing glyphs) | Check-time label-script coverage over the actual label characters; render-time warnings; `--strict` fails on warnings | `scaffold::check_seal_refuses_to_seal_a_broken_bundle` (CJK label, no font → refusal, and no seal written) | | 10 | Pathological template data exhausts CPU or memory | Serves render in a supervised worker process: killed at the configured timeout, `RLIMIT_AS` on Linux, recycled on panic; bounded concurrency; request body and output caps with hard ceilings | `serve::pathological_renders_are_bounded_and_the_service_recovers` (timeout or memory wall is audited; a fresh worker immediately serves a healthy render) | | 11 | Assets smuggle executables or balloons | base64-decoded, JPEG/PNG magic sniffed, per-asset 2 MiB / per-request 8 MiB caps, exact bytes covered by `dataSha256` | `golden::wrong_media_type_and_oversize_assets_are_refused` | diff --git a/products/render/integrations/openfn/JOURNEY.md b/products/render/integrations/openfn/JOURNEY.md index 7e709ed4c..4347fa587 100644 --- a/products/render/integrations/openfn/JOURNEY.md +++ b/products/render/integrations/openfn/JOURNEY.md @@ -94,8 +94,9 @@ is the profile's, not the caller's. ## Audit ledger The walk predates the shared audit writer. Render now writes a `request` -and a `response` entry per render, joined by `correlation` (the job's -`eventEffectId` here), and has no `audit-verify` command or audit key; a +and a `response` entry per render, joined by a `correlation` the server +draws for each call, with the job's `eventEffectId` recorded as +`correlationId`, and has no `audit-verify` command or audit key; a re-walk reads the audit file directly. What the walk recorded: `registry-render audit-verify` (after graceful shutdown; a live writer From b85ceafdb333ec7abb7d69922e4074a3d8a05e18 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:00:38 +0000 Subject: [PATCH 06/32] fix(breg): answer every audited request entry Every attempt is held through the shared AuditRequest handle until its terminal or refusal entry answers it; a request that ends first, through an error, a timeout, or a dropped caller, writes an unfinished response under the same correlation. Reads that fail after their rows were read write the Refused terminal and log a terminal the destination refuses. Plain record_pre_io_audit now accepts only refusals, so an attempt can only be written through the handle. - A reviewed apply records its attempt before the receipt preflight's reads and the review authority, and holds it across the action. - Migration reconciliation answers a transition that fails after its request entry with a failed response. - Request detail erasure answers a refused or failed erasure, deletes external attachment objects before its response and records how many remain, and reports an erasure that committed without its response as request_retention.erasure.unaudited. Attachment cleanup is held too. - Evidence retention erasure is audited under breg-evidence-retention-audit/v1 with its cutoff and erased count. - An ingestion call refused after its ingestion request entry is answered in the ingestion schema and not recorded again as a general refusal. Refs #1592 #1587 #1597 #1590 #1593 #1507 Signed-off-by: Jeremi Joslin --- .../src/action_evidence_maintenance.rs | 72 ++++ crates/registry-breg/src/api/ingestion.rs | 42 ++- crates/registry-breg/src/audit.rs | 165 ++++++++- crates/registry-breg/src/ingestion_store.rs | 60 +++- .../registry-breg/src/migration_reconcile.rs | 96 ++++-- crates/registry-breg/src/mutation.rs | 58 +++- crates/registry-breg/src/mutation/action.rs | 66 ++-- crates/registry-breg/src/mutation/request.rs | 88 +++-- .../src/postgres/history_read.rs | 53 ++- crates/registry-breg/src/postgres/mutation.rs | 326 +++++++++++++----- crates/registry-breg/src/postgres/read.rs | 168 +++++---- .../src/postgres/revision_read.rs | 46 ++- crates/registry-breg/src/request_retention.rs | 180 +++++++--- .../postgres_action_evidence_retention.rs | 33 ++ .../tests/postgres_change_requests.rs | 23 ++ .../tests/postgres_ingestion_runs.rs | 102 ++++++ .../registry-breg/tests/postgres_migration.rs | 50 ++- crates/registry-breg/tests/postgres_read.rs | 37 +- .../tests/postgres_request_read_retention.rs | 96 ++++++ .../postgres_request_upgrade_retention.rs | 18 +- crates/registry-bregctl/src/lib.rs | 4 + .../registry-bregctl/src/request_retention.rs | 11 + 22 files changed, 1412 insertions(+), 382 deletions(-) diff --git a/crates/registry-breg/src/action_evidence_maintenance.rs b/crates/registry-breg/src/action_evidence_maintenance.rs index 31d85286a..718a54404 100644 --- a/crates/registry-breg/src/action_evidence_maintenance.rs +++ b/crates/registry-breg/src/action_evidence_maintenance.rs @@ -3,6 +3,11 @@ use std::{path::Path, time::Duration}; +use registry_platform_audit::AuditEntry; +use serde_json::{json, Value}; +use uuid::Uuid; + +use crate::audit::RegistryAudit; use crate::mutation::{erase_expired_action_evidence, MutationError}; use crate::postgres::{ verify_catalog_identity_for_catalog, verify_migration_role, ConnectionConfig, @@ -10,6 +15,10 @@ use crate::postgres::{ }; use crate::runtime_config::load_runtime_config; +/// Schema of the request and response entries of one expired-Evidence +/// erasure. +pub const EVIDENCE_RETENTION_AUDIT_SCHEMA: &str = "breg-evidence-retention-audit/v1"; + /// Package-bound authority for erasing expired protected Evidence material. /// The migration identity, actual target catalog and registry interlock are /// checked again in the same transaction that deletes the retained material. @@ -22,6 +31,7 @@ pub struct ActionEvidenceRetentionOperatorService { runtime_role: SqlIdentifier, lock_timeout: Duration, statement_timeout: Duration, + audit: RegistryAudit, } impl ActionEvidenceRetentionOperatorService { @@ -56,6 +66,9 @@ impl ActionEvidenceRetentionOperatorService { runtime_role: config.database().roles().runtime().clone(), lock_timeout: config.operational_timeouts().migration_lock, statement_timeout: config.operational_timeouts().migration_statement, + audit: RegistryAudit::open_companion(&config) + .await + .map_err(|_| MutationError::Unavailable)?, }) } @@ -68,6 +81,7 @@ impl ActionEvidenceRetentionOperatorService { migration_connection: ConnectionConfig, migration_role: SqlIdentifier, runtime_role: SqlIdentifier, + audit: RegistryAudit, ) -> Self { Self { expected, @@ -78,9 +92,16 @@ impl ActionEvidenceRetentionOperatorService { runtime_role, lock_timeout: Duration::from_secs(5), statement_timeout: Duration::from_secs(10), + audit, } } + /// Erase the retained Evidence material whose expiry is before `before`. + /// + /// The request entry, naming the threshold, is accepted before the + /// erasure transaction opens, so an audit outage erases nothing. Its + /// response records the erased count once the transaction commits, or + /// the failure when it does not. pub async fn erase_expired( &self, before: chrono::DateTime, @@ -88,6 +109,57 @@ impl ActionEvidenceRetentionOperatorService { if before > chrono::Utc::now() { return Err(MutationError::InvalidRequest); } + let correlation = Uuid::new_v4().to_string(); + let record = |phase: &str, outcome: &str| -> Value { + json!({ + "kind": "evidenceRetention", + "phase": phase, + "outcome": outcome, + "packageRevision": self.expected.package_revision, + "actor": "breg:evidence-retention-operator", + "before": before.to_rfc3339_opts(chrono::SecondsFormat::Secs, true), + "correlation": correlation, + }) + }; + let mut attempt = self + .audit + .begin( + AuditEntry::request( + EVIDENCE_RETENTION_AUDIT_SCHEMA, + correlation.clone(), + record("attempt", "started"), + ), + record("terminal", "unfinished"), + ) + .await + .map_err(|_| MutationError::Unavailable)?; + match self.erase_in_transaction(before).await { + Ok(erased) => { + let mut response = record("terminal", "erased"); + response["erased"] = json!(erased); + // The erasure committed; a refused entry reports the command + // unavailable, and the writer then refuses every later entry. + attempt + .respond(response) + .await + .map_err(|_| MutationError::Unavailable)?; + Ok(erased) + } + Err(error) => { + if attempt.respond(record("terminal", "failed")).await.is_err() { + tracing::error!( + "the failed Evidence retention's response audit entry was not recorded" + ); + } + Err(error) + } + } + } + + async fn erase_in_transaction( + &self, + before: chrono::DateTime, + ) -> Result { let pool = self .migration_connection .build_pool() diff --git a/crates/registry-breg/src/api/ingestion.rs b/crates/registry-breg/src/api/ingestion.rs index 42e788a65..46675d0c2 100644 --- a/crates/registry-breg/src/api/ingestion.rs +++ b/crates/registry-breg/src/api/ingestion.rs @@ -177,15 +177,17 @@ async fn create_run( .await { Ok(run) => ingestion_response(StatusCode::CREATED, json!({ "run": run })), - // A refusal after the request entry was accepted owes the journal - // its response entry, as a refused chunk submission does. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -395,15 +397,17 @@ async fn cancel_run( .await { Ok(run) => ingestion_response(StatusCode::OK, json!({ "run": run })), - // A refusal after the request entry was accepted owes the journal - // its response entry, as a refused chunk submission does. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -545,16 +549,17 @@ async fn submit_chunk( .await { Ok(answer) => ingestion_response(StatusCode::OK, answer), - // A submission the run refuses after parsing owes the journal the - // same durable refusal envelope pre-parse failures write, so an audit - // outage gates the refusal instead of passing silently. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -631,16 +636,17 @@ async fn chunk_receipt( .await { Ok(receipt) => ingestion_response(StatusCode::OK, receipt), - // A recovery the run refuses after its request entry was accepted - // owes the journal the same durable refusal envelope a refused - // cancellation or chunk submission owes. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await diff --git a/crates/registry-breg/src/audit.rs b/crates/registry-breg/src/audit.rs index 4cad4e569..78c073290 100644 --- a/crates/registry-breg/src/audit.rs +++ b/crates/registry-breg/src/audit.rs @@ -6,10 +6,15 @@ //! `response` entry with its outcome, both through the one platform //! [`AuditWriter`] the process opened at startup and both correlated by the //! request id Base Registry Engine minted. A refusal is one `response` entry. +//! A request that goes on to protected I/O holds its attempt as an +//! [`AuditRequest`] until its terminal or refusal entry answers it; one that +//! ends first writes an `unfinished` response, so no attempt stays unpaired. //! Entries carry keyed references and closed-vocabulary terms, never a raw //! principal, record id, selector, token, or free text. -use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile, AuditWriter}; +use registry_platform_audit::{ + AuditEntry, AuditKeyHasher, AuditProfile, AuditRequest, AuditWriter, +}; use serde_json::{json, Value}; use uuid::Uuid; @@ -82,6 +87,28 @@ impl RegistryAudit { Ok(Self::new(profile, writer)) } + /// Append a `request` entry and return the handle that owes its + /// `response`. Dropped unanswered, the handle writes `unfinished` as the + /// response under the entry's schema and correlation. + pub(crate) async fn begin( + &self, + entry: AuditEntry, + unfinished: Value, + ) -> Result { + if entry.phase() != registry_platform_audit::AuditPhase::Request { + return Err(RegistryAuditError::InvalidContext); + } + self.writer + .begin( + entry.schema(), + entry.correlation(), + entry.record().clone(), + unfinished, + ) + .await + .map_err(|_| RegistryAuditError::Unavailable) + } + /// Append one entry. A refused append is the audit-unavailable refusal: /// the caller performs no protected I/O and releases no disclosure. pub async fn append(&self, entry: AuditEntry) -> Result<(), RegistryAuditError> { @@ -258,6 +285,8 @@ pub(crate) enum WebhookAuditOutcome { PayloadExpired, WorkerInterrupted, ReplayRequested, + ReplayCommitted, + ReplayRefused, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -281,17 +310,50 @@ pub(crate) struct WebhookAudit<'a> { pub disposition: WebhookAuditDisposition, } -/// Append one minimized attempt or refusal before protected record I/O. -/// -/// An attempt is the `request` entry of the request it names and must be -/// accepted before any protected read or write starts. A refusal is the single -/// `response` entry of a request that performs no protected I/O. +/// Append one minimized refusal before protected record I/O: the single +/// `response` entry of a request that performs no protected I/O, or the one +/// that answers the attempt a request holds. An attempt goes through +/// [`begin_pre_io_audit`], which owes its response, so this refuses one. pub async fn record_pre_io_audit( audit: &RegistryAudit, expected: &ExpectedRegistryIdentity, claims: &ClaimContext, event: PreIoAudit<'_>, ) -> Result<(), RegistryAuditError> { + if event.kind != PreIoAuditKind::Refusal { + return Err(RegistryAuditError::InvalidContext); + } + let record = pre_io_record(audit, expected, claims, &event)?; + audit + .append(pre_io_entry(event.kind, event.correlation, record)) + .await +} + +/// Append the attempt `request` entry of a request that goes on to protected +/// I/O, and return the handle that owes its `response`. The terminal or +/// refusal entry the request writes under its request id answers it; a +/// request that returns, fails, or is canceled first writes an `unfinished` +/// response naming only the operation and the request when the handle is +/// dropped, so no attempt is left unpaired. +pub(crate) async fn begin_pre_io_audit( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ClaimContext, + event: PreIoAudit<'_>, +) -> Result { + if event.kind != PreIoAuditKind::Attempt { + return Err(RegistryAuditError::InvalidContext); + } + let record = pre_io_record(audit, expected, claims, &event)?; + begin_attempt(audit, event.correlation, record).await +} + +fn pre_io_record( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ClaimContext, + event: &PreIoAudit<'_>, +) -> Result { let profile = audit.profile(); if event.operation_id.is_empty() || !profile_is_keyed(profile) { return Err(RegistryAuditError::InvalidContext); @@ -341,17 +403,45 @@ pub async fn record_pre_io_audit( } } insert_refusal_reason(&mut record, event.refusal_reason); + Ok(record) +} + +/// [`record_pre_io_audit`] for a governed action request. +pub(crate) async fn record_action_pre_io_audit( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ActionClaimContext, + event: PreIoAudit<'_>, +) -> Result<(), RegistryAuditError> { + if event.kind != PreIoAuditKind::Refusal { + return Err(RegistryAuditError::InvalidContext); + } + let record = action_pre_io_record(audit, expected, claims, &event)?; audit .append(pre_io_entry(event.kind, event.correlation, record)) .await } -pub(crate) async fn record_action_pre_io_audit( +/// [`begin_pre_io_audit`] for a governed action request. +pub(crate) async fn begin_action_pre_io_audit( audit: &RegistryAudit, expected: &ExpectedRegistryIdentity, claims: &ActionClaimContext, event: PreIoAudit<'_>, -) -> Result<(), RegistryAuditError> { +) -> Result { + if event.kind != PreIoAuditKind::Attempt { + return Err(RegistryAuditError::InvalidContext); + } + let record = action_pre_io_record(audit, expected, claims, &event)?; + begin_attempt(audit, event.correlation, record).await +} + +fn action_pre_io_record( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ActionClaimContext, + event: &PreIoAudit<'_>, +) -> Result { let profile = audit.profile(); if event.operation_id.is_empty() || event.target_record.is_some() @@ -384,9 +474,47 @@ pub(crate) async fn record_action_pre_io_audit( "actionId": claims.action_id(), }); insert_refusal_reason(&mut record, event.refusal_reason); + Ok(record) +} + +/// Append `record` as the attempt `request` entry of `correlation`. +async fn begin_attempt( + audit: &RegistryAudit, + correlation: &RequestCorrelation, + record: Value, +) -> Result { + let unfinished = unfinished_record(&record); audit - .append(pre_io_entry(event.kind, event.correlation, record)) + .writer() + .begin( + AUDIT_SCHEMA, + correlation.request_id().to_string(), + record, + unfinished, + ) .await + .map_err(|_| RegistryAuditError::Unavailable) +} + +/// The `response` an attempt writes when its request ends without a +/// terminal or refusal entry: the operation and request it answers, and +/// nothing the request read or was about to write. +fn unfinished_record(attempt: &Value) -> Value { + let mut record = serde_json::Map::new(); + record.insert("phase".to_owned(), json!("unfinished")); + for field in [ + "method", + "operationId", + "requestId", + "traceId", + "packageRevision", + "actionId", + ] { + if let Some(value) = attempt.get(field) { + record.insert(field.to_owned(), value.clone()); + } + } + Value::Object(record) } fn pre_io_phase_name(kind: PreIoAuditKind) -> &'static str { @@ -680,8 +808,13 @@ pub(crate) fn webhook_entry( ) => event.attempt >= 0, ( WebhookAuditPhase::Replay, - WebhookAuditOutcome::ReplayRequested, + WebhookAuditOutcome::ReplayRequested | WebhookAuditOutcome::ReplayCommitted, WebhookAuditDisposition::ReplayPending, + ) + | ( + WebhookAuditPhase::Replay, + WebhookAuditOutcome::ReplayRefused, + WebhookAuditDisposition::DeadLettered, ) => event.attempt == 0, _ => false, }; @@ -723,11 +856,15 @@ pub(crate) fn webhook_entry( "generation": event.generation, "attempt": event.attempt, }); - Ok(match event.phase { - WebhookAuditPhase::Attempt => { + // An attempt's start and an operator's replay request are requests; + // the terminal disposition and the replay's committed or refused reset + // answer them under the same correlation. + Ok(match (event.phase, event.outcome) { + (WebhookAuditPhase::Attempt, _) + | (WebhookAuditPhase::Replay, WebhookAuditOutcome::ReplayRequested) => { AuditEntry::request(WEBHOOK_AUDIT_SCHEMA, correlation, record) } - WebhookAuditPhase::Terminal | WebhookAuditPhase::Replay => { + (WebhookAuditPhase::Terminal | WebhookAuditPhase::Replay, _) => { AuditEntry::response(WEBHOOK_AUDIT_SCHEMA, correlation, record) } }) @@ -761,6 +898,8 @@ fn webhook_outcome_name(outcome: WebhookAuditOutcome) -> &'static str { WebhookAuditOutcome::PayloadExpired => "payload_expired", WebhookAuditOutcome::WorkerInterrupted => "worker_interrupted", WebhookAuditOutcome::ReplayRequested => "replay_requested", + WebhookAuditOutcome::ReplayCommitted => "replay_committed", + WebhookAuditOutcome::ReplayRefused => "replay_refused", } } diff --git a/crates/registry-breg/src/ingestion_store.rs b/crates/registry-breg/src/ingestion_store.rs index 3a6c8d2da..c56019b00 100644 --- a/crates/registry-breg/src/ingestion_store.rs +++ b/crates/registry-breg/src/ingestion_store.rs @@ -1221,16 +1221,46 @@ pub(crate) struct RunRequest<'a> { pub(crate) correlation: &'a str, } +/// The accepted `request` entry of one run transition, which owes its +/// `response` in the ingestion schema. +pub(crate) struct RunAttempt { + request: registry_platform_audit::AuditRequest, + record: Value, +} + +impl RunAttempt { + /// Whether the transition's `response` entry was accepted. + pub(crate) fn is_answered(&self) -> bool { + self.request.is_answered() + } + + /// Answer the request with the refusal the transition ended in, + /// reporting whether the destination accepted it. + pub(crate) async fn refuse(mut self) -> bool { + let record = outcome_record(&self.record, "refused"); + self.request.respond(record).await.is_ok() + } +} + +fn outcome_record(request: &Value, outcome: &str) -> Value { + let mut record = request.clone(); + record["phase"] = json!("terminal"); + record["outcome"] = json!(outcome); + record +} + /// Append the value-free `request` entry of one run transition, correlated by -/// the request that drives it. Callers append it before the transition's -/// first protected read or write and perform neither unless the append is -/// accepted; the transition's `response` entry, appended after commit, shares -/// the correlation. The entry names only what that response entry already +/// the request that drives it, and return the handle that owes its +/// `response`. Callers append it before the transition's first protected +/// write and perform none unless it is accepted; the transition's `response` +/// entry, appended after commit, shares the correlation and answers it. A +/// transition that ends without one writes the `unfinished` outcome when the +/// handle is dropped. The entry names only what that response entry already /// records. -pub(crate) async fn append_run_request( +pub(crate) async fn begin_run_request( audit: &crate::audit::RegistryAudit, request: RunRequest<'_>, -) -> Result<(), IngestionStoreError> { +) -> Result { if !crate::audit::profile_is_keyed(audit.profile()) { return Err(IngestionStoreError::Unavailable); } @@ -1250,14 +1280,18 @@ pub(crate) async fn append_run_request( if let Some(chunk_index) = request.chunk_index { record["chunkIndex"] = json!(chunk_index); } - audit - .append(AuditEntry::request( - INGESTION_AUDIT_SCHEMA, - request.correlation.to_owned(), - record, - )) + let request = audit + .begin( + AuditEntry::request( + INGESTION_AUDIT_SCHEMA, + request.correlation.to_owned(), + record.clone(), + ), + outcome_record(&record, "unfinished"), + ) .await - .map_err(|_| IngestionStoreError::Unavailable) + .map_err(|_| IngestionStoreError::Unavailable)?; + Ok(RunAttempt { request, record }) } /// The canonical lifecycle record for one run transition. diff --git a/crates/registry-breg/src/migration_reconcile.rs b/crates/registry-breg/src/migration_reconcile.rs index e587f535e..b2dbd6581 100644 --- a/crates/registry-breg/src/migration_reconcile.rs +++ b/crates/registry-breg/src/migration_reconcile.rs @@ -25,8 +25,8 @@ use std::time::Duration; -use registry_platform_audit::AuditEntry; -use serde_json::json; +use registry_platform_audit::{AuditEntry, AuditRequest}; +use serde_json::{json, Value}; use crate::audit::RegistryAudit; use crate::history_maintenance::profile_is_keyed; @@ -344,12 +344,8 @@ async fn reconcile_under_lock( match report.outcome { ReconcileOutcome::Completable => { let entry = audit_entry(request, target, ledger, "completed", &report)?; - append_request( - request.audit, - request_entry(request, target, ledger, "completed")?, - ) - .await?; - connection + let mut attempt = begin_request(request, target, ledger, "completed").await?; + let transition = connection .activate_verified_package( Some(request.current), target, @@ -360,17 +356,17 @@ async fn reconcile_under_lock( runtime_role: request.runtime_role, }, ) - .await?; + .await; + if let Err(error) = transition { + respond_failed(&mut attempt, request, target, ledger, "completed").await; + return Err(error.into()); + } append_after_commit(request.audit, entry).await?; } ReconcileOutcome::Revertible => { let entry = audit_entry(request, target, ledger, "reverted", &report)?; - append_request( - request.audit, - request_entry(request, target, ledger, "reverted")?, - ) - .await?; - connection + let mut attempt = begin_request(request, target, ledger, "reverted").await?; + let transition = connection .revert_failed_package( request.current, &target.package_revision, @@ -381,7 +377,11 @@ async fn reconcile_under_lock( runtime_role: request.runtime_role, }, ) - .await?; + .await; + if let Err(error) = transition { + respond_failed(&mut attempt, request, target, ledger, "reverted").await; + return Err(error.into()); + } append_after_commit(request.audit, entry).await?; } outcome @ (ReconcileOutcome::Ready @@ -427,16 +427,68 @@ fn unresolvable_reason(progress: Option) -> &'static } } -/// Append the reconciliation's request entry before its transition runs. A -/// refused entry reports the reconciliation unavailable and leaves the pinned -/// target exactly as the assessment found it. -async fn append_request(audit: &RegistryAudit, entry: AuditEntry) -> Result<(), ReconcileError> { - audit - .append(entry) +/// Append the reconciliation's request entry before its transition runs and +/// return the handle that owes its response. A refused entry reports the +/// reconciliation unavailable and leaves the pinned target exactly as the +/// assessment found it. A reconciliation that ends without a response +/// writes the `unfinished` outcome when the handle is dropped. +async fn begin_request( + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, +) -> Result { + request + .audit + .begin( + request_entry(request, target, ledger, action)?, + outcome_record(request, target, ledger, action, "unfinished")?, + ) .await .map_err(|_| ReconcileError::Unavailable) } +/// Answer the request entry of a transition that did not commit. The +/// reconciliation already failed, so a refused entry is only logged; the +/// held request then writes its `unfinished` outcome instead. +async fn respond_failed( + attempt: &mut AuditRequest, + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, +) { + let recorded = match outcome_record(request, target, ledger, action, "failed") { + Ok(record) => attempt.respond(record).await.is_ok(), + Err(_) => false, + }; + if !recorded { + tracing::error!("the failed reconciliation's response audit entry was not recorded"); + } +} + +/// The `response` of a transition that did not commit: the request's +/// identities and plan shape with the outcome, and no count or finding. +fn outcome_record( + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, + outcome: &'static str, +) -> Result { + Ok(json!({ + "phase": "terminal", + "outcome": outcome, + "operationId": AUDIT_OPERATION_ID, + "action": action, + "packageRevision": request.current.package_revision, + "targetPackageRevision": target.package_revision, + "packageSequence": target.package_sequence, + "planKind": ledger.plan_kind.as_str(), + "operatorReference": operator_reference(request)?, + })) +} + /// Append the reconciliation's response entry after its transition committed. /// A refused entry reports the reconciliation unavailable even though the /// transition is durable; a rerun then finds the Registry ready. diff --git a/crates/registry-breg/src/mutation.rs b/crates/registry-breg/src/mutation.rs index bf9b9cdc7..2e28e9db4 100644 --- a/crates/registry-breg/src/mutation.rs +++ b/crates/registry-breg/src/mutation.rs @@ -14,7 +14,7 @@ use std::sync::Arc; use std::time::Duration; use deadpool_postgres::Client; -use registry_platform_audit::{AuditEntry, AuditProfile}; +use registry_platform_audit::{AuditEntry, AuditProfile, AuditRequest}; use registry_platform_canonical_json::canonicalize_json; use registry_platform_crypto::field_encryption::{ envelope_member_json, FieldCryptoError, MAX_FIELD_PLAINTEXT_BYTES, @@ -29,9 +29,9 @@ use uuid::Uuid; use crate::artifacts::event_data_schema_binding; use crate::audit::{ - action_terminal_entry, profile_is_keyed, record_action_pre_io_audit, record_pre_io_audit, - terminal_entry, PreIoAudit, PreIoAuditKind, RegistryAudit, RegistryAuditError, TerminalAudit, - TerminalAuditOutcome, + action_terminal_entry, begin_action_pre_io_audit, begin_pre_io_audit, profile_is_keyed, + record_action_pre_io_audit, record_pre_io_audit, terminal_entry, PreIoAudit, PreIoAuditKind, + RegistryAudit, RegistryAuditError, TerminalAudit, TerminalAuditOutcome, }; use crate::compiler::{ WEBHOOK_ATTEMPT_TIMEOUT_MS, WEBHOOK_BACKOFF_MULTIPLIER, WEBHOOK_INITIAL_BACKOFF_MS, @@ -1134,8 +1134,7 @@ impl MutationCoordinator { .await?; return Err(error); } - self.record_boundary_audit(&request, PreIoAuditKind::Attempt) - .await?; + let _attempt = self.begin_boundary_audit(&request).await?; if let Err(error) = self.stage_attachment(client, &request).await { self.record_boundary_audit(&request, PreIoAuditKind::Refusal) .await?; @@ -1187,8 +1186,7 @@ impl MutationCoordinator { .await?; return Err(error); } - self.record_batch_boundary_audit(&request, PreIoAuditKind::Attempt) - .await?; + let _attempt = self.begin_batch_boundary_audit(&request).await?; let result = self .execute_batch_after_attempt(client, &request, fault) .await; @@ -1202,6 +1200,50 @@ impl MutationCoordinator { }) } + /// Append the batch's attempt and hold it until its terminal or refusal + /// entry answers it. + async fn begin_batch_boundary_audit( + &self, + request: &BatchMutationRequest<'_>, + ) -> Result { + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + request.claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: request.plan.route.method, + operation_id: &request.plan.route.id, + target_record: None, + refusal_reason: None, + correlation: &request.correlation, + }, + ) + .await?) + } + + /// Append the mutation's attempt and hold it until its terminal or + /// refusal entry answers it. + async fn begin_boundary_audit( + &self, + request: &MutationRequest<'_>, + ) -> Result { + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + request.claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: request.plan.route.method, + operation_id: &request.plan.route.id, + target_record: request.record_id, + refusal_reason: None, + correlation: &request.correlation, + }, + ) + .await?) + } + async fn record_batch_boundary_audit( &self, request: &BatchMutationRequest<'_>, diff --git a/crates/registry-breg/src/mutation/action.rs b/crates/registry-breg/src/mutation/action.rs index 1108ce2e5..24c2b5511 100644 --- a/crates/registry-breg/src/mutation/action.rs +++ b/crates/registry-breg/src/mutation/action.rs @@ -173,13 +173,9 @@ impl MutationCoordinator { validate_action_claims(action, claims, Operation::Invoke)?; let normalized_input = validate_action_input(action, input.input)?; validate_precondition_set(action, &input.preconditions)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + let _attempt = self + .begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?; let request_digest = canonical_action_request_digest(action, &normalized_input, &input.preconditions)?; let binding = resolve_action_binding( @@ -712,14 +708,10 @@ impl MutationCoordinator { }; let application_id = Uuid::new_v4(); let correlation = RequestCorrelation::breg_created(); - self.record_action_boundary_audit( - &claims, - &route_id, - &correlation, - PreIoAuditKind::Attempt, - ) - .await - .map_err(|_| UncertainApply)?; + let _attempt = self + .begin_action_boundary_audit(&claims, &route_id, &correlation) + .await + .map_err(|_| UncertainApply)?; let deadline = tokio::time::Instant::now() + HOOK_PROPOSAL_APPLY_BUDGET; let fault = FaultControl::Disabled; @@ -840,13 +832,9 @@ impl MutationCoordinator { )?; validate_action_claims(action, claims, Operation::Invoke)?; let refs = validate_condition_inputs(action, input.input)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + let _attempt = self + .begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?; let result = self .action_target_conditions_after_attempt( client, @@ -871,6 +859,30 @@ impl MutationCoordinator { result } + /// Append the action's attempt and hold it until its terminal or refusal + /// entry answers it. + pub(crate) async fn begin_action_boundary_audit( + &self, + claims: &ActionClaimContext, + operation_id: &str, + correlation: &RequestCorrelation, + ) -> Result { + Ok(begin_action_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: HttpMethod::Post, + operation_id, + target_record: None, + refusal_reason: None, + correlation, + }, + ) + .await?) + } + pub(crate) async fn record_action_boundary_audit( &self, claims: &ActionClaimContext, @@ -2708,13 +2720,9 @@ impl MutationCoordinator { validate_action_claims(action, claims, Operation::Invoke)?; let normalized = validate_action_input(action, input.input)?; validate_precondition_set(action, &input.preconditions)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + let _attempt = self + .begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?; let binding = resolve_action_binding( self.audit.profile(), &ActionIdempotencyBinding { diff --git a/crates/registry-breg/src/mutation/request.rs b/crates/registry-breg/src/mutation/request.rs index 9110e6d72..73b9c8bdc 100644 --- a/crates/registry-breg/src/mutation/request.rs +++ b/crates/registry-breg/src/mutation/request.rs @@ -285,6 +285,38 @@ impl MutationCoordinator { }) } + /// Append the attempt of one request action and return the handle that + /// owes its response. An orchestrating caller that runs protected reads + /// before the action itself, such as a reviewed apply's receipt + /// preflight, holds it across all of them. + pub(crate) async fn begin_request_action_audit( + &self, + registry: &CompiledRegistry, + input: &RequestActionInput<'_>, + claims: &ClaimContext, + ) -> Result { + let route = registry + .routes() + .routes + .iter() + .find(|route| route.id == input.route_id) + .ok_or(MutationError::InvalidRequest)?; + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: route.method, + operation_id: &route.id, + target_record: Some(input.record_id), + refusal_reason: None, + correlation: input.correlation, + }, + ) + .await?) + } + pub(crate) async fn record_request_boundary_refusal( &self, registry: &CompiledRegistry, @@ -321,6 +353,7 @@ impl MutationCoordinator { input: &RequestActionInput<'_>, claims: &ClaimContext, deadline: tokio::time::Instant, + attempt: &mut Option, ) -> Result { let route = registry .routes() @@ -365,20 +398,26 @@ impl MutationCoordinator { { return Err(MutationError::InvalidRequest); } - record_pre_io_audit( - &self.audit, - &self.expected, - claims, - PreIoAudit { - kind: PreIoAuditKind::Attempt, - method: route.method, - operation_id: &route.id, - target_record: Some(input.record_id), - refusal_reason: None, - correlation: input.correlation, - }, - ) - .await?; + // The caller holds the attempt across the remote Evidence I/O and the + // action transaction that follow this preflight. + if attempt.is_none() { + *attempt = Some( + begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: route.method, + operation_id: &route.id, + target_record: Some(input.record_id), + refusal_reason: None, + correlation: input.correlation, + }, + ) + .await?, + ); + } let binding = resolve_binding( self.audit.profile(), &IdempotencyBinding { @@ -668,15 +707,20 @@ impl MutationCoordinator { refusal_reason, correlation: input.correlation, }; - if !attempt_recorded { - record_pre_io_audit( - &self.audit, - &self.expected, - claims, - audit(PreIoAuditKind::Attempt, None), + // A caller that recorded the attempt holds it until this returns. + let _attempt = if attempt_recorded { + None + } else { + Some( + begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + audit(PreIoAuditKind::Attempt, None), + ) + .await?, ) - .await?; - } + }; let deadline = tokio::time::Instant::now() + REQUEST_ACTION_TIMEOUT; // Capture the exact intake under request RLS, close that transaction, // then run the bounded planner exactly once outside retry and target diff --git a/crates/registry-breg/src/postgres/history_read.rs b/crates/registry-breg/src/postgres/history_read.rs index 8d48dfd24..2d9f33181 100644 --- a/crates/registry-breg/src/postgres/history_read.rs +++ b/crates/registry-breg/src/postgres/history_read.rs @@ -20,8 +20,8 @@ use crate::api::{ SnapshotReadRequest, SnapshotReadService, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation}; use crate::cursor::{ @@ -154,7 +154,7 @@ impl PostgresSnapshotReadService { } }; - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -174,25 +174,18 @@ impl PostgresSnapshotReadService { let materialized = match materialized { Ok(materialized) => materialized, Err(error) => { - let _ = self - .record_terminal( - &claims, - &request, - &plan, - None, - TerminalAuditOutcome::Refused, - 0, - ) - .await; - return Err(error); + return Err(self.refused(&claims, &request, &plan, error).await); } }; - let held = SnapshotReadResult::from_materialized( + let held = match SnapshotReadResult::from_materialized( &self.registry, &plan.entity, request.plan.cursor_binding.representation, materialized, - )?; + ) { + Ok(held) => held, + Err(error) => return Err(self.refused(&claims, &request, &plan, error).await), + }; self.fault .fail_at(SnapshotReadFaultPoint::BeforeTerminalAudit)?; let outcome = if held.result_count == 0 { @@ -213,6 +206,34 @@ impl PostgresSnapshotReadService { Ok(held) } + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused( + &self, + claims: &ClaimContext, + request: &SnapshotReadRequest, + plan: &SnapshotReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + if self + .record_terminal( + claims, + request, + plan, + None, + TerminalAuditOutcome::Refused, + 0, + ) + .await + .is_err() + { + tracing::error!("the refused history read's terminal audit entry was not recorded"); + } + error + } + async fn read_rows( &self, client: &mut deadpool_postgres::Client, diff --git a/crates/registry-breg/src/postgres/mutation.rs b/crates/registry-breg/src/postgres/mutation.rs index 31a36a69f..9bd94edc2 100644 --- a/crates/registry-breg/src/postgres/mutation.rs +++ b/crates/registry-breg/src/postgres/mutation.rs @@ -87,6 +87,24 @@ pub struct IngestionRunListQuery { pub limit: i64, } +/// A refused ingestion-run call. `answered` is true when the call's refusal +/// is already on record as the `response` of the ingestion `request` entry it +/// wrote, so the caller must not record it again in another schema. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct IngestionRefusal { + pub error: IngestionServiceError, + pub answered: bool, +} + +/// What one ingestion-run call has recorded so far: the ingestion `request` +/// entry it wrote, or whether it handed the chunk to the batch mutation, +/// which records its own attempt and refusal. +#[derive(Default)] +struct IngestionAudit { + run: Option, + batch: bool, +} + /// The closed refusal vocabulary of the ingestion-run service. It is bounded /// and value-free: no chunk bytes, row values, or bearer material appear. #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -385,7 +403,9 @@ impl PostgresRecordMutationService { .await; } if is_evidence_apply { - return self.request_evidence_apply(input, &claims, None).await; + return self + .request_evidence_apply(input, &claims, None, None) + .await; } let client = self .pool @@ -429,7 +449,12 @@ impl PostgresRecordMutationService { input: crate::api::RequestActionInput<'_>, claims: &ClaimContext, review_evidence: Option<&crate::review_integration::AcceptedReviewEvidence>, + attempt: Option, ) -> Result { + // The request's attempt, held until the action answers it: the + // reviewed apply that routes here has already recorded it, and the + // preflight records it otherwise. + let mut attempt = attempt; let evaluator = self .evidence_evaluator .as_ref() @@ -456,6 +481,7 @@ impl PostgresRecordMutationService { &input, claims, deadline, + &mut attempt, ) .await; if result.is_err() && tokio::time::Instant::now() < deadline { @@ -543,6 +569,15 @@ impl PostgresRecordMutationService { MutationFaultControl::At(point) => crate::mutation::FaultControl::At(point), _ => crate::mutation::FaultControl::Disabled, }; + // The attempt precedes the receipt preflight's reads, the review + // authority, and the action transaction, and is held across them. + let attempt = tokio::time::timeout_at( + deadline, + self.coordinator + .begin_request_action_audit(&self.registry, &input, claims), + ) + .await + .map_err(|_| MutationError::Unavailable)??; let receipt_preflight = { let client = self .pool @@ -561,9 +596,23 @@ impl PostgresRecordMutationService { ) .await; match result { - Ok(result) => { + Ok(Ok(preflight)) => { guard.disarm(); - result? + preflight + } + Ok(Err(error)) => { + guard.disarm(); + tokio::time::timeout_at( + deadline, + self.coordinator.record_request_boundary_refusal( + &self.registry, + &input, + claims, + ), + ) + .await + .map_err(|_| MutationError::Unavailable)??; + return Err(error); } Err(_) => { guard.cancel_and_discard().await; @@ -592,7 +641,7 @@ impl PostgresRecordMutationService { fault, None, None, - false, + true, ), ) .await; @@ -648,8 +697,8 @@ impl PostgresRecordMutationService { } .await; if let Err(error) = task_authority { - // The refusal happens before any journaled attempt, so it is - // recorded here, the same as a refused evidence-apply preflight. + // The refusal answers the held attempt, the same as a refused + // evidence-apply preflight. tokio::time::timeout_at( deadline, self.coordinator @@ -666,7 +715,7 @@ impl PostgresRecordMutationService { let review_evidence = source.approved_evidence(&authority, &accepted).await?; if needs_action_evidence { return self - .request_evidence_apply(input, claims, Some(&review_evidence)) + .request_evidence_apply(input, claims, Some(&review_evidence), Some(attempt)) .await; } let client = self @@ -685,7 +734,7 @@ impl PostgresRecordMutationService { fault, Some(&review_evidence), None, - false, + true, ), ) .await; @@ -1024,14 +1073,101 @@ impl PostgresRecordMutationService { run.response_json(active.0, active.1) } + /// Open one ingestion run. + pub async fn create_ingestion_run( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + input: IngestionRunCreateInput, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .create_ingestion_run_in(context, correlation, input, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Cancel one open ingestion run. + pub async fn cancel_ingestion_run( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + entity_id: &str, + run_id: Uuid, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .cancel_ingestion_run_in(context, correlation, entity_id, run_id, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Submit one chunk of an open ingestion run. + pub async fn submit_ingestion_chunk( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + input: IngestionChunkSubmitInput, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .submit_ingestion_chunk_in(context, correlation, input, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Recover the retained receipt of one committed chunk. + pub async fn ingestion_chunk_receipt( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + entity_id: &str, + run_id: Uuid, + chunk_index: i64, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .ingestion_chunk_receipt_in( + context, + correlation, + entity_id, + run_id, + chunk_index, + &mut attempt, + ) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Answer the ingestion `request` entry a refused call wrote, in the + /// ingestion schema, so the refusal is never recorded only in another + /// schema. A call refused before it wrote one leaves the refusal to its + /// caller. + async fn settle_ingestion( + result: Result, + audit: IngestionAudit, + ) -> Result { + let error = match result { + Ok(value) => return Ok(value), + Err(error) => error, + }; + let answered = match audit.run { + Some(attempt) if attempt.is_answered() => true, + Some(attempt) => attempt.refuse().await, + None => audit.batch, + }; + Err(IngestionRefusal { error, answered }) + } + /// Create a durable ingestion run bound to the active package revision, /// schema fingerprint, entity, profile, operation, input digest, chunking /// algorithm, and announced counts. - pub async fn create_ingestion_run( + async fn create_ingestion_run_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, input: IngestionRunCreateInput, + attempt: &mut IngestionAudit, ) -> Result { if !crate::audit::profile_is_keyed(self.audit.profile()) { return Err(IngestionServiceError::Unavailable); @@ -1107,21 +1243,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the run is opened: an audit // outage refuses the creation instead of opening a run nobody // recorded asking for. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "create", - run_id: None, - chunk_index: None, - package_revision: &self.expected.package_revision, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &run.created_principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "create", + run_id: None, + chunk_index: None, + package_revision: &self.expected.package_revision, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &run.created_principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut client = self.client().await?; // The run binding must name the package the database still holds // active, so creation takes the same guarded transaction ordinary @@ -1316,12 +1454,13 @@ impl PostgresRecordMutationService { /// Cancel an open or blocked run, preserving the committed prefix, the /// counts, and the audit trail. - pub async fn cancel_ingestion_run( + async fn cancel_ingestion_run_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, entity_id: &str, run_id: Uuid, + attempt: &mut IngestionAudit, ) -> Result { if !crate::audit::profile_is_keyed(self.audit.profile()) { return Err(IngestionServiceError::Unavailable); @@ -1334,21 +1473,23 @@ impl PostgresRecordMutationService { let request_correlation = correlation.request_id().to_string(); // The request entry is accepted before the run is read or closed: an // audit outage leaves the run open and resumable. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "cancel", - run_id: Some(run_id), - chunk_index: None, - package_revision: &self.expected.package_revision, - entity_id, - profile_id: claims.access_profile(), - principal_reference: &self.ingestion_principal_reference(principal)?, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "cancel", + run_id: Some(run_id), + chunk_index: None, + package_revision: &self.expected.package_revision, + entity_id, + profile_id: claims.access_profile(), + principal_reference: &self.ingestion_principal_reference(principal)?, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut client = self.client().await?; let run = self .visible_run(&**client, context, entity_id, run_id) @@ -1411,11 +1552,12 @@ impl PostgresRecordMutationService { /// Submit the next exact chunk of one run. The server derives the /// idempotency key from the run binding, so an interrupted submission /// replays the original receipt without a duplicate mutation. - pub async fn submit_ingestion_chunk( + async fn submit_ingestion_chunk_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, input: IngestionChunkSubmitInput, + attempt: &mut IngestionAudit, ) -> Result { let claims = strict_claim_context(&self.registry, context, &input.entity_id) .map_err(|_| IngestionServiceError::RequestInvalid)?; @@ -1505,21 +1647,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the release transaction // opens: an audit outage moves no attempt marker and releases // nothing. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "submitChunk", - run_id: Some(run.run_id), - chunk_index: Some(input.chunk_index), - package_revision: &self.expected.package_revision, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "submitChunk", + run_id: Some(run.run_id), + chunk_index: Some(input.chunk_index), + package_revision: &self.expected.package_revision, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut disclosure_writer = self.client().await?; let disclosure_transaction = begin_record_transaction( &mut disclosure_writer, @@ -1650,21 +1794,23 @@ impl PostgresRecordMutationService { let request_correlation = correlation.request_id().to_string(); // The request entry is accepted before the blocking transition // opens: an audit outage leaves the run open. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "submitChunk", - run_id: Some(run.run_id), - chunk_index: Some(input.chunk_index), - package_revision: &active.0, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "submitChunk", + run_id: Some(run.run_id), + chunk_index: Some(input.chunk_index), + package_revision: &active.0, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut writer = self.client().await?; let transaction = writer .transaction() @@ -1789,6 +1935,9 @@ impl PostgresRecordMutationService { ingestion: Some(&chunk_binding), }; let mut writer = self.client().await?; + // From here the batch mutation records the chunk's attempt and its + // refusal under this request's correlation. + attempt.batch = true; #[cfg(feature = "postgres-test")] if let MutationFaultControl::At(fault) = self.fault { return self @@ -2010,13 +2159,14 @@ impl PostgresRecordMutationService { /// Recover the stored receipt of one committed chunk after a lost /// response. The receipt is erased with the record history it describes. - pub async fn ingestion_chunk_receipt( + async fn ingestion_chunk_receipt_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, entity_id: &str, run_id: Uuid, chunk_index: i64, + attempt: &mut IngestionAudit, ) -> Result { if chunk_index < 0 { return Err(IngestionServiceError::RequestInvalid); @@ -2034,21 +2184,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the run or the stored chunk is // read: an audit outage refuses the recovery instead of releasing a // receipt nobody recorded asking for. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "chunkReceipt", - run_id: Some(run_id), - chunk_index: Some(chunk_index), - package_revision: &self.expected.package_revision, - entity_id, - profile_id: claims.access_profile(), - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "chunkReceipt", + run_id: Some(run_id), + chunk_index: Some(chunk_index), + package_revision: &self.expected.package_revision, + entity_id, + profile_id: claims.access_profile(), + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let client = self.client().await?; let run = self .visible_run(&**client, context, entity_id, run_id) diff --git a/crates/registry-breg/src/postgres/read.rs b/crates/registry-breg/src/postgres/read.rs index cbfc4a3ba..5adb321d8 100644 --- a/crates/registry-breg/src/postgres/read.rs +++ b/crates/registry-breg/src/postgres/read.rs @@ -23,8 +23,8 @@ use crate::api::{ RowBoundaryOperator as ApiRowBoundaryOperator, ServiceFuture, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation}; use crate::cursor::{ @@ -238,7 +238,7 @@ impl PostgresRecordReadService { return Ok(ReadResult::empty_get()); } - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -257,22 +257,7 @@ impl PostgresRecordReadService { let materialized = self.read_rows(&mut client, &request, &claims, &plan).await; let materialized = match materialized { Ok(materialized) => materialized, - Err(error) => { - let _ = self - .record_read_terminal_audit( - &request, - self.terminal( - &request, - &claims, - &plan, - TerminalAuditOutcome::Refused, - 0, - None, - )?, - ) - .await; - return Err(error); - } + Err(error) => return Err(self.refused_read(&request, &claims, &plan, error).await), }; let attachment_verification = materialized.rows.first().and_then(|record| { crate::mutation::attachment_verification_etag_fields(&plan.entity, &record.data) @@ -287,47 +272,21 @@ impl PostgresRecordReadService { .and_then(|result| result.enforce_spatial_response_budget(&request)) { Ok(held) => held, - Err(error) => { - let _ = self - .record_read_terminal_audit( - &request, - self.terminal( - &request, - &claims, - &plan, - TerminalAuditOutcome::Refused, - 0, - None, - )?, - ) - .await; - return Err(error); - } + Err(error) => return Err(self.refused_read(&request, &claims, &plan, error).await), }; if plan.operation == Operation::Get && request.representation != CursorRepresentation::GeoJson && held.response.is_some() { - let response = held.response.take().ok_or(ReadServiceError::Unavailable)?; - let record_id = target_record.ok_or(ReadServiceError::Unavailable)?; - let record_revision = held.record_revision.ok_or(ReadServiceError::Unavailable)?; - let representation = match request.representation { - CursorRepresentation::Json => RecordRepresentation::Json, - CursorRepresentation::JsonLd => RecordRepresentation::JsonLd, - CursorRepresentation::GeoJson => return Err(ReadServiceError::Unavailable), - }; - let etag = strong_record_etag_for_representation( - self.audit.profile(), + if let Err(error) = self.attach_strong_etag( + &mut held, + &request, &claims, - &self.expected.package_revision, - record_id, - record_revision, - &request.selected_fields, - representation, + target_record, attachment_verification.as_ref(), - ) - .map_err(|_| ReadServiceError::Unavailable)?; - held.response = Some(response.with_strong_etag(etag)); + ) { + return Err(self.refused_read(&request, &claims, &plan, error).await); + } } self.fault.fail_at(ReadFaultPoint::BeforeTerminalAudit)?; let outcome = match (plan.operation, held.result_count) { @@ -388,28 +347,27 @@ impl PostgresRecordReadService { let mut request = request; request.operation_id = crate::attachment::operation_id(&request.operation_id, &slot_id, request.method); - record_pre_io_audit( - &self.audit, - &self.expected, - &claims, - PreIoAudit { - kind: if valid { - PreIoAuditKind::Attempt - } else { - PreIoAuditKind::Refusal - }, - method: request.method, - operation_id: &request.operation_id, - target_record: target_record(&request.kind), - refusal_reason: None, - correlation: &request.correlation, + let event = PreIoAudit { + kind: if valid { + PreIoAuditKind::Attempt + } else { + PreIoAuditKind::Refusal }, - ) - .await - .map_err(|_| ReadServiceError::Unavailable)?; + method: request.method, + operation_id: &request.operation_id, + target_record: target_record(&request.kind), + refusal_reason: None, + correlation: &request.correlation, + }; if !valid { + record_pre_io_audit(&self.audit, &self.expected, &claims, event) + .await + .map_err(|_| ReadServiceError::Unavailable)?; return Ok(None); } + let _attempt = begin_pre_io_audit(&self.audit, &self.expected, &claims, event) + .await + .map_err(|_| ReadServiceError::Unavailable)?; let plan = plan.map_err(|_| ReadServiceError::Unavailable)?; let transaction = begin_record_transaction( &mut client, @@ -590,6 +548,70 @@ impl PostgresRecordReadService { /// Append the read's `response` entry. The caller releases the result /// only after this returns `Ok`. + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused_read( + &self, + request: &RecordReadRequest, + claims: &ClaimContext, + plan: &ReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + let recorded = match self.terminal( + request, + claims, + plan, + TerminalAuditOutcome::Refused, + 0, + None, + ) { + Ok(terminal) => self + .record_read_terminal_audit(request, terminal) + .await + .map_err(|_| ()), + Err(_) => Err(()), + }; + if recorded.is_err() { + tracing::error!("the refused read's terminal audit entry was not recorded"); + } + error + } + + /// Bind the strong entity tag of a single-record read to its response. + fn attach_strong_etag( + &self, + held: &mut ReadResult, + request: &RecordReadRequest, + claims: &ClaimContext, + target_record: Option<&str>, + attachment_verification: Option<&Value>, + ) -> Result<(), ReadServiceError> { + self.fault.fail_at(ReadFaultPoint::StrongEtag)?; + let response = held.response.take().ok_or(ReadServiceError::Unavailable)?; + let record_id = target_record.ok_or(ReadServiceError::Unavailable)?; + let record_revision = held.record_revision.ok_or(ReadServiceError::Unavailable)?; + let representation = match request.representation { + CursorRepresentation::Json => RecordRepresentation::Json, + CursorRepresentation::JsonLd => RecordRepresentation::JsonLd, + CursorRepresentation::GeoJson => return Err(ReadServiceError::Unavailable), + }; + let etag = strong_record_etag_for_representation( + self.audit.profile(), + claims, + &self.expected.package_revision, + record_id, + record_revision, + &request.selected_fields, + representation, + attachment_verification, + ) + .map_err(|_| ReadServiceError::Unavailable)?; + held.response = Some(response.with_strong_etag(etag)); + Ok(()) + } + async fn record_read_terminal_audit( &self, request: &RecordReadRequest, @@ -4090,12 +4112,16 @@ mod tests { #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum ReadFaultPoint { BeforeTerminalAudit, + /// The strong entity tag of a single-record read cannot be bound, after + /// the rows were read. + StrongEtag, } #[cfg(not(feature = "postgres-test"))] #[derive(Clone, Copy, Debug, Eq, PartialEq)] enum ReadFaultPoint { BeforeTerminalAudit, + StrongEtag, } #[derive(Clone, Copy)] diff --git a/crates/registry-breg/src/postgres/revision_read.rs b/crates/registry-breg/src/postgres/revision_read.rs index bfd2055ca..4e6b6f279 100644 --- a/crates/registry-breg/src/postgres/revision_read.rs +++ b/crates/registry-breg/src/postgres/revision_read.rs @@ -19,8 +19,8 @@ use crate::api::{ RevisionReadService, RowBoundaryOperator as ApiRowBoundaryOperator, ServiceFuture, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation, ProvenanceFieldSource}; use crate::cursor::CursorRepresentation; @@ -135,7 +135,7 @@ impl PostgresRevisionReadService { } }; - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -155,26 +155,19 @@ impl PostgresRevisionReadService { let materialized = match materialized { Ok(materialized) => materialized, Err(error) => { - let _ = self - .record_terminal( - &claims, - &request, - &plan, - TerminalAuditOutcome::Refused, - 0, - &[], - ) - .await; - return Err(error); + return Err(self.refused(&claims, &request, &plan, error).await); } }; - let held = RevisionReadResult::from_rows( + let held = match RevisionReadResult::from_rows( &self.registry, &plan.entity, request.representation, plan.kind, materialized, - )?; + ) { + Ok(held) => held, + Err(error) => return Err(self.refused(&claims, &request, &plan, error).await), + }; self.fault .fail_at(RevisionReadFaultPoint::BeforeTerminalAudit)?; let outcome = if held.result_count == 0 { @@ -195,6 +188,27 @@ impl PostgresRevisionReadService { Ok(held) } + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused( + &self, + claims: &ClaimContext, + request: &RevisionReadRequest, + plan: &RevisionReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + if self + .record_terminal(claims, request, plan, TerminalAuditOutcome::Refused, 0, &[]) + .await + .is_err() + { + tracing::error!("the refused revision read's terminal audit entry was not recorded"); + } + error + } + async fn read_rows( &self, client: &mut deadpool_postgres::Client, diff --git a/crates/registry-breg/src/request_retention.rs b/crates/registry-breg/src/request_retention.rs index c6e0fabc0..e152cbbe4 100644 --- a/crates/registry-breg/src/request_retention.rs +++ b/crates/registry-breg/src/request_retention.rs @@ -8,6 +8,7 @@ use std::time::Duration; use registry_platform_audit::{AuditEntry, AuditProfile}; use serde::Serialize; +use serde_json::{json, Value}; use tokio_postgres::GenericClient; use uuid::Uuid; @@ -48,6 +49,12 @@ pub enum RequestRetentionError { AttachmentStorageBindingMismatch, #[error("request retention state is unavailable")] Unavailable, + /// The erasure committed, but the audit destination refused the entry + /// recording it. The erased detail is gone; restore the destination, + /// then reconcile the erasure against the database before relying on + /// the audit journal for it. + #[error("the request detail erasure committed but its audit entry was not recorded; restore the audit destination")] + ErasureUnaudited, } pub type Result = std::result::Result; @@ -451,19 +458,82 @@ impl RequestRetentionOperatorService { .await .map_err(|_| RequestRetentionError::Unavailable)?; // The request entry is accepted before the erasure transaction opens, - // so an audit outage erases nothing; the committed response shares - // its correlation. + // so an audit outage erases nothing; the response shares its + // correlation. An erasure that ends without one writes the + // unfinished outcome when the held request is dropped. let correlation = RequestCorrelation::breg_created(); - self.audit - .append(retention_request_entry( - self.audit.profile(), - &self.expected, - scope.clone(), - &correlation, - )?) + let request_entry = retention_request_entry( + self.audit.profile(), + &self.expected, + scope.clone(), + &correlation, + )?; + let unfinished = retention_outcome_record(&request_entry, "unfinished"); + let request_record = request_entry.clone(); + let mut attempt = self + .audit + .begin(request_entry, unfinished) .await .map_err(|_| RequestRetentionError::Unavailable)?; - let transaction = self.begin_verified_transaction(&mut client).await?; + let erased = self + .erase_in_transaction(&mut client, scope.clone(), correlation) + .await; + let (plan, erasure, entry) = match erased { + Ok(erased) => erased, + Err(error) => { + // Nothing committed: answer the request with the refusal or + // the failure. + let outcome = if error == RequestRetentionError::Unavailable { + "failed" + } else { + "refused" + }; + let answer = retention_outcome_record(&request_record, outcome); + if attempt.respond(answer).await.is_err() { + tracing::error!("the failed erasure's response audit entry was not recorded"); + } + return Err(error); + } + }; + // External objects are deleted before the response is recorded, so + // the entry states how many remain instead of claiming a finished + // erasure while objects still exist. + let external = self.retry_external_deletions(&mut client).await; + let mut record = entry.record().clone(); + if let Some(fields) = record.as_object_mut() { + let (pending, tombstones) = match &external { + Ok((pending, tombstones)) => (json!(pending), json!(tombstones)), + Err(_) => (Value::Null, Value::Null), + }; + fields.insert("pendingExternalDeletions".to_owned(), pending); + fields.insert("externalDeletionTombstones".to_owned(), tombstones); + } + attempt + .respond(record) + .await + .map_err(|_| RequestRetentionError::ErasureUnaudited)?; + let (pending_external_deletions, external_deletion_tombstones) = external?; + Ok(RequestRetentionErase { + request_entity_id: scope.request_entity_id.to_owned(), + request_id: scope.request_id.to_string(), + proposal_version: scope.proposal_version, + request_state: plan.current_state, + retention_mode: retention_mode_name(plan.retention_mode), + erasure, + pending_external_deletions, + external_deletion_tombstones, + }) + } + + /// Erase one request's detail in one committed transaction and build the + /// terminal entry that records it. + async fn erase_in_transaction( + &self, + client: &mut deadpool_postgres::Client, + scope: RequestDetailErasureScope<'_>, + correlation: RequestCorrelation, + ) -> Result<(RequestErasurePlan, RequestDetailErasure, AuditEntry)> { + let transaction = self.begin_verified_transaction(client).await?; let plan = load_erasure_plan(&transaction, &self.registry, scope.clone(), true).await?; let (erasure, current_revision) = erase_request_detail_in_transaction(&transaction, &self.registry, scope.clone(), &plan) @@ -500,22 +570,7 @@ impl RequestRetentionOperatorService { .commit() .await .map_err(|_| RequestRetentionError::Unavailable)?; - self.audit - .append(entry) - .await - .map_err(|_| RequestRetentionError::Unavailable)?; - let (pending_external_deletions, external_deletion_tombstones) = - self.retry_external_deletions(&mut client).await?; - Ok(RequestRetentionErase { - request_entity_id: scope.request_entity_id.to_owned(), - request_id: scope.request_id.to_string(), - proposal_version: scope.proposal_version, - request_state: plan.current_state, - retention_mode: retention_mode_name(plan.retention_mode), - erasure, - pending_external_deletions, - external_deletion_tombstones, - }) + Ok((plan, erasure, entry)) } /// Retry orphaned external objects even when every request is active or @@ -535,36 +590,50 @@ impl RequestRetentionOperatorService { // audit state. let transaction = self.begin_verified_transaction(&mut client).await?; transaction.commit().await.map_err(map_retention_error)?; - self.audit - .append(AuditEntry::request( - ATTACHMENT_CLEANUP_AUDIT_SCHEMA, - correlation.clone(), - serde_json::json!({ - "kind":"attachmentCleanup", "phase":"attempt", "outcome":"started", - "packageRevision":self.expected.package_revision, - "actor":"breg:request-retention-operator", "correlation":correlation, - }), - )) + let record = |phase: &str, outcome: &str| { + serde_json::json!({ + "kind":"attachmentCleanup", "phase":phase, "outcome":outcome, + "packageRevision":self.expected.package_revision, + "actor":"breg:request-retention-operator", "correlation":correlation, + }) + }; + // A cleanup that ends before its response writes the unfinished + // outcome when the held request is dropped. + let mut attempt = self + .audit + .begin( + AuditEntry::request( + ATTACHMENT_CLEANUP_AUDIT_SCHEMA, + correlation.clone(), + record("attempt", "started"), + ), + record("terminal", "unfinished"), + ) .await .map_err(|_| RequestRetentionError::Unavailable)?; - let (pending_external_deletions, external_deletion_tombstones) = - self.retry_external_deletions(&mut client).await?; + let (pending_external_deletions, external_deletion_tombstones) = match self + .retry_external_deletions(&mut client) + .await + { + Ok(counts) => counts, + Err(error) => { + if attempt.respond(record("terminal", "failed")).await.is_err() { + tracing::error!("the failed cleanup's response audit entry was not recorded"); + } + return Err(error); + } + }; let result = AttachmentCleanup { pending_external_deletions, external_deletion_tombstones, }; - self.audit - .append(AuditEntry::response( - ATTACHMENT_CLEANUP_AUDIT_SCHEMA, - correlation.clone(), - serde_json::json!({ - "kind":"attachmentCleanup", "phase":"terminal", "outcome":"completed", - "packageRevision":self.expected.package_revision, - "actor":"breg:request-retention-operator", "correlation":correlation, - "pendingExternalDeletions":result.pending_external_deletions, - "externalDeletionTombstones":result.external_deletion_tombstones, - }), - )) + let mut completed = record("terminal", "completed"); + completed["pendingExternalDeletions"] = + serde_json::json!(result.pending_external_deletions); + completed["externalDeletionTombstones"] = + serde_json::json!(result.external_deletion_tombstones); + attempt + .respond(completed) .await .map_err(|_| RequestRetentionError::Unavailable)?; Ok(result) @@ -1466,6 +1535,17 @@ fn retention_request_entry( )) } +/// The `response` of an erasure that did not record its committed terminal: +/// the request's identities with `outcome`, and no count. +fn retention_outcome_record(request: &AuditEntry, outcome: &str) -> Value { + let mut record = request.record().clone(); + if let Some(fields) = record.as_object_mut() { + fields.insert("phase".to_owned(), json!("terminal")); + fields.insert("outcome".to_owned(), json!(outcome)); + } + record +} + fn retention_terminal_entry( profile: &AuditProfile, expected: &ExpectedRegistryIdentity, diff --git a/crates/registry-breg/tests/postgres_action_evidence_retention.rs b/crates/registry-breg/tests/postgres_action_evidence_retention.rs index 85886443d..7bec3944a 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_retention.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_retention.rs @@ -84,9 +84,40 @@ fn service( connection, database.migration_role.clone(), database.runtime_role.clone(), + database.audit( + registry_platform_audit::AuditProfile::production_from_secret_bytes( + vec![0x5e; 32].into(), + ) + .unwrap(), + ), )) } +/// Every erasure is one request entry naming its threshold, answered under +/// its correlation: `outcomes` lists each answer's outcome and erased count. +fn assert_retention_audited(database: &TestDatabase, outcomes: &[(&str, Option)]) { + let entries = database + .audit_entries() + .into_iter() + .filter(|entry| entry["schema"] == "breg-evidence-retention-audit/v1") + .collect::>(); + assert_eq!(entries.len(), outcomes.len() * 2, "{entries:?}"); + for (pair, (outcome, erased)) in entries.chunks(2).zip(outcomes) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); + assert!(pair[0]["record"]["before"].is_string()); + assert_eq!(pair[1]["record"]["before"], pair[0]["record"]["before"]); + assert_eq!(pair[1]["record"]["outcome"], *outcome); + assert_eq!( + pair[1]["record"] + .get("erased") + .and_then(serde_json::Value::as_u64), + *erased + ); + } +} + async fn sentinel(database: &TestDatabase) { database.admin.batch_execute("INSERT INTO registry_internal.registry_idempotency (key_reference,binding_reference,result_kind,result_count,response_status,response_body,response_headers) @@ -168,6 +199,7 @@ async fn expired_request_evidence_erases_only_retained_uses() { assert_eq!(remaining.get::<_, i64>(0), 1); assert_eq!(remaining.get::<_, i64>(1), 1); assert_eq!(operator.erase_expired(cutoff()).await.unwrap(), 0); + assert_retention_audited(&database, &[("erased", Some(1)), ("erased", Some(0))]); drop(operator); database.cleanup().await; } @@ -236,6 +268,7 @@ async fn retention_refuses_misbound_database_with_identical_roles_and_catalog_dr let wrong = service(&original, ®istry, &expected, other_connection.clone()); assert!(wrong.erase_expired(cutoff()).await.is_err(), "a verified runtime identity cannot authorize deletion in another database sharing its role names"); assert_eq!(count(&other).await, 1); + assert_retention_audited(&original, &[("failed", None)]); let correct = service(&original, ®istry, &other_expected, other_connection); other .admin diff --git a/crates/registry-breg/tests/postgres_change_requests.rs b/crates/registry-breg/tests/postgres_change_requests.rs index 10732e187..3a787a836 100644 --- a/crates/registry-breg/tests/postgres_change_requests.rs +++ b/crates/registry-breg/tests/postgres_change_requests.rs @@ -209,6 +209,7 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt .await; let apply = action(&before.body, "apply_request", None); + let entries_before = database.audit_entries().len(); let unavailable = send_action( &app, &apply, @@ -218,6 +219,9 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt ) .await; assert_eq!(unavailable.status, StatusCode::SERVICE_UNAVAILABLE); + // The attempt precedes the receipt preflight's reads and the review + // authority, so even this refusal is a request answered in order. + assert_requested_then_answered(&database.audit_entries()[entries_before..]); assert_eq!(application_result_count(&database).await, 0); assert_eq!(authority_state.lookups.load(Ordering::SeqCst), 1); @@ -248,6 +252,7 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt assert_eq!(authority_state.lookups.load(Ordering::SeqCst), 3); authority_state.mode.store(0, Ordering::SeqCst); + let entries_before = database.audit_entries().len(); let replay = send_action( &app, &apply, @@ -257,6 +262,8 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt ) .await; assert_eq!(replay.status, StatusCode::OK, "{}", replay.body); + // The receipt branch records its attempt once, before its preflight. + assert_requested_then_answered(&database.audit_entries()[entries_before..]); assert_eq!(replay.body, applied.body); assert_eq!( authority_state.lookups.load(Ordering::SeqCst), @@ -268,6 +275,22 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt database.cleanup().await; } +/// One call's general audit entries: its single attempt request entry first, +/// then at least one response under the same correlation. +fn assert_requested_then_answered(entries: &[serde_json::Value]) { + let entries = entries + .iter() + .filter(|entry| entry["schema"] == "breg-audit/v2") + .collect::>(); + assert!(entries.len() >= 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request", "{entries:?}"); + assert_eq!(entries[0]["record"]["phase"], "attempt"); + for entry in &entries[1..] { + assert_eq!(entry["phase"], "response", "{entries:?}"); + assert_eq!(entry["correlation"], entries[0]["correlation"]); + } +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn review_submissions_bind_the_subject_to_the_registrys_request_entity() { let database = TestDatabase::create(8).await; diff --git a/crates/registry-breg/tests/postgres_ingestion_runs.rs b/crates/registry-breg/tests/postgres_ingestion_runs.rs index 485fa5f26..e94216e8c 100644 --- a/crates/registry-breg/tests/postgres_ingestion_runs.rs +++ b/crates/registry-breg/tests/postgres_ingestion_runs.rs @@ -710,6 +710,108 @@ async fn cancel_closes_the_run_and_preserves_the_committed_prefix() { assert_eq!(body_json(second).await["code"], "ingestion.run_not_open"); } +/// A run creation refused after its ingestion request entry is answered in +/// the ingestion schema under the same correlation, and a chunk the batch +/// mutation refuses is answered by that mutation's own refusal; neither is +/// recorded a second time as a general refusal. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema() { + let harness = IngestionHarness::create().await; + let claims = operator_claims(PRINCIPAL, "zone-a"); + let items = announce_items("refused-after-request", 4); + let chunks = plan_chunks(&items, 2); + + refuse_inserts(&harness, "registry_ingestion_runs").await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + "/v1/records/widgets/ingestion-runs", + &claims, + harness.run_body("create", &chunks), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_inserts(&harness, "registry_ingestion_runs").await; + assert_answered_in_the_ingestion_schema(&harness.database.audit_entries()[before..], "create"); + + let run_id = harness.create_run(&claims, &chunks).await; + refuse_inserts(&harness, "registry_ingestion_run_chunks").await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + &format!("/v1/records/widgets/ingestion-runs/{run_id}/chunks"), + &claims, + chunk_body(&chunks, 0), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_inserts(&harness, "registry_ingestion_run_chunks").await; + let entries = &harness.database.audit_entries()[before..]; + assert_eq!( + entries + .iter() + .map(|entry| ( + entry["schema"].as_str().expect("schema"), + entry["phase"].as_str().expect("phase") + )) + .collect::>(), + [("breg-audit/v2", "request"), ("breg-audit/v2", "response")], + "{entries:?}" + ); + assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); +} + +/// The refused call wrote one ingestion request entry and one response +/// entry answering it, both in the ingestion schema, and nothing else. +fn assert_answered_in_the_ingestion_schema(entries: &[Value], transition: &str) { + let ingestion = entries + .iter() + .filter(|entry| entry["record"]["transition"] == transition) + .collect::>(); + assert_eq!(ingestion.len(), 2, "{entries:?}"); + for entry in &ingestion { + assert_eq!(entry["schema"], "breg-ingestion-audit/v1"); + } + assert_eq!(ingestion[0]["phase"], "request"); + assert_eq!(ingestion[1]["phase"], "response"); + assert_eq!(ingestion[1]["record"]["outcome"], "refused"); + assert_eq!(ingestion[0]["correlation"], ingestion[1]["correlation"]); + assert!( + entries + .iter() + .all(|entry| entry["correlation"] != ingestion[0]["correlation"] + || entry["schema"] == "breg-ingestion-audit/v1"), + "the refusal is not recorded again in another schema: {entries:?}" + ); +} + +async fn refuse_inserts(harness: &IngestionHarness, table: &str) { + harness + .database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_insert() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN RAISE EXCEPTION 'test refuses this insert'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_insert() TO PUBLIC; + CREATE TRIGGER test_refuse_insert BEFORE INSERT ON registry_internal.{table} + FOR EACH ROW EXECUTE FUNCTION public.test_refuse_insert();" + )) + .await + .expect("administrator installs the insert refusal"); +} + +async fn allow_inserts(harness: &IngestionHarness, table: &str) { + harness + .database + .admin + .batch_execute(&format!( + "DROP TRIGGER test_refuse_insert ON registry_internal.{table}; + DROP FUNCTION public.test_refuse_insert();" + )) + .await + .expect("administrator removes the insert refusal"); +} + /// Cancellation is itself the run's last attempt, and it is not a chunk /// attempt: the metadata renders the refused outcome with no chunk index, /// so a cancelled run never reports an earlier chunk's index as its last. diff --git a/crates/registry-breg/tests/postgres_migration.rs b/crates/registry-breg/tests/postgres_migration.rs index b524ace52..0cf50cbea 100644 --- a/crates/registry-breg/tests/postgres_migration.rs +++ b/crates/registry-breg/tests/postgres_migration.rs @@ -1091,6 +1091,39 @@ async fn reconciliation_completes_a_target_the_catalog_already_reached() { assert_eq!(durable_snapshot(&database).await, before_refusal); database.audit_capture().restore(); + // A transition that fails after its request entry answers that entry + // with a failed response under the same correlation. + database + .admin + .batch_execute( + "BEGIN; SELECT 1 FROM registry_internal.registry_state WHERE singleton FOR UPDATE", + ) + .await + .expect("administrator holds the Registry state row"); + let entries_before = database.audit_entries().len(); + reconcile(&database, &package, &active, &base, true) + .await + .expect_err("the activation cannot take the held Registry state row"); + database + .admin + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the Registry state row"); + let failed = database.audit_entries().split_off(entries_before); + assert_eq!( + failed + .iter() + .map(|entry| ( + entry["phase"].as_str().expect("phase"), + entry["record"]["outcome"].as_str().expect("outcome") + )) + .collect::>(), + [("request", "started"), ("response", "failed")], + "{failed:?}" + ); + assert_eq!(failed[0]["correlation"], failed[1]["correlation"]); + assert!(!failed[1].to_string().contains(RECONCILE_OPERATOR_CANARY)); + let completed = reconcile(&database, &package, &active, &base, true) .await .expect("the missing activation transition completes"); @@ -4061,14 +4094,15 @@ async fn assert_reconcile_audit_is_minimized(database: &TestDatabase, action: &s matched.push(entry); } } - assert_eq!( - matched - .iter() - .map(|entry| entry["phase"].as_str().expect("phase is a string")) - .collect::>(), - ["request", "response"] - ); - assert_eq!(matched[0]["correlation"], matched[1]["correlation"]); + // Every execution is a request answered by one response under its + // correlation; the last one committed. + assert!(!matched.is_empty() && matched.len() % 2 == 0, "{matched:?}"); + for pair in matched.chunks(2) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); + } + assert_eq!(matched[matched.len() - 1]["record"]["outcome"], "committed"); } fn assert_value_free(actual: Option, expected: MigrationError) { diff --git a/crates/registry-breg/tests/postgres_read.rs b/crates/registry-breg/tests/postgres_read.rs index 2c854dd5c..edca9aecc 100644 --- a/crates/registry-breg/tests/postgres_read.rs +++ b/crates/registry-breg/tests/postgres_read.rs @@ -570,10 +570,35 @@ async fn real_postgres_read_is_authorized_bounded_minimized_and_audit_gated() { assert!(!faulted_body.to_string().contains("label-001")); assert_eq!( audit_count(&database).await, - before_fault + 1, - "a terminal audit fault releases no protected data and commits only the prior attempt" + before_fault + 2, + "a read ended before its terminal releases no protected data and answers its attempt \ + as unfinished" ); + // A failure after the rows were read, binding the strong entity tag, + // answers the attempt with the Refused terminal and releases nothing. + let before_etag_fault = audit_count(&database).await; + let etag_faulting_app = read_router( + pool.clone(), + compiled.clone(), + identity.clone(), + lock_key, + profile.clone(), + Some(ReadFaultPoint::StrongEtag), + ); + let etag_faulted = send( + &etag_faulting_app, + &format!("/v1/records/widgets/{VISIBLE_RECORD}?$select=label"), + Some(read_claims(["zone-a"])), + ) + .await; + assert_eq!(etag_faulted.status(), StatusCode::SERVICE_UNAVAILABLE); + assert!(!body_json(etag_faulted) + .await + .to_string() + .contains("label-001")); + assert_eq!(audit_count(&database).await, before_etag_fault + 2); + assert_read_audit_is_ordered_paired_and_minimized(&database, &compiled); // A restarted process over the recovered destination. A writer that // refused an append stays failed, so recovery is a fresh writer. @@ -1642,7 +1667,10 @@ fn assert_read_audit_is_ordered_paired_and_minimized( assert_eq!(entry["schema"], registry_breg::audit::AUDIT_SCHEMA); } for pair in entries.windows(2) { - if pair[0]["phase"] == "request" && pair[1]["record"]["phase"] == "terminal" { + if pair[0]["phase"] == "request" + && (pair[1]["record"]["phase"] == "terminal" + || pair[1]["record"]["phase"] == "unfinished") + { assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); } } @@ -1688,6 +1716,9 @@ fn assert_read_audit_is_ordered_paired_and_minimized( ("terminal", Some("empty")), ("refusal", None), ("attempt", None), + ("unfinished", None), + ("attempt", None), + ("terminal", Some("refused")), ], "durable read audit records bracket release in order" ); diff --git a/crates/registry-breg/tests/postgres_request_read_retention.rs b/crates/registry-breg/tests/postgres_request_read_retention.rs index dc29bdbc1..cfb15341d 100644 --- a/crates/registry-breg/tests/postgres_request_read_retention.rs +++ b/crates/registry-breg/tests/postgres_request_read_retention.rs @@ -762,6 +762,11 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it assert_eq!(phases, ["request", "response"]); assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); assert_eq!(entries[0]["record"]["phase"], "attempt"); + // The response is recorded after the external deletions ran, and states + // how many objects still wait for deletion. + assert_eq!(entries[1]["record"]["outcome"], "committed"); + assert_eq!(entries[1]["record"]["pendingExternalDeletions"], 0); + assert_eq!(entries[1]["record"]["externalDeletionTombstones"], 0); assert_eq!( entries[0]["record"]["recordReference"], entries[1]["record"]["recordReference"] @@ -773,6 +778,97 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it database.cleanup().await; } +/// An erasure whose transaction fails after its request entry answers that +/// entry with a failed response and erases nothing. One whose committed +/// erasure the destination refuses to record reports that distinctly: the +/// detail is gone, and the journal holds only the request. A recorded +/// erasure states the external objects still pending deletion. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn request_detail_erasure_pairs_its_request_entry_on_every_outcome() { + let database = TestDatabase::create(6).await; + let registry = Arc::new(compiled_registry()); + let identity = install_registry(&database, ®istry).await; + let app = request_router(&database, registry.clone(), identity.clone()); + let operator = claims("operator", "operator-principal"); + let request_id = applied_correction_request(&app, operator, "paired-erasure").await; + let request_uuid = Uuid::parse_str(&request_id).expect("request id parses"); + let scope = RequestDetailErasureScope { + request_entity_id: "correction-request", + request_id: request_uuid, + proposal_version: 1, + }; + let profile = || { + AuditProfile::production_from_secret_bytes(vec![0x8d; 32].into()) + .expect("test audit profile is keyed") + }; + let retention_with = |audit| { + RequestRetentionOperatorService::new_for_test( + registry.as_ref().clone(), + identity.clone(), + ExpectedManagedCatalog::compiled(®istry), + RegistryLockKey::derive(PACKAGE_ID).expect("lock key derives"), + database.migration_config.clone(), + database.migration_role.clone(), + database.runtime_role.clone(), + audit, + ) + }; + let (audit, capture) = registry_breg::audit::test_support::capturing(profile()); + + // The request row is held, so the erasure transaction cannot lock it. + database + .admin + .batch_execute(&format!( + "BEGIN; SELECT 1 FROM registry_internal.registry_request_state + WHERE request_id = '{request_uuid}' FOR UPDATE" + )) + .await + .expect("administrator holds the request row"); + retention_with(audit) + .erase(scope.clone()) + .await + .expect_err("the erasure cannot lock the held request"); + database + .admin + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the request row"); + let failed = capture.entries(); + assert_eq!(failed.len(), 2, "{failed:?}"); + assert_eq!(failed[0]["phase"], "request"); + assert_eq!(failed[1]["phase"], "response"); + assert_eq!(failed[1]["record"]["outcome"], "failed"); + assert_eq!(failed[0]["correlation"], failed[1]["correlation"]); + let retained = retention_with(capture.audit(profile())); + assert!( + !retained + .dry_run(scope.clone()) + .await + .expect("the detail still plans") + .detail_erased, + "the failed erasure erased nothing" + ); + + // The destination accepts the request entry and refuses the response. + capture.fail_after(1); + assert_eq!( + retained.erase(scope.clone()).await, + Err(RequestRetentionError::ErasureUnaudited) + ); + capture.restore(); + assert!( + retention_with(capture.audit(profile())) + .dry_run(scope.clone()) + .await + .expect("the erased detail still plans") + .detail_erased, + "the unaudited erasure committed" + ); + assert_eq!(capture.entries().len(), failed.len() + 1); + + database.cleanup().await; +} + /// Create, submit, and apply one correction request, returning its id. async fn applied_correction_request( app: &axum::Router, diff --git a/crates/registry-breg/tests/postgres_request_upgrade_retention.rs b/crates/registry-breg/tests/postgres_request_upgrade_retention.rs index 840849a47..b1885d5aa 100644 --- a/crates/registry-breg/tests/postgres_request_upgrade_retention.rs +++ b/crates/registry-breg/tests/postgres_request_upgrade_retention.rs @@ -1426,8 +1426,8 @@ async fn operator_retention_service_counts_pages_erases_under_forced_rls_and_aud let audit_entries = database.audit_entries(); assert_eq!( audit_entries.len(), - 5, - "the refused erasure records its request, and cleanup appends a request and a \ + 6, + "the refused erasure answers its request, and cleanup appends a request and a \ response entry independently of erasure eligibility" ); assert_eq!(audit_entries[2]["phase"], "request"); @@ -1435,14 +1435,20 @@ async fn operator_retention_service_counts_pages_erases_under_forced_rls_and_aud audit_entries[2]["correlation"], audit_entries[1]["correlation"], "each erasure invocation has its own correlation" ); - assert_eq!(audit_entries[3]["phase"], "request"); - assert_eq!(audit_entries[4]["phase"], "response"); + assert_eq!(audit_entries[3]["phase"], "response"); + assert_eq!(audit_entries[3]["record"]["outcome"], "refused"); assert_eq!( - audit_entries[3]["correlation"], audit_entries[4]["correlation"], + audit_entries[2]["correlation"], audit_entries[3]["correlation"], + "the refusal answers the erasure's request" + ); + assert_eq!(audit_entries[4]["phase"], "request"); + assert_eq!(audit_entries[5]["phase"], "response"); + assert_eq!( + audit_entries[4]["correlation"], audit_entries[5]["correlation"], "the cleanup response shares its request's correlation" ); assert_eq!( - audit_entries[3]["schema"], + audit_entries[4]["schema"], "breg-attachment-cleanup-audit/v1" ); diff --git a/crates/registry-bregctl/src/lib.rs b/crates/registry-bregctl/src/lib.rs index 51860a4c5..c8e2b900a 100644 --- a/crates/registry-bregctl/src/lib.rs +++ b/crates/registry-bregctl/src/lib.rs @@ -2160,6 +2160,10 @@ fn request_retention_failure( "request_retention.mode.retain", "the request retention policy does not permit operator erasure", ), + RequestRetentionCliError::ErasureUnaudited => ( + "request_retention.erasure.unaudited", + "the erasure committed but its audit entry was not recorded; restore the audit destination, then reconcile the erased request against the database", + ), RequestRetentionCliError::AttachmentStorageBindingMismatch => ( "request_retention.attachment_storage.binding_mismatch", "restore the original attachment storage binding and verification policy before retrying; the registry pin, retained content, or deletion tombstones still require them", diff --git a/crates/registry-bregctl/src/request_retention.rs b/crates/registry-bregctl/src/request_retention.rs index 3adb4d312..f6a906cd8 100644 --- a/crates/registry-bregctl/src/request_retention.rs +++ b/crates/registry-bregctl/src/request_retention.rs @@ -22,6 +22,8 @@ pub(crate) enum RequestRetentionCliError { ActiveDetailPinned, RetainMode, AttachmentStorageBindingMismatch, + /// The erasure committed without its audit entry. + ErasureUnaudited, } #[derive(Clone, Debug, Eq, PartialEq, Serialize)] @@ -151,6 +153,7 @@ fn map_error(error: RequestRetentionError) -> RequestRetentionCliError { RequestRetentionError::AttachmentStorageBindingMismatch => { RequestRetentionCliError::AttachmentStorageBindingMismatch } + RequestRetentionError::ErasureUnaudited => RequestRetentionCliError::ErasureUnaudited, RequestRetentionError::ActiveProposalRequiresRebase | RequestRetentionError::Unavailable => RequestRetentionCliError::Operator, } @@ -166,6 +169,14 @@ fn operator_runtime() -> Result Date: Sat, 26 Sep 2026 14:00:38 +0000 Subject: [PATCH 07/32] fix(hooks): record delivery outcomes only after they commit The delivery worker appended terminal and expiry entries inside the transaction it was about to commit, so a failed commit left an entry for a transition that never happened. Terminal dispositions and payload expiries are now recorded after their transaction commits. An attempt's request stays ahead of its lease commit, so it is on record before egress; if that commit fails the worker answers it with worker_interrupted. An operator replay is a replay_requested request before the reset and a replay_committed or replay_refused response after it. The seam no longer receives the transaction. This changes the shared seam, so Scheduling's hook audit follows the same order. Refs #1592 #1588 Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/webhook.rs | 19 +- .../tests/postgres_webhook_delivery.rs | 170 +++++++++++- .../src/delivery/seams.rs | 38 ++- .../src/delivery/service.rs | 256 ++++++++++++------ crates/registry-scheduling/src/hooks.rs | 38 +-- 5 files changed, 378 insertions(+), 143 deletions(-) diff --git a/crates/registry-breg/src/webhook.rs b/crates/registry-breg/src/webhook.rs index ae39b2e7f..1e5aca2b8 100644 --- a/crates/registry-breg/src/webhook.rs +++ b/crates/registry-breg/src/webhook.rs @@ -403,17 +403,12 @@ impl DeliverySeams for BregDeliverySeams { self.handlers.handler(binding) } - /// The platform worker calls this seam only after the guarded transition - /// the entry reports has already succeeded, immediately before it commits - /// the transaction: a refused append still rolls that transition back, - /// but a failed commit after an accepted append leaves an entry for a - /// transition that did not happen, since this durable append cannot be - /// rolled back with the transaction. - async fn record_audit( - &self, - _transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { + /// The platform worker records an attempt's start before its lease + /// commits, answering it with a worker interruption if that commit + /// fails, and records a terminal disposition, an expiry, and a replay's + /// outcome only after the transition commits, so an entry never stands + /// for a transition that rolled back. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { let entry = webhook_entry( self.audit.profile(), WebhookAudit { @@ -696,6 +691,8 @@ fn audit_outcome(outcome: DeliveryAuditOutcome) -> WebhookAuditOutcome { DeliveryAuditOutcome::PayloadExpired => WebhookAuditOutcome::PayloadExpired, DeliveryAuditOutcome::WorkerInterrupted => WebhookAuditOutcome::WorkerInterrupted, DeliveryAuditOutcome::ReplayRequested => WebhookAuditOutcome::ReplayRequested, + DeliveryAuditOutcome::ReplayCommitted => WebhookAuditOutcome::ReplayCommitted, + DeliveryAuditOutcome::ReplayRefused => WebhookAuditOutcome::ReplayRefused, DeliveryAuditOutcome::HandlerBindingRefused => WebhookAuditOutcome::HandlerBindingRefused, DeliveryAuditOutcome::HandlerDeadline => WebhookAuditOutcome::HandlerDeadline, DeliveryAuditOutcome::HandlerResource => WebhookAuditOutcome::HandlerResource, diff --git a/crates/registry-breg/tests/postgres_webhook_delivery.rs b/crates/registry-breg/tests/postgres_webhook_delivery.rs index 5aa91bc6f..a8cd3bdc3 100644 --- a/crates/registry-breg/tests/postgres_webhook_delivery.rs +++ b/crates/registry-breg/tests/postgres_webhook_delivery.rs @@ -384,16 +384,13 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun header(&replay_request, "idempotency-key"), "operator replay changes the deterministic generation binding" ); - assert_exact_audit_outcome( - &database, - &audit_profile, - &timeout_event, - 2, - 0, - "replay", - "replay_requested", - ) - .await; + // The replay is a request before its reset and a response once the + // reset commits, under one correlation. + assert_eq!( + audit_outcomes(&database, &audit_profile, &timeout_event, 2, 0, "replay").await, + ["replay_requested", "replay_committed"], + "an operator replay is answered once its reset commits" + ); assert_exact_audit_outcome( &database, &audit_profile, @@ -799,6 +796,100 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun "delivered", ) .await; + // A disposition whose commit fails is never recorded as done: the + // journal keeps only the attempt, and expiry recovery answers it. + let commit_egress_before = receiver.count().await; + receiver.enqueue(ResponsePlan::Status(204)).await; + let commit_refused = create_event( + &database, + &coordinator, + &mut mutation_client, + &plan, + &claims, + "delivery-terminal-commit-refused", + "terminal-commit", + ) + .await; + refuse_delivery_state_commit(&database, "delivered").await; + assert_eq!( + service.deliver_once().await, + Err(WebhookDeliveryError::Unavailable) + ); + allow_delivery_state_commit(&database).await; + receiver.wait_for_count(commit_egress_before + 1).await; + assert_eq!( + delivery_state(&database, &commit_refused).await, + (1, "leased".to_owned(), 1), + "the rolled-back disposition leaves the lease for expiry recovery" + ); + assert_no_audit_outcome(&database, &audit_profile, &commit_refused, 1, 1, "terminal").await; + expire_lease(&database, &commit_refused).await; + service + .deliver_once() + .await + .expect("expiry recovery answers the interrupted attempt"); + assert_exact_audit_outcome( + &database, + &audit_profile, + &commit_refused, + 1, + 1, + "terminal", + "worker_interrupted", + ) + .await; + + // A lease whose commit fails after its attempt was recorded sends + // nothing, and its attempt is answered as interrupted. + let lease_egress_before = receiver.count().await; + let lease_refused = create_event( + &database, + &coordinator, + &mut mutation_client, + &plan, + &claims, + "delivery-lease-commit-refused", + "lease-commit", + ) + .await; + refuse_delivery_state_commit(&database, "leased").await; + assert_eq!( + service.deliver_once().await, + Err(WebhookDeliveryError::Unavailable) + ); + allow_delivery_state_commit(&database).await; + assert_eq!(receiver.count().await, lease_egress_before, "no egress"); + assert_eq!( + delivery_state(&database, &lease_refused).await, + (1, "pending".to_owned(), 0) + ); + assert_exact_audit_outcome( + &database, + &audit_profile, + &lease_refused, + 1, + 1, + "attempt", + "attempt_started", + ) + .await; + assert_exact_audit_outcome( + &database, + &audit_profile, + &lease_refused, + 1, + 1, + "terminal", + "worker_interrupted", + ) + .await; + receiver.enqueue(ResponsePlan::Status(204)).await; + assert_eq!( + service.deliver_once().await, + Ok(WebhookWorkOutcome::Delivered), + "the refused lease is claimed again once the commit succeeds" + ); + let terminal_egress_before = receiver.count().await; let terminal_response_release = Arc::new(Notify::new()); @@ -828,10 +919,13 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun Err(WebhookDeliveryError::Unavailable) ); database.audit_capture().restore(); + // The terminal is recorded only after its disposition commits, so the + // refused entry leaves the committed disposition without it: the writer + // then refuses every later entry until the destination is repaired. assert_eq!( delivery_state(&database, &terminal_audit_refused).await, - (1, "leased".to_owned(), 1), - "terminal audit refusal leaves the committed lease for expiry recovery" + (1, "delivered".to_owned(), 1), + "the terminal entry follows the committed disposition" ); assert_no_audit_outcome( &database, @@ -852,6 +946,58 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun database.cleanup().await; } +/// Make the commit of any delivery transition into `state` fail, after +/// every statement in its transaction succeeded. +async fn refuse_delivery_state_commit(database: &TestDatabase, state: &str) { + database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_delivery_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.state = '{state}' THEN + RAISE EXCEPTION 'test refuses this delivery commit'; + END IF; + RETURN NEW; + END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_delivery_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_delivery_commit + AFTER UPDATE ON registry_internal.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_delivery_commit();" + )) + .await + .expect("administrator installs the commit refusal"); +} + +async fn allow_delivery_state_commit(database: &TestDatabase) { + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_delivery_commit + ON registry_internal.registry_webhook_delivery_state; + DROP FUNCTION public.test_refuse_delivery_commit();", + ) + .await + .expect("administrator removes the commit refusal"); +} + +async fn expire_lease(database: &TestDatabase, event: &CapturedEvent) { + database + .admin + .execute( + // Move the whole lease into the past, keeping its captured length. + "UPDATE registry_internal.registry_webhook_delivery_state + SET attempt_started_at = attempt_started_at + - (lease_expires_at - attempt_started_at) - interval '1 second', + lease_expires_at = attempt_started_at - interval '1 second' + WHERE event_id = $1 AND state = 'leased'", + &[&event.event_id], + ) + .await + .expect("administrator expires the lease"); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn real_postgres_webhook_delivery_finishes_prior_package_work_after_compatible_upgrade() { let receiver = HttpsReceiver::start().await; diff --git a/crates/registry-platform-hooks/src/delivery/seams.rs b/crates/registry-platform-hooks/src/delivery/seams.rs index a0db3f393..031a034f0 100644 --- a/crates/registry-platform-hooks/src/delivery/seams.rs +++ b/crates/registry-platform-hooks/src/delivery/seams.rs @@ -138,17 +138,27 @@ pub trait DeliverySeams: Send + Sync + 'static { ) -> Result, DeliveryError>; /// Record one neutral delivery-audit event in the product's audit - /// journal. The worker calls this only after every guarded transition the - /// event reports has already succeeded, immediately before it commits the - /// transaction: a failed commit after an accepted append still leaves an - /// entry for a transition that did not happen, since a durable append - /// cannot be rolled back with the transaction. Every audited occurrence - /// and every audited field of the moved worker arrives here. - async fn record_audit( - &self, - transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError>; + /// journal. Every audited occurrence and every audited field of the + /// moved worker arrives here, in an order that keeps the journal from + /// claiming more than the database committed: + /// + /// - An attempt's start is its request, recorded inside the lease + /// transaction immediately before that transaction commits, so it is + /// on record before the request can leave the process and a refusal + /// rolls the lease back. If that commit then fails, the worker records + /// the attempt's `WorkerInterrupted` terminal, so the request is + /// answered and no egress follows. + /// - A terminal disposition and a payload expiry are recorded only after + /// the transaction that made them commits, so an entry never stands + /// for a transition that rolled back. + /// - An operator replay is a request, `ReplayRequested`, recorded before + /// the reset, and a response recorded after it: `ReplayCommitted` once + /// the reset commits, `ReplayRefused` when it does not. + /// + /// A refused append after a commit leaves that committed transition + /// without its entry; the writer then refuses every later entry until + /// the operator repairs the destination, which readiness reports. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError>; /// Report one operational event through the product's vocabulary. fn operational_event(&self, event: DeliveryOperationalEvent); @@ -418,7 +428,13 @@ pub enum DeliveryAuditOutcome { PayloadRefused, PayloadExpired, WorkerInterrupted, + /// An operator asked for a dead-lettered delivery to be replayed: the + /// request of a replay. ReplayRequested, + /// The replay's reset committed, and the delivery is pending again. + ReplayCommitted, + /// The replay's reset did not commit; the delivery stays dead-lettered. + ReplayRefused, } impl DeliveryAuditOutcome { diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index c5f15aa58..5f103b999 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -61,6 +61,34 @@ pub struct DeliveryConfig { pub delivery_source: String, } +/// A delivery-audit event held by the worker until it is recorded, such as +/// one whose transition must commit first. +struct PendingAudit { + event_id: Uuid, + compiled_delivery_id: String, + package_revision: String, + generation: i64, + attempt: i16, + phase: DeliveryAuditPhase, + outcome: DeliveryAuditOutcome, + disposition: DeliveryAuditDisposition, +} + +impl PendingAudit { + fn record(&self) -> DeliveryAuditRecord<'_> { + DeliveryAuditRecord { + event_id: self.event_id, + compiled_delivery_id: &self.compiled_delivery_id, + package_revision: &self.package_revision, + generation: self.generation, + attempt: self.attempt, + phase: self.phase, + outcome: self.outcome, + disposition: self.disposition, + } + } +} + /// The delivery worker over one product's seams. #[derive(Clone)] pub struct DeliveryService { @@ -323,6 +351,61 @@ impl DeliveryService { { return Err(DeliveryError::Unavailable); } + let next_generation = generation + .checked_add(1) + .ok_or(DeliveryError::Unavailable)?; + // The replay's request is on record before the reset, and its + // response records whether the reset committed. + let replay = PendingAudit { + event_id, + compiled_delivery_id: compiled_delivery_id.to_owned(), + package_revision: package_revision.clone(), + generation: next_generation, + attempt: 0, + phase: DeliveryAuditPhase::Replay, + outcome: DeliveryAuditOutcome::ReplayRequested, + disposition: DeliveryAuditDisposition::ReplayPending, + }; + self.seams.record_audit(replay.record()).await?; + let reset = self + .reset_for_replay(transaction, event_id, compiled_delivery_id, generation) + .await; + let (outcome, disposition) = if reset.is_ok() { + ( + DeliveryAuditOutcome::ReplayCommitted, + DeliveryAuditDisposition::ReplayPending, + ) + } else { + ( + DeliveryAuditOutcome::ReplayRefused, + DeliveryAuditDisposition::DeadLettered, + ) + }; + let recorded = self + .seams + .record_audit( + PendingAudit { + outcome, + disposition, + ..replay + } + .record(), + ) + .await; + reset?; + recorded?; + Ok(next_generation) + } + + /// Reset one dead-lettered delivery to pending under its next generation + /// and commit. + async fn reset_for_replay( + &self, + transaction: Transaction<'_>, + event_id: Uuid, + compiled_delivery_id: &str, + generation: i64, + ) -> Result<(), DeliveryError> { let next_generation = generation .checked_add(1) .ok_or(DeliveryError::Unavailable)?; @@ -357,23 +440,8 @@ impl DeliveryService { if changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id, - package_revision: &package_revision, - generation: next_generation, - attempt: 0, - phase: DeliveryAuditPhase::Replay, - outcome: DeliveryAuditOutcome::ReplayRequested, - disposition: DeliveryAuditDisposition::ReplayPending, - }, - ) - .await?; transaction.commit().await?; - Ok(next_generation) + Ok(()) } async fn claim(&self) -> Result, DeliveryError> { @@ -383,11 +451,19 @@ impl DeliveryService { self.refused(DeliveryTransitionCode::ClaimIdentityRefused); return Err(DeliveryError::Unavailable); } - if self.reap_expired_leases(&transaction).await.is_err() { + // Recovered and expired deliveries are recorded once this + // transaction commits. + let mut committed = Vec::new(); + if self + .reap_expired_leases(&transaction, &mut committed) + .await + .is_err() + { self.refused(DeliveryTransitionCode::ClaimRecoveryFailed); return Err(DeliveryError::Unavailable); } - self.expire_retained_payload(&transaction).await?; + self.expire_retained_payload(&transaction, &mut committed) + .await?; let row = transaction .query_opt( &self.sql( @@ -422,6 +498,7 @@ impl DeliveryService { })?; let Some(row) = row else { transaction.commit().await?; + self.record_committed(committed).await?; return Ok(None); }; let event_id = row.try_get::<_, Uuid>(0)?; @@ -498,31 +575,34 @@ impl DeliveryService { .query_one("SELECT transaction_timestamp()", &[]) .await? .try_get::<_, SystemTime>(0)?; - if self - .seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Attempt, - outcome: DeliveryAuditOutcome::AttemptStarted, - disposition: DeliveryAuditDisposition::Leased, - }, - ) - .await - .is_err() - { + let started = PendingAudit { + event_id, + compiled_delivery_id: compiled_delivery_id.clone(), + package_revision: package_revision.clone(), + generation, + attempt, + phase: DeliveryAuditPhase::Attempt, + outcome: DeliveryAuditOutcome::AttemptStarted, + disposition: DeliveryAuditDisposition::Leased, + }; + if self.seams.record_audit(started.record()).await.is_err() { self.refused(DeliveryTransitionCode::ClaimAuditFailed); return Err(DeliveryError::Unavailable); } - transaction.commit().await.map_err(|_| { + if transaction.commit().await.is_err() { self.refused(DeliveryTransitionCode::ClaimCommitFailed); - DeliveryError::Unavailable - })?; + // The attempt's request is on record but its lease rolled back: + // answer it, so the journal shows the attempt never ran. + let interrupted = PendingAudit { + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: DeliveryAuditDisposition::RetryPending, + ..started + }; + let _ = self.seams.record_audit(interrupted.record()).await; + return Err(DeliveryError::Unavailable); + } + self.record_committed(committed).await?; Ok(Some(DeliveryClaim { event_id, compiled_delivery_id, @@ -540,6 +620,7 @@ impl DeliveryService { async fn reap_expired_leases( &self, transaction: &Transaction<'_>, + committed: &mut Vec, ) -> Result<(), DeliveryError> { let row = transaction .query_opt( @@ -672,31 +753,27 @@ impl DeliveryService { if changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Terminal, - outcome: DeliveryAuditOutcome::WorkerInterrupted, - disposition: if dead_lettered { - DeliveryAuditDisposition::DeadLettered - } else { - DeliveryAuditDisposition::RetryPending - }, - }, - ) - .await?; + committed.push(PendingAudit { + event_id, + compiled_delivery_id, + package_revision, + generation, + attempt, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: if dead_lettered { + DeliveryAuditDisposition::DeadLettered + } else { + DeliveryAuditDisposition::RetryPending + }, + }); Ok(()) } async fn expire_retained_payload( &self, transaction: &Transaction<'_>, + committed: &mut Vec, ) -> Result<(), DeliveryError> { let row = transaction .query_opt( @@ -762,21 +839,16 @@ impl DeliveryService { if state_changed != 1 || payload_changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Terminal, - outcome: DeliveryAuditOutcome::PayloadExpired, - disposition: DeliveryAuditDisposition::Expired, - }, - ) - .await?; + committed.push(PendingAudit { + event_id, + compiled_delivery_id, + package_revision, + generation, + attempt, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::PayloadExpired, + disposition: DeliveryAuditDisposition::Expired, + }); Ok(()) } @@ -1277,25 +1349,32 @@ impl DeliveryService { return Err(DeliveryError::Unavailable); } } + transaction.commit().await?; + // Recorded once the disposition committed, so the journal never + // names a disposition the database does not hold. self.seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id: claim.event_id, - compiled_delivery_id: &claim.compiled_delivery_id, - package_revision: &claim.package_revision, - generation: claim.generation, - attempt: claim.attempt, - phase: DeliveryAuditPhase::Terminal, - outcome, - disposition, - }, - ) + .record_audit(DeliveryAuditRecord { + event_id: claim.event_id, + compiled_delivery_id: &claim.compiled_delivery_id, + package_revision: &claim.package_revision, + generation: claim.generation, + attempt: claim.attempt, + phase: DeliveryAuditPhase::Terminal, + outcome, + disposition, + }) .await?; - transaction.commit().await?; Ok(work_outcome) } + /// Record the events of a transaction that has committed. + async fn record_committed(&self, committed: Vec) -> Result<(), DeliveryError> { + for event in committed { + self.seams.record_audit(event.record()).await?; + } + Ok(()) + } + async fn update_terminal_state( &self, transaction: &Transaction<'_>, @@ -1926,7 +2005,6 @@ mod tests { async fn record_audit( &self, - _transaction: &Transaction<'_>, _record: DeliveryAuditRecord<'_>, ) -> Result<(), DeliveryError> { Err(DeliveryError::Unavailable) @@ -2406,7 +2484,6 @@ mod tests { async fn record_audit( &self, - _transaction: &Transaction<'_>, _record: DeliveryAuditRecord<'_>, ) -> Result<(), DeliveryError> { Err(DeliveryError::Unavailable) @@ -2770,7 +2847,6 @@ mod tests { async fn record_audit( &self, - _transaction: &Transaction<'_>, _record: DeliveryAuditRecord<'_>, ) -> Result<(), DeliveryError> { *self.audit_calls.lock().expect("audit calls lock") += 1; diff --git a/crates/registry-scheduling/src/hooks.rs b/crates/registry-scheduling/src/hooks.rs index d8233b90c..a18db6ab4 100644 --- a/crates/registry-scheduling/src/hooks.rs +++ b/crates/registry-scheduling/src/hooks.rs @@ -760,23 +760,20 @@ impl DeliverySeams for SchedulingDeliverySeams { Ok(None) } - /// The platform worker calls this inside the transaction it is about to - /// commit, so the entry is written to the audit destination before that - /// commit: an attempt is on record before its request can leave the - /// process, and a destination that refuses it fails the transition, which - /// rolls back without egress. A transaction that rolls back after the - /// entry was accepted leaves an entry naming a transition that did not - /// commit, and a later pass that takes the transition writes its own. + /// The platform worker records an attempt's start inside the lease + /// transaction before it commits, so an attempt is on record before its + /// request can leave the process and a destination that refuses it rolls + /// the lease back without egress; a lease whose commit then fails is + /// answered with a worker interruption. A terminal disposition, an + /// expiry, and a replay's outcome are recorded only after the transition + /// commits, so no entry names a transition that rolled back. /// - /// An attempt's start is its `request` entry; its terminal disposition - /// and an operator replay are `response` entries. One attempt's entries - /// share a correlation built from the hook event, the compiled delivery, - /// the generation, and the attempt. - async fn record_audit( - &self, - _transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { + /// An attempt's start and an operator's replay request are `request` + /// entries; the terminal disposition and the replay's committed or + /// refused reset are `response` entries. One attempt's entries share a + /// correlation built from the hook event, the compiled delivery, the + /// generation, and the attempt. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { let audit = json!({ "event": "scheduling.hook-delivery", "hookEventId": record.event_id, @@ -792,11 +789,12 @@ impl DeliverySeams for SchedulingDeliverySeams { "{}/{}/{}/{}", record.event_id, record.compiled_delivery_id, record.generation, record.attempt ); - let entry = match record.phase { - DeliveryAuditPhase::Attempt => { + let entry = match (record.phase, record.outcome) { + (DeliveryAuditPhase::Attempt, _) + | (DeliveryAuditPhase::Replay, DeliveryAuditOutcome::ReplayRequested) => { AuditEntry::request(SCHEDULING_AUDIT_SCHEMA, correlation, audit) } - DeliveryAuditPhase::Terminal | DeliveryAuditPhase::Replay => { + (DeliveryAuditPhase::Terminal | DeliveryAuditPhase::Replay, _) => { AuditEntry::response(SCHEDULING_AUDIT_SCHEMA, correlation, audit) } }; @@ -1089,6 +1087,8 @@ fn audit_outcome(value: DeliveryAuditOutcome) -> &'static str { DeliveryAuditOutcome::PayloadExpired => "payload_expired", DeliveryAuditOutcome::WorkerInterrupted => "worker_interrupted", DeliveryAuditOutcome::ReplayRequested => "replay_requested", + DeliveryAuditOutcome::ReplayCommitted => "replay_committed", + DeliveryAuditOutcome::ReplayRefused => "replay_refused", } } From 368ce72a8255e02ce2a6df9920e9829ebf426423 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:00:38 +0000 Subject: [PATCH 08/32] docs: describe the audit request pairing guarantee Signed-off-by: Jeremi Joslin --- .../content/docs/operate/breg-retention.mdx | 16 +++++++++- products/breg/CHANGELOG.md | 30 +++++++++++++++++++ products/casework/CHANGELOG.md | 7 +++++ products/platform/CHANGELOG.md | 8 +++++ 4 files changed, 60 insertions(+), 1 deletion(-) diff --git a/docs/site/src/content/docs/operate/breg-retention.mdx b/docs/site/src/content/docs/operate/breg-retention.mdx index db60c26b2..6b716fbca 100644 --- a/docs/site/src/content/docs/operate/breg-retention.mdx +++ b/docs/site/src/content/docs/operate/breg-retention.mdx @@ -312,7 +312,21 @@ entry before disclosure or after a mutation commits. The entries share a correla keyed references and closed vocabulary, not raw record values. A refusal can have only a response entry. The file destination acknowledges an append after fsync; `stdout` flushes each line and provides best-effort delivery. A failed request entry blocks protected I/O, and a failed response -entry blocks disclosure. A crash can leave a request entry without a response. +entry blocks disclosure. A request that ends before its outcome is known, because it failed, timed +out, or its caller went away, or an operator command that exits early, still answers its request +entry with a response whose outcome is `unfinished`. Only a crash, or a destination that already +stopped accepting entries, can leave a request entry without a response. + +Operator commands follow the same rule. `evidence-retention erase-expired` records the cutoff it +was given in its request entry and the number of assertions it erased in its response. +`request-retention erase` deletes the external attachment objects before it records its response, +which states how many objects still wait for deletion. If the erasure committed but its response +entry could not be written, the command reports `request_retention.erasure.unaudited`: the detail is +gone, so restore the audit destination and reconcile the erased request against the database. An +event delivery's attempt is recorded before its request leaves, and its outcome only once the +delivery's state has committed, so the journal never records a delivery that rolled back; an +operator replay is recorded as a request before the reset and a response once it commits or is +refused. A crash or a killed process can also tear the file's final line itself, leaving it without its closing newline. The writer refuses to open a destination in that state, so the next start of the diff --git a/products/breg/CHANGELOG.md b/products/breg/CHANGELOG.md index 8fee4f0f5..6e426bb1f 100644 --- a/products/breg/CHANGELOG.md +++ b/products/breg/CHANGELOG.md @@ -2,6 +2,36 @@ ## Unreleased +- Answer every audited request entry. A read, mutation, action, or request + action that ends after its attempt without a terminal or refusal entry, + because it failed, timed out, or its caller went away, writes a response + with the phase `unfinished` under the same correlation. A read that fails + after its rows were read writes the Refused terminal, and a terminal the + destination refuses is logged instead of discarded. + - A reviewed change-request apply records its attempt before the receipt + preflight's reads and the review authority, and holds it through the + action. + - `migration reconcile` answers a transition that fails after its request + entry with a `failed` response. + - `request-retention erase` answers a refused or failed erasure with a + `refused` or `failed` response, deletes the external attachment objects + before it records a committed erasure, and records how many objects still + wait for deletion. An erasure that committed without its response entry + reports `request_retention.erasure.unaudited`. `request-retention + cleanup-attachments` answers a failed cleanup with a `failed` response. + - `evidence-retention erase-expired` is audited under + `breg-evidence-retention-audit/v1`: a request entry naming the cutoff + before the erasure, and a response with the erased count or `failed`. + - An ingestion run creation, cancellation, chunk replay, or receipt + recovery refused after its `breg-ingestion-audit/v1` request entry is + answered in that schema with a `refused` response, and not recorded again + as a general refusal. + - Event delivery records a terminal outcome and a payload expiry only after + the delivery state commits. An attempt whose lease commit fails is + answered with `worker_interrupted`. An operator replay writes a + `replay_requested` request before the reset and a `replay_committed` or + `replay_refused` response after it. + - BREAKING: write audit through the platform audit writer instead of a hash-chained journal in PostgreSQL. Each process opens one writer at startup and writes JSON Lines entries `{schema, correlation, phase, time, diff --git a/products/casework/CHANGELOG.md b/products/casework/CHANGELOG.md index 3e60a2d3b..d88db33c9 100644 --- a/products/casework/CHANGELOG.md +++ b/products/casework/CHANGELOG.md @@ -2,6 +2,13 @@ ## Unreleased +- Answer every audited request entry. An operation that ends after its + request entry without committing, a refusal, a failure, or a canceled + request, writes `{event, outcome: "unfinished"}` as its response under the + same correlation. Adding a review note is audited as + `casework.review_note_added`, naming the note's history event but never its + text or audience. + - BREAKING: write audit through the shared platform audit writer instead of a hash-chained journal published from a PostgreSQL outbox. - The `audit` block takes `hashKeyRef`, `destination` (`file`, the diff --git a/products/platform/CHANGELOG.md b/products/platform/CHANGELOG.md index 1a20e0186..1864e979a 100644 --- a/products/platform/CHANGELOG.md +++ b/products/platform/CHANGELOG.md @@ -2,6 +2,14 @@ ## Unreleased +- Add `AuditWriter::begin`, which appends a `request` entry and returns an + `AuditRequest` that owes its `response`. A response the handle writes, or + one appended under the same schema and correlation, answers it; a handle + dropped unanswered, by an early return, a panic, or a canceled future, + writes the product's `unfinished` record as the response, so no request + entry stays unpaired. A file destination flushes that line on the runtime, + or when the last reference to the writer is dropped at shutdown. + - Refuse to reopen an audit file ending in an incomplete JSONL entry, preserving its bytes for operator archival before starting a fresh stream. - Keep queued stream appends stopped after an earlier write fails, and finish From 94e761c21f3b9938fe8e15279b4d057d3791f75b Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:13:46 +0000 Subject: [PATCH 09/32] test(breg): expect a faulted mutation to answer its attempt A mutation that stops after its attempt now answers it with an unfinished response, so each fault leaves two audit entries. Refs #1593 Signed-off-by: Jeremi Joslin --- crates/registry-breg/tests/postgres_mutation.rs | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/crates/registry-breg/tests/postgres_mutation.rs b/crates/registry-breg/tests/postgres_mutation.rs index 4a34c7345..36831bed3 100644 --- a/crates/registry-breg/tests/postgres_mutation.rs +++ b/crates/registry-breg/tests/postgres_mutation.rs @@ -304,10 +304,10 @@ async fn real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable() assert_eq!( durable_counts(&database, table).await, DurableCounts { - audit: before.audit + 1, + audit: before.audit + 2, ..before }, - "fault {fault:?} retains only its unavoidable durable attempt" + "fault {fault:?} retains only its attempt and the unfinished answer to it" ); } @@ -2257,10 +2257,11 @@ async fn real_postgres_http_mutations_are_guarded_and_exactly_replayable() { assert_eq!( durable_counts(&database, &table).await, DurableCounts { - audit: before_fault.audit + 1, + audit: before_fault.audit + 2, ..before_fault }, - "terminal audit failure releases no success bytes and commits no mutation packet" + "terminal audit failure releases no success bytes, commits no mutation packet, and \ + answers its attempt as unfinished" ); assert_journals_are_minimized_and_paired(&database).await; From 747a32f819c1f7a7b40e1f2a03d59cdebf44e442 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:25:48 +0000 Subject: [PATCH 10/32] test(breg): expect faulted reads and tombstones to answer their attempt A read or tombstone that stops after its attempt now answers it with an unfinished response, so each fault leaves the attempt and its answer. Refs #1593 #1507 Signed-off-by: Jeremi Joslin --- crates/registry-breg/tests/postgres_revision_http.rs | 10 +++++++--- crates/registry-breg/tests/postgres_spatial_read.rs | 10 ++++++---- .../registry-breg/tests/postgres_tombstone_revision.rs | 3 ++- 3 files changed, 15 insertions(+), 8 deletions(-) diff --git a/crates/registry-breg/tests/postgres_revision_http.rs b/crates/registry-breg/tests/postgres_revision_http.rs index a05d395ff..ffa1d87fc 100644 --- a/crates/registry-breg/tests/postgres_revision_http.rs +++ b/crates/registry-breg/tests/postgres_revision_http.rs @@ -301,8 +301,9 @@ async fn real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gate assert_eq!(body_json(faulted).await["code"], "source.unavailable"); assert_eq!( audit_count(&database).await, - before_fault + 1, - "terminal audit gate failure releases no held revision and leaves only the attempt" + before_fault + 2, + "terminal audit gate failure releases no held revision and answers the attempt as \ + unfinished" ); let unkeyed = revision_router( @@ -779,7 +780,10 @@ async fn assert_revision_audit_is_ordered_and_minimized( && window[1]["phase"] == "terminal" && window[1]["outcome"] == "returned" })); - assert_eq!(records.last().expect("fault attempt")["phase"], "attempt"); + // The faulted read's attempt is answered as unfinished. + let fault_answer = records.len() - 1; + assert_eq!(records[fault_answer]["phase"], "unfinished"); + assert_eq!(records[fault_answer - 1]["phase"], "attempt"); assert!(records.iter().any(|record| record["phase"] == "refusal")); assert!(records.iter().any(|record| { record["phase"] == "terminal" diff --git a/crates/registry-breg/tests/postgres_spatial_read.rs b/crates/registry-breg/tests/postgres_spatial_read.rs index 7ccf1c1b1..80fa66f51 100644 --- a/crates/registry-breg/tests/postgres_spatial_read.rs +++ b/crates/registry-breg/tests/postgres_spatial_read.rs @@ -613,8 +613,9 @@ async fn real_postgres_spatial_bbox_reads_preserve_authority_and_geojson_audit() assert!(!faulted.to_string().contains("edge-west")); assert_eq!( audit_count(&harness.database).await, - before_fault + 1, - "terminal audit failure releases no held GeoJSON bytes and commits only the attempt" + before_fault + 2, + "terminal audit failure releases no held GeoJSON bytes and answers the attempt as \ + unfinished" ); let before_adapter_fault = audit_count(&harness.database).await; @@ -631,8 +632,9 @@ async fn real_postgres_spatial_bbox_reads_preserve_authority_and_geojson_audit() assert!(!adapter_faulted.to_string().contains("zero-area")); assert_eq!( audit_count(&harness.database).await, - before_adapter_fault + 1, - "GIS adapter terminal audit failure releases no held GeoJSON bytes" + before_adapter_fault + 2, + "GIS adapter terminal audit failure releases no held GeoJSON bytes and answers the \ + attempt as unfinished" ); assert_pool_context_clean(&harness.pool, &harness.database.runtime_role).await; diff --git a/crates/registry-breg/tests/postgres_tombstone_revision.rs b/crates/registry-breg/tests/postgres_tombstone_revision.rs index 14801cfa9..a010bd38d 100644 --- a/crates/registry-breg/tests/postgres_tombstone_revision.rs +++ b/crates/registry-breg/tests/postgres_tombstone_revision.rs @@ -337,7 +337,8 @@ async fn tombstone_refusals_faults_and_concurrency_have_no_duplicate_effects() { assert_eq!( durable_counts(&fixture.database, &fixture.table).await, DurableCounts { - audit: before.audit + 1, + // The attempt and the unfinished answer to it. + audit: before.audit + 2, ..before } ); From 8c05a59aed3b3b942eab30f5cdee61519a3e0797 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:33:27 +0000 Subject: [PATCH 11/32] docs(breg): enforce BREG-V1-27 Every caller-requested operation now answers its audit request entry, so the row returns to enforced with the fault-injection tests that prove the paths its gap listed. Signed-off-by: Jeremi Joslin --- products/breg/DEFINITION-OF-DONE.md | 5 ----- products/breg/contracts/definition-of-done.yaml | 2 +- 2 files changed, 1 insertion(+), 6 deletions(-) diff --git a/products/breg/DEFINITION-OF-DONE.md b/products/breg/DEFINITION-OF-DONE.md index 5ab044e6c..f2d7ae05e 100644 --- a/products/breg/DEFINITION-OF-DONE.md +++ b/products/breg/DEFINITION-OF-DONE.md @@ -86,11 +86,6 @@ rebaseline, or reconciliation finds the state the first run committed rather than replaying its terminal entry. Both recoveries are operational, not a second audit mechanism. -`BREG-V1-27` is `partial`: most caller-requested operations pair their -audit request entry with a response entry, but the paths its `gap` lists can -still leave a request entry unpaired or write no entry, so the row makes no -completion claim until they do. - The HTTP record contract is also explicit: caller-filtered and generated OpenAPI artifacts assign every record-related route to the shared single or collection Registry Record profile, or to a named BReg-specific shape. diff --git a/products/breg/contracts/definition-of-done.yaml b/products/breg/contracts/definition-of-done.yaml index 13bd82829..8026eab42 100644 --- a/products/breg/contracts/definition-of-done.yaml +++ b/products/breg/contracts/definition-of-done.yaml @@ -48,7 +48,7 @@ requirements: - {id: BREG-V1-24, phase: W3, state: enforced, doneWhen: "Application authorization and RLS agree for positive, negative, malformed, and pooled authority.", journeys: [BREG-J05, BREG-J11], evidence: [{path: crates/registry-breg/tests/postgres_compiled_schema.rs, name: compiled_postgres_schema_enforces_context_rls_and_exact_catalog}, {path: crates/registry-breg/tests/postgres_kernel.rs, name: real_postgres_kernel_proves_roles_rls_interlock_and_pool_isolation}, {path: crates/registry-breg/tests/postgres_pilot_acceptance.rs, name: real_postgres_five_domain_pilot_is_configured_production_closed_and_source_neutral}]} - {id: BREG-V1-25, phase: W3, state: enforced, doneWhen: "Provenance stays distinct from minimized value-free audit, logs, metrics, and traces.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_tombstone_revision.rs, name: tombstone_revisions_survive_package_upgrade_and_replay_exactly}, {path: crates/registry-breg/tests/startup_http.rs, name: operational_log_level_is_a_closed_vocabulary}, {path: crates/registry-breg/tests/startup_http.rs, name: every_operational_event_renders_exact_closed_value_free_json_fields}, {path: crates/registry-breg/tests/startup_http.rs, name: provenance_operational_logs_metrics_and_traces_are_separate_closed_and_value_free}]} - {id: BREG-V1-26, phase: W3, state: enforced, doneWhen: "Exact media-specific Registry Record bytes release or replay only after successful attempt and terminal audit gates.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_http_mutations_are_guarded_and_exactly_replayable}]} - - {id: BREG-V1-27, phase: W3, state: partial, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], gap: "Some refusal and failure paths still return after the request entry without a paired response entry: migration reconciliation (#1592), reviewed-apply preflight reads before the request entry (#1587), ingestion refusals closed in the general schema (#1597), read post-processing failures (#1593), and evidence-retention erasure, which writes no entry (#1590).", evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}]} + - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}]} - {id: BREG-V1-28, phase: W4, state: enforced, doneWhen: "Production packages capture the governed closure and sign exact canonical bytes with monotonic identity.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_builder_is_deterministic_and_local_publication_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: production_package_requires_exact_trust_anchor_threshold_and_signature}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_layout_contract_conditional_manifest_projection_is_in_projected_closure}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_omits_manifest_projection_from_signed_closure_and_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_refuses_claimed_manifest_artifacts}]} - {id: BREG-V1-29, phase: W4, state: enforced, doneWhen: "Activation verifies trust, identity, inventory, filesystem safety, artifacts, and schema before readiness.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_binding_refuses_wrong_environment_instance_database_sequence_and_prior}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_refuses_symlinks_and_production_writable_permissions}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_manifest_refuses_ddl_checksum_path_and_canonical_json_tampering}, {path: crates/registry-breg/tests/postgres_package.rs, name: signed_schema_fingerprint_mismatch_is_durably_failed_and_never_ready}]} - {id: BREG-V1-30, phase: W4, state: enforced, doneWhen: "Apply retains the lock through migrations, catalog verification, activation, and maintenance clearing.", journeys: [BREG-J13, BREG-J15], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: real_postgres_package_startup_apply_failure_and_old_process_are_closed}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_backfill_and_destructive_recovery_are_bounded_resumable_and_activation_closed}]} From 68157f5ff13c80f5a93790bc8b1bcd4032104078 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 14:40:04 +0000 Subject: [PATCH 12/32] docs: anchor the audit pairing claims in the retention page Signed-off-by: Jeremi Joslin --- docs/site/src/content/docs/operate/breg-retention.mdx | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/site/src/content/docs/operate/breg-retention.mdx b/docs/site/src/content/docs/operate/breg-retention.mdx index 6b716fbca..9abc401a4 100644 --- a/docs/site/src/content/docs/operate/breg-retention.mdx +++ b/docs/site/src/content/docs/operate/breg-retention.mdx @@ -375,7 +375,10 @@ outside the new writer's retention directory. {/* Evidence: crates/registry-breg/src/audit.rs; crates/registry-breg/src/runtime_config.rs, AuditConfig; - crates/registry-platform-audit/src/writer.rs; + crates/registry-platform-audit/src/writer.rs, AuditRequest; + crates/registry-breg/src/request_retention.rs, ErasureUnaudited; + crates/registry-breg/src/action_evidence_maintenance.rs, EVIDENCE_RETENTION_AUDIT_SCHEMA; + crates/registry-platform-hooks/src/delivery/seams.rs, record_audit; crates/registry-breg/src/postgres/schema.rs; crates/registry-bregctl/src/dev/mod.rs. */} From d7161db394fe3d70e5f93796086f80edee6e8570 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 16:06:24 +0000 Subject: [PATCH 13/32] fix(breg): answer maintenance request entries that end early History rebaseline, a standalone history erasure, and a field-encryption erase-and-rebaseline run appended their request entry and then returned through refusals and database errors without a response. They now hold the shared request handle through the run, so an early end writes an unfinished response under the same correlation. The attachment verification worker holds its attempt the same way, which also pairs a job its time budget cancels. Refs #1592 Signed-off-by: Jeremi Joslin --- .../src/attachment_verification_worker.rs | 43 ++++++++++++++++--- .../src/field_encryption_backfill.rs | 22 ++++++---- crates/registry-breg/src/history_erasure.rs | 18 ++++---- .../registry-breg/src/history_maintenance.rs | 22 +++++++++- .../registry-breg/src/history_rebaseline.rs | 11 ++--- .../tests/postgres_history_erasure.rs | 9 ++++ .../tests/postgres_history_rebaseline.rs | 17 ++++++-- products/breg/CHANGELOG.md | 4 ++ .../breg/contracts/definition-of-done.yaml | 2 +- 9 files changed, 111 insertions(+), 37 deletions(-) diff --git a/crates/registry-breg/src/attachment_verification_worker.rs b/crates/registry-breg/src/attachment_verification_worker.rs index b2f0fb855..0d1db164c 100644 --- a/crates/registry-breg/src/attachment_verification_worker.rs +++ b/crates/registry-breg/src/attachment_verification_worker.rs @@ -113,7 +113,9 @@ impl AttachmentVerificationWorker { }; transaction.commit().await.map_err(unavailable)?; drop(client); - self.audit_job(&job, "attempt", "started").await?; + // Held until the terminal entry answers it; a run that ends first, + // including one its time budget cancels, answers it as unfinished. + let _attempt = self.begin_job(&job).await?; let verdict = match self.content(&job).await { Ok(bytes) => verifier @@ -240,10 +242,42 @@ impl AttachmentVerificationWorker { Ok(transaction) } + /// Append the attempt `request` entry of one leased job and return the + /// handle that owes its terminal `response`. + async fn begin_job( + &self, + job: &VerificationJob, + ) -> Result { + let (reference, record) = self.job_record(job, "attempt", "started")?; + let (_, unfinished) = self.job_record(job, "terminal", "unfinished")?; + self.audit + .begin( + AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record), + unfinished, + ) + .await + .map_err(unavailable) + } + /// Append one verification entry. The attempt is the `request` entry and /// the terminal outcome is the `response` entry; both are correlated by /// the keyed verification reference of the leased job. async fn audit_job(&self, job: &VerificationJob, phase: &str, outcome: &str) -> Result<()> { + let (reference, record) = self.job_record(job, phase, outcome)?; + let entry = if phase == "attempt" { + AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) + } else { + AuditEntry::response(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) + }; + self.audit.append(entry).await.map_err(unavailable) + } + + fn job_record( + &self, + job: &VerificationJob, + phase: &str, + outcome: &str, + ) -> Result<(String, serde_json::Value)> { let hasher = self.audit.profile().key_hasher(); let reference = hasher .audit_reference_hash( @@ -257,12 +291,7 @@ impl AttachmentVerificationWorker { "packageRevision": self.expected.package_revision, "actor": "breg:attachment-verifier", "verificationReference": reference, }); - let entry = if phase == "attempt" { - AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) - } else { - AuditEntry::response(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) - }; - self.audit.append(entry).await.map_err(unavailable) + Ok((reference, record)) } } diff --git a/crates/registry-breg/src/field_encryption_backfill.rs b/crates/registry-breg/src/field_encryption_backfill.rs index a0ca647f3..73a1663da 100644 --- a/crates/registry-breg/src/field_encryption_backfill.rs +++ b/crates/registry-breg/src/field_encryption_backfill.rs @@ -34,8 +34,8 @@ use crate::history_erasure::{ RecordHistoryErasureTarget, MAX_ERASURE_REVISIONS, }; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceTimeouts, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceTimeouts, }; use crate::history_rebaseline::{ history_rebaseline_entry, rebaseline_history_coverage_in_transaction, HistoryRebaselineError, @@ -588,8 +588,11 @@ pub async fn erase_field_encryption_history( // copies that still carry plaintext for a recorded erase-and-rebaseline // flip. The transaction also marks coverage incomplete, which is the // existing durable retry signal if this lifecycle stops before rebaseline. + // The lifecycle's request entry, when this run is one, is held until the + // run ends, so a run that stops before its terminal still answers it. + let mut lifecycle_attempt = None; let (_, _, needs_rebaseline, lifecycle_reference) = - scrub_plaintext_request_snapshots(client, &request).await?; + scrub_plaintext_request_snapshots(client, &request, &mut lifecycle_attempt).await?; let mut erased_any = false; loop { let targets = pending_erase_targets(client, &request).await?; @@ -702,6 +705,7 @@ pub async fn erase_field_encryption_history( async fn scrub_plaintext_request_snapshots( client: &mut Client, request: &FieldEncryptionHistoryErasureRequest<'_>, + lifecycle_attempt: &mut Option, ) -> Result<(u64, u64, bool, String), FieldEncryptionHistoryErasureError> { let transaction = client .transaction() @@ -754,11 +758,13 @@ async fn scrub_plaintext_request_snapshots( // A recorded erase-and-rebaseline flip makes this a lifecycle run: its // request entry must be accepted before any request snapshot, record // history, or coverage state is read for erasure or changed. - append_maintenance_entries( - request.audit, - vec![lifecycle_request_entry(request, &lifecycle_reference)?], - ) - .await?; + *lifecycle_attempt = Some( + begin_maintenance_request( + request.audit, + lifecycle_request_entry(request, &lifecycle_reference)?, + ) + .await?, + ); let (terminal_exists, correlated_progress_exists) = lifecycle_progress_state(&transaction, &lifecycle_reference).await?; let unresolved_provenance: bool = transaction diff --git a/crates/registry-breg/src/history_erasure.rs b/crates/registry-breg/src/history_erasure.rs index 859d6d2b7..a2d769869 100644 --- a/crates/registry-breg/src/history_erasure.rs +++ b/crates/registry-breg/src/history_erasure.rs @@ -23,8 +23,8 @@ use uuid::Uuid; use crate::audit::RegistryAudit; use crate::history_commit::{lock_history_head, HistoryCommitError}; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceError, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceError, }; use crate::idempotency::{tombstone_erased_cached_responses, IdempotencyError}; use crate::postgres::{ @@ -203,13 +203,13 @@ async fn erase_record_history_scoped( // A standalone erasure's request entry is accepted before its // transaction opens, so an audit outage erases nothing. A lifecycle // erasure runs under the request entry its parent lifecycle appended. - if lifecycle_reference.is_none() { - append_maintenance_entries( - request.audit, - vec![history_erasure_request_entry(&request)?], - ) - .await?; - } + let _attempt = match lifecycle_reference { + None => Some( + begin_maintenance_request(request.audit, history_erasure_request_entry(&request)?) + .await?, + ), + Some(_) => None, + }; let transaction = client .transaction() diff --git a/crates/registry-breg/src/history_maintenance.rs b/crates/registry-breg/src/history_maintenance.rs index a9a6d7858..41db26f4a 100644 --- a/crates/registry-breg/src/history_maintenance.rs +++ b/crates/registry-breg/src/history_maintenance.rs @@ -15,7 +15,7 @@ use std::time::Duration; -use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile}; +use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile, AuditRequest}; use crate::audit::RegistryAudit; use crate::postgres::{ExpectedRegistryIdentity, PostgresKernelError}; @@ -136,6 +136,26 @@ pub(crate) fn profile_is_keyed(profile: &AuditProfile) -> bool { matches!(profile.key_hasher(), AuditKeyHasher::Keyed(_)) } +/// Append a maintenance `request` entry before its transaction opens and +/// return the handle that owes its response. The terminal entry appended +/// after commit, under the same schema and correlation, answers it; a +/// maintenance run that returns first writes the request's fields with the +/// `unfinished` outcome when the handle is dropped. +pub(crate) async fn begin_maintenance_request( + audit: &RegistryAudit, + entry: AuditEntry, +) -> Result { + let mut unfinished = entry.record().clone(); + if let Some(fields) = unfinished.as_object_mut() { + fields.insert("phase".to_owned(), "terminal".into()); + fields.insert("outcome".to_owned(), "unfinished".into()); + } + audit + .begin(entry, unfinished) + .await + .map_err(|_| HistoryMaintenanceError::Unavailable) +} + /// Append maintenance entries after the transaction that made their change /// durable has committed. A refused entry reports the maintenance path /// unavailable even though its change already committed: the operator sees diff --git a/crates/registry-breg/src/history_rebaseline.rs b/crates/registry-breg/src/history_rebaseline.rs index 616042094..d74fd404d 100644 --- a/crates/registry-breg/src/history_rebaseline.rs +++ b/crates/registry-breg/src/history_rebaseline.rs @@ -24,8 +24,8 @@ use crate::history_commit::{ allocate_coverage_baseline_commit, lock_history_head, HistoryCommitError, }; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceError, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceError, }; use crate::history_migration::{verify_live_rows_match_journal_heads, HistoryMigrationError}; use crate::model::CompiledRegistry; @@ -156,12 +156,9 @@ pub async fn rebaseline_history_coverage( // The baseline position is known only once the transaction allocates it, // so one invocation correlates its two entries by a fresh identifier. let correlation = Uuid::new_v4().to_string(); - append_maintenance_entries( + let _attempt = begin_maintenance_request( request.audit, - vec![history_rebaseline_request_entry( - &request, - correlation.clone(), - )?], + history_rebaseline_request_entry(&request, correlation.clone())?, ) .await?; diff --git a/crates/registry-breg/tests/postgres_history_erasure.rs b/crates/registry-breg/tests/postgres_history_erasure.rs index fba965239..19990fea9 100644 --- a/crates/registry-breg/tests/postgres_history_erasure.rs +++ b/crates/registry-breg/tests/postgres_history_erasure.rs @@ -1092,6 +1092,14 @@ async fn field_encryption_erasure_uses_flip_provenance_for_structured_plaintext( .expect("completed lifecycle replay state resolves"); assert!(replay_state.get::<_, bool>(0)); assert_eq!(field_encryption_terminal_entries(&database).len(), 1); + // The refused replay still answers the lifecycle request it wrote. + let last = database + .audit_entries() + .pop() + .expect("the replay's entries"); + assert_eq!(last["schema"], FIELD_ENCRYPTION_AUDIT_SCHEMA); + assert_eq!(last["phase"], "response"); + assert_eq!(last["record"]["outcome"], "unfinished"); migration_task.abort(); database.cleanup().await; @@ -2366,6 +2374,7 @@ fn field_encryption_terminal_entries(database: &TestDatabase) -> Vec>(); assert_eq!( rebaseline_entries, - ["request"], - "a refused rebaseline records its request and no committed response" + ["request", "response"], + "a refused rebaseline answers its request without a committed response" ); + let answer = database + .audit_entries() + .into_iter() + .rfind(|entry| entry["schema"] == HISTORY_REBASELINE_AUDIT_SCHEMA) + .expect("the refused rebaseline's answer"); + assert_eq!(answer["record"]["outcome"], "unfinished"); migration_task.abort(); database.cleanup().await; @@ -852,8 +858,8 @@ fn assert_rebaseline_audit_is_minimized(database: &TestDatabase) { }) .collect::>(); // The erasure, the completed rebaseline, and the second rebaseline that - // had nothing to do each record their request; the two that committed - // record their response under the same correlation. + // had nothing to do each record their request and a response under the + // same correlation; the one that had nothing to do answers unfinished. assert_eq!( shape, [ @@ -862,10 +868,13 @@ fn assert_rebaseline_audit_is_minimized(database: &TestDatabase) { (HISTORY_REBASELINE_AUDIT_SCHEMA, "request"), (HISTORY_REBASELINE_AUDIT_SCHEMA, "response"), (HISTORY_REBASELINE_AUDIT_SCHEMA, "request"), + (HISTORY_REBASELINE_AUDIT_SCHEMA, "response"), ] ); assert_eq!(entries[2]["correlation"], entries[3]["correlation"]); assert_ne!(entries[2]["correlation"], entries[4]["correlation"]); + assert_eq!(entries[4]["correlation"], entries[5]["correlation"]); + assert_eq!(entries[5]["record"]["outcome"], "unfinished"); let audit_text = serde_json::Value::Array(entries[2..].to_vec()).to_string(); assert!(audit_text.contains("history-rebaseline-maintenance")); assert!(!audit_text.contains(OPERATOR_CANARY)); diff --git a/products/breg/CHANGELOG.md b/products/breg/CHANGELOG.md index 6e426bb1f..244c0d7e4 100644 --- a/products/breg/CHANGELOG.md +++ b/products/breg/CHANGELOG.md @@ -19,6 +19,10 @@ wait for deletion. An erasure that committed without its response entry reports `request_retention.erasure.unaudited`. `request-retention cleanup-attachments` answers a failed cleanup with a `failed` response. + - `history rebaseline`, a standalone `history erase`, and a + field-encryption erase-and-rebaseline run answer their request entry with + an `unfinished` response when they are refused or fail before their + terminal entry, as does an attachment verification job that stops early. - `evidence-retention erase-expired` is audited under `breg-evidence-retention-audit/v1`: a request entry naming the cutoff before the erasure, and a response with the erased count or `failed`. diff --git a/products/breg/contracts/definition-of-done.yaml b/products/breg/contracts/definition-of-done.yaml index 8026eab42..c9016662d 100644 --- a/products/breg/contracts/definition-of-done.yaml +++ b/products/breg/contracts/definition-of-done.yaml @@ -48,7 +48,7 @@ requirements: - {id: BREG-V1-24, phase: W3, state: enforced, doneWhen: "Application authorization and RLS agree for positive, negative, malformed, and pooled authority.", journeys: [BREG-J05, BREG-J11], evidence: [{path: crates/registry-breg/tests/postgres_compiled_schema.rs, name: compiled_postgres_schema_enforces_context_rls_and_exact_catalog}, {path: crates/registry-breg/tests/postgres_kernel.rs, name: real_postgres_kernel_proves_roles_rls_interlock_and_pool_isolation}, {path: crates/registry-breg/tests/postgres_pilot_acceptance.rs, name: real_postgres_five_domain_pilot_is_configured_production_closed_and_source_neutral}]} - {id: BREG-V1-25, phase: W3, state: enforced, doneWhen: "Provenance stays distinct from minimized value-free audit, logs, metrics, and traces.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_tombstone_revision.rs, name: tombstone_revisions_survive_package_upgrade_and_replay_exactly}, {path: crates/registry-breg/tests/startup_http.rs, name: operational_log_level_is_a_closed_vocabulary}, {path: crates/registry-breg/tests/startup_http.rs, name: every_operational_event_renders_exact_closed_value_free_json_fields}, {path: crates/registry-breg/tests/startup_http.rs, name: provenance_operational_logs_metrics_and_traces_are_separate_closed_and_value_free}]} - {id: BREG-V1-26, phase: W3, state: enforced, doneWhen: "Exact media-specific Registry Record bytes release or replay only after successful attempt and terminal audit gates.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_http_mutations_are_guarded_and_exactly_replayable}]} - - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}]} + - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}, {path: crates/registry-breg/tests/postgres_history_rebaseline.rs, name: rebaseline_refuses_while_maintenance_is_not_ready}]} - {id: BREG-V1-28, phase: W4, state: enforced, doneWhen: "Production packages capture the governed closure and sign exact canonical bytes with monotonic identity.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_builder_is_deterministic_and_local_publication_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: production_package_requires_exact_trust_anchor_threshold_and_signature}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_layout_contract_conditional_manifest_projection_is_in_projected_closure}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_omits_manifest_projection_from_signed_closure_and_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_refuses_claimed_manifest_artifacts}]} - {id: BREG-V1-29, phase: W4, state: enforced, doneWhen: "Activation verifies trust, identity, inventory, filesystem safety, artifacts, and schema before readiness.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_binding_refuses_wrong_environment_instance_database_sequence_and_prior}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_refuses_symlinks_and_production_writable_permissions}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_manifest_refuses_ddl_checksum_path_and_canonical_json_tampering}, {path: crates/registry-breg/tests/postgres_package.rs, name: signed_schema_fingerprint_mismatch_is_durably_failed_and_never_ready}]} - {id: BREG-V1-30, phase: W4, state: enforced, doneWhen: "Apply retains the lock through migrations, catalog verification, activation, and maintenance clearing.", journeys: [BREG-J13, BREG-J15], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: real_postgres_package_startup_apply_failure_and_old_process_are_closed}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_backfill_and_destructive_recovery_are_bounded_resumable_and_activation_closed}]} From eab4baf1d260a1148bb569bf4b97dffffb5c2c3b Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 16:31:58 +0000 Subject: [PATCH 14/32] fix(audit): keep request pairing through cancellation Review findings on the AuditRequest handle: - A caller canceled while its request entry was written could leave the request accepted with no handle to answer it. The write and the handle's registration now run in one task that outlives the caller, and a caller that stopped waiting drops the finished handle, which answers it. - A response accepted after its caller was canceled was not marked as answering its request, so the drop wrote a second, false unfinished response. append and respond now record the answer inside the task that writes it; a drop while a response is in flight leaves that response to settle the request, and writes the unfinished record only if it is refused. - The unfinished entry is validated before the request is written, so a request is never accepted with a response it could not write. - A stream destination writes a dropped request's unfinished entry on a dedicated thread, joined when the writer is dropped, instead of blocking a runtime thread on a stalled stream. wait_for_detached_entries lets tests and shutdown paths read after a drop; the product test captures use it. Refs #1561 Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/audit.rs | 12 + crates/registry-casework/src/audit.rs | 8 + crates/registry-platform-audit/src/writer.rs | 531 ++++++++++++++++--- crates/registry-render/src/audit.rs | 7 + crates/registry-render/src/server.rs | 1 + crates/registry-scheduling/src/audit.rs | 8 + 6 files changed, 495 insertions(+), 72 deletions(-) diff --git a/crates/registry-breg/src/audit.rs b/crates/registry-breg/src/audit.rs index 78c073290..023a44b05 100644 --- a/crates/registry-breg/src/audit.rs +++ b/crates/registry-breg/src/audit.rs @@ -1078,6 +1078,9 @@ pub mod test_support { /// The envelope schema and record phase of the first entry refused /// regardless of `remaining`. refuse: Option<(String, String)>, + /// Every writer recording here, so a read can wait for the entries + /// they write when a request handle is dropped. + writers: Vec, } /// The entries one [`RegistryAudit`] accepted, in append order. @@ -1127,6 +1130,10 @@ pub mod test_support { /// Every accepted entry, parsed. #[must_use] pub fn entries(&self) -> Vec { + let writers = self.0.lock().expect("audit capture lock").writers.clone(); + for writer in &writers { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture lock"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -1162,6 +1169,11 @@ pub mod test_support { #[must_use] pub fn audit(&self, profile: AuditProfile) -> RegistryAudit { let writer = AuditWriter::from_line_sink(Box::new(CaptureSink(Arc::clone(&self.0)))); + self.0 + .lock() + .expect("audit capture lock") + .writers + .push(writer.clone()); RegistryAudit::new(profile, writer) } } diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index 22300c3df..ee8390a05 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -406,6 +406,9 @@ mod capture { struct CaptureState { bytes: Vec, accepted_lines: Option, + /// The writer recording here, so a read can wait for the entries it + /// writes when a request handle is dropped. + writer: Option, } /// The lines a test audit destination accepted, and a switch that makes @@ -417,6 +420,10 @@ mod capture { /// Every accepted entry, parsed, in write order. #[must_use] pub fn entries(&self) -> Vec { + let writer = self.0.lock().expect("audit capture").writer.clone(); + if let Some(writer) = writer { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -482,6 +489,7 @@ mod capture { pub fn capture() -> (Self, AuditCapture) { let capture = AuditCapture::default(); let writer = AuditWriter::from_line_sink(Box::new(capture.clone())); + capture.0.lock().expect("audit capture").writer = Some(writer.clone()); ( Self::new(writer, AuditKeyHasher::unkeyed_dev_only()), capture, diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index cf393a067..a36c10710 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -583,7 +583,7 @@ impl OpenRequests { enum WriterInner { File(Arc), - Stream(Arc), + Stream(Arc, DetachedLines), } impl std::fmt::Debug for AuditWriter { @@ -593,7 +593,7 @@ impl std::fmt::Debug for AuditWriter { WriterInner::File(file) => debug .field("destination", &"file") .field("path", &file.file.path), - WriterInner::Stream(_) => debug.field("destination", &"stdout"), + WriterInner::Stream(..) => debug.field("destination", &"stdout"), }; debug.finish_non_exhaustive() } @@ -610,12 +610,8 @@ impl AuditWriter { .map_err(|error| AuditError::Io(io::Error::other(error)))??; WriterInner::File(Arc::new(GroupCommitFile::new(opened))) } - AuditDestination::Stdout => { - WriterInner::Stream(Arc::new(LineStream::new(Box::new(io::stdout())))) - } - AuditDestination::Stderr => { - WriterInner::Stream(Arc::new(LineStream::new(Box::new(io::stderr())))) - } + AuditDestination::Stdout => WriterInner::stream(Box::new(io::stdout())), + AuditDestination::Stderr => WriterInner::stream(Box::new(io::stderr())), }; Ok(Self { inner: Arc::new(inner), @@ -628,7 +624,7 @@ impl AuditWriter { #[must_use] pub fn from_line_sink(sink: Box) -> Self { Self { - inner: Arc::new(WriterInner::Stream(Arc::new(LineStream::new(sink)))), + inner: Arc::new(WriterInner::stream(sink)), open: Arc::default(), } } @@ -640,11 +636,19 @@ impl AuditWriter { /// An accepted `response` entry answers the oldest [`AuditRequest`] still /// open under the same schema and correlation. pub async fn append(&self, entry: AuditEntry) -> Result<(), AuditUnavailable> { - self.write(&entry).await?; - if entry.phase == AuditPhase::Response { - self.open.answer(&(entry.schema, entry.correlation)); - } - Ok(()) + // The write and the bookkeeping it implies run in one task that + // outlives a canceled caller, so an accepted response always answers + // its request, whether or not the caller is still waiting. + let writer = self.clone(); + tokio::spawn(async move { + writer.write(&entry).await?; + if entry.phase == AuditPhase::Response { + writer.open.answer(&(entry.schema, entry.correlation)); + } + Ok(()) + }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? } async fn write(&self, entry: &AuditEntry) -> Result<(), AuditUnavailable> { @@ -656,7 +660,7 @@ impl AuditWriter { .await .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? } - WriterInner::Stream(stream) => { + WriterInner::Stream(stream, _) => { let stream = Arc::clone(stream); tokio::task::spawn_blocking(move || stream.append(&line)) .await @@ -685,23 +689,50 @@ impl AuditWriter { ) -> Result { let schema = schema.into(); let correlation = correlation.into(); - if !unfinished.is_object() { - return Err(AuditUnavailable::new(AuditUnavailableReason::InvalidEntry)); - } - self.write(&AuditEntry::request( - schema.clone(), - correlation.clone(), - request, - )) - .await?; - let answered = self.open.open((schema.clone(), correlation.clone())); - Ok(AuditRequest { - writer: self.clone(), - schema, - correlation, - answered, - unfinished, + // The unfinished response must be writable before the request is: + // a request accepted with a response it could never write would stay + // unpaired. + AuditEntry::response(schema.clone(), correlation.clone(), unfinished.clone()).to_line()?; + // The request is written and its handle registered in one task that + // outlives a canceled caller. A caller that stops waiting drops the + // finished handle with the task's output, which writes the + // unfinished response. + let writer = self.clone(); + tokio::spawn(async move { + writer + .write(&AuditEntry::request( + schema.clone(), + correlation.clone(), + request, + )) + .await?; + let answered = writer.open.open((schema.clone(), correlation.clone())); + Ok(AuditRequest { + writer, + schema, + correlation, + answered, + state: Arc::new(StdMutex::new(RequestState { + unfinished: Some(unfinished), + in_flight: 0, + dropped: false, + })), + }) }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? + } + + /// Block until every response entry a dropped [`AuditRequest`] handed + /// to a stream destination has been written. Those entries are written + /// on a dedicated thread so a drop never blocks the runtime; this is for + /// tests and shutdown paths that read the stream right after a drop. A + /// file destination queues its entries into the group commit instead, + /// and this returns at once for it. + pub fn wait_for_detached_entries(&self) { + if let WriterInner::Stream(_, detached) = self.inner.as_ref() { + detached.wait(); + } } /// Write `entry` without waiting for the destination to accept it. @@ -715,14 +746,10 @@ impl AuditWriter { }; match self.inner.as_ref() { WriterInner::File(file) => file.enqueue_detached(line), - WriterInner::Stream(stream) => { - // Written inline, unlike `write`: a blocking task queued from - // a drop during runtime shutdown may never run, and the - // unfinished response is the entry that must not be lost. - if stream.append(&line).is_err() { - tracing::error!("an unfinished response entry was not accepted"); - } - } + // A dedicated thread writes it, so a drop never blocks a runtime + // thread on a slow stream, and the thread is joined when the + // writer is dropped, so shutdown does not lose it. + WriterInner::Stream(_, detached) => detached.send(line), } } @@ -731,7 +758,7 @@ impl AuditWriter { pub async fn ready(&self) -> bool { match self.inner.as_ref() { WriterInner::File(file) => file.ready().await, - WriterInner::Stream(stream) => stream.healthy(), + WriterInner::Stream(stream, _) => stream.healthy(), } } @@ -739,7 +766,7 @@ impl AuditWriter { pub fn kind(&self) -> AuditDestinationKind { match self.inner.as_ref() { WriterInner::File(_) => AuditDestinationKind::File, - WriterInner::Stream(_) => AuditDestinationKind::Stdout, + WriterInner::Stream(..) => AuditDestinationKind::Stdout, } } @@ -748,7 +775,7 @@ impl AuditWriter { pub fn path(&self) -> Option<&Path> { match self.inner.as_ref() { WriterInner::File(file) => Some(&file.file.path), - WriterInner::Stream(_) => None, + WriterInner::Stream(..) => None, } } @@ -758,7 +785,7 @@ impl AuditWriter { pub fn durable_writes(&self) -> u64 { match self.inner.as_ref() { WriterInner::File(file) => file.durable_writes.load(Ordering::Relaxed), - WriterInner::Stream(_) => 0, + WriterInner::Stream(..) => 0, } } } @@ -779,8 +806,18 @@ pub struct AuditRequest { /// Set once a response entry under this schema and correlation was /// accepted. answered: Arc, - /// The record written if the handle is dropped unanswered. - unfinished: Value, + /// The unfinished record and the responses still being written, shared + /// with those writes so the last one to settle owns the drop's duty. + state: Arc>, +} + +struct RequestState { + /// The record written if the request ends unanswered. + unfinished: Option, + /// Responses whose write has started and not yet settled. + in_flight: usize, + /// The handle was dropped while a response was in flight. + dropped: bool, } impl std::fmt::Debug for AuditRequest { @@ -807,22 +844,40 @@ impl AuditRequest { } /// Append one `response` entry. A refused entry leaves the request - /// unanswered, so a later drop still writes the unfinished record. + /// unanswered, so a later drop still writes the unfinished record. The + /// write and its bookkeeping outlive a canceled caller: a response + /// accepted after the caller stopped waiting still answers the request, + /// and a drop while it is in flight writes the unfinished record only if + /// that response is refused. pub async fn respond(&mut self, record: Value) -> Result<(), AuditUnavailable> { - self.writer - .write(&AuditEntry::response( - self.schema.clone(), - self.correlation.clone(), - record, - )) - .await?; - // Answer this request, not an older one open under its correlation. - self.writer.open.close( - &(self.schema.clone(), self.correlation.clone()), - &self.answered, - ); - self.answered.store(true, Ordering::Release); - Ok(()) + if let Ok(mut state) = self.state.lock() { + state.in_flight += 1; + } + let writer = self.writer.clone(); + let key = (self.schema.clone(), self.correlation.clone()); + let answered = Arc::clone(&self.answered); + let state = Arc::clone(&self.state); + tokio::spawn(async move { + let result = writer + .write(&AuditEntry::response(key.0.clone(), key.1.clone(), record)) + .await; + if result.is_ok() { + // Answer this request, not an older one open under its + // correlation. + writer.open.close(&key, &answered); + answered.store(true, Ordering::Release); + } + let owes_unfinished = state.lock().is_ok_and(|mut state| { + state.in_flight -= 1; + state.in_flight == 0 && state.dropped + }); + if owes_unfinished { + settle_unanswered(&writer, key, &answered, &state); + } + result + }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? } /// Append `record` as the only `response` entry and release the handle. @@ -831,19 +886,147 @@ impl AuditRequest { } } +/// Write the unfinished record of a request still owed a response. +fn settle_unanswered( + writer: &AuditWriter, + key: RequestKey, + answered: &Arc, + state: &StdMutex, +) { + if !writer.open.close(&key, answered) { + return; + } + let unfinished = state + .lock() + .ok() + .and_then(|mut state| state.unfinished.take()); + if let Some(unfinished) = unfinished { + writer.append_detached(&AuditEntry::response(key.0, key.1, unfinished)); + } +} + impl Drop for AuditRequest { fn drop(&mut self) { + // A response still being written settles the request itself. + let in_flight = self.state.lock().is_ok_and(|mut state| { + state.dropped = true; + state.in_flight > 0 + }); + if in_flight { + return; + } let key = ( std::mem::take(&mut self.schema), std::mem::take(&mut self.correlation), ); - if self.writer.open.close(&key, &self.answered) { - let entry = AuditEntry::response(key.0, key.1, std::mem::take(&mut self.unfinished)); - self.writer.append_detached(&entry); + settle_unanswered(&self.writer, key, &self.answered, &self.state); + } +} + +/// The unfinished response entries dropped requests hand to a stream +/// destination, written in order on one dedicated thread. +struct DetachedLines { + stream: Arc, + sender: StdMutex>>, + thread: StdMutex>>, + /// Lines handed over and not yet written, with a signal on each write. + pending: Arc<(StdMutex, std::sync::Condvar)>, +} + +impl DetachedLines { + fn new(stream: Arc) -> Self { + Self { + stream, + sender: StdMutex::new(None), + thread: StdMutex::new(None), + pending: Arc::new((StdMutex::new(0), std::sync::Condvar::new())), + } + } + + fn send(&self, line: String) { + let Ok(mut sender) = self.sender.lock() else { + tracing::error!("an unfinished response entry was not written"); + return; + }; + if sender.is_none() { + let (lines, received) = std::sync::mpsc::channel::(); + let stream = Arc::clone(&self.stream); + let pending = Arc::clone(&self.pending); + let spawned = std::thread::Builder::new() + .name("audit-detached".to_owned()) + .spawn(move || { + for line in received { + if stream.append(&line).is_err() { + tracing::error!("an unfinished response entry was not accepted"); + } + let (count, written) = &*pending; + if let Ok(mut count) = count.lock() { + *count = count.saturating_sub(1); + } + written.notify_all(); + } + }); + match spawned { + Ok(thread) => { + if let Ok(mut slot) = self.thread.lock() { + *slot = Some(thread); + } + *sender = Some(lines); + } + Err(error) => { + tracing::error!(%error, "an unfinished response entry was not written"); + return; + } + } + } + if let Ok(mut count) = self.pending.0.lock() { + *count += 1; + } + if sender + .as_ref() + .is_some_and(|lines| lines.send(line).is_err()) + { + tracing::error!("an unfinished response entry was not written"); + if let Ok(mut count) = self.pending.0.lock() { + *count = count.saturating_sub(1); + } + } + } + + fn wait(&self) { + let (count, written) = &*self.pending; + let Ok(mut count) = count.lock() else { + return; + }; + while *count > 0 { + count = match written.wait(count) { + Ok(count) => count, + Err(_) => return, + }; } } } +impl Drop for DetachedLines { + /// Close the queue and let the thread write what is left. + fn drop(&mut self) { + if let Ok(mut sender) = self.sender.lock() { + sender.take(); + } + let thread = self.thread.lock().ok().and_then(|mut thread| thread.take()); + if let Some(thread) = thread { + let _ = thread.join(); + } + } +} + +impl WriterInner { + fn stream(out: Box) -> Self { + let stream = Arc::new(LineStream::new(out)); + Self::Stream(Arc::clone(&stream), DetachedLines::new(stream)) + } +} + struct LineStream { out: StdMutex>, healthy: AtomicBool, @@ -3067,6 +3250,13 @@ mod tests { .collect() } + /// The lines accepted once every detached unfinished response is + /// written. + fn settled_lines(writer: &AuditWriter, buffer: &SharedBuffer) -> Vec { + writer.wait_for_detached_entries(); + buffered_lines(buffer) + } + fn assert_paired(entries: &[Value], outcome: &str) { assert_eq!(entries.len(), 2, "{entries:?}"); assert_eq!(entries[0]["phase"], "request"); @@ -3100,7 +3290,7 @@ mod tests { .await .expect("second response"); drop(request); - let entries = buffered_lines(&buffer); + let entries = settled_lines(&writer, &buffer); assert_eq!(entries.len(), 3); assert!(entries[1..] .iter() @@ -3131,7 +3321,7 @@ mod tests { .expect("response"); assert!(request.is_answered()); drop(request); - assert_paired(&buffered_lines(&buffer), "returned"); + assert_paired(&settled_lines(&writer, &buffer), "returned"); } #[tokio::test] @@ -3151,7 +3341,7 @@ mod tests { .expect("other schema"); assert!(!request.is_answered()); drop(request); - let entries = buffered_lines(&buffer); + let entries = settled_lines(&writer, &buffer); assert_eq!(entries.len(), 3); assert_eq!(entries[2]["schema"], SCHEMA); assert_eq!(entries[2]["correlation"], "req-1"); @@ -3176,7 +3366,7 @@ mod tests { assert!(!first.is_answered(), "the second's response is its own"); drop(second); drop(first); - let outcomes: Vec<_> = buffered_lines(&buffer) + let outcomes: Vec<_> = settled_lines(&writer, &buffer) .iter() .filter(|entry| entry["phase"] == "response") .map(|entry| entry["record"]["outcome"].clone()) @@ -3197,7 +3387,7 @@ mod tests { .await .expect("request"); drop(request); - assert_paired(&buffered_lines(&buffer), "unfinished"); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); } #[tokio::test] @@ -3217,7 +3407,7 @@ mod tests { } let (writer, buffer) = buffered(); assert_eq!(refused(&writer).await, Err("not found")); - assert_paired(&buffered_lines(&buffer), "unfinished"); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); } #[tokio::test] @@ -3239,7 +3429,7 @@ mod tests { ); assert!(!request.is_answered()); drop(request); - assert_paired(&buffered_lines(&buffer), "unfinished"); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); } #[tokio::test] @@ -3254,7 +3444,7 @@ mod tests { ) .await; assert!(refused.is_err()); - assert!(buffered_lines(&buffer).is_empty()); + assert!(settled_lines(&writer, &buffer).is_empty()); } #[tokio::test] @@ -3314,7 +3504,7 @@ mod tests { } }); assert!(operation.await.expect_err("panicked").is_panic()); - assert_paired(&buffered_lines(&buffer), "unfinished"); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); } #[test] @@ -3362,4 +3552,201 @@ mod tests { .is_err()); assert!(!writer.ready().await); } + + /// A line sink that holds each write until the test releases it. + #[derive(Clone)] + struct GatedSink { + buffer: SharedBuffer, + open: Arc<(Mutex, std::sync::Condvar)>, + entered: Arc, + } + + impl GatedSink { + fn new() -> Self { + Self { + buffer: SharedBuffer::default(), + open: Arc::new((Mutex::new(false), std::sync::Condvar::new())), + entered: Arc::default(), + } + } + + fn release(&self) { + *self.open.0.lock().expect("gate") = true; + self.open.1.notify_all(); + } + + async fn wait_entered(&self, writes: usize) { + for _ in 0..500 { + if self.entered.load(Ordering::SeqCst) >= writes { + return; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + panic!("the sink never received write {writes}"); + } + } + + impl Write for GatedSink { + fn write(&mut self, bytes: &[u8]) -> io::Result { + self.entered.fetch_add(1, Ordering::SeqCst); + let mut open = self.open.0.lock().expect("gate"); + while !*open { + open = self.open.1.wait(open).expect("gate"); + } + drop(open); + self.buffer.write(bytes) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + + /// The accepted lines once `count` of them arrived. + async fn lines_eventually(buffer: &SharedBuffer, count: usize) -> Vec { + for _ in 0..500 { + let lines = buffered_lines(buffer); + if lines.len() >= count { + return lines; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + buffered_lines(buffer) + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_request_canceled_while_its_entry_is_written_is_still_paired() { + let sink = GatedSink::new(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let begun = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + } + }); + sink.wait_entered(1).await; + // The caller goes away while its request entry is being written. + begun.abort(); + sink.release(); + assert_paired(&lines_eventually(&sink.buffer, 2).await, "unfinished"); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_accepted_after_its_caller_left_is_the_only_answer() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let responding = tokio::spawn(async move { + let mut request = request; + request.respond(json!({"outcome": "returned"})).await + }); + sink.wait_entered(2).await; + // The caller and its handle go away while the response is written. + responding.abort(); + sink.release(); + let lines = lines_eventually(&sink.buffer, 2).await; + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + assert_eq!(buffered_lines(&sink.buffer).len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_appended_after_its_caller_left_still_answers_the_request() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The caller goes away while the response is written. + appending.abort(); + sink.release(); + lines_eventually(&sink.buffer, 2).await; + for _ in 0..500 { + if request.is_answered() { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + drop(request); + writer.wait_for_detached_entries(); + assert_paired(&buffered_lines(&sink.buffer), "returned"); + } + + #[tokio::test] + async fn an_unfinished_record_too_large_to_write_refuses_the_request() { + let (writer, buffer) = buffered(); + let oversized = json!({"outcome": "unfinished", "padding": "x".repeat(MAX_ENTRY_BYTES)}); + assert!(writer + .begin(SCHEMA, "req-1", json!({"operationId": "read"}), oversized) + .await + .is_err()); + assert!(buffered_lines(&buffer).is_empty()); + assert!(writer.ready().await, "a refused request stops nothing"); + } + + #[test] + fn dropping_a_request_never_waits_on_a_stalled_stream() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + // The stream stalls; the drop must still return to the runtime. + let started = std::time::Instant::now(); + runtime.block_on(async move { drop(request) }); + assert!(started.elapsed() < Duration::from_secs(1)); + sink.release(); + writer.wait_for_detached_entries(); + assert_paired(&buffered_lines(&sink.buffer), "unfinished"); + } } diff --git a/crates/registry-render/src/audit.rs b/crates/registry-render/src/audit.rs index c7b7b6d05..6f3000909 100644 --- a/crates/registry-render/src/audit.rs +++ b/crates/registry-render/src/audit.rs @@ -188,6 +188,13 @@ impl RenderAudit { }) } + /// Wait for the unfinished response entries dropped calls handed to a + /// stream destination. + #[cfg(test)] + pub(crate) fn wait_for_detached_entries(&self) { + self.writer.wait_for_detached_entries(); + } + /// Readiness: the destination still accepts entries. pub async fn ready(&self) -> bool { self.writer.ready().await diff --git a/crates/registry-render/src/server.rs b/crates/registry-render/src/server.rs index fd7b8380a..ca088b1dd 100644 --- a/crates/registry-render/src/server.rs +++ b/crates/registry-render/src/server.rs @@ -1034,6 +1034,7 @@ mod tests { pending.is_err(), "with no render slot free the request must be waiting at the render step" ); + service.audit.wait_for_detached_entries(); let accepted = lines.accepted(); // The call dropped while it waited for a render slot, as a canceled // request is: its request entry is paired with an unfinished diff --git a/crates/registry-scheduling/src/audit.rs b/crates/registry-scheduling/src/audit.rs index e65d6f224..af1c6b66f 100644 --- a/crates/registry-scheduling/src/audit.rs +++ b/crates/registry-scheduling/src/audit.rs @@ -139,6 +139,9 @@ mod capture { struct CaptureState { bytes: Vec, accepted_lines: Option, + /// The writer recording here, so a read can wait for the entries it + /// writes when a request handle is dropped. + writer: Option, } /// The lines a test audit destination accepted, and a switch that makes @@ -150,6 +153,10 @@ mod capture { /// Every accepted entry, parsed, in write order. #[must_use] pub fn entries(&self) -> Vec { + let writer = self.0.lock().expect("audit capture").writer.clone(); + if let Some(writer) = writer { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -197,6 +204,7 @@ mod capture { pub fn capture() -> (Self, AuditCapture) { let capture = AuditCapture::default(); let writer = AuditWriter::from_line_sink(Box::new(capture.clone())); + capture.0.lock().expect("audit capture").writer = Some(writer.clone()); (Self::new(writer), capture) } } From 474ac057741403ae350a2deeaeb7bca7fa49c5ab Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 16:31:58 +0000 Subject: [PATCH 15/32] fix(hooks): resolve an unacknowledged commit before auditing it A PostgreSQL commit error does not prove the transaction rolled back. After a failed commit the worker now reads the delivery state on a fresh connection: a terminal disposition or replay reset that did commit is recorded, and one that rolled back is left to expiry recovery or recorded as refused. A lease commit whose fate cannot be read is answered as interrupted, since a second interrupted answer is harmless and a missing one is not. Refs #1592 #1588 Signed-off-by: Jeremi Joslin --- .../src/delivery/service.rs | 162 +++++++++++++++--- 1 file changed, 138 insertions(+), 24 deletions(-) diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index 5f103b999..b2d94f9cb 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -63,6 +63,7 @@ pub struct DeliveryConfig { /// A delivery-audit event held by the worker until it is recorded, such as /// one whose transition must commit first. +#[derive(Clone)] struct PendingAudit { event_id: Uuid, compiled_delivery_id: String, @@ -367,9 +368,24 @@ impl DeliveryService { disposition: DeliveryAuditDisposition::ReplayPending, }; self.seams.record_audit(replay.record()).await?; - let reset = self + let mut reset = self .reset_for_replay(transaction, event_id, compiled_delivery_id, generation) .await; + if reset.is_err() + && self + .transition_committed( + &PendingAudit { + outcome: DeliveryAuditOutcome::ReplayCommitted, + ..replay.clone() + }, + None, + ) + .await + == Some(true) + { + // The reset committed although its acknowledgement was lost. + reset = Ok(()); + } let (outcome, disposition) = if reset.is_ok() { ( DeliveryAuditOutcome::ReplayCommitted, @@ -497,7 +513,10 @@ impl DeliveryService { DeliveryError::Unavailable })?; let Some(row) = row else { - transaction.commit().await?; + if let Err(error) = transaction.commit().await { + self.record_resolved(committed).await; + return Err(error.into()); + } self.record_committed(committed).await?; return Ok(None); }; @@ -591,15 +610,24 @@ impl DeliveryService { } if transaction.commit().await.is_err() { self.refused(DeliveryTransitionCode::ClaimCommitFailed); - // The attempt's request is on record but its lease rolled back: - // answer it, so the journal shows the attempt never ran. - let interrupted = PendingAudit { - phase: DeliveryAuditPhase::Terminal, - outcome: DeliveryAuditOutcome::WorkerInterrupted, - disposition: DeliveryAuditDisposition::RetryPending, - ..started - }; - let _ = self.seams.record_audit(interrupted.record()).await; + // A failed commit acknowledgement does not prove a rollback, so + // read what the database holds. A lease that did commit sends + // nothing from here and is answered when it expires; one that + // rolled back, or one whose fate cannot be read, is answered now, + // since a second interrupted answer is harmless and none is not. + if self.transition_committed(&started, Some(lease_token)).await == Some(true) { + self.record_resolved(committed).await; + } else { + let interrupted = PendingAudit { + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: DeliveryAuditDisposition::RetryPending, + ..started + }; + // A refused entry has already stopped the product's writer, + // which reports it; the claim fails either way. + let _ = self.seams.record_audit(interrupted.record()).await; + } return Err(DeliveryError::Unavailable); } self.record_committed(committed).await?; @@ -1349,24 +1377,110 @@ impl DeliveryService { return Err(DeliveryError::Unavailable); } } - transaction.commit().await?; + let terminal = PendingAudit { + event_id: claim.event_id, + compiled_delivery_id: claim.compiled_delivery_id.clone(), + package_revision: claim.package_revision.clone(), + generation: claim.generation, + attempt: claim.attempt, + phase: DeliveryAuditPhase::Terminal, + outcome, + disposition, + }; + if let Err(error) = transaction.commit().await { + // A lost acknowledgement does not prove a rollback, and a + // terminal row is never reaped, so a disposition that did commit + // is recorded here or never. One that rolled back leaves the + // lease for expiry recovery, which answers the attempt. + if self.transition_committed(&terminal, None).await != Some(true) { + return Err(error.into()); + } + } // Recorded once the disposition committed, so the journal never // names a disposition the database does not hold. - self.seams - .record_audit(DeliveryAuditRecord { - event_id: claim.event_id, - compiled_delivery_id: &claim.compiled_delivery_id, - package_revision: &claim.package_revision, - generation: claim.generation, - attempt: claim.attempt, - phase: DeliveryAuditPhase::Terminal, - outcome, - disposition, - }) - .await?; + self.seams.record_audit(terminal.record()).await?; Ok(work_outcome) } + /// Record the events of a transaction whose commit returned an error, + /// each only if the transition it names is durable. + async fn record_resolved(&self, events: Vec) { + for event in events { + // A refused entry has already stopped the product's writer, + // which reports it; the caller is failing either way. + if self.transition_committed(&event, None).await == Some(true) { + let _ = self.seams.record_audit(event.record()).await; + } + } + } + + /// Whether the transition `event` records is durable, read on a fresh + /// connection after its commit returned an error, since a failed commit + /// acknowledgement does not prove the transaction rolled back. `None` + /// when the state cannot be read. An attempt's lease is matched on the + /// token this worker wrote. + async fn transition_committed( + &self, + event: &PendingAudit, + lease_token: Option, + ) -> Option { + for _ in 0..3 { + let Ok(client) = self.seams.connection().await else { + continue; + }; + let Ok(row) = client + .query_opt( + &self.sql( + "SELECT state, generation, attempt, lease_token, expired_at IS NOT NULL + FROM {schema}.registry_webhook_delivery_state + WHERE event_id = $1 AND compiled_delivery_id = $2", + ), + &[&event.event_id, &event.compiled_delivery_id], + ) + .await + else { + continue; + }; + let Some(row) = row else { + return Some(false); + }; + let (Ok(state), Ok(generation), Ok(attempt), Ok(token), Ok(expired)) = ( + row.try_get::<_, String>(0), + row.try_get::<_, i64>(1), + row.try_get::<_, i16>(2), + row.try_get::<_, Option>(3), + row.try_get::<_, bool>(4), + ) else { + return None; + }; + let current = generation == event.generation; + let at_attempt = current && attempt == event.attempt; + return Some(match (event.phase, event.disposition) { + (DeliveryAuditPhase::Attempt, _) => { + at_attempt && state == "leased" && token.is_some() && token == lease_token + } + (DeliveryAuditPhase::Replay, _) => current && state == "pending", + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Expired) => { + current && expired + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Delivered) => { + at_attempt && state == "delivered" + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::DeadLettered) => { + at_attempt && state == "dead_lettered" + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::RetryPending) => { + at_attempt && state == "pending" && token.is_none() + } + ( + DeliveryAuditPhase::Terminal, + DeliveryAuditDisposition::Leased | DeliveryAuditDisposition::ReplayPending, + ) => false, + }); + } + None + } + /// Record the events of a transaction that has committed. async fn record_committed(&self, committed: Vec) -> Result<(), DeliveryError> { for event in committed { From 4c5ed4d2acbe53221864eaffd785599a5b1c04d7 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 17:20:01 +0000 Subject: [PATCH 16/32] fix(hooks): answer an operator replay whose caller left An operator replay now runs to its response in a task of its own, so a caller that times out or disconnects after the replay_requested entry is accepted still leaves that request answered. A reset whose commit acknowledgement was lost is recognized by its replacement generation, which only a replay writes, rather than by the pending state the worker may already have moved it out of. Signed-off-by: Jeremi Joslin --- .../tests/postgres_webhook_delivery.rs | 60 +++++++- .../src/delivery/service.rs | 145 +++++++++++++++--- 2 files changed, 175 insertions(+), 30 deletions(-) diff --git a/crates/registry-breg/tests/postgres_webhook_delivery.rs b/crates/registry-breg/tests/postgres_webhook_delivery.rs index a8cd3bdc3..c564eddc8 100644 --- a/crates/registry-breg/tests/postgres_webhook_delivery.rs +++ b/crates/registry-breg/tests/postgres_webhook_delivery.rs @@ -353,14 +353,38 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun ) .await; - service - .replay( - timeout_event.event_id, - &timeout_event.compiled_delivery_id, - 1, + // The operator stops waiting while the reset's commit is in flight: + // the replay still runs to its response, so its accepted request is + // answered once the reset commits. + slow_delivery_state_commit(&database, "pending").await; + assert!( + tokio::time::timeout( + Duration::from_millis(300), + service.replay( + timeout_event.event_id, + &timeout_event.compiled_delivery_id, + 1, + ), ) .await - .expect("compiled operator replay resets one terminal generation"); + .is_err(), + "the caller leaves before the reset commits" + ); + allow_delivery_state_commit(&database).await; + let mut answered = Vec::new(); + for _ in 0..50 { + answered = audit_outcomes(&database, &audit_profile, &timeout_event, 2, 0, "replay").await; + if answered.len() == 2 { + break; + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + assert_eq!( + answered, + ["replay_requested", "replay_committed"], + "a replay whose caller left is still answered" + ); + assert_eq!(delivery_state(&database, &timeout_event).await.0, 2); assert_eq!( service .replay( @@ -970,6 +994,30 @@ async fn refuse_delivery_state_commit(database: &TestDatabase, state: &str) { .expect("administrator installs the commit refusal"); } +/// Hold the commit of any delivery transition into `state` for a second, +/// after every statement in its transaction succeeded. +async fn slow_delivery_state_commit(database: &TestDatabase, state: &str) { + database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_delivery_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.state = '{state}' THEN + PERFORM pg_sleep(1); + END IF; + RETURN NEW; + END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_delivery_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_delivery_commit + AFTER UPDATE ON registry_internal.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_delivery_commit();" + )) + .await + .expect("administrator installs the commit delay"); +} + async fn allow_delivery_state_commit(database: &TestDatabase) { database .admin diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index b2d94f9cb..f1006e231 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -275,6 +275,29 @@ impl DeliveryService { event_id: Uuid, compiled_delivery_id: &str, expected_generation: i64, + ) -> Result + where + S: Clone, + { + // The replay runs to its response in a task of its own: once its + // request entry is accepted, a caller that stops waiting (a timeout + // or a disconnect) cannot leave that request unanswered. + let service = self.clone(); + let compiled_delivery_id = compiled_delivery_id.to_owned(); + tokio::spawn(async move { + service + .replay_in(event_id, &compiled_delivery_id, expected_generation) + .await + }) + .await + .map_err(|_| DeliveryError::Unavailable)? + } + + async fn replay_in( + &self, + event_id: Uuid, + compiled_delivery_id: &str, + expected_generation: i64, ) -> Result { if compiled_delivery_id.is_empty() || compiled_delivery_id.len() > 256 @@ -1453,30 +1476,17 @@ impl DeliveryService { ) else { return None; }; - let current = generation == event.generation; - let at_attempt = current && attempt == event.attempt; - return Some(match (event.phase, event.disposition) { - (DeliveryAuditPhase::Attempt, _) => { - at_attempt && state == "leased" && token.is_some() && token == lease_token - } - (DeliveryAuditPhase::Replay, _) => current && state == "pending", - (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Expired) => { - current && expired - } - (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Delivered) => { - at_attempt && state == "delivered" - } - (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::DeadLettered) => { - at_attempt && state == "dead_lettered" - } - (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::RetryPending) => { - at_attempt && state == "pending" && token.is_none() - } - ( - DeliveryAuditPhase::Terminal, - DeliveryAuditDisposition::Leased | DeliveryAuditDisposition::ReplayPending, - ) => false, - }); + return Some(transition_holds( + event, + &ObservedDelivery { + state: &state, + generation, + attempt, + lease_token: token, + expired, + }, + lease_token, + )); } None } @@ -1565,6 +1575,55 @@ impl DeliveryService { } } +/// One delivery row as read back after a commit returned an error. +struct ObservedDelivery<'a> { + state: &'a str, + generation: i64, + attempt: i16, + lease_token: Option, + expired: bool, +} + +/// Whether `observed` shows the transition `event` records as durable. An +/// attempt's lease is matched on the token this worker wrote. +fn transition_holds( + event: &PendingAudit, + observed: &ObservedDelivery<'_>, + lease_token: Option, +) -> bool { + let current = observed.generation == event.generation; + let at_attempt = current && observed.attempt == event.attempt; + let state = observed.state; + match (event.phase, event.disposition) { + (DeliveryAuditPhase::Attempt, _) => { + at_attempt + && state == "leased" + && observed.lease_token.is_some() + && observed.lease_token == lease_token + } + // Only a replay's reset writes a generation, so the replacement + // generation proves the reset committed whatever state the worker + // has since moved it to. + (DeliveryAuditPhase::Replay, _) => current, + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Expired) => { + current && observed.expired + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Delivered) => { + at_attempt && state == "delivered" + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::DeadLettered) => { + at_attempt && state == "dead_lettered" + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::RetryPending) => { + at_attempt && state == "pending" && observed.lease_token.is_none() + } + ( + DeliveryAuditPhase::Terminal, + DeliveryAuditDisposition::Leased | DeliveryAuditDisposition::ReplayPending, + ) => false, + } +} + /// The post-commit delivery loop: claim, send, finalize, until shutdown. pub struct DeliveryWorker { service: DeliveryService, @@ -2185,6 +2244,44 @@ mod tests { } } + fn replay_of(generation: i64) -> PendingAudit { + PendingAudit { + event_id: Uuid::nil(), + compiled_delivery_id: "delivery".to_owned(), + package_revision: "revision".to_owned(), + generation, + attempt: 0, + phase: DeliveryAuditPhase::Replay, + outcome: DeliveryAuditOutcome::ReplayCommitted, + disposition: DeliveryAuditDisposition::ReplayPending, + } + } + + fn observed(state: &str, generation: i64, attempt: i16) -> ObservedDelivery<'_> { + ObservedDelivery { + state, + generation, + attempt, + lease_token: None, + expired: false, + } + } + + #[test] + fn a_replay_whose_reset_committed_holds_after_the_worker_moves_it() { + let replay = replay_of(2); + for state in ["pending", "leased", "delivered", "dead_lettered"] { + assert!( + transition_holds(&replay, &observed(state, 2, 1), None), + "the replacement generation proves the reset in state {state}" + ); + } + assert!( + !transition_holds(&replay, &observed("dead_lettered", 1, 2), None), + "the prior generation proves the reset rolled back" + ); + } + #[test] fn the_worker_accepts_the_stored_envelope_it_will_deliver_unchanged() { let body = stored_envelope() From ba583674f35a0cafa5e051d04928c452b61236f6 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 17:20:01 +0000 Subject: [PATCH 17/32] fix(breg): resolve erasure and receipt outcomes before auditing them A request-detail erasure whose commit returned an error now reads the detail back on a fresh connection and answers its request with the erasure, failed, or unfinished, instead of assuming a rollback. A replayed ingestion chunk builds its answer inside the release transaction, before its disclosure entry, so an accepted disclosure is never followed by a failure that returns no receipt. Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/postgres/mutation.rs | 32 +++++----- crates/registry-breg/src/request_retention.rs | 61 +++++++++++++++---- .../tests/postgres_request_read_retention.rs | 46 ++++++++++++++ 3 files changed, 114 insertions(+), 25 deletions(-) diff --git a/crates/registry-breg/src/postgres/mutation.rs b/crates/registry-breg/src/postgres/mutation.rs index 9bd94edc2..6d7c0e0a6 100644 --- a/crates/registry-breg/src/postgres/mutation.rs +++ b/crates/registry-breg/src/postgres/mutation.rs @@ -1755,29 +1755,33 @@ impl PostgresRecordMutationService { &principal_reference, Some(&request_correlation), ); - disclosure_transaction - .commit() - .await - .map_err(|_| IngestionServiceError::Unavailable)?; - ingestion_store::append_run_audit(&self.audit, disclosure_record) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; // The attempt row above moved the run's last-attempt marker, so // the answer describes the run as it now stands, not as this - // request found it. The guarded transaction proved the durable - // binding equals this process's identity, so the replayed run - // renders under it. - let run = ingestion_store::load_run(&**client, run.run_id) + // request found it. It is read inside the release transaction, + // which sees that row, and the answer is built before the + // disclosure entry: once that entry is accepted, nothing fallible + // stands between it and the caller. The guarded transaction + // proved the durable binding equals this process's identity, so + // the replayed run renders under it. + let current = ingestion_store::load_run(tx, run.run_id) .await .map_err(|_| IngestionServiceError::Unavailable)? .ok_or(IngestionServiceError::Unavailable)?; - return Ok(json!({ + let answer = json!({ "run": Self::run_response( - &run, + ¤t, (&self.expected.package_revision, &self.expected.schema_fingerprint), ), "receipt": receipt_json(input.chunk_index, &input.digest, true, false, batch), - })); + }); + disclosure_transaction + .commit() + .await + .map_err(|_| IngestionServiceError::Unavailable)?; + ingestion_store::append_run_audit(&self.audit, disclosure_record) + .await + .map_err(|_| IngestionServiceError::Unavailable)?; + return Ok(answer); } // A terminal run stays terminal when the active package later // changes: the blocking transition belongs to open runs alone, so a diff --git a/crates/registry-breg/src/request_retention.rs b/crates/registry-breg/src/request_retention.rs index e152cbbe4..232acac58 100644 --- a/crates/registry-breg/src/request_retention.rs +++ b/crates/registry-breg/src/request_retention.rs @@ -479,10 +479,33 @@ impl RequestRetentionOperatorService { .erase_in_transaction(&mut client, scope.clone(), correlation) .await; let (plan, erasure, entry) = match erased { - Ok(erased) => erased, + Ok((plan, erasure, entry, true)) => (plan, erasure, entry), + Ok((plan, erasure, entry, false)) => { + // A commit that returned an error may still have committed, + // so the outcome recorded for this destructive operation is + // the one the database holds, read on a fresh connection. + match self.erasure_committed(&pool, scope.clone()).await { + Some(true) => (plan, erasure, entry), + resolved => { + let outcome = if resolved == Some(false) { + "failed" + } else { + "unfinished" + }; + let answer = retention_outcome_record(&request_record, outcome); + if attempt.respond(answer).await.is_err() { + tracing::error!( + "the unacknowledged erasure's response audit entry was not recorded" + ); + } + return Err(RequestRetentionError::Unavailable); + } + } + } Err(error) => { - // Nothing committed: answer the request with the refusal or - // the failure. + // The erasure failed before its commit, so nothing + // committed: answer the request with the refusal or the + // failure. let outcome = if error == RequestRetentionError::Unavailable { "failed" } else { @@ -525,14 +548,33 @@ impl RequestRetentionOperatorService { }) } - /// Erase one request's detail in one committed transaction and build the - /// terminal entry that records it. + /// Whether the detail `scope` names is erased, read on a fresh + /// connection after an erasure commit returned an error. `None` when the + /// state cannot be read. + async fn erasure_committed( + &self, + pool: &crate::postgres::RuntimePool, + scope: RequestDetailErasureScope<'_>, + ) -> Option { + let mut client = pool.get().await.ok()?; + let transaction = self.begin_verified_transaction(&mut client).await.ok()?; + let plan = load_erasure_plan(&transaction, &self.registry, scope, false) + .await + .ok()?; + transaction.commit().await.ok()?; + Some(plan.detail_erased) + } + + /// Erase one request's detail in one transaction and build the terminal + /// entry that records it. The flag is false when the commit returned an + /// error, which does not prove the transaction rolled back; every + /// earlier error is returned as one. async fn erase_in_transaction( &self, client: &mut deadpool_postgres::Client, scope: RequestDetailErasureScope<'_>, correlation: RequestCorrelation, - ) -> Result<(RequestErasurePlan, RequestDetailErasure, AuditEntry)> { + ) -> Result<(RequestErasurePlan, RequestDetailErasure, AuditEntry, bool)> { let transaction = self.begin_verified_transaction(client).await?; let plan = load_erasure_plan(&transaction, &self.registry, scope.clone(), true).await?; let (erasure, current_revision) = @@ -566,11 +608,8 @@ impl RequestRetentionOperatorService { erasure, correlation, )?; - transaction - .commit() - .await - .map_err(|_| RequestRetentionError::Unavailable)?; - Ok((plan, erasure, entry)) + let acknowledged = transaction.commit().await.is_ok(); + Ok((plan, erasure, entry, acknowledged)) } /// Retry orphaned external objects even when every request is active or diff --git a/crates/registry-breg/tests/postgres_request_read_retention.rs b/crates/registry-breg/tests/postgres_request_read_retention.rs index cfb15341d..57b1da26c 100644 --- a/crates/registry-breg/tests/postgres_request_read_retention.rs +++ b/crates/registry-breg/tests/postgres_request_read_retention.rs @@ -849,6 +849,52 @@ async fn request_detail_erasure_pairs_its_request_entry_on_every_outcome() { "the failed erasure erased nothing" ); + // The erasure's commit is refused after every statement succeeded, so + // the outcome is read back from the database: nothing was erased. + database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_erasure_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'test refuses this erasure commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_erasure_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_erasure_commit + AFTER UPDATE ON registry_internal.registry_request_proposals + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_erasure_commit();", + ) + .await + .expect("administrator installs the erasure commit refusal"); + assert_eq!( + retained.erase(scope.clone()).await, + Err(RequestRetentionError::Unavailable) + ); + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_erasure_commit + ON registry_internal.registry_request_proposals; + DROP FUNCTION public.test_refuse_erasure_commit();", + ) + .await + .expect("administrator removes the erasure commit refusal"); + let unacknowledged = &capture.entries()[failed.len()..]; + assert_eq!(unacknowledged.len(), 2, "{unacknowledged:?}"); + assert_eq!(unacknowledged[1]["record"]["outcome"], "failed"); + assert_eq!( + unacknowledged[0]["correlation"], + unacknowledged[1]["correlation"] + ); + assert!( + !retention_with(capture.audit(profile())) + .dry_run(scope.clone()) + .await + .expect("the detail still plans") + .detail_erased, + "the refused commit erased nothing" + ); + let failed = capture.entries(); + // The destination accepts the request entry and refuses the response. capture.fail_after(1); assert_eq!( From c39868102a1d0b38afd2df1086a1df746b2a4f8c Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 17:39:31 +0000 Subject: [PATCH 18/32] fix(audit): claim a request for a response appended in its name A response appended through AuditWriter::append now claims the oldest open request under its schema and correlation before its write starts, so a handle dropped while that response is written leaves the request to it instead of also writing its unfinished record. Signed-off-by: Jeremi Joslin --- crates/registry-platform-audit/src/writer.rs | 138 +++++++++++++++---- 1 file changed, 109 insertions(+), 29 deletions(-) diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index a36c10710..c9eb1e46e 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -532,33 +532,40 @@ pub struct AuditWriter { /// The schema and correlation a request entry and its responses share. type RequestKey = (String, String); +/// One request entry still owed a response: whether one was accepted, and +/// the state its handle shares with the responses being written. +#[derive(Clone)] +struct OpenRequest { + answered: Arc, + state: Arc>, +} + /// The request entries whose [`AuditRequest`] still owes a response, by /// schema and correlation, oldest first. #[derive(Default)] -struct OpenRequests(StdMutex>>>); +struct OpenRequests(StdMutex>>); impl OpenRequests { - fn open(&self, key: RequestKey) -> Arc { - let answered = Arc::new(AtomicBool::new(false)); + fn open(&self, key: RequestKey, request: OpenRequest) { if let Ok(mut open) = self.0.lock() { - open.entry(key).or_default().push(Arc::clone(&answered)); + open.entry(key).or_default().push(request); } - answered } - /// Mark the oldest open request under `key` answered. - fn answer(&self, key: &RequestKey) { - let Ok(mut open) = self.0.lock() else { - return; - }; - if let Some(waiting) = open.get_mut(key) { - if !waiting.is_empty() { - waiting.remove(0).store(true, Ordering::Release); - } - if waiting.is_empty() { - open.remove(key); + /// Claim the oldest open request under `key` that no appended response + /// has claimed yet, counting that response as in flight on it, so its + /// handle dropped meanwhile leaves the request to that response. + fn claim(&self, key: &RequestKey) -> Option { + let open = self.0.lock().ok()?; + open.get(key)?.iter().find_map(|request| { + let mut state = request.state.lock().ok()?; + if state.claimed { + return None; } - } + state.claimed = true; + state.in_flight += 1; + Some(request.clone()) + }) } /// Close `answered` under `key`, reporting whether it is still owed a @@ -572,7 +579,7 @@ impl OpenRequests { return false; } if let Some(waiting) = open.get_mut(key) { - waiting.retain(|candidate| !Arc::ptr_eq(candidate, answered)); + waiting.retain(|candidate| !Arc::ptr_eq(&candidate.answered, answered)); if waiting.is_empty() { open.remove(key); } @@ -638,14 +645,33 @@ impl AuditWriter { pub async fn append(&self, entry: AuditEntry) -> Result<(), AuditUnavailable> { // The write and the bookkeeping it implies run in one task that // outlives a canceled caller, so an accepted response always answers - // its request, whether or not the caller is still waiting. + // its request, whether or not the caller is still waiting. The + // request is claimed before that task starts, so a handle dropped + // while the response is written leaves the request to it. + let claimed = if entry.phase == AuditPhase::Response { + let key = (entry.schema.clone(), entry.correlation.clone()); + self.open.claim(&key).map(|request| (key, request)) + } else { + None + }; let writer = self.clone(); tokio::spawn(async move { - writer.write(&entry).await?; - if entry.phase == AuditPhase::Response { - writer.open.answer(&(entry.schema, entry.correlation)); + let result = writer.write(&entry).await; + if let Some((key, request)) = claimed { + if result.is_ok() { + writer.open.close(&key, &request.answered); + request.answered.store(true, Ordering::Release); + } + let owes_unfinished = request.state.lock().is_ok_and(|mut state| { + state.claimed = false; + state.in_flight -= 1; + state.in_flight == 0 && state.dropped + }); + if owes_unfinished { + settle_unanswered(&writer, key, &request.answered, &request.state); + } } - Ok(()) + result }) .await .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? @@ -706,17 +732,26 @@ impl AuditWriter { request, )) .await?; - let answered = writer.open.open((schema.clone(), correlation.clone())); + let answered = Arc::new(AtomicBool::new(false)); + let state = Arc::new(StdMutex::new(RequestState { + unfinished: Some(unfinished), + in_flight: 0, + claimed: false, + dropped: false, + })); + writer.open.open( + (schema.clone(), correlation.clone()), + OpenRequest { + answered: Arc::clone(&answered), + state: Arc::clone(&state), + }, + ); Ok(AuditRequest { writer, schema, correlation, answered, - state: Arc::new(StdMutex::new(RequestState { - unfinished: Some(unfinished), - in_flight: 0, - dropped: false, - })), + state, }) }) .await @@ -816,6 +851,9 @@ struct RequestState { unfinished: Option, /// Responses whose write has started and not yet settled. in_flight: usize, + /// A response appended through [`AuditWriter::append`] claimed this + /// request and has not settled. + claimed: bool, /// The handle was dropped while a response was in flight. dropped: bool, } @@ -3711,6 +3749,48 @@ mod tests { assert_paired(&buffered_lines(&sink.buffer), "returned"); } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_request_dropped_while_an_appended_response_is_written_is_answered_once() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The caller and its handle go away while the appended response is + // written: that response answers the request, and the drop writes + // nothing more. + appending.abort(); + drop(request); + sink.release(); + lines_eventually(&sink.buffer, 2).await; + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + let lines = buffered_lines(&sink.buffer); + assert_eq!(lines.len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + #[tokio::test] async fn an_unfinished_record_too_large_to_write_refuses_the_request() { let (writer, buffer) = buffered(); From 91ece6281f7399f209d9cc3f4f441a08c428479d Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 17:39:31 +0000 Subject: [PATCH 19/32] fix(breg): read back retention and reconcile outcomes after commit errors An Evidence retention erasure whose commit returned an error now checks on a fresh connection whether expired material remains, and a reconciliation transition that returned an error reads the maintenance state back. Each answers its request with the durable outcome, failed, or unfinished when that state cannot be read, instead of assuming a rollback. Signed-off-by: Jeremi Joslin --- .../src/action_evidence_maintenance.rs | 79 +++++++++++++++++-- .../registry-breg/src/migration_reconcile.rs | 49 +++++++++--- .../postgres_action_evidence_retention.rs | 35 +++++++- 3 files changed, 146 insertions(+), 17 deletions(-) diff --git a/crates/registry-breg/src/action_evidence_maintenance.rs b/crates/registry-breg/src/action_evidence_maintenance.rs index 718a54404..67efa7f8c 100644 --- a/crates/registry-breg/src/action_evidence_maintenance.rs +++ b/crates/registry-breg/src/action_evidence_maintenance.rs @@ -133,7 +133,32 @@ impl ActionEvidenceRetentionOperatorService { ) .await .map_err(|_| MutationError::Unavailable)?; - match self.erase_in_transaction(before).await { + let erased = match self.erase_in_transaction(before).await { + Ok(Erasure::Committed(erased)) => Ok(erased), + // A commit that returned an error may still have committed, so + // the outcome recorded for this destructive operation is the one + // the database holds, read on a fresh connection. + Ok(Erasure::Unacknowledged { erased, cutoff }) => { + match self.expired_evidence_remains(cutoff).await { + Some(false) => Ok(erased), + resolved => { + let outcome = if resolved == Some(true) { + "failed" + } else { + "unfinished" + }; + if attempt.respond(record("terminal", outcome)).await.is_err() { + tracing::error!( + "the unacknowledged Evidence retention's response audit entry was not recorded" + ); + } + return Err(MutationError::Unavailable); + } + } + } + Err(error) => Err(error), + }; + match erased { Ok(erased) => { let mut response = record("terminal", "erased"); response["erased"] = json!(erased); @@ -156,10 +181,36 @@ impl ActionEvidenceRetentionOperatorService { } } + /// Whether retained Evidence expiring at or before `cutoff` remains, read + /// on a fresh connection after an erasure commit returned an error. + /// `None` when it cannot be read. + async fn expired_evidence_remains( + &self, + cutoff: chrono::DateTime, + ) -> Option { + let pool = self.migration_connection.build_pool().ok()?; + let client = pool.get().await.ok()?; + let row = client + .query_one( + "SELECT EXISTS (SELECT 1 FROM registry_internal.registry_action_evidence_uses + WHERE expires_at <= $1) + OR EXISTS (SELECT 1 FROM registry_internal.registry_request_evidence_uses + WHERE expires_at <= $1)", + &[&cutoff], + ) + .await + .ok()?; + row.try_get(0).ok() + } + + /// Erase the expired material in one transaction. A commit that returned + /// an error, which does not prove the transaction rolled back, is + /// reported with the count it would have erased and the cutoff it erased + /// through; every earlier error is returned as one. async fn erase_in_transaction( &self, before: chrono::DateTime, - ) -> Result { + ) -> Result { let pool = self .migration_connection .build_pool() @@ -199,15 +250,29 @@ impl ActionEvidenceRetentionOperatorService { if !ready { return Err(MutationError::Unavailable); } - let erased = erase_expired_action_evidence(&transaction, before).await?; - transaction - .commit() + // The cutoff the deletion applies, fixed by the transaction's start. + let cutoff: chrono::DateTime = transaction + .query_one("SELECT LEAST($1, CURRENT_TIMESTAMP)", &[&before]) .await - .map_err(|_| MutationError::Unavailable)?; - Ok(erased) + .map_err(|_| MutationError::Unavailable)? + .get(0); + let erased = erase_expired_action_evidence(&transaction, before).await?; + if transaction.commit().await.is_err() { + return Ok(Erasure::Unacknowledged { erased, cutoff }); + } + Ok(Erasure::Committed(erased)) } } +/// How an erasure transaction ended once every statement in it succeeded. +enum Erasure { + Committed(u64), + Unacknowledged { + erased: u64, + cutoff: chrono::DateTime, + }, +} + /// Erase only material whose declared expiry has passed. Diagnostics contain /// no connection, selector, assertion or provider values. pub async fn erase_expired(path: &Path, before: &str) -> Result { diff --git a/crates/registry-breg/src/migration_reconcile.rs b/crates/registry-breg/src/migration_reconcile.rs index b2dbd6581..80ffb3a43 100644 --- a/crates/registry-breg/src/migration_reconcile.rs +++ b/crates/registry-breg/src/migration_reconcile.rs @@ -358,8 +358,13 @@ async fn reconcile_under_lock( ) .await; if let Err(error) = transition { - respond_failed(&mut attempt, request, target, ledger, "completed").await; - return Err(error.into()); + let landed = + transition_landed(connection, |snapshot| snapshot.identity == *target).await; + if landed != Some(true) { + respond_unlanded(&mut attempt, request, target, ledger, "completed", landed) + .await; + return Err(error.into()); + } } append_after_commit(request.audit, entry).await?; } @@ -379,8 +384,14 @@ async fn reconcile_under_lock( ) .await; if let Err(error) = transition { - respond_failed(&mut attempt, request, target, ledger, "reverted").await; - return Err(error.into()); + let landed = + transition_landed(connection, |snapshot| snapshot.identity == *request.current) + .await; + if landed != Some(true) { + respond_unlanded(&mut attempt, request, target, ledger, "reverted", landed) + .await; + return Err(error.into()); + } } append_after_commit(request.audit, entry).await?; } @@ -448,17 +459,37 @@ async fn begin_request( .map_err(|_| ReconcileError::Unavailable) } -/// Answer the request entry of a transition that did not commit. The -/// reconciliation already failed, so a refused entry is only logged; the -/// held request then writes its `unfinished` outcome instead. -async fn respond_failed( +/// Whether a transition that returned an error nonetheless landed: the +/// maintenance target is cleared and the active identity is the one +/// `landed` expects. An error does not prove the transaction rolled back, so +/// the durable state decides; `None` when it cannot be read. +async fn transition_landed( + connection: &mut VerifiedPackageApplyConnection, + landed: impl FnOnce(&MaintenanceSnapshot) -> bool, +) -> Option { + let snapshot = connection.maintenance_snapshot().await.ok()?; + Some(snapshot.maintenance_target_revision.is_none() && landed(&snapshot)) +} + +/// Answer the request entry of a transition that did not land: `failed` +/// when the durable state shows it did not, `unfinished` when that state +/// could not be read. The reconciliation already failed, so a refused entry +/// is only logged; the held request then writes its `unfinished` outcome +/// instead. +async fn respond_unlanded( attempt: &mut AuditRequest, request: &ReconcileRequest<'_>, target: &ExpectedRegistryIdentity, ledger: &MigrationLedgerEntry, action: &'static str, + landed: Option, ) { - let recorded = match outcome_record(request, target, ledger, action, "failed") { + let outcome = if landed == Some(false) { + "failed" + } else { + "unfinished" + }; + let recorded = match outcome_record(request, target, ledger, action, outcome) { Ok(record) => attempt.respond(record).await.is_ok(), Err(_) => false, }; diff --git a/crates/registry-breg/tests/postgres_action_evidence_retention.rs b/crates/registry-breg/tests/postgres_action_evidence_retention.rs index 7bec3944a..94c6d106e 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_retention.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_retention.rs @@ -10,6 +10,7 @@ use registry_breg::{ action_evidence_maintenance::ActionEvidenceRetentionOperatorService, compiler::{compile_project_with_assets, CompileProfile}, contract::{parse_project_yaml, ModuleAssetSource}, + mutation::MutationError, postgres::{ initialize_registry_state_for_catalog_test, install_compiled_schema, ConnectionConfig, ExpectedManagedCatalog, ExpectedRegistryIdentity, RegistryLockKey, @@ -185,6 +186,35 @@ async fn expired_request_evidence_erases_only_retained_uses() { &expected, database.migration_config.clone(), ); + // The erasure's commit is refused after every statement succeeded, so + // its outcome is read back from the database: nothing was erased. + database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_evidence_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'test refuses this erasure commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_evidence_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_evidence_commit + AFTER DELETE ON registry_internal.registry_request_evidence_uses + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_evidence_commit();", + ) + .await + .unwrap(); + assert!(matches!( + operator.erase_expired(cutoff()).await, + Err(MutationError::Unavailable) + )); + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_evidence_commit + ON registry_internal.registry_request_evidence_uses; + DROP FUNCTION public.test_refuse_evidence_commit();", + ) + .await + .unwrap(); assert_eq!(operator.erase_expired(cutoff()).await.unwrap(), 1); let remaining = database .admin @@ -199,7 +229,10 @@ async fn expired_request_evidence_erases_only_retained_uses() { assert_eq!(remaining.get::<_, i64>(0), 1); assert_eq!(remaining.get::<_, i64>(1), 1); assert_eq!(operator.erase_expired(cutoff()).await.unwrap(), 0); - assert_retention_audited(&database, &[("erased", Some(1)), ("erased", Some(0))]); + assert_retention_audited( + &database, + &[("failed", None), ("erased", Some(1)), ("erased", Some(0))], + ); drop(operator); database.cleanup().await; } From f2c8221d4e4c0ae55ca769b2581d7695555b7464 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 18:05:25 +0000 Subject: [PATCH 20/32] fix(audit): queue a dropped request's file response outside a runtime A request dropped outside a Tokio runtime while the file state is busy now waits for the lock to queue its unfinished response, instead of discarding that response after a bounded number of attempts. No holder keeps the lock across an await, and the file's drop or the next group commit writes the queued line. Signed-off-by: Jeremi Joslin --- crates/registry-platform-audit/src/writer.rs | 60 +++++++++++++++++++- 1 file changed, 58 insertions(+), 2 deletions(-) diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index c9eb1e46e..ccf104a62 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -1222,8 +1222,20 @@ impl GroupCommitFile { std::thread::yield_now(); } let Some(runtime) = runtime else { - if line.is_some() { - tracing::error!("audit file state is busy outside a runtime; an unfinished response entry was not written"); + // Outside a runtime nothing can flush the line later, so it is + // queued under a blocking wait for the lock, which no holder + // keeps across an await; the file's drop or the next group + // commit writes it. + if let Some(line) = line { + let mut state = self.state.blocking_lock(); + if state.stopped { + tracing::error!( + "audit writer stopped; an unfinished response entry was not written" + ); + return; + } + state.pending.push(line); + state.enqueued = state.enqueued.saturating_add(1); } return; }; @@ -3576,6 +3588,50 @@ mod tests { assert_paired(&lines(&path), "unfinished"); } + #[test] + fn a_request_dropped_outside_a_runtime_while_the_file_is_busy_still_pairs() { + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let (writer, request) = runtime.block_on(async move { + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "erase"}), + unfinished(), + ) + .await + .expect("request"); + (writer, request) + }); + drop(runtime); + let WriterInner::File(file) = writer.inner.as_ref() else { + panic!("a file destination"); + }; + let file = Arc::clone(file); + let (held, holding) = std::sync::mpsc::channel(); + // Another thread holds the file state for longer than any bounded + // wait while the handle is dropped outside a runtime. + let holder = std::thread::spawn(move || { + let _state = file.state.blocking_lock(); + held.send(()).expect("signal"); + std::thread::sleep(Duration::from_millis(200)); + }); + holding.recv().expect("the state is held"); + drop(request); + holder.join().expect("holder"); + drop(writer); + assert_paired(&lines(&path), "unfinished"); + } + #[tokio::test] async fn a_stopped_writer_refuses_the_request_and_writes_nothing() { let writer = AuditWriter::from_line_sink(Box::new(FailingSink)); From f94e9a70e96310ac07ca40606c1cd60de5004546 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 18:05:25 +0000 Subject: [PATCH 21/32] fix(hooks): recognize retry and dead-letter commits after later progress A scheduled retry whose commit acknowledgement was lost is recognized by a later attempt of the same generation, and a dead letter by a later replay generation, rather than only by the state the worker may already have moved on from. Signed-off-by: Jeremi Joslin --- .../src/delivery/service.rs | 60 ++++++++++++++++++- 1 file changed, 58 insertions(+), 2 deletions(-) diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index f1006e231..ed10da8c0 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -1611,11 +1611,17 @@ fn transition_holds( (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Delivered) => { at_attempt && state == "delivered" } + // A dead letter leaves that state only through a replay, which + // writes a later generation. (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::DeadLettered) => { - at_attempt && state == "dead_lettered" + (at_attempt && state == "dead_lettered") || observed.generation > event.generation } + // A scheduled retry is claimed again under a later attempt of the + // same generation, which a rolled-back retry reaches only after its + // lease expires and the reaper answers the attempt itself. (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::RetryPending) => { - at_attempt && state == "pending" && observed.lease_token.is_none() + (at_attempt && state == "pending" && observed.lease_token.is_none()) + || (current && observed.attempt > event.attempt) } ( DeliveryAuditPhase::Terminal, @@ -2267,6 +2273,56 @@ mod tests { } } + fn terminal_of(disposition: DeliveryAuditDisposition) -> PendingAudit { + PendingAudit { + attempt: 1, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition, + ..replay_of(1) + } + } + + #[test] + fn a_committed_retry_holds_after_its_next_attempt_is_claimed() { + let retry = terminal_of(DeliveryAuditDisposition::RetryPending); + assert!(transition_holds(&retry, &observed("pending", 1, 1), None)); + assert!( + transition_holds(&retry, &observed("leased", 1, 2), None), + "the next attempt proves the retry committed" + ); + assert!( + !transition_holds( + &retry, + &ObservedDelivery { + lease_token: Some(Uuid::nil()), + ..observed("leased", 1, 1) + }, + None + ), + "a lease still at the attempt proves the retry rolled back" + ); + } + + #[test] + fn a_committed_dead_letter_holds_after_it_is_replayed() { + let dead_letter = terminal_of(DeliveryAuditDisposition::DeadLettered); + assert!(transition_holds( + &dead_letter, + &observed("dead_lettered", 1, 1), + None + )); + assert!( + transition_holds(&dead_letter, &observed("pending", 2, 0), None), + "a replay generation proves the dead letter committed" + ); + assert!(!transition_holds( + &dead_letter, + &observed("leased", 1, 1), + None + )); + } + #[test] fn a_replay_whose_reset_committed_holds_after_the_worker_moves_it() { let replay = replay_of(2); From 6cad72874dc7b0fcee560ec4dee7420bd06b0d19 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 18:56:20 +0000 Subject: [PATCH 22/32] fix(audit): answer a request once when a claim races its drop or shutdown A response claimed between a handle's in-flight check and its close no longer lets the drop write unfinished as a second answer, and a response task dropped by a runtime shutdown releases its in-flight count so the handle still pairs the request. A poisoned open-request map and a detached line lost at shutdown are logged. Signed-off-by: Jeremi Joslin --- crates/registry-platform-audit/src/writer.rs | 305 ++++++++++++++++--- 1 file changed, 265 insertions(+), 40 deletions(-) diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index ccf104a62..88843ff19 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -547,8 +547,14 @@ struct OpenRequests(StdMutex open.entry(key).or_default().push(request), + // The request is still owed by its handle, which writes its + // unfinished response; only a response appended elsewhere can + // no longer answer it. + Err(_) => tracing::error!( + "the open audit requests are poisoned; a response appended elsewhere will not answer this request" + ), } } @@ -568,23 +574,92 @@ impl OpenRequests { }) } - /// Close `answered` under `key`, reporting whether it is still owed a - /// response. Checked and removed under one lock, so a response cannot be - /// counted after the owner decided it had none. - fn close(&self, key: &RequestKey, answered: &Arc) -> bool { + /// Close `answered` under `key` once a response to it was accepted. + fn answer(&self, key: &RequestKey, answered: &Arc) { + if let Ok(mut open) = self.0.lock() { + Self::remove(&mut open, key, answered); + } + } + + /// Close `answered` under `key` for its unfinished response, reporting + /// whether it is still owed one: not when a response was accepted, and + /// not when one is in flight on it, since that response settles the + /// request itself. Checked and removed under the lock [`Self::claim`] + /// takes, so a response cannot be claimed after the owner decided it had + /// none. + fn close_unanswered( + &self, + key: &RequestKey, + answered: &Arc, + state: &StdMutex, + ) -> bool { let Ok(mut open) = self.0.lock() else { return !answered.load(Ordering::Acquire); }; if answered.load(Ordering::Acquire) { return false; } + if state.lock().is_ok_and(|state| state.in_flight > 0) { + return false; + } + Self::remove(&mut open, key, answered); + true + } + + fn remove( + open: &mut std::collections::HashMap>, + key: &RequestKey, + answered: &Arc, + ) { if let Some(waiting) = open.get_mut(key) { waiting.retain(|candidate| !Arc::ptr_eq(&candidate.answered, answered)); if waiting.is_empty() { open.remove(key); } } - true + } +} + +/// One response counted in flight on the request it answers. Dropping it +/// settles that count, whether its write finished or its task was dropped +/// before it could, such as by a runtime shutting down, so the request's +/// handle is never left waiting on a response that will not come. +struct InFlightResponse { + writer: AuditWriter, + key: RequestKey, + answered: Arc, + state: Arc>, + /// The response was appended through [`AuditWriter::append`] and holds + /// the request's claim. + claimed: bool, +} + +impl InFlightResponse { + /// Record that the response was accepted, answering this request and + /// not an older one open under its correlation. + fn accepted(&self) { + self.writer.open.answer(&self.key, &self.answered); + self.answered.store(true, Ordering::Release); + } +} + +impl Drop for InFlightResponse { + fn drop(&mut self) { + let owes_unfinished = self.state.lock().is_ok_and(|mut state| { + if self.claimed { + state.claimed = false; + } + state.in_flight -= 1; + state.in_flight == 0 && state.dropped + }); + if owes_unfinished { + settle_unanswered( + &self.writer, + std::mem::take(&mut self.key), + &self.answered, + &self.state, + ); + } } } @@ -650,25 +725,22 @@ impl AuditWriter { // while the response is written leaves the request to it. let claimed = if entry.phase == AuditPhase::Response { let key = (entry.schema.clone(), entry.correlation.clone()); - self.open.claim(&key).map(|request| (key, request)) + self.open.claim(&key).map(|request| InFlightResponse { + writer: self.clone(), + key, + answered: request.answered, + state: request.state, + claimed: true, + }) } else { None }; let writer = self.clone(); tokio::spawn(async move { let result = writer.write(&entry).await; - if let Some((key, request)) = claimed { + if let Some(claimed) = claimed { if result.is_ok() { - writer.open.close(&key, &request.answered); - request.answered.store(true, Ordering::Release); - } - let owes_unfinished = request.state.lock().is_ok_and(|mut state| { - state.claimed = false; - state.in_flight -= 1; - state.in_flight == 0 && state.dropped - }); - if owes_unfinished { - settle_unanswered(&writer, key, &request.answered, &request.state); + claimed.accepted(); } } result @@ -706,6 +778,10 @@ impl AuditWriter { /// correlation. A response appended through [`Self::append`] under the /// same schema and correlation answers it too, so an operation whose /// outcome is written elsewhere only holds the handle until it returns. + /// + /// That pairing holds while the process runs. A process killed or + /// exited without unwinding, or a runtime shut down while an entry is + /// being written, can leave a request entry without its response. pub async fn begin( &self, schema: impl Into, @@ -891,26 +967,19 @@ impl AuditRequest { if let Ok(mut state) = self.state.lock() { state.in_flight += 1; } - let writer = self.writer.clone(); - let key = (self.schema.clone(), self.correlation.clone()); - let answered = Arc::clone(&self.answered); - let state = Arc::clone(&self.state); + // A poisoned state skips both the count and its release. + let in_flight = InFlightResponse { + writer: self.writer.clone(), + key: (self.schema.clone(), self.correlation.clone()), + answered: Arc::clone(&self.answered), + state: Arc::clone(&self.state), + claimed: false, + }; + let entry = AuditEntry::response(self.schema.clone(), self.correlation.clone(), record); tokio::spawn(async move { - let result = writer - .write(&AuditEntry::response(key.0.clone(), key.1.clone(), record)) - .await; + let result = in_flight.writer.write(&entry).await; if result.is_ok() { - // Answer this request, not an older one open under its - // correlation. - writer.open.close(&key, &answered); - answered.store(true, Ordering::Release); - } - let owes_unfinished = state.lock().is_ok_and(|mut state| { - state.in_flight -= 1; - state.in_flight == 0 && state.dropped - }); - if owes_unfinished { - settle_unanswered(&writer, key, &answered, &state); + in_flight.accepted(); } result }) @@ -931,7 +1000,7 @@ fn settle_unanswered( answered: &Arc, state: &StdMutex, ) { - if !writer.open.close(&key, answered) { + if !writer.open.close_unanswered(&key, answered, state) { return; } let unfinished = state @@ -1154,6 +1223,12 @@ impl GroupCommitFile { state.enqueued = state.enqueued.saturating_add(1); state.enqueued }; + self.wait_durable(position).await + } + + /// Wait until the line queued at `position` is durable, flushing the + /// queue when no other append is. + async fn wait_durable(&self, position: u64) -> Result<(), AuditUnavailable> { loop { if self.durable.load(Ordering::Acquire) >= position { return Ok(()); @@ -1240,9 +1315,27 @@ impl GroupCommitFile { return; }; let file = Arc::clone(self); + // A line not queued yet is lost if the runtime drops this task + // before it runs, such as at shutdown; that loss is reported. + let unqueued = line.map(|line| UnqueuedLine(Some(line))); runtime.spawn(async move { - let result = match line { - Some(line) => file.append(line).await, + let result = match unqueued { + Some(mut unqueued) => { + let position = { + let mut state = file.state.lock().await; + if state.stopped { + Err(AuditUnavailable::new(AuditUnavailableReason::Stopped)) + } else { + state.pending.extend(unqueued.0.take()); + state.enqueued = state.enqueued.saturating_add(1); + Ok(state.enqueued) + } + }; + match position { + Ok(position) => file.wait_durable(position).await, + Err(error) => Err(error), + } + } None => { let _writer = file.flush.lock().await; file.flush_once().await @@ -1255,6 +1348,18 @@ impl GroupCommitFile { } } +/// A detached line not yet handed to the group commit, which reports its +/// loss if dropped still holding it. +struct UnqueuedLine(Option); + +impl Drop for UnqueuedLine { + fn drop(&mut self) { + if self.0.is_some() { + tracing::error!("an unfinished response entry was dropped before it was written"); + } + } +} + impl Drop for GroupCommitFile { /// Write the lines still queued, such as an unfinished response whose /// flush was canceled when the runtime shut down. @@ -3885,4 +3990,124 @@ mod tests { writer.wait_for_detached_entries(); assert_paired(&buffered_lines(&sink.buffer), "unfinished"); } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_claimed_while_its_request_is_dropped_is_the_only_answer() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + // The drop's first half: it marks the handle dropped and finds no + // response in flight. + let in_flight = request.state.lock().is_ok_and(|mut state| { + state.dropped = true; + state.in_flight > 0 + }); + assert!(!in_flight); + // A response appended elsewhere claims the request before the drop + // settles it. + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The drop's second half. + settle_unanswered( + &writer, + (SCHEMA.to_owned(), "req-1".to_owned()), + &request.answered, + &request.state, + ); + sink.release(); + appending + .await + .expect("append task") + .expect("the claimed response"); + drop(request); + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + let lines = buffered_lines(&sink.buffer); + assert_eq!(lines.len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + + #[test] + fn a_response_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle() { + let (writer, buffer) = buffered(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let mut request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + // The response's task is spawned and never runs: the runtime shuts + // down first and drops it. + runtime.block_on(async { + tokio::select! { + biased; + _ = request.respond(json!({"outcome": "returned"})) => { + panic!("the response task never ran") + } + () = std::future::ready(()) => {} + } + }); + drop(runtime); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[test] + fn an_append_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle() { + let (writer, buffer) = buffered(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + // The appended response claims the request, and its task is dropped + // by the runtime's shutdown before it runs. + runtime.block_on(async { + tokio::select! { + biased; + _ = writer.append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) => panic!("the append task never ran"), + () = std::future::ready(()) => {} + } + }); + drop(runtime); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } } From ebd09b568b7680e2dc125413b5076b39025f5af0 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:03:26 +0000 Subject: [PATCH 23/32] fix(breg): hold an Evidence action's attempt until its terminal answers it Preflight dropped the attempt handle when it returned, so an admitted Evidence action was answered unfinished while evaluation still ran and answered again by its terminal or refusal. The admission now carries the held attempt to finalize, and the tests assert every audit request has exactly one response. Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/mutation/action.rs | 60 ++++++++++++++++++- crates/registry-breg/src/postgres/mutation.rs | 10 ---- .../tests/postgres_action_evidence.rs | 11 ++-- .../postgres_action_evidence_retention.rs | 4 ++ .../tests/postgres_action_evidence_targets.rs | 1 + .../registry-breg/tests/postgres_mutation.rs | 1 + .../tests/support/postgres_harness.rs | 31 ++++++++++ 7 files changed, 101 insertions(+), 17 deletions(-) diff --git a/crates/registry-breg/src/mutation/action.rs b/crates/registry-breg/src/mutation/action.rs index 24c2b5511..1495b8f7e 100644 --- a/crates/registry-breg/src/mutation/action.rs +++ b/crates/registry-breg/src/mutation/action.rs @@ -2695,9 +2695,16 @@ pub(crate) struct PreparedEvidenceAction { binding: crate::idempotency::ResolvedIdempotencyBinding, reserved_creates: BTreeMap, application_id: Uuid, + /// The action's attempt, held across Evidence evaluation until the + /// finalize terminal or refusal answers it. Dropping the admission + /// before that answers the attempt as unfinished. + attempt: Option, } impl MutationCoordinator { + /// Admit an Evidence action, recording its refusal when admission fails + /// before the deadline. A successful admission carries the held attempt + /// to [`Self::finalize_evidence_action`]; a recovered receipt answers it. #[allow(clippy::too_many_arguments)] pub(crate) async fn preflight_evidence_action( &self, @@ -2707,6 +2714,51 @@ impl MutationCoordinator { claims: &ActionClaimContext, target_authority: &BTreeMap>, deadline: tokio::time::Instant, + ) -> Result, MutationError> { + let route_id = input.route_id; + let correlation = input.correlation; + let mut attempt = None; + let result = self + .preflight_evidence_action_holding( + client, + registry, + input, + claims, + target_authority, + deadline, + &mut attempt, + ) + .await; + if result.is_err() && tokio::time::Instant::now() < deadline { + self.record_action_boundary_audit( + claims, + route_id, + correlation, + PreIoAuditKind::Refusal, + ) + .await?; + } + Ok(match result? { + Ok(mut prepared) => { + prepared.attempt = attempt; + Ok(prepared) + } + Err(receipt) => Err(receipt), + }) + } + + /// Admission itself. The attempt it begins is left in `attempt`, so the + /// caller answers it on every path. + #[allow(clippy::too_many_arguments)] + async fn preflight_evidence_action_holding( + &self, + client: &mut Client, + registry: &CompiledRegistry, + input: ImmediateActionInput<'_>, + claims: &ActionClaimContext, + target_authority: &BTreeMap>, + deadline: tokio::time::Instant, + attempt: &mut Option, ) -> Result, MutationError> { if !profile_is_keyed(self.audit.profile()) { return Err(MutationError::Unavailable); @@ -2720,9 +2772,10 @@ impl MutationCoordinator { validate_action_claims(action, claims, Operation::Invoke)?; let normalized = validate_action_input(action, input.input)?; validate_precondition_set(action, &input.preconditions)?; - let _attempt = self - .begin_action_boundary_audit(claims, input.route_id, input.correlation) - .await?; + *attempt = Some( + self.begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?, + ); let binding = resolve_action_binding( self.audit.profile(), &ActionIdempotencyBinding { @@ -2819,6 +2872,7 @@ impl MutationCoordinator { binding, reserved_creates: reserve_action_create_ids(action)?, application_id, + attempt: None, })) } diff --git a/crates/registry-breg/src/postgres/mutation.rs b/crates/registry-breg/src/postgres/mutation.rs index 6d7c0e0a6..261010b32 100644 --- a/crates/registry-breg/src/postgres/mutation.rs +++ b/crates/registry-breg/src/postgres/mutation.rs @@ -285,16 +285,6 @@ impl PostgresRecordMutationService { deadline, ) .await; - if result.is_err() && tokio::time::Instant::now() < deadline { - self.coordinator - .record_action_boundary_audit( - claims, - route_id, - correlation, - crate::audit::PreIoAuditKind::Refusal, - ) - .await?; - } guard.disarm(); match result? { Ok(prepared) => prepared, diff --git a/crates/registry-breg/tests/postgres_action_evidence.rs b/crates/registry-breg/tests/postgres_action_evidence.rs index 164a58ffc..c67f53665 100644 --- a/crates/registry-breg/tests/postgres_action_evidence.rs +++ b/crates/registry-breg/tests/postgres_action_evidence.rs @@ -162,10 +162,8 @@ fn app_with_client( fault: Option, ) -> (axum::Router, RuntimePool) { let pool = database.runtime_config.build_pool().unwrap(); - let audit = registry_breg::audit::test_support::capturing( - AuditProfile::production_from_secret_bytes(vec![0x42; 32].into()).unwrap(), - ) - .0; + let audit = + database.audit(AuditProfile::production_from_secret_bytes(vec![0x42; 32].into()).unwrap()); let lock = RegistryLockKey::derive(PACKAGE).unwrap(); let cursors = Arc::new( CursorCodec::new(Zeroizing::new(vec![0x63; 32]), Duration::from_secs(300)).unwrap(), @@ -570,6 +568,7 @@ async fn signed_evidence_actions_release_postgres_and_commit_atomic_transcripts( committed, "retention failure rolls back all operation material" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -596,6 +595,7 @@ async fn caught_failed_helper_cannot_commit_or_make_another_disclosure() { ); assert_eq!(counts(&database, ®istry).await, vec![0, 0, 0, 0, 0, 0]); assert!(!failed.1.to_string().contains("FR-12345")); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -682,6 +682,7 @@ async fn concurrent_receipt_overrides_failed_acquisition_and_recovers_ambiguous_ "ambiguous commit recovery uses retained receipt" ); assert_eq!(counts(&database, ®istry).await, committed); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -734,6 +735,7 @@ async fn verified_acquisition_expiring_during_sql_wait_cannot_commit() { 2, "only a later caller attempt obtains fresh evidence" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -879,5 +881,6 @@ async fn real_evidence_service_resolves_exact_selector_and_commits_verified_post ); assert_eq!(provider.requests().len(), calls + 1); assert_eq!(counts(&database, ®istry).await, committed); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } diff --git a/crates/registry-breg/tests/postgres_action_evidence_retention.rs b/crates/registry-breg/tests/postgres_action_evidence_retention.rs index 94c6d106e..5891fa5e5 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_retention.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_retention.rs @@ -234,6 +234,7 @@ async fn expired_request_evidence_erases_only_retained_uses() { &[("failed", None), ("erased", Some(1)), ("erased", Some(0))], ); drop(operator); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } fn cutoff() -> chrono::DateTime { @@ -327,7 +328,9 @@ async fn retention_refuses_misbound_database_with_identical_roles_and_catalog_dr assert_eq!(correct.erase_expired(cutoff()).await.unwrap(), 1); assert_eq!(count(&other).await, 0); drop((wrong, correct)); + other.assert_every_audit_request_answered_once(); other.cleanup().await; + original.assert_every_audit_request_answered_once(); original.cleanup().await; } @@ -420,5 +423,6 @@ async fn retention_serializes_activation_and_holds_identity_lock_through_deletio assert_eq!(erase.await.unwrap().unwrap(), 1); assert_eq!(count(&database).await, 0); drop(operator); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } diff --git a/crates/registry-breg/tests/postgres_action_evidence_targets.rs b/crates/registry-breg/tests/postgres_action_evidence_targets.rs index 9acc1fd31..76b26acef 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_targets.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_targets.rs @@ -348,6 +348,7 @@ async fn final_local_target_checks_refuse_changes_during_external_wait_even_for_ !row.get::<_, bool>(0), "the script omitted the declared patch slot" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } } diff --git a/crates/registry-breg/tests/postgres_mutation.rs b/crates/registry-breg/tests/postgres_mutation.rs index 36831bed3..ddc0bd727 100644 --- a/crates/registry-breg/tests/postgres_mutation.rs +++ b/crates/registry-breg/tests/postgres_mutation.rs @@ -3112,6 +3112,7 @@ async fn assert_patch_preserved_omitted_field( } async fn assert_journals_are_minimized_and_paired(database: &TestDatabase) { + database.assert_every_audit_request_answered_once(); let ordered = database.audit_entries(); for entry in &ordered { let expected = if entry["record"]["phase"] == "attempt" { diff --git a/crates/registry-breg/tests/support/postgres_harness.rs b/crates/registry-breg/tests/support/postgres_harness.rs index a65ba967e..75c45dbb2 100644 --- a/crates/registry-breg/tests/support/postgres_harness.rs +++ b/crates/registry-breg/tests/support/postgres_harness.rs @@ -139,6 +139,37 @@ impl TestDatabase { self.audit.entries() } + /// Assert that every audit request accepted so far has exactly one + /// response, written after it under the same schema and correlation. + /// The only response allowed without a request is a refusal for a + /// correlation that never opened one: a request refused before its + /// attempt was recorded. + #[allow(dead_code)] // Not every integration target reads audit entries. + pub fn assert_every_audit_request_answered_once(&self) { + let entries = self.audit.entries(); + let mut open = std::collections::BTreeMap::<(String, String), usize>::new(); + for entry in &entries { + let key = ( + entry["schema"].as_str().unwrap_or_default().to_owned(), + entry["correlation"].as_str().unwrap_or_default().to_owned(), + ); + match entry["phase"].as_str() { + Some("request") => *open.entry(key).or_default() += 1, + Some("response") => match open.get_mut(&key) { + Some(pending) if *pending > 0 => *pending -= 1, + None if entry["record"]["phase"] == "refusal" => {} + _ => panic!("a response answers no open request: {entry}\n{entries:#?}"), + }, + other => panic!("an audit entry has no pairing phase {other:?}: {entry}"), + } + } + let unanswered: Vec<_> = open.iter().filter(|(_, pending)| **pending > 0).collect(); + assert!( + unanswered.is_empty(), + "audit requests without a response: {unanswered:?}\n{entries:#?}" + ); + } + /// The record of every audit entry accepted so far. #[allow(dead_code)] // Not every integration target reads audit entries. pub fn audit_records(&self) -> Vec { From 62d28e23e768cd1320e4810b36ff1c3617bb8456 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:10:38 +0000 Subject: [PATCH 24/32] fix(breg): record a committed erasure before retrying external deletions The response of a committed request-detail erasure waited on up to thirty seconds of external-deletion retries, so a slow backend or an interrupted process left the committed erasure answered unfinished. The response is now recorded once the commit is confirmed, and the operation result alone reports the external deletions still pending. Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/request_retention.rs | 22 +-- .../tests/postgres_request_read_retention.rs | 140 +++++++++++++++++- 2 files changed, 141 insertions(+), 21 deletions(-) diff --git a/crates/registry-breg/src/request_retention.rs b/crates/registry-breg/src/request_retention.rs index 232acac58..f5b8fd2cc 100644 --- a/crates/registry-breg/src/request_retention.rs +++ b/crates/registry-breg/src/request_retention.rs @@ -518,24 +518,16 @@ impl RequestRetentionOperatorService { return Err(error); } }; - // External objects are deleted before the response is recorded, so - // the entry states how many remain instead of claiming a finished - // erasure while objects still exist. - let external = self.retry_external_deletions(&mut client).await; - let mut record = entry.record().clone(); - if let Some(fields) = record.as_object_mut() { - let (pending, tombstones) = match &external { - Ok((pending, tombstones)) => (json!(pending), json!(tombstones)), - Err(_) => (Value::Null, Value::Null), - }; - fields.insert("pendingExternalDeletions".to_owned(), pending); - fields.insert("externalDeletionTombstones".to_owned(), tombstones); - } + // The committed erasure is recorded before external objects are + // retried, so a slow or failing backend cannot hold its response + // back. The result, not the entry, reports the registry-wide + // external deletions still pending. attempt - .respond(record) + .respond(entry.record().clone()) .await .map_err(|_| RequestRetentionError::ErasureUnaudited)?; - let (pending_external_deletions, external_deletion_tombstones) = external?; + let (pending_external_deletions, external_deletion_tombstones) = + self.retry_external_deletions(&mut client).await?; Ok(RequestRetentionErase { request_entity_id: scope.request_entity_id.to_owned(), request_id: scope.request_id.to_string(), diff --git a/crates/registry-breg/tests/postgres_request_read_retention.rs b/crates/registry-breg/tests/postgres_request_read_retention.rs index 57b1da26c..080ae5a8a 100644 --- a/crates/registry-breg/tests/postgres_request_read_retention.rs +++ b/crates/registry-breg/tests/postgres_request_read_retention.rs @@ -762,11 +762,7 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it assert_eq!(phases, ["request", "response"]); assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); assert_eq!(entries[0]["record"]["phase"], "attempt"); - // The response is recorded after the external deletions ran, and states - // how many objects still wait for deletion. assert_eq!(entries[1]["record"]["outcome"], "committed"); - assert_eq!(entries[1]["record"]["pendingExternalDeletions"], 0); - assert_eq!(entries[1]["record"]["externalDeletionTombstones"], 0); assert_eq!( entries[0]["record"]["recordReference"], entries[1]["record"]["recordReference"] @@ -778,11 +774,143 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it database.cleanup().await; } +/// A committed request-detail erasure is recorded as soon as its commit is +/// confirmed. The external-deletion retry that follows cannot hold the +/// committed erasure's response back: the retry here waits for the registry +/// lock while the response is already on record. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn request_detail_erasure_records_its_commit_before_retrying_external_deletions() { + let database = TestDatabase::create(6).await; + let registry = Arc::new(compiled_registry()); + let identity = install_registry(&database, ®istry).await; + let app = request_router(&database, registry.clone(), identity.clone()); + let operator = claims("operator", "operator-principal"); + let request_id = applied_correction_request(&app, operator, "recorded-erasure").await; + let request_uuid = Uuid::parse_str(&request_id).expect("request id parses"); + let scope = RequestDetailErasureScope { + request_entity_id: "correction-request", + request_id: request_uuid, + proposal_version: 1, + }; + let (audit, capture) = registry_breg::audit::test_support::capturing( + AuditProfile::production_from_secret_bytes(vec![0x8e; 32].into()) + .expect("test audit profile is keyed"), + ); + let lock_key = RegistryLockKey::derive(PACKAGE_ID).expect("lock key derives"); + let retention = RequestRetentionOperatorService::new_for_test( + registry.as_ref().clone(), + identity.clone(), + ExpectedManagedCatalog::compiled(®istry), + lock_key, + database.migration_config.clone(), + database.migration_role.clone(), + database.runtime_role.clone(), + audit, + ); + // Other tests share the cluster, so only this database's sessions count. + let waiting = |wait_event: &'static str| { + let admin = &database.admin; + async move { + tokio::time::timeout(Duration::from_secs(4), async { + loop { + let waiting: i64 = admin + .query_one( + "SELECT count(*) FROM pg_catalog.pg_stat_activity + WHERE datname = current_database() + AND wait_event_type = 'Lock' AND wait_event = $1", + &[&wait_event], + ) + .await + .expect("administrator reads lock waits") + .get(0); + if waiting > 0 { + return; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("a session waits for the held lock"); + } + }; + + // The erasure holds the registry lock while it waits for the request + // row. A second session queues for the registry lock behind it, so it + // takes the lock the moment the erasure commits, and the + // external-deletion retry that follows waits for it. + let (holder, holder_task) = database.connect_admin().await; + holder + .batch_execute(&format!( + "BEGIN; SELECT 1 FROM registry_internal.registry_request_state + WHERE request_id = '{request_uuid}' FOR UPDATE" + )) + .await + .expect("administrator holds the request row"); + let erasure = tokio::spawn(async move { retention.erase(scope).await }); + waiting("transactionid").await; + let (queued, queued_task) = database.connect_admin().await; + let queued = Arc::new(queued); + let registry_lock = tokio::spawn({ + let queued = Arc::clone(&queued); + async move { + queued + .execute("SELECT pg_catalog.pg_advisory_lock($1)", &[&lock_key.get()]) + .await + .expect("the queued session takes the registry lock"); + } + }); + waiting("advisory").await; + holder + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the request row"); + + let recorded = tokio::time::timeout(Duration::from_secs(4), async { + loop { + let entries = capture.entries(); + if entries.len() >= 2 { + return entries; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await; + let still_retrying = !erasure.is_finished(); + registry_lock + .await + .expect("the queued session holds the registry lock"); + queued + .execute( + "SELECT pg_catalog.pg_advisory_unlock($1)", + &[&lock_key.get()], + ) + .await + .expect("the queued session releases the registry lock"); + let entries = recorded.expect("the committed erasure is recorded while the retry waits"); + assert!(still_retrying, "the retry was still waiting for the lock"); + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["record"]["outcome"], "committed"); + assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); + let erased = erasure + .await + .expect("the erasure task completes") + .expect("the erasure succeeds once the retry proceeds"); + assert_eq!(erased.pending_external_deletions, 0); + assert_eq!(erased.external_deletion_tombstones, 0); + assert_eq!(capture.entries().len(), 2); + + drop(holder); + drop(queued); + holder_task.abort(); + queued_task.abort(); + database.cleanup().await; +} + /// An erasure whose transaction fails after its request entry answers that /// entry with a failed response and erases nothing. One whose committed /// erasure the destination refuses to record reports that distinctly: the -/// detail is gone, and the journal holds only the request. A recorded -/// erasure states the external objects still pending deletion. +/// detail is gone, and the journal holds only the request. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn request_detail_erasure_pairs_its_request_entry_on_every_outcome() { let database = TestDatabase::create(6).await; From d8da69bfaf5d5d35f40013de2df417c3ef8ed285 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:12:36 +0000 Subject: [PATCH 25/32] fix(breg): answer an ingestion transition unfinished when its commit fails A run creation, cancellation, blocking, or receipt release whose commit returned an error answered its request entry refused, although the transition may have committed. Those commit errors now answer the request unfinished. Signed-off-by: Jeremi Joslin --- crates/registry-breg/src/ingestion_store.rs | 8 ++ crates/registry-breg/src/postgres/mutation.rs | 23 +++-- .../tests/postgres_ingestion_runs.rs | 88 +++++++++++++++++++ 3 files changed, 114 insertions(+), 5 deletions(-) diff --git a/crates/registry-breg/src/ingestion_store.rs b/crates/registry-breg/src/ingestion_store.rs index c56019b00..179702ed5 100644 --- a/crates/registry-breg/src/ingestion_store.rs +++ b/crates/registry-breg/src/ingestion_store.rs @@ -1240,6 +1240,14 @@ impl RunAttempt { let record = outcome_record(&self.record, "refused"); self.request.respond(record).await.is_ok() } + + /// Answer the request as unfinished: the transition's commit returned + /// an error, which does not prove it rolled back, so its outcome is + /// unknown. Reports whether the destination accepted the answer. + pub(crate) async fn abandon(mut self) -> bool { + let record = outcome_record(&self.record, "unfinished"); + self.request.respond(record).await.is_ok() + } } fn outcome_record(request: &Value, outcome: &str) -> Value { diff --git a/crates/registry-breg/src/postgres/mutation.rs b/crates/registry-breg/src/postgres/mutation.rs index 261010b32..619c7c790 100644 --- a/crates/registry-breg/src/postgres/mutation.rs +++ b/crates/registry-breg/src/postgres/mutation.rs @@ -103,6 +103,18 @@ pub struct IngestionRefusal { struct IngestionAudit { run: Option, batch: bool, + /// The transition's commit returned an error, so whether it committed + /// is unknown and the request is answered unfinished, never refused. + commit_unknown: bool, +} + +impl IngestionAudit { + /// Mark the transition's commit outcome unknown and refuse the call as + /// an outage. + fn commit_failed(&mut self) -> IngestionServiceError { + self.commit_unknown = true; + IngestionServiceError::Unavailable + } } /// The closed refusal vocabulary of the ingestion-run service. It is bounded @@ -1143,6 +1155,7 @@ impl PostgresRecordMutationService { }; let answered = match audit.run { Some(attempt) if attempt.is_answered() => true, + Some(attempt) if audit.commit_unknown => attempt.abandon().await, Some(attempt) => attempt.refuse().await, None => audit.batch, }; @@ -1278,7 +1291,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; // The run exists once the transaction commits; its answer leaves only // after the audit entry is accepted. ingestion_store::append_run_audit(&self.audit, audit_record) @@ -1523,7 +1536,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, audit_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; @@ -1767,7 +1780,7 @@ impl PostgresRecordMutationService { disclosure_transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, disclosure_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; @@ -1863,7 +1876,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, blocked_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; @@ -2279,7 +2292,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, disclosure_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; diff --git a/crates/registry-breg/tests/postgres_ingestion_runs.rs b/crates/registry-breg/tests/postgres_ingestion_runs.rs index e94216e8c..72569b6ef 100644 --- a/crates/registry-breg/tests/postgres_ingestion_runs.rs +++ b/crates/registry-breg/tests/postgres_ingestion_runs.rs @@ -761,6 +761,94 @@ async fn a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_sche assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); } +/// A run transition whose commit returns an error may still have committed, +/// so its request entry is answered unfinished rather than refused. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_transition_whose_commit_fails_is_answered_unfinished() { + let harness = IngestionHarness::create().await; + let claims = operator_claims(PRINCIPAL, "zone-a"); + let items = announce_items("unacknowledged-commit", 4); + let chunks = plan_chunks(&items, 2); + + refuse_run_commits(&harness).await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + "/v1/records/widgets/ingestion-runs", + &claims, + harness.run_body("create", &chunks), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_run_commits(&harness).await; + assert_unfinished_in_the_ingestion_schema( + &harness.database.audit_entries()[before..], + "create", + ); + + let run_id = harness.create_run(&claims, &chunks).await; + refuse_run_commits(&harness).await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_empty( + &format!("/v1/records/widgets/ingestion-runs/{run_id}/cancel"), + &claims, + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_run_commits(&harness).await; + assert_unfinished_in_the_ingestion_schema( + &harness.database.audit_entries()[before..], + "cancel", + ); + harness.database.assert_every_audit_request_answered_once(); +} + +fn assert_unfinished_in_the_ingestion_schema(entries: &[Value], transition: &str) { + let ingestion = entries + .iter() + .filter(|entry| entry["record"]["transition"] == transition) + .collect::>(); + assert_eq!(ingestion.len(), 2, "{entries:?}"); + assert_eq!(ingestion[0]["phase"], "request"); + assert_eq!(ingestion[1]["phase"], "response"); + assert_eq!( + ingestion[1]["record"]["outcome"], "unfinished", + "{entries:?}" + ); + assert_eq!(ingestion[0]["correlation"], ingestion[1]["correlation"]); +} + +/// Refuse every commit that wrote a run row, after all its statements ran. +async fn refuse_run_commits(harness: &IngestionHarness) { + harness + .database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_run_commit() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN RAISE EXCEPTION 'test refuses this commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_run_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_run_commit + AFTER INSERT OR UPDATE ON registry_internal.registry_ingestion_runs + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_run_commit();", + ) + .await + .expect("administrator installs the commit refusal"); +} + +async fn allow_run_commits(harness: &IngestionHarness) { + harness + .database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_run_commit ON registry_internal.registry_ingestion_runs; + DROP FUNCTION public.test_refuse_run_commit();", + ) + .await + .expect("administrator removes the commit refusal"); +} + /// The refused call wrote one ingestion request entry and one response /// entry answering it, both in the ingestion schema, and nothing else. fn assert_answered_in_the_ingestion_schema(entries: &[Value], transition: &str) { From 487a4f8f239256e9ec5b5d460db9671347173c0c Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:19:02 +0000 Subject: [PATCH 26/32] fix(hooks): answer every delivery transition whose commit fate is unknown A replay whose reset changed no row is refused without a read-back, so a generation another replay wrote is never taken for its own. A claim whose lease commit failed records each durable recovered or expired delivery whatever the lease's fate, and a finalize whose terminal commit cannot be read back answers its attempt as interrupted. The read-backs wait a short bounded backoff between attempts, and a replay read-back accepts any later generation. Signed-off-by: Jeremi Joslin --- .../src/delivery/service.rs | 579 ++++++++++++++++-- 1 file changed, 523 insertions(+), 56 deletions(-) diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index ed10da8c0..e0c219538 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -40,6 +40,10 @@ impl From for DeliveryError { const LEASE_FINALIZATION_ALLOWANCE: Duration = Duration::from_secs(5); const WORKER_POLL_INTERVAL: Duration = Duration::from_millis(100); +/// The waits between the read-backs of a transition whose commit returned an +/// error: one fewer than the read-backs, and short, since a caller is failing +/// while they run. +const READ_BACK_BACKOFF: [Duration; 2] = [Duration::from_millis(50), Duration::from_millis(100)]; pub const MAX_DELIVERY_STATUS_RESULTS: u16 = 100; /// The product-supplied constants the worker cannot derive. @@ -391,24 +395,34 @@ impl DeliveryService { disposition: DeliveryAuditDisposition::ReplayPending, }; self.seams.record_audit(replay.record()).await?; - let mut reset = self - .reset_for_replay(transaction, event_id, compiled_delivery_id, generation) - .await; - if reset.is_err() - && self - .transition_committed( - &PendingAudit { - outcome: DeliveryAuditOutcome::ReplayCommitted, - ..replay.clone() - }, - None, - ) - .await - == Some(true) + let reset = match self + .reset_for_replay(&transaction, event_id, compiled_delivery_id, generation) + .await { - // The reset committed although its acknowledgement was lost. - reset = Ok(()); - } + // A reset that failed or changed no row did not commit. + Err(error) => Err(error), + Ok(()) => match transaction.commit().await { + Ok(()) => Ok(()), + Err(error) => { + // The commit's acknowledgement may be all that was lost. + if self + .transition_committed( + &PendingAudit { + outcome: DeliveryAuditOutcome::ReplayCommitted, + ..replay.clone() + }, + None, + ) + .await + == Some(true) + { + Ok(()) + } else { + Err(error.into()) + } + } + }, + }; let (outcome, disposition) = if reset.is_ok() { ( DeliveryAuditOutcome::ReplayCommitted, @@ -436,11 +450,11 @@ impl DeliveryService { Ok(next_generation) } - /// Reset one dead-lettered delivery to pending under its next generation - /// and commit. + /// Reset one dead-lettered delivery to pending under its next generation, + /// leaving the commit to the caller. async fn reset_for_replay( &self, - transaction: Transaction<'_>, + transaction: &Transaction<'_>, event_id: Uuid, compiled_delivery_id: &str, generation: i64, @@ -479,7 +493,6 @@ impl DeliveryService { if changed != 1 { return Err(DeliveryError::Unavailable); } - transaction.commit().await?; Ok(()) } @@ -634,13 +647,15 @@ impl DeliveryService { if transaction.commit().await.is_err() { self.refused(DeliveryTransitionCode::ClaimCommitFailed); // A failed commit acknowledgement does not prove a rollback, so - // read what the database holds. A lease that did commit sends - // nothing from here and is answered when it expires; one that - // rolled back, or one whose fate cannot be read, is answered now, - // since a second interrupted answer is harmless and none is not. - if self.transition_committed(&started, Some(lease_token)).await == Some(true) { - self.record_resolved(committed).await; - } else { + // read what the database holds. Each recovered or expired + // delivery is recorded if it reads back as durable, whatever the + // lease's own fate. A lease that did commit sends nothing from + // here and is answered when it expires; one that rolled back, or + // one whose fate cannot be read, is answered now, since a second + // interrupted answer is harmless and none is not. + let lease_committed = self.transition_committed(&started, Some(lease_token)).await; + self.record_resolved(committed).await; + if lease_committed != Some(true) { let interrupted = PendingAudit { phase: DeliveryAuditPhase::Terminal, outcome: DeliveryAuditOutcome::WorkerInterrupted, @@ -1415,8 +1430,24 @@ impl DeliveryService { // terminal row is never reaped, so a disposition that did commit // is recorded here or never. One that rolled back leaves the // lease for expiry recovery, which answers the attempt. - if self.transition_committed(&terminal, None).await != Some(true) { - return Err(error.into()); + match self.transition_committed(&terminal, None).await { + Some(true) => {} + Some(false) => return Err(error.into()), + None => { + // The disposition's fate cannot be read, and one that + // did commit is never answered by expiry recovery, so the + // attempt is answered now as interrupted: a second + // interrupted answer is harmless and none is not. + let interrupted = PendingAudit { + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: DeliveryAuditDisposition::RetryPending, + ..terminal + }; + // A refused entry has already stopped the product's + // writer, which reports it; finalize fails either way. + let _ = self.seams.record_audit(interrupted.record()).await; + return Err(error.into()); + } } } // Recorded once the disposition committed, so the journal never @@ -1447,7 +1478,13 @@ impl DeliveryService { event: &PendingAudit, lease_token: Option, ) -> Option { - for _ in 0..3 { + for read_back in 0..=READ_BACK_BACKOFF.len() { + if let Some(wait) = read_back + .checked_sub(1) + .and_then(|index| READ_BACK_BACKOFF.get(index)) + { + tokio::time::sleep(*wait).await; + } let Ok(client) = self.seams.connection().await else { continue; }; @@ -1601,10 +1638,11 @@ fn transition_holds( && observed.lease_token.is_some() && observed.lease_token == lease_token } - // Only a replay's reset writes a generation, so the replacement - // generation proves the reset committed whatever state the worker - // has since moved it to. - (DeliveryAuditPhase::Replay, _) => current, + // Only a replay's reset writes a generation, and generations only + // grow, so the replacement generation or a later one proves the reset + // committed whatever state the worker or a later replay has since + // moved it to. + (DeliveryAuditPhase::Replay, _) => observed.generation >= event.generation, (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Expired) => { current && observed.expired } @@ -2332,6 +2370,10 @@ mod tests { "the replacement generation proves the reset in state {state}" ); } + assert!( + transition_holds(&replay, &observed("pending", 3, 0), None), + "a later replay's generation proves this reset committed before it" + ); assert!( !transition_holds(&replay, &observed("dead_lettered", 1, 2), None), "the prior generation proves the reset rolled back" @@ -3078,23 +3120,89 @@ mod tests { client } + type RecordedAudit = ( + DeliveryAuditPhase, + DeliveryAuditOutcome, + DeliveryAuditDisposition, + ); + + /// A local handler that only answers to its reviewed digest, so a replay + /// can find the binding its row was written against. + struct DigestHandler(String); + + #[async_trait::async_trait] + impl HookHandler for DigestHandler { + fn handler_digest(&self) -> &str { + &self.0 + } + + fn attempt_timeout(&self) -> Duration { + Duration::ZERO + } + + fn maximum_attempts(&self) -> u8 { + 1 + } + + async fn run( + &self, + _envelope: &[u8], + _remaining: Duration, + ) -> Result, crate::delivery::HandlerRunFailure> { + unreachable!("no real-database test runs a handler") + } + } + /// A seam set that opens its own connection against a real PostgreSQL - /// test database and counts every audit record it is given, so a test can - /// prove no record was written for a transition that never happened. + /// test database and records every audit record it is given, so a test can + /// prove which transitions were written. Connections are numbered from + /// one in the order they are opened: a test may refuse chosen ones, and + /// may run one statement on a chosen one before the service uses it, to + /// stand for another session acting at that moment. struct RealDbSeams { url: String, - audit_calls: Arc>, + audit: Arc>>, + connections: Mutex, + refused_connections: Vec, + interleave: Option<(u32, String)>, + handler_digest: Option, + } + + impl RealDbSeams { + fn new(url: &str, audit: &Arc>>) -> Self { + Self { + url: url.to_owned(), + audit: Arc::clone(audit), + connections: Mutex::new(0), + refused_connections: Vec::new(), + interleave: None, + handler_digest: None, + } + } } #[async_trait::async_trait] impl DeliverySeams for RealDbSeams { type Destination = UnusedDestination; - type Handler = UnusedHandler; + type Handler = DigestHandler; async fn connection(&self) -> Result { - Ok(Box::new(DirectClient( - connect_test_database(&self.url).await, - ))) + let number = { + let mut connections = self.connections.lock().expect("connections lock"); + *connections += 1; + *connections + }; + if self.refused_connections.contains(&number) { + return Err(DeliveryError::Unavailable); + } + let client = connect_test_database(&self.url).await; + if let Some((_, statement)) = self.interleave.as_ref().filter(|(at, _)| *at == number) { + client + .batch_execute(statement) + .await + .expect("the interleaved statement runs"); + } + Ok(Box::new(DirectClient(client))) } async fn verify_transaction( @@ -3108,15 +3216,19 @@ mod tests { None } - fn handler(&self, _binding: HookHandlerBinding<'_>) -> Option { - None + fn handler(&self, binding: HookHandlerBinding<'_>) -> Option { + self.handler_digest + .as_deref() + .filter(|digest| *digest == binding.handler_digest) + .map(|digest| DigestHandler(digest.to_owned())) } - async fn record_audit( - &self, - _record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { - *self.audit_calls.lock().expect("audit calls lock") += 1; + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { + self.audit.lock().expect("audit lock").push(( + record.phase, + record.outcome, + record.disposition, + )); Ok(()) } @@ -3252,12 +3364,9 @@ mod tests { .expect("simulate an active lease"); assert_eq!(changed, 1, "the inserted delivery state row exists"); - let audit_calls = Arc::new(Mutex::new(0u32)); + let audit = Arc::new(Mutex::new(Vec::new())); let service = DeliveryService::new( - RealDbSeams { - url: url.clone(), - audit_calls: Arc::clone(&audit_calls), - }, + RealDbSeams::new(&url, &audit), DeliveryConfig { schema: schema.to_owned(), idempotency_domain: b"hooks-delivery-lease-test-v1".to_vec(), @@ -3288,9 +3397,367 @@ mod tests { "a lease-guarded update that changes zero rows must fail closed" ); assert_eq!( - *audit_calls.lock().expect("audit calls lock"), - 0, + *audit.lock().expect("audit lock"), + Vec::::new(), "no audit entry for a transition that did not happen" ); } + + const REAL_DELIVERY_ID: &str = "events.permit.granted.webhook"; + const REAL_PACKAGE_REVISION: &str = + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + const REAL_HANDLER_DIGEST: &str = + "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; + const REAL_SCHEMA_FINGERPRINT: &str = + "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc"; + const OTHER_EVENT_ID: &str = "9b1c3a5e-2d4f-4e6a-8b0c-1d3e5f7a9b2c"; + + fn real_database_url() -> String { + std::env::var("HOOKS_TEST_DATABASE_URL") + .expect("HOOKS_TEST_DATABASE_URL is required for the real PostgreSQL tests") + } + + /// Install the delivery schema afresh under `schema` and return an + /// administrative connection to the test database. + async fn fresh_delivery_schema(url: &str, schema: &str) -> tokio_postgres::Client { + let client = connect_test_database(url).await; + client + .batch_execute(&format!( + "DROP SCHEMA IF EXISTS {schema} CASCADE; CREATE SCHEMA {schema};" + )) + .await + .expect("reset the test schema"); + delivery_schema::install(&client, schema) + .await + .expect("install the delivery schema"); + client + } + + /// Insert one pending local-handler delivery of `event_id` whose stored + /// payload expires after `payload_lifetime`, a PostgreSQL interval. + async fn insert_real_delivery( + client: &mut tokio_postgres::Client, + schema: &str, + event_id: Uuid, + payload_lifetime: &str, + operator_replay: bool, + ) { + client + .execute( + &format!( + "INSERT INTO {schema}.registry_outbox + (event_id, event_type, trigger, entity_id, record_reference, + record_revision, package_revision, schema_fingerprint, payload, + payload_expires_at) + VALUES ($1, 'permit.granted', 'test', 'permit', $2, 3, $3, $4, $5, + transaction_timestamp() + interval '{payload_lifetime}')", + ), + &[ + &event_id, + &"8f14e45fceea467a9cc18b2a4b9e2a1103b41d5ad4c88c8f2a0a9ac4f4c0d2b7", + &REAL_PACKAGE_REVISION, + &REAL_SCHEMA_FINGERPRINT, + &b"{}".as_slice(), + ], + ) + .await + .expect("insert the outbox row"); + let transaction = client.transaction().await.expect("insert transaction"); + insert_delivery( + &transaction, + schema, + event_id, + DeliveryCapture { + compiled_delivery_id: REAL_DELIVERY_ID, + handler_kind: HookHandlerKind::Rhai, + logical_destination_id: None, + destination_binding_digest: REAL_HANDLER_DIGEST, + package_revision: REAL_PACKAGE_REVISION, + schema_fingerprint: REAL_SCHEMA_FINGERPRINT, + data_schema: STORED_DATA_SCHEMA, + classification_ceiling: "public", + authentication_profile: "hmac_sha256_v1", + delivery_mode: "after_commit", + attempt_timeout_ms: 5_000, + initial_backoff_ms: 1_000, + maximum_backoff_ms: 60_000, + exponential_backoff_multiplier: 2, + maximum_attempts: 3, + retry_delays_ms: &[1_000, 2_000], + maximum_payload_bytes: 1_024, + payload: b"{}", + deployed_attempt_timeout_ms: 5_000, + deployed_maximum_attempts: 3, + dead_letter: "required", + operator_replay, + }, + ) + .await + .expect("insert the delivery row"); + transaction.commit().await.expect("commit the insert"); + } + + fn real_service(seams: RealDbSeams, schema: &str) -> DeliveryService { + DeliveryService::new( + seams, + DeliveryConfig { + schema: schema.to_owned(), + idempotency_domain: b"hooks-delivery-real-test-v1".to_vec(), + delivery_source: STORED_SOURCE.to_owned(), + }, + ) + } + + /// A replay whose reset changed no row did not commit, so its response + /// is a refusal even when the row reads back under the generation the + /// replay would have written, here because another replay wrote it. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn a_replay_whose_reset_changed_no_row_is_refused() { + let url = real_database_url(); + let schema = "hooks_delivery_replay_refused_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let event_id = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, event_id, "1 day", true).await; + client + .batch_execute(&format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'dead_lettered', attempt = 3, next_attempt_at = NULL, + dead_lettered_at = transaction_timestamp(); + CREATE FUNCTION {schema}.skip_reset() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + IF current_setting('hooks_test.allow_reset', true) = 'on' THEN + RETURN NEW; + END IF; + RETURN NULL; + END $$; + CREATE TRIGGER skip_reset + BEFORE UPDATE ON {schema}.registry_webhook_delivery_state + FOR EACH ROW WHEN (NEW.generation > OLD.generation) + EXECUTE FUNCTION {schema}.skip_reset();" + )) + .await + .expect("make the reset change no row"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + handler_digest: Some(REAL_HANDLER_DIGEST.to_owned()), + // The first connection after the replay's own stands for a + // concurrent replay that commits the next generation. + interleave: Some(( + 2, + format!( + "SET hooks_test.allow_reset = 'on'; + UPDATE {schema}.registry_webhook_delivery_state + SET generation = 2, state = 'pending', attempt = 0, + next_attempt_at = transaction_timestamp(), + dead_lettered_at = NULL" + ), + )), + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + + let result = service.replay_in(event_id, REAL_DELIVERY_ID, 1).await; + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a reset that changed no row fails: {result:?}" + ); + assert_eq!( + *audit.lock().expect("audit lock"), + vec![ + ( + DeliveryAuditPhase::Replay, + DeliveryAuditOutcome::ReplayRequested, + DeliveryAuditDisposition::ReplayPending, + ), + ( + DeliveryAuditPhase::Replay, + DeliveryAuditOutcome::ReplayRefused, + DeliveryAuditDisposition::DeadLettered, + ), + ] + ); + } + + /// A claim whose lease commit failed still records every transition of + /// that transaction that reads back as durable, even when the lease's + /// own fate cannot be read. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn a_claim_whose_lease_commit_failed_records_its_durable_transitions() { + let url = real_database_url(); + let schema = "hooks_delivery_claim_commit_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let expiring = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + let claimable = Uuid::parse_str(OTHER_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, expiring, "-1 second", false).await; + insert_real_delivery(&mut client, schema, claimable, "1 day", false).await; + client + .batch_execute(&format!( + "CREATE FUNCTION {schema}.refuse_lease() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + RAISE EXCEPTION 'lease commit refused'; + END $$; + CREATE CONSTRAINT TRIGGER refuse_lease + AFTER UPDATE ON {schema}.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED + FOR EACH ROW WHEN (NEW.state = 'leased') + EXECUTE FUNCTION {schema}.refuse_lease();" + )) + .await + .expect("make the lease commit fail"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + // Every read-back of the lease fails, so its fate is unknown. + refused_connections: vec![2, 3, 4], + // The next connection reads back the payload expiry, which + // this stands for having committed. + interleave: Some(( + 5, + format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'expired', next_attempt_at = NULL, + expired_at = transaction_timestamp() + WHERE event_id = '{expiring}'; + UPDATE {schema}.registry_outbox SET payload = NULL + WHERE event_id = '{expiring}'" + ), + )), + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + + let result = service.claim().await; + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a claim whose commit failed fails" + ); + let recorded = audit.lock().expect("audit lock").clone(); + assert_eq!( + recorded.first(), + Some(&( + DeliveryAuditPhase::Attempt, + DeliveryAuditOutcome::AttemptStarted, + DeliveryAuditDisposition::Leased, + )) + ); + assert!( + recorded.contains(&( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::PayloadExpired, + DeliveryAuditDisposition::Expired, + )), + "the durable expiry is recorded: {recorded:?}" + ); + assert!( + recorded.contains(&( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::WorkerInterrupted, + DeliveryAuditDisposition::RetryPending, + )), + "the attempt of unknown fate is answered: {recorded:?}" + ); + assert_eq!(recorded.len(), 3, "{recorded:?}"); + } + + /// A terminal transition whose commit failed and whose fate cannot be + /// read back still answers the attempt, after a bounded backoff between + /// the read-backs. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn finalize_answers_an_attempt_whose_commit_cannot_be_read_back() { + let url = real_database_url(); + let schema = "hooks_delivery_finalize_unknown_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let event_id = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, event_id, "1 day", false).await; + let lease_token = Uuid::new_v4(); + let changed = client + .execute( + &format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'leased', + attempt = 1, + next_attempt_at = NULL, + attempt_started_at = transaction_timestamp(), + lease_expires_at = transaction_timestamp() + interval '30 seconds', + lease_token = $1 + WHERE event_id = $2", + ), + &[&lease_token, &event_id], + ) + .await + .expect("lease the delivery"); + assert_eq!(changed, 1); + client + .batch_execute(&format!( + "CREATE FUNCTION {schema}.refuse_delivered() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + RAISE EXCEPTION 'terminal commit refused'; + END $$; + CREATE CONSTRAINT TRIGGER refuse_delivered + AFTER UPDATE ON {schema}.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED + FOR EACH ROW WHEN (NEW.state = 'delivered') + EXECUTE FUNCTION {schema}.refuse_delivered();" + )) + .await + .expect("make the terminal commit fail"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + refused_connections: vec![2, 3, 4], + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + let claim = DeliveryClaim { + event_id, + compiled_delivery_id: REAL_DELIVERY_ID.to_owned(), + generation: 1, + attempt: 1, + attempt_started_at: SystemTime::now(), + lease_token, + deployed_maximum_attempts: 3, + retry_delays_ms: vec![1_000, 2_000], + package_revision: REAL_PACKAGE_REVISION.to_owned(), + handler_kind: HookHandlerKind::Rhai, + }; + + let started = Instant::now(); + let result = service + .finalize(&claim, AttemptResult::from(DeliveryAuditOutcome::Delivered)) + .await; + let elapsed = started.elapsed(); + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a terminal commit that failed fails: {result:?}" + ); + assert_eq!( + *audit.lock().expect("audit lock"), + vec![( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::WorkerInterrupted, + DeliveryAuditDisposition::RetryPending, + )] + ); + let backoff: Duration = READ_BACK_BACKOFF.iter().sum(); + assert!( + elapsed >= backoff, + "the read-backs wait {backoff:?} between them, took {elapsed:?}" + ); + } } From 476e2137b858190629e43eaf2b7c08650939fc53 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:19:54 +0000 Subject: [PATCH 27/32] docs(breg): scope audit pairing to a running process in BREG-V1-27 Signed-off-by: Jeremi Joslin --- products/breg/DEFINITION-OF-DONE.md | 7 +++++++ products/breg/contracts/definition-of-done.yaml | 2 +- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/products/breg/DEFINITION-OF-DONE.md b/products/breg/DEFINITION-OF-DONE.md index f2d7ae05e..dd2332f72 100644 --- a/products/breg/DEFINITION-OF-DONE.md +++ b/products/breg/DEFINITION-OF-DONE.md @@ -86,6 +86,13 @@ rebaseline, or reconciliation finds the state the first run committed rather than replaying its terminal entry. Both recoveries are operational, not a second audit mechanism. +The pairing of a request entry with its response is a property of a running +process. A request whose operation is abandoned is answered `unfinished`, and +so is an ingestion transition whose commit returned an error, since that error +does not prove a rollback. A process that is killed, exits, or shuts its +runtime down while a write is in flight may leave a request without its +response, and `BREG-V1-27` claims no more than that. + The HTTP record contract is also explicit: caller-filtered and generated OpenAPI artifacts assign every record-related route to the shared single or collection Registry Record profile, or to a named BReg-specific shape. diff --git a/products/breg/contracts/definition-of-done.yaml b/products/breg/contracts/definition-of-done.yaml index c9016662d..3c76ed2a2 100644 --- a/products/breg/contracts/definition-of-done.yaml +++ b/products/breg/contracts/definition-of-done.yaml @@ -48,7 +48,7 @@ requirements: - {id: BREG-V1-24, phase: W3, state: enforced, doneWhen: "Application authorization and RLS agree for positive, negative, malformed, and pooled authority.", journeys: [BREG-J05, BREG-J11], evidence: [{path: crates/registry-breg/tests/postgres_compiled_schema.rs, name: compiled_postgres_schema_enforces_context_rls_and_exact_catalog}, {path: crates/registry-breg/tests/postgres_kernel.rs, name: real_postgres_kernel_proves_roles_rls_interlock_and_pool_isolation}, {path: crates/registry-breg/tests/postgres_pilot_acceptance.rs, name: real_postgres_five_domain_pilot_is_configured_production_closed_and_source_neutral}]} - {id: BREG-V1-25, phase: W3, state: enforced, doneWhen: "Provenance stays distinct from minimized value-free audit, logs, metrics, and traces.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_tombstone_revision.rs, name: tombstone_revisions_survive_package_upgrade_and_replay_exactly}, {path: crates/registry-breg/tests/startup_http.rs, name: operational_log_level_is_a_closed_vocabulary}, {path: crates/registry-breg/tests/startup_http.rs, name: every_operational_event_renders_exact_closed_value_free_json_fields}, {path: crates/registry-breg/tests/startup_http.rs, name: provenance_operational_logs_metrics_and_traces_are_separate_closed_and_value_free}]} - {id: BREG-V1-26, phase: W3, state: enforced, doneWhen: "Exact media-specific Registry Record bytes release or replay only after successful attempt and terminal audit gates.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_http_mutations_are_guarded_and_exactly_replayable}]} - - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}, {path: crates/registry-breg/tests/postgres_history_rebaseline.rs, name: rebaseline_refuses_while_maintenance_is_not_ready}]} + - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup. The pairing holds while that process runs; a process that is killed, exits, or shuts its runtime down while a write is in flight may leave a request without its response.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-platform-audit/src/writer.rs, name: a_response_claimed_while_its_request_is_dropped_is_the_only_answer}, {path: crates/registry-platform-audit/src/writer.rs, name: a_response_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle}, {path: crates/registry-platform-audit/src/writer.rs, name: an_append_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_transition_whose_commit_fails_is_answered_unfinished}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_records_its_commit_before_retrying_external_deletions}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}, {path: crates/registry-breg/tests/postgres_history_rebaseline.rs, name: rebaseline_refuses_while_maintenance_is_not_ready}]} - {id: BREG-V1-28, phase: W4, state: enforced, doneWhen: "Production packages capture the governed closure and sign exact canonical bytes with monotonic identity.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_builder_is_deterministic_and_local_publication_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: production_package_requires_exact_trust_anchor_threshold_and_signature}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_layout_contract_conditional_manifest_projection_is_in_projected_closure}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_omits_manifest_projection_from_signed_closure_and_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_refuses_claimed_manifest_artifacts}]} - {id: BREG-V1-29, phase: W4, state: enforced, doneWhen: "Activation verifies trust, identity, inventory, filesystem safety, artifacts, and schema before readiness.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_binding_refuses_wrong_environment_instance_database_sequence_and_prior}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_refuses_symlinks_and_production_writable_permissions}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_manifest_refuses_ddl_checksum_path_and_canonical_json_tampering}, {path: crates/registry-breg/tests/postgres_package.rs, name: signed_schema_fingerprint_mismatch_is_durably_failed_and_never_ready}]} - {id: BREG-V1-30, phase: W4, state: enforced, doneWhen: "Apply retains the lock through migrations, catalog verification, activation, and maintenance clearing.", journeys: [BREG-J13, BREG-J15], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: real_postgres_package_startup_apply_failure_and_old_process_are_closed}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_backfill_and_destructive_recovery_are_bounded_resumable_and_activation_closed}]} From fcf54eb9875990aec7002fb3a9ae21b97f373c02 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:00:52 +0000 Subject: [PATCH 28/32] fix(casework): write an operation's whole response tail even when its caller is canceled Signed-off-by: Jeremi Joslin --- crates/registry-casework/src/audit.rs | 158 ++++++++++++++++++++++---- 1 file changed, 139 insertions(+), 19 deletions(-) diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index ee8390a05..83b66305c 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -281,6 +281,8 @@ impl AuditOperation { /// terminal outcome of a caller-requested operation that recorded none. /// The caller's transaction has committed, so a refusal here reports the /// destination unavailable while the committed change stays in place. + /// The entries are written in one task that outlives a canceled caller, + /// so a caller that stops waiting cannot leave part of them unwritten. pub(crate) async fn complete(self) -> Result<(), StoreError> { self.ensure_terminal()?; let Self { @@ -296,23 +298,27 @@ impl AuditOperation { .push(audit.minimized(json!({"event": event, "outcome": outcome.as_str()}))?); } } - for record in records { - match &mut pairing { - Pairing::Requested { request, .. } => request.respond(record).await, - Pairing::Background { correlation } => { - audit - .writer - .append(AuditEntry::response( - CASEWORK_AUDIT_SCHEMA, - correlation.clone(), - record, - )) - .await + let writer = audit.writer; + tokio::spawn(async move { + for record in records { + match &mut pairing { + Pairing::Requested { request, .. } => request.respond(record).await, + Pairing::Background { correlation } => { + writer + .append(AuditEntry::response( + CASEWORK_AUDIT_SCHEMA, + correlation.clone(), + record, + )) + .await + } } + .map_err(|_| StoreError::AuditUnavailable)?; } - .map_err(|_| StoreError::AuditUnavailable)?; - } - Ok(()) + Ok(()) + }) + .await + .map_err(|_| StoreError::AuditUnavailable)? } } @@ -395,7 +401,8 @@ fn audit_record_with_event_id(event_id: Uuid, mut record: Value) -> Result, } - /// The lines a test audit destination accepted, and a switch that makes - /// it refuse every line past a count. + /// A switch that holds the write of one line until it is released, kept + /// apart from the accepted lines so a held write blocks no reader. + #[derive(Default)] + struct Gate { + state: Mutex, + changed: Condvar, + } + + #[derive(Default)] + struct GateState { + /// The index of the line whose write is held. + hold_at: Option, + /// Whether that write has started and is waiting. + holding: bool, + } + + /// The lines a test audit destination accepted, a switch that makes it + /// refuse every line past a count, and one that holds a line's write. #[derive(Clone, Default)] - pub struct AuditCapture(Arc>); + pub struct AuditCapture(Arc>, Arc); impl AuditCapture { /// Every accepted entry, parsed, in write order. @@ -465,10 +488,56 @@ mod capture { pub fn refuse_after(&self, lines: usize) { self.0.lock().expect("audit capture").accepted_lines = Some(lines); } + + /// Hold the write of the line at index `line` until [`Self::release`]. + pub fn hold_line(&self, line: usize) { + self.1.state.lock().expect("audit gate").hold_at = Some(line); + } + + /// Wait until the held line's write has started, and report whether + /// it did within `timeout`. + #[must_use] + pub fn wait_until_held(&self, timeout: Duration) -> bool { + let state = self.1.state.lock().expect("audit gate"); + let (state, _) = self + .1 + .changed + .wait_timeout_while(state, timeout, |state| !state.holding) + .expect("audit gate"); + state.holding + } + + /// Let a held write proceed. + pub fn release(&self) { + let mut state = self.1.state.lock().expect("audit gate"); + state.hold_at = None; + state.holding = false; + self.1.changed.notify_all(); + } + + /// Wait while a gate holds the write of line `index`. + fn pass_gate(&self, index: usize) { + let mut state = self.1.state.lock().expect("audit gate"); + if state.hold_at != Some(index) { + return; + } + state.holding = true; + self.1.changed.notify_all(); + let _released = self + .1 + .changed + .wait_while(state, |state| state.hold_at == Some(index)) + .expect("audit gate"); + } } impl Write for AuditCapture { fn write(&mut self, bytes: &[u8]) -> io::Result { + let index = { + let state = self.0.lock().expect("audit capture"); + state.bytes.iter().filter(|byte| **byte == b'\n').count() + }; + self.pass_gate(index); let mut state = self.0.lock().expect("audit capture"); let written = state.bytes.iter().filter(|byte| **byte == b'\n').count(); if state.accepted_lines.is_some_and(|limit| written >= limit) { @@ -503,6 +572,8 @@ pub use capture::AuditCapture; #[cfg(test)] mod tests { + use std::time::Duration; + use registry_casework_core::{ActorContext, CaseworkRole, IssuerPrincipal}; use super::*; @@ -576,6 +647,55 @@ mod tests { } } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_canceled_completion_still_writes_every_response_entry() { + let (audit, capture) = CaseworkAudit::capture(); + let mut operation = audit + .begin(request_record("claimed", None, "officer", json!({}))) + .await + .unwrap(); + for event in [ + "casework.claimed", + "casework.task_invalidated", + "casework.task_invalidated", + ] { + operation + .record( + Uuid::new_v4(), + json!({"event": event, "profileId": "officer"}), + ) + .unwrap(); + } + // Hold the first response entry's write, then cancel the caller + // while it waits. + capture.hold_line(1); + let completion = tokio::spawn(operation.complete()); + assert!(capture.wait_until_held(Duration::from_secs(10))); + completion.abort(); + assert!(completion.await.unwrap_err().is_cancelled()); + capture.release(); + + let deadline = std::time::Instant::now() + Duration::from_secs(5); + let mut entries = capture.entries(); + while entries.len() < 4 && std::time::Instant::now() < deadline { + tokio::time::sleep(Duration::from_millis(10)).await; + entries = capture.entries(); + } + let phases: Vec<_> = entries.iter().map(|entry| entry["phase"].clone()).collect(); + assert_eq!( + phases, + ["request", "response", "response", "response"], + "{entries:?}" + ); + let correlation = &entries[0]["correlation"]; + assert!(entries + .iter() + .all(|entry| &entry["correlation"] == correlation)); + assert!(entries + .iter() + .all(|entry| entry["record"].get("outcome").is_none())); + } + #[tokio::test] async fn a_refused_request_entry_starts_no_operation() { let (audit, capture) = CaseworkAudit::capture(); From 2c65e431af92ebc4a78feece11c7242105257300 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:04:53 +0000 Subject: [PATCH 29/32] fix(casework): read back a commit whose acknowledgment was lost before recording it unfinished Signed-off-by: Jeremi Joslin --- crates/registry-casework/src/audit.rs | 125 ++++++++++- crates/registry-casework/src/store.rs | 2 + .../tests/postgres_transactions.rs | 199 +++++++++++++++++- products/casework/CHANGELOG.md | 12 +- 4 files changed, 320 insertions(+), 18 deletions(-) diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index 83b66305c..e40536233 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -7,9 +7,13 @@ //! operation that recorded no domain event, such as an idempotent replay, //! appends one `response` entry naming its terminal outcome instead, so no //! caller-requested result is released without an accepted `response` entry. -//! One that returns without committing, a refusal, a failure, or a canceled -//! request, appends `{event, outcome: "unfinished"}` as its `response` entry -//! when its operation is dropped, so no `request` entry stays unpaired. +//! One whose change is not known to have committed, a refusal, a failure, a +//! canceled request, or a commit whose acknowledgment never arrived and whose +//! outcome could not be read back as committed, appends +//! `{event, outcome: "unfinished"}` as its `response` entry when its +//! operation is dropped, so no `request` entry stays unpaired. A commit whose +//! acknowledgment never arrived is read back first: one that took effect is +//! recorded by its domain events like any other. //! Entries carry only event metadata and keyed references, never source //! selectors, free-text reasons, receipts, or issuer and subject identities. @@ -31,6 +35,10 @@ const TASK_GRANT_SYSTEM_PROFILE: &str = "system:task-grants"; pub struct CaseworkAudit { writer: AuditWriter, identifiers: AuditKeyHasher, + /// A test switch that reports the next commit that took effect as one + /// whose acknowledgment never arrived. + #[cfg(any(test, feature = "postgres-test"))] + lose_acknowledgment: std::sync::Arc, } impl std::fmt::Debug for CaseworkAudit { @@ -48,6 +56,8 @@ impl CaseworkAudit { Self { writer, identifiers, + #[cfg(any(test, feature = "postgres-test"))] + lose_acknowledgment: std::sync::Arc::default(), } } @@ -81,6 +91,7 @@ impl CaseworkAudit { pairing: Pairing::Requested { request, event }, outcome: None, responses: Vec::new(), + read_back: None, }) } @@ -99,6 +110,7 @@ impl CaseworkAudit { }, outcome: None, responses: Vec::new(), + read_back: None, }) } @@ -111,6 +123,19 @@ impl CaseworkAudit { fn minimized(&self, record: Value) -> Result { published_audit_record(record, &self.identifiers).map_err(|()| StoreError::Corrupt) } + + /// Fail a commit that took effect, as a connection lost before its + /// acknowledgment arrived does, when the test switch asks for it. + #[cfg(any(test, feature = "postgres-test"))] + fn acknowledged(&self) -> Result<(), StoreError> { + if self + .lose_acknowledgment + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err(StoreError::Unavailable); + } + Ok(()) + } } /// `identifiers` is a JSON object of the identifiers the request names, keyed @@ -174,6 +199,17 @@ pub(crate) struct AuditOperation { pairing: Pairing, outcome: Option, responses: Vec<(Uuid, Value)>, + /// The pool a commit whose acknowledgment never arrived reads its + /// transaction's status back through. + read_back: Option, +} + +/// What reading back an unacknowledged commit found. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum ReadBack { + Committed, + RolledBack, + Unknown, } /// How an operation's `response` entries are correlated. @@ -189,6 +225,13 @@ enum Pairing { } impl AuditOperation { + /// Read the outcome of a commit whose acknowledgment never arrived + /// through `pool`, on a connection of its own. + pub(crate) fn with_read_back(mut self, pool: deadpool_postgres::Pool) -> Self { + self.read_back = Some(pool); + self + } + /// Name the terminal outcome of a caller-requested operation that may /// record no domain event. It is appended as the operation's `response` /// entry only when no domain event was recorded; background work ignores @@ -267,16 +310,64 @@ impl AuditOperation { /// append one `response` entry per recorded event, or the terminal /// outcome when none was recorded. A caller-requested operation with /// neither is refused before `transaction` commits. + /// + /// A commit that fails may still have taken effect, as when the + /// connection is lost after `COMMIT` reached the database. The + /// transaction's status is then read back on another connection: a + /// committed change is answered and appended like any other, and one + /// that rolled back or whose status cannot be read returns the commit + /// error, so the dropped operation appends its unfinished response. pub(crate) async fn commit( mut self, transaction: deadpool_postgres::Transaction<'_>, ) -> Result<(), StoreError> { self.collect_task_invalidations(&transaction).await?; self.ensure_terminal()?; - transaction.commit().await?; + let transaction_id: String = transaction + .query_one("SELECT pg_current_xact_id()::text", &[]) + .await? + .get(0); + let committed = transaction.commit().await.map_err(StoreError::from); + #[cfg(any(test, feature = "postgres-test"))] + let committed = committed.and_then(|()| self.audit.acknowledged()); + if let Err(error) = committed { + match self.read_back(&transaction_id).await { + ReadBack::Committed => tracing::warn!( + "a Casework commit was not acknowledged but took effect; its response entries are appended" + ), + ReadBack::RolledBack => return Err(error), + ReadBack::Unknown => { + tracing::error!( + "a Casework commit was not acknowledged and its outcome could not be read back; its response entry is unfinished" + ); + return Err(error); + } + } + } self.complete().await } + /// Read whether the transaction `transaction_id` committed, on a + /// connection outside any transaction. It writes nothing. + async fn read_back(&self, transaction_id: &str) -> ReadBack { + let Some(pool) = &self.read_back else { + return ReadBack::Unknown; + }; + let Ok(client) = pool.get().await else { + return ReadBack::Unknown; + }; + let status = client + .query_one("SELECT pg_xact_status($1::text::xid8)", &[&transaction_id]) + .await + .map(|row| row.get::<_, Option>(0)); + match status.as_ref().map(|status| status.as_deref()) { + Ok(Some("committed")) => ReadBack::Committed, + Ok(Some("aborted")) => ReadBack::RolledBack, + // Still in progress, too old to report, or unreadable. + _ => ReadBack::Unknown, + } + } + /// Append one `response` entry per recorded event, or one naming the /// terminal outcome of a caller-requested operation that recorded none. /// The caller's transaction has committed, so a refusal here reports the @@ -290,6 +381,7 @@ impl AuditOperation { mut pairing, outcome, responses, + .. } = self; let mut records: Vec = responses.into_iter().map(|(_, record)| record).collect(); if records.is_empty() { @@ -416,6 +508,9 @@ mod capture { /// The writer recording here, so a read can wait for the entries it /// writes when a request handle is dropped. writer: Option, + /// The switch the audit recording here reads before it treats a + /// commit as acknowledged. + lose_acknowledgment: Arc, } /// A switch that holds the write of one line until it is released, kept @@ -507,6 +602,16 @@ mod capture { state.holding } + /// Report the next commit of an audited operation as unacknowledged + /// after it took effect, as a connection lost during `COMMIT` does. + pub fn lose_next_commit_acknowledgment(&self) { + self.0 + .lock() + .expect("audit capture") + .lose_acknowledgment + .store(true, std::sync::atomic::Ordering::SeqCst); + } + /// Let a held write proceed. pub fn release(&self) { let mut state = self.1.state.lock().expect("audit gate"); @@ -558,11 +663,13 @@ mod capture { pub fn capture() -> (Self, AuditCapture) { let capture = AuditCapture::default(); let writer = AuditWriter::from_line_sink(Box::new(capture.clone())); - capture.0.lock().expect("audit capture").writer = Some(writer.clone()); - ( - Self::new(writer, AuditKeyHasher::unkeyed_dev_only()), - capture, - ) + let audit = Self::new(writer.clone(), AuditKeyHasher::unkeyed_dev_only()); + { + let mut state = capture.0.lock().expect("audit capture"); + state.writer = Some(writer); + state.lose_acknowledgment = Arc::clone(&audit.lose_acknowledgment); + } + (audit, capture) } } } diff --git a/crates/registry-casework/src/store.rs b/crates/registry-casework/src/store.rs index 7da7d3267..7478c5912 100644 --- a/crates/registry-casework/src/store.rs +++ b/crates/registry-casework/src/store.rs @@ -319,6 +319,7 @@ impl PostgresStore { .ok_or(StoreError::AuditUnavailable)? .begin_background() .await + .map(|operation| operation.with_read_back(self.pool.clone())) } /// Append the `request` entry of one audited operation. Call it before @@ -332,6 +333,7 @@ impl PostgresStore { .ok_or(StoreError::AuditUnavailable)? .begin(request) .await + .map(|operation| operation.with_read_back(self.pool.clone())) } /// Start an audited operation that a caller requested when `actor` names diff --git a/crates/registry-casework/tests/postgres_transactions.rs b/crates/registry-casework/tests/postgres_transactions.rs index 0b918d7ce..a82224e0c 100644 --- a/crates/registry-casework/tests/postgres_transactions.rs +++ b/crates/registry-casework/tests/postgres_transactions.rs @@ -1783,8 +1783,20 @@ struct SettlementFixture { binding_reference: String, } -async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFixture { - let (store, client, _schema) = isolated_schema(prefix).await; +/// One open item in a schema of its own, audited into a capture, and the +/// staff member who may claim it. +struct OpenItemFixture { + store: PostgresStore, + audit: registry_casework::AuditCapture, + client: tokio_postgres::Client, + schema: String, + holder: ActorContext, + item_id: uuid::Uuid, + revision: i64, +} + +async fn open_item_fixture(prefix: &str) -> OpenItemFixture { + let (store, client, schema) = isolated_schema(prefix).await; let (audit, audit_capture) = registry_casework::CaseworkAudit::capture(); let store = store.with_audit(audit); store.migrate().await.expect("migrate"); @@ -1827,8 +1839,29 @@ async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFix .await .expect("initial observation") .expect("item opened"); + OpenItemFixture { + store, + audit: audit_capture, + client, + schema, + holder, + item_id: item.item_id, + revision: item.revision, + } +} + +async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFixture { + let OpenItemFixture { + store, + audit, + client, + holder, + item_id, + revision, + .. + } = open_item_fixture(prefix).await; let claimed = store - .claim(&holder, item.item_id, item.revision, "claim-settlement") + .claim(&holder, item_id, revision, "claim-settlement") .await .expect("holder claims the item"); let prepared = PreparedSourceAttempt { @@ -1859,7 +1892,7 @@ async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFix } SettlementFixture { store, - audit: audit_capture, + audit, client, holder, item_id: claimed.item_id, @@ -1921,7 +1954,12 @@ impl SettlementFixture { entry["phase"] == "response" && entry["record"]["outcome"] != "unfinished" }) .count()); - snapshot["unpairedRequests"] = serde_json::json!(unpaired_requests(&self.audit.entries())); + let unpaired = unpaired_requests(&self.audit.entries()); + assert!( + unpaired.is_empty(), + "unpaired request entries: {unpaired:?}" + ); + snapshot["unpairedRequests"] = serde_json::json!(unpaired); snapshot } @@ -2214,6 +2252,157 @@ async fn a_refusal_after_the_request_entry_pairs_it_with_an_unfinished_response( } } +/// A claim whose `COMMIT` took effect but whose acknowledgment never arrived +/// is read back as committed: the caller gets the claim, and its response +/// entry records the claim rather than an unfinished outcome. +#[tokio::test] +async fn a_claim_whose_commit_acknowledgment_is_lost_is_read_back_as_committed() { + let fixture = open_item_fixture("claim_lost_ack").await; + let written = fixture.audit.entries().len(); + fixture.audit.lose_next_commit_acknowledgment(); + let claimed = fixture + .store + .claim( + &fixture.holder, + fixture.item_id, + fixture.revision, + "claim-lost-ack", + ) + .await + .expect("the committed claim is answered"); + assert_eq!(claimed.revision, fixture.revision + 1); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!(current.revision, claimed.revision); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!(entries[1]["record"]["event"], "casework.claimed"); + assert!( + entries[1]["record"].get("outcome").is_none(), + "{entries:#?}" + ); + assert!(unpaired_requests(&fixture.audit.entries()).is_empty()); +} + +/// A claim whose `COMMIT` itself is refused rolls back, and the read-back +/// finds it rolled back: the caller gets the error and the request entry is +/// paired with an unfinished response, never with the claim. +#[tokio::test] +async fn a_claim_refused_at_commit_is_read_back_as_not_committed() { + let fixture = open_item_fixture("claim_refused_at_commit").await; + fixture + .client + .batch_execute( + "CREATE FUNCTION refuse_at_commit() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'refused at commit'; END $$; + CREATE CONSTRAINT TRIGGER refuse_claim_at_commit AFTER UPDATE ON casework_items + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION refuse_at_commit();", + ) + .await + .expect("install a trigger that refuses the claim at COMMIT"); + let written = fixture.audit.entries().len(); + let refused = fixture + .store + .claim( + &fixture.holder, + fixture.item_id, + fixture.revision, + "claim-refused-at-commit", + ) + .await; + assert!( + matches!(refused, Err(StoreError::Postgres(_))), + "{refused:?}" + ); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!(current.revision, fixture.revision, "the claim rolled back"); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); +} + +/// A claim whose future is dropped while it waits inside its transaction +/// pairs its request entry with exactly one unfinished response and changes +/// nothing. +#[tokio::test] +async fn a_claim_dropped_inside_its_transaction_writes_one_unfinished_response() { + let fixture = open_item_fixture("claim_dropped").await; + let written = fixture.audit.entries().len(); + let mut locker = connect_scoped(&fixture.schema).await; + let lock = locker.transaction().await.expect("lock transaction"); + lock.execute( + "SELECT 1 FROM casework_items WHERE item_id=$1 FOR UPDATE", + &[&fixture.item_id], + ) + .await + .expect("hold the item row"); + let locker_pid: i32 = lock + .query_one("SELECT pg_backend_pid()", &[]) + .await + .expect("lock holder pid") + .get(0); + + let store = fixture.store.clone(); + let holder = fixture.holder.clone(); + let (item_id, revision) = (fixture.item_id, fixture.revision); + let claim = tokio::spawn(async move { + store + .claim(&holder, item_id, revision, "claim-dropped") + .await + }); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + let waiting: i64 = fixture + .client + .query_one( + "SELECT count(*) FROM pg_stat_activity WHERE $1=ANY(pg_blocking_pids(pid))", + &[&locker_pid], + ) + .await + .expect("read lock waits") + .get(0); + if waiting > 0 { + break; + } + assert!( + std::time::Instant::now() < deadline, + "the claim never waited on the item row" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + claim.abort(); + assert!(claim + .await + .expect_err("the claim was dropped") + .is_cancelled()); + lock.rollback().await.expect("release the item row"); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!( + current.revision, fixture.revision, + "the dropped claim changed nothing" + ); +} + #[tokio::test] async fn a_live_execution_lease_refuses_settlement_and_writes_nothing() { let fixture = settlement_fixture("settle_live_lease", true).await; diff --git a/products/casework/CHANGELOG.md b/products/casework/CHANGELOG.md index d88db33c9..ff050672b 100644 --- a/products/casework/CHANGELOG.md +++ b/products/casework/CHANGELOG.md @@ -2,10 +2,14 @@ ## Unreleased -- Answer every audited request entry. An operation that ends after its - request entry without committing, a refusal, a failure, or a canceled - request, writes `{event, outcome: "unfinished"}` as its response under the - same correlation. Adding a review note is audited as +- Answer every audited request entry. An operation whose change is not + known to have committed (a refusal, a failure, a canceled request, or a + commit whose acknowledgment was lost and whose outcome could not be read + back) writes `{event, outcome: "unfinished"}` as its response under the + same correlation. A commit whose acknowledgment was lost is read back + first, and one that took effect is answered and recorded like any other. + A committed operation writes all of its response entries even when its + caller disconnects while they are written. Adding a review note is audited as `casework.review_note_added`, naming the note's history event but never its text or audience. From 4108d345f52a8f4171ce4df6cc1ce5da1c46ff5d Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:05:51 +0000 Subject: [PATCH 30/32] fix(casework): name the review request a review note's request entry concerns by its pseudonym Signed-off-by: Jeremi Joslin --- crates/registry-casework/src/audit.rs | 4 +++- crates/registry-casework/src/review.rs | 2 +- .../tests/review_postgres.rs | 19 ++++++++++++++++--- products/casework/CHANGELOG.md | 7 ++++--- 4 files changed, 24 insertions(+), 8 deletions(-) diff --git a/crates/registry-casework/src/audit.rs b/crates/registry-casework/src/audit.rs index e40536233..ab0d0f293 100644 --- a/crates/registry-casework/src/audit.rs +++ b/crates/registry-casework/src/audit.rs @@ -140,7 +140,8 @@ impl CaseworkAudit { /// `identifiers` is a JSON object of the identifiers the request names, keyed /// by the record field that carries them (`itemId`, `grantId`, `teamId`, -/// `queueId`); minimization keeps only their keyed pseudonyms. +/// `queueId`, `reviewRequestId`); minimization keeps only their keyed +/// pseudonyms. /// /// The `request` fields an audited operation names before it opens its /// transaction: the event it performs, the caller's profile and pseudonymized @@ -438,6 +439,7 @@ fn published_audit_record(record: Value, identifiers: &AuditKeyHasher) -> Result ("grantId", "grantPseudonym"), ("teamId", "teamPseudonym"), ("queueId", "queuePseudonym"), + ("reviewRequestId", "reviewRequestPseudonym"), ] { if let Some(value) = raw.get(field) { let value = value.as_str().ok_or(())?; diff --git a/crates/registry-casework/src/review.rs b/crates/registry-casework/src/review.rs index 17b833ba9..5d2bcd93a 100644 --- a/crates/registry-casework/src/review.rs +++ b/crates/registry-casework/src/review.rs @@ -3747,7 +3747,7 @@ impl PostgresStore { "review_note_added", Some(actor), &actor.profile_id, - json!({}), + json!({"reviewRequestId": request_id}), )) .await?; if request.note.trim().is_empty() diff --git a/crates/registry-casework/tests/review_postgres.rs b/crates/registry-casework/tests/review_postgres.rs index 5b7036c9d..cf7d059ea 100644 --- a/crates/registry-casework/tests/review_postgres.rs +++ b/crates/registry-casework/tests/review_postgres.rs @@ -3352,10 +3352,23 @@ async fn review_notes_are_audited_without_their_text() { "eventId", &added.event_id.to_string(), ); - assert!(!serde_json::to_string(&audit.entries()) - .expect("audit JSON") - .contains("stays out of the audit")); + let audit_text = serde_json::to_string(&audit.entries()).expect("audit JSON"); + assert!(!audit_text.contains("stays out of the audit")); + assert!(!audit_text.contains(&request_id.to_string())); assert!(record.get("principalPseudonym").is_some()); + // The request entry names the review request only by its pseudonym. + let requested: Vec<_> = audit + .entries() + .into_iter() + .filter(|entry| { + entry["phase"] == "request" && entry["record"]["event"] == "casework.review_note_added" + }) + .collect(); + assert_eq!(requested.len(), 1, "{requested:?}"); + assert_eq!( + requested[0]["record"]["reviewRequestPseudonym"], + audit.reference("reviewRequestId", &request_id.to_string()) + ); let (replay_service, replay_audit) = service_with_audit(&fixture, project("1")); replay_service diff --git a/products/casework/CHANGELOG.md b/products/casework/CHANGELOG.md index ff050672b..2bd0cb3aa 100644 --- a/products/casework/CHANGELOG.md +++ b/products/casework/CHANGELOG.md @@ -9,9 +9,10 @@ same correlation. A commit whose acknowledgment was lost is read back first, and one that took effect is answered and recorded like any other. A committed operation writes all of its response entries even when its - caller disconnects while they are written. Adding a review note is audited as - `casework.review_note_added`, naming the note's history event but never its - text or audience. + caller disconnects while they are written. Adding a review note is audited + as `casework.review_note_added`, naming the review request by its keyed + pseudonym and the note's history event, but never the note's text or + audience. - BREAKING: write audit through the shared platform audit writer instead of a hash-chained journal published from a PostgreSQL outbox. From d7010a44eb43a1c336b981b096342b8d8bb80888 Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:15:12 +0000 Subject: [PATCH 31/32] fix(scheduling): read back a capacity commit whose acknowledgment was lost before recording it Signed-off-by: Jeremi Joslin --- crates/registry-scheduling/src/audit.rs | 13 +- crates/registry-scheduling/src/service.rs | 18 +- crates/registry-scheduling/src/store.rs | 142 ++++++++++++- .../tests/postgres_commitments.rs | 193 +++++++++++++++++- products/scheduling/CHANGELOG.md | 16 +- products/scheduling/RUNTIME-CONFIG.md | 21 +- .../contracts/security-invariant-matrix.yaml | 17 +- .../contracts/security-test-traceability.yaml | 4 + 8 files changed, 391 insertions(+), 33 deletions(-) diff --git a/crates/registry-scheduling/src/audit.rs b/crates/registry-scheduling/src/audit.rs index af1c6b66f..d215e98e0 100644 --- a/crates/registry-scheduling/src/audit.rs +++ b/crates/registry-scheduling/src/audit.rs @@ -5,10 +5,15 @@ //! opens and one `response` entry once the decision is known: after commit //! for an allowed commitment, after rollback for a refused one. Both share a //! correlation, which is also the `eventId` the response record carries. -//! A commitment nothing decided, because its transaction failed, the -//! environment records moved under it, or its idempotency key was refused, -//! still answers its request entry with an `unfinished` response naming the -//! reason; one that returns or is canceled without answering writes the +//! A commitment nothing decided, because its transaction rolled back on a +//! failure, the environment records moved under it, or its idempotency key +//! was refused, still answers its request entry with an `unfinished` +//! response naming the reason. A capacity commit that is not acknowledged is +//! read back from the database before it is recorded: one that took effect is +//! answered and recorded as committed, one that rolled back as +//! `commitment.failed`, and one whose status cannot be read is recorded as +//! `commitment.unfinished`, never as failed, because it may have taken +//! effect. One that returns or is canceled without answering writes the //! `commitment.unfinished` response when its request handle is dropped. //! Entries carry only pseudonymized references and closed codes, never a raw //! principal, grant, claim identifier, or free-text reason. diff --git a/crates/registry-scheduling/src/service.rs b/crates/registry-scheduling/src/service.rs index 1aaca3f40..cb46a6e0d 100644 --- a/crates/registry-scheduling/src/service.rs +++ b/crates/registry-scheduling/src/service.rs @@ -1414,7 +1414,9 @@ impl SchedulingService { /// a replay of that key answers the same, and is answered only once its /// `response` entry is accepted: denied for a decision the ledger took, /// `unfinished` with its reason for a failed transaction, a replaced - /// environment, or a refused idempotency key. Every entry carries the + /// environment, a refused idempotency key, or a commit whose outcome + /// could not be read back. A commit that took effect though its + /// acknowledgment was lost reaches here as minted. Every entry carries the /// `correlation` of the commitment's `request` entry. #[allow(clippy::too_many_arguments)] async fn commitment_outcome( @@ -1465,6 +1467,13 @@ impl SchedulingService { tracing::error!(%error, "the Scheduling store failed mid-commitment"); Unanswered::Unfinished("commitment.failed") } + // The commit was not acknowledged and its read-back + // failed, so it may have taken effect: this records the + // outcome as unknown, never as failed. + CommitError::Unacknowledged => { + tracing::error!(%error, "a Scheduling commitment's outcome is unknown"); + Unanswered::Unfinished("commitment.unfinished") + } // A records replacement moved under this request, and the // caller retries; the swap is an expected operator act, so // this is a warning, not a failure. @@ -1489,7 +1498,8 @@ impl SchedulingService { }; // A refusal writes its receipt so a replay of the key answers // the same. A failed transaction and a replaced environment - // decided nothing, and an expired receipt cannot be + // decided nothing, an unacknowledged commit may already hold + // the key's receipt, and an expired receipt cannot be // recreated. A reused key still passes through the // insert-or-replay path: a concurrent identical winner is // replayed, while a different request hash remains key-reused. @@ -1500,6 +1510,7 @@ impl SchedulingService { | CommitError::Hooks(_) | CommitError::FactsStale | CommitError::KeyExpired + | CommitError::Unacknowledged ); if receipted { let receipt = self.commitment( @@ -2165,6 +2176,9 @@ fn problem_of(error: &CommitError) -> ProblemCode { CommitError::CutoffPassed => ProblemCode::CancellationCutoffPassed, // Nothing was decided: the caller retries against the current records. CommitError::FactsStale => ProblemCode::ServiceUnavailable, + // The caller retries under the same key, which replays the receipt + // if the commit took effect. + CommitError::Unacknowledged => ProblemCode::ServiceUnavailable, } } diff --git a/crates/registry-scheduling/src/store.rs b/crates/registry-scheduling/src/store.rs index 75e094c5e..b862524aa 100644 --- a/crates/registry-scheduling/src/store.rs +++ b/crates/registry-scheduling/src/store.rs @@ -34,6 +34,10 @@ use std::collections::HashMap; use std::str::FromStr; +#[cfg(feature = "postgres-test")] +use std::sync::atomic::{AtomicBool, Ordering}; +#[cfg(feature = "postgres-test")] +use std::sync::Arc; use std::time::Duration; use chrono::{DateTime, TimeDelta, Utc}; @@ -177,6 +181,12 @@ pub enum StoreError { pub struct PostgresStore { pool: Pool, clock: SharedClock, + /// Test switches that report the next capacity commit that took effect + /// as unacknowledged, and the next read-back of one as unreadable. + #[cfg(feature = "postgres-test")] + lose_acknowledgment: Arc, + #[cfg(feature = "postgres-test")] + fail_read_back: Arc, } impl PostgresStore { @@ -210,6 +220,117 @@ impl PostgresStore { .write() .expect("the store clock is never held across a panic") = clock; } + + /// Report the next capacity commit as unacknowledged after it took + /// effect, as a connection lost during `COMMIT` does. Test-only, and + /// shared by every handle cloned from this store. + #[cfg(feature = "postgres-test")] + pub fn lose_next_commit_acknowledgment(&self) { + self.lose_acknowledgment.store(true, Ordering::SeqCst); + } + + /// Make the next read-back of an unacknowledged commit unreadable, as a + /// database that cannot be reached again does. Test-only. + #[cfg(feature = "postgres-test")] + pub fn fail_next_read_back(&self) { + self.fail_read_back.store(true, Ordering::SeqCst); + } + + /// Whether the test switch reports this commit as unacknowledged. Always + /// false outside tests. + fn acknowledgment_lost(&self) -> bool { + #[cfg(feature = "postgres-test")] + { + self.lose_acknowledgment.swap(false, Ordering::SeqCst) + } + #[cfg(not(feature = "postgres-test"))] + { + false + } + } + + /// Whether the test switch makes this read-back unreadable. Always false + /// outside tests. + fn read_back_fails(&self) -> bool { + #[cfg(feature = "postgres-test")] + { + self.fail_read_back.swap(false, Ordering::SeqCst) + } + #[cfg(not(feature = "postgres-test"))] + { + false + } + } + + /// Commit a capacity transaction whose decision is already written. + /// + /// A `COMMIT` whose acknowledgment never arrives may still have taken + /// effect, as when the connection is lost after it reached the database. + /// Its transaction's status is then read back on another connection, + /// outside any transaction: the read writes nothing and takes no + /// capacity lock. A commit that took effect is answered as committed, + /// one that rolled back as the query failure it was, and one whose + /// status cannot be read as [`CommitError::Unacknowledged`]. + async fn commit_capacity( + &self, + transaction: deadpool_postgres::Transaction<'_>, + ) -> Result<(), CommitError> { + let transaction_id: String = transaction + .query_one("SELECT pg_current_xact_id()::text", &[]) + .await? + .get(0); + let failure = match transaction.commit().await { + Ok(()) if !self.acknowledgment_lost() => return Ok(()), + Ok(()) => None, + Err(error) => Some(error), + }; + match self.read_back(&transaction_id).await { + ReadBack::Committed => { + tracing::warn!( + "a Scheduling capacity commit was not acknowledged but took effect; it is answered as committed" + ); + Ok(()) + } + ReadBack::RolledBack => { + Err(failure.map_or(CommitError::Unacknowledged, CommitError::Query)) + } + ReadBack::Unknown => { + tracing::error!( + "a Scheduling capacity commit was not acknowledged and its outcome could not be read back" + ); + Err(CommitError::Unacknowledged) + } + } + } + + /// Read whether the transaction `transaction_id` committed, on a pooled + /// connection outside any transaction. + async fn read_back(&self, transaction_id: &str) -> ReadBack { + if self.read_back_fails() { + return ReadBack::Unknown; + } + let Ok(client) = self.client().await else { + return ReadBack::Unknown; + }; + let status = client + .query_one("SELECT pg_xact_status($1::text::xid8)", &[&transaction_id]) + .await + .map(|row| row.get::<_, Option>(0)); + match status.as_ref().map(|status| status.as_deref()) { + Ok(Some("committed")) => ReadBack::Committed, + Ok(Some("aborted")) => ReadBack::RolledBack, + // Still in progress, too old to report, or unreadable. + _ => ReadBack::Unknown, + } + } +} + +/// What reading back an unacknowledged capacity commit found. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum ReadBack { + Committed, + RolledBack, + Unknown, } impl std::fmt::Debug for PostgresStore { @@ -410,6 +531,11 @@ pub enum CommitError { /// current records. #[error("the environment records were replaced while the request was in flight")] FactsStale, + /// The capacity transaction's `COMMIT` was not acknowledged and reading + /// its status back failed, so whether it took effect is unknown. A retry + /// under the same idempotency key replays its receipt if it did. + #[error("the capacity commit was not acknowledged and its outcome is unknown")] + Unacknowledged, } /// One due or delivered outbox intent. @@ -512,6 +638,10 @@ impl PostgresStore { Ok(Self { pool, clock: system_clock(), + #[cfg(feature = "postgres-test")] + lose_acknowledgment: Arc::default(), + #[cfg(feature = "postgres-test")] + fail_read_back: Arc::default(), }) } @@ -1366,7 +1496,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Hold(claim)) } @@ -1474,7 +1604,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(claim)) } @@ -1620,7 +1750,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(claim)) } @@ -1687,7 +1817,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Released) } @@ -1845,7 +1975,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(moved)) } @@ -1971,7 +2101,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Cancelled(cancelled)) } diff --git a/crates/registry-scheduling/tests/postgres_commitments.rs b/crates/registry-scheduling/tests/postgres_commitments.rs index 3f552b241..88dbc16b9 100644 --- a/crates/registry-scheduling/tests/postgres_commitments.rs +++ b/crates/registry-scheduling/tests/postgres_commitments.rs @@ -6591,6 +6591,192 @@ async fn a_failed_capacity_transaction_pairs_its_request_entry() { assert_eq!(response["operation"], "appointment.create"); } +/// SCHEDULING-SEC-14: a capacity commit that took effect though its +/// acknowledgment was lost is read back as committed. The caller gets the +/// appointment, its response entry records it allowed, and a retry under the +/// same key replays the same appointment. +#[tokio::test] +async fn a_commitment_whose_commit_acknowledgment_is_lost_is_read_back_as_committed() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let before = fx.capture.entries().len(); + fx.store.lose_next_commit_acknowledgment(); + let (status, appointment) = fx + .post("/v1/appointments", &fx.agent, "lost-ack", body.clone()) + .await; + assert_eq!(status, StatusCode::CREATED, "{appointment}"); + let response = one_request_and_one_response(&fx.capture.entries().split_off(before)); + assert_eq!(response["operation"], "appointment.create"); + assert_eq!(response["outcome"], "allowed"); + assert_eq!(response["reason"], "authorization.allowed"); + + let (status, replayed) = fx + .post("/v1/appointments", &fx.agent, "lost-ack", body) + .await; + assert_eq!(status, StatusCode::CREATED, "{replayed}"); + assert_eq!(replayed["appointmentId"], appointment["appointmentId"]); +} + +/// SCHEDULING-SEC-14: a capacity commit that was not acknowledged and whose +/// status cannot be read back is recorded as unfinished, never as failed, +/// because it may have taken effect. Here it did, so a retry under the same +/// key replays the appointment it committed. +#[tokio::test] +async fn a_commitment_whose_outcome_cannot_be_read_back_is_recorded_unfinished() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let before = fx.capture.entries().len(); + fx.store.lose_next_commit_acknowledgment(); + fx.store.fail_next_read_back(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "unknown-commit", + body.clone(), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); + let response = one_unfinished_response( + &fx.capture.entries().split_off(before), + "commitment.unfinished", + ); + assert_eq!(response["operation"], "appointment.create"); + + let (status, replayed) = fx + .post("/v1/appointments", &fx.agent, "unknown-commit", body) + .await; + assert_eq!(status, StatusCode::CREATED, "{replayed}"); +} + +/// SCHEDULING-SEC-14: a capacity transaction refused at `COMMIT` itself rolls +/// back, and the read-back finds it rolled back, so its request entry is +/// answered as failed. +#[tokio::test] +async fn a_commitment_refused_at_commit_is_read_back_as_failed() { + let fx = fixture().await; + fx.admin + .batch_execute( + "CREATE FUNCTION test_refuse_at_commit() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'refused at commit'; END $$; + CREATE CONSTRAINT TRIGGER test_refuse_claim_at_commit AFTER INSERT ON scheduling_claims + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION test_refuse_at_commit();", + ) + .await + .expect("install a trigger that refuses the commitment at COMMIT"); + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "refused-at-commit", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + let response = + one_unfinished_response(&fx.capture.entries().split_off(before), "commitment.failed"); + assert_eq!(response["operation"], "appointment.create"); + let claims: i64 = fx + .admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + assert_eq!(claims, 0, "the refused commitment rolled back"); +} + +/// SCHEDULING-SEC-14: a commitment whose caller goes away while its capacity +/// transaction waits for the supply lock pairs its request entry with exactly +/// one unfinished response, and commits nothing. +#[tokio::test] +async fn a_commitment_dropped_inside_its_capacity_transaction_writes_one_unfinished_response() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let (http, agent, capture) = (fx.http.clone(), fx.agent.clone(), fx.capture.clone()); + let before = capture.entries().len(); + let mut admin = fx.admin; + let claims_before: i64 = admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + let holder = admin + .transaction() + .await + .expect("the stand-in capacity transaction opens"); + holder + .execute("SELECT supply_id FROM scheduling_supply FOR UPDATE", &[]) + .await + .expect("the stand-in holds every supply anchor"); + let holder_pid: i32 = holder + .query_one("SELECT pg_backend_pid()", &[]) + .await + .expect("the stand-in's backend") + .get(0); + + let commitment = tokio::spawn(send( + http, + "POST".to_owned(), + "/v1/appointments".to_owned(), + agent, + Some("dropped-commitment".to_owned()), + Some(body), + )); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + let waiting: i64 = holder + .query_one( + "SELECT count(DISTINCT pid) FROM pg_locks WHERE $1=ANY(pg_blocking_pids(pid))", + &[&holder_pid], + ) + .await + .expect("read the waiting commitment") + .get(0); + if waiting > 0 { + break; + } + assert!( + !commitment.is_finished(), + "the commitment finished before reaching the supply lock" + ); + assert!( + std::time::Instant::now() < deadline, + "the commitment never waited on the supply lock" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + commitment.abort(); + assert!(commitment + .await + .expect_err("the commitment was dropped") + .is_cancelled()); + holder + .rollback() + .await + .expect("the stand-in releases the supply"); + + let response = one_unfinished_response( + &capture.entries().split_off(before), + "commitment.unfinished", + ); + assert_eq!(response["operation"], "appointment.create"); + let claims_after: i64 = admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + assert_eq!( + claims_after, claims_before, + "the dropped commitment committed nothing" + ); +} + /// SCHEDULING-SEC-14: an idempotency key refused as reused, and one refused /// as expired, each answer the request entry the commitment wrote, without /// recording the key refusal as an authorization decision. @@ -6708,9 +6894,10 @@ async fn a_records_swap_under_a_commitment_pairs_its_request_entry() { } /// SCHEDULING-SEC-14: a refusal is answered only once its response entry is -/// accepted, like an allowed commitment. When the destination refuses it, the -/// caller is told the service is unavailable, for a permission mismatch -/// decided before the transaction and for a refusal the ledger decided. +/// accepted, like an allowed commitment. When the destination refuses the +/// response entry of a refusal the ledger decided, the caller is told the +/// service is unavailable. The permission mismatch decided before the +/// transaction is covered by the test that follows. #[tokio::test] async fn a_refusal_whose_response_entry_is_refused_answers_service_unavailable() { let fx = fixture().await; diff --git a/products/scheduling/CHANGELOG.md b/products/scheduling/CHANGELOG.md index f00af1eb7..2a9817f88 100644 --- a/products/scheduling/CHANGELOG.md +++ b/products/scheduling/CHANGELOG.md @@ -24,11 +24,17 @@ - Every commitment request entry is answered. A refusal, decided by the ledger or by the permission check, is answered only once its response entry is accepted, and `service.unavailable` otherwise, where it was - previously answered with the write failure only logged. A failed - transaction, records replaced under a commitment, and a reused or expired - idempotency key now write a response with the outcome `unfinished` and a - closed reason, and a commitment that returns or is canceled before - answering writes `commitment.unfinished`. + previously answered with the write failure only logged. A transaction + rolled back on a failure, records replaced under a commitment, and a + reused or expired idempotency key now write a response with the outcome + `unfinished` and a closed reason, and a commitment that returns or is + canceled before answering writes `commitment.unfinished`. + - A capacity commit that is not acknowledged is read back by its + transaction identifier before it is recorded: one that took effect is + answered and recorded as committed, one that rolled back writes + `commitment.failed`, and one whose status cannot be read writes + `commitment.unfinished` and answers `service.unavailable`, never + `commitment.failed`. - Hook delivery writes an attempt's request entry before egress and its terminal response entry with the same correlation; a refused entry leaves the delivery pending and sends nothing. diff --git a/products/scheduling/RUNTIME-CONFIG.md b/products/scheduling/RUNTIME-CONFIG.md index 88a5f27ee..60a607431 100644 --- a/products/scheduling/RUNTIME-CONFIG.md +++ b/products/scheduling/RUNTIME-CONFIG.md @@ -104,13 +104,20 @@ the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation past its cutoff) or the permission check refused it before the transaction opened, reaches the caller only once its `response` entry is accepted, and answers `service.unavailable` otherwise. A commitment nothing decided still -answers its `request` entry: a failed transaction, records replaced under it, -or a reused or expired idempotency key writes a `response` with the outcome -`unfinished` and the reason `commitment.failed`, `commitment.facts-stale`, -`idempotency.key-reused`, or `idempotency.expired`, and one that returns or is -canceled before answering writes `commitment.unfinished`. `/readyz` reports unavailable -while the destination refuses writes. An expired hold writes its history entry -as `system` and no audit entry. +answers its `request` entry: a transaction rolled back on a failure, records +replaced under it, or a reused or expired idempotency key writes a `response` +with the outcome `unfinished` and the reason `commitment.failed`, +`commitment.facts-stale`, `idempotency.key-reused`, or `idempotency.expired`, +and one that returns or is canceled before answering writes +`commitment.unfinished`. A capacity commit that is not acknowledged is read +back by its transaction identifier on a separate connection that changes +nothing: one that took effect is answered and recorded as committed, one that +rolled back writes `commitment.failed`, and one whose status cannot be read +writes `commitment.unfinished` and answers `service.unavailable`, because it +may have taken effect; a retry under the same idempotency key then replays +whatever committed. `/readyz` reports unavailable while the destination +refuses writes. An expired hold writes its history entry as `system` and no +audit entry. `destinations` is optional. `destinations.reminders` is the one place due reminder intents are delivered, as CloudEvents 1.0 events over HTTPS POST with diff --git a/products/scheduling/contracts/security-invariant-matrix.yaml b/products/scheduling/contracts/security-invariant-matrix.yaml index 78107bb9b..e214ecdfb 100644 --- a/products/scheduling/contracts/security-invariant-matrix.yaml +++ b/products/scheduling/contracts/security-invariant-matrix.yaml @@ -255,12 +255,17 @@ invariants: the ledger decided: an admission refusal, the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation past its cutoff. Every request entry is answered. A commitment nothing decided, because - its transaction failed, the environment records were replaced under it, - or its idempotency key was refused as reused or expired, writes one - response entry with the outcome unfinished and a closed reason, and - never an authorization verdict. One that returns or is canceled before - it answers writes the commitment.unfinished response when its request - handle is dropped. The match that selects between them is exhaustive, + its transaction rolled back on a failure, the environment records were + replaced under it, or its idempotency key was refused as reused or + expired, writes one response entry with the outcome unfinished and a + closed reason, and never an authorization verdict. A capacity commit + that is not acknowledged is read back by its transaction identifier on + a separate connection that mutates nothing: one that took effect is + answered and recorded as committed, one that rolled back is recorded + commitment.failed, and one whose status cannot be read is recorded + commitment.unfinished, never failed. One that returns or is canceled + before it answers writes the commitment.unfinished response when its + request handle is dropped. The match that selects between them is exhaustive, so a new outcome states its side. A replayed receipt, including one a concurrent identical request won, is answered only after its own response entry is accepted, and that entry records the diff --git a/products/scheduling/contracts/security-test-traceability.yaml b/products/scheduling/contracts/security-test-traceability.yaml index 6b92cfe38..a2eb2efe5 100644 --- a/products/scheduling/contracts/security-test-traceability.yaml +++ b/products/scheduling/contracts/security-test-traceability.yaml @@ -126,6 +126,10 @@ entries: - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_replayed_refusal_is_recorded_as_the_refusal_it_replays} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_racing_replay_is_not_released_when_its_response_entry_is_refused} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_failed_capacity_transaction_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_whose_commit_acknowledgment_is_lost_is_read_back_as_committed} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_whose_outcome_cannot_be_read_back_is_recorded_unfinished} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_refused_at_commit_is_read_back_as_failed} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_dropped_inside_its_capacity_transaction_writes_one_unfinished_response} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: an_idempotency_key_refusal_pairs_its_request_entry} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_records_swap_under_a_commitment_pairs_its_request_entry} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_refusal_whose_response_entry_is_refused_answers_service_unavailable} From 3dd1be247af77bd84647edc54b631413b173320d Mon Sep 17 00:00:00 2001 From: Jeremi Joslin Date: Sat, 26 Sep 2026 19:19:05 +0000 Subject: [PATCH 32/32] docs(site): state what the BReg and Render audit journals record when a request ends early Signed-off-by: Jeremi Joslin --- .../content/docs/operate/breg-retention.mdx | 22 ++++++++++++------- .../content/docs/operate/registry-render.mdx | 2 ++ 2 files changed, 16 insertions(+), 8 deletions(-) diff --git a/docs/site/src/content/docs/operate/breg-retention.mdx b/docs/site/src/content/docs/operate/breg-retention.mdx index 9abc401a4..f178ce779 100644 --- a/docs/site/src/content/docs/operate/breg-retention.mdx +++ b/docs/site/src/content/docs/operate/breg-retention.mdx @@ -314,18 +314,22 @@ entry. The file destination acknowledges an append after fsync; `stdout` flushes provides best-effort delivery. A failed request entry blocks protected I/O, and a failed response entry blocks disclosure. A request that ends before its outcome is known, because it failed, timed out, or its caller went away, or an operator command that exits early, still answers its request -entry with a response whose outcome is `unfinished`. Only a crash, or a destination that already -stopped accepting entries, can leave a request entry without a response. +entry. A runtime read or mutation answers with a response whose `phase` is `unfinished`, carrying +only the operation and request it answers; an operator command answers with a `terminal` response +whose `outcome` is `unfinished`. Only a crash, or a destination that already stopped accepting +entries, can leave a request entry without a response. Operator commands follow the same rule. `evidence-retention erase-expired` records the cutoff it was given in its request entry and the number of assertions it erased in its response. `request-retention erase` deletes the external attachment objects before it records its response, -which states how many objects still wait for deletion. If the erasure committed but its response +which states how many objects still wait for deletion; when that deletion pass itself fails, the +count is `null` and the command reports the failure. If the erasure committed but its response entry could not be written, the command reports `request_retention.erasure.unaudited`: the detail is gone, so restore the audit destination and reconcile the erased request against the database. An -event delivery's attempt is recorded before its request leaves, and its outcome only once the -delivery's state has committed, so the journal never records a delivery that rolled back; an -operator replay is recorded as a request before the reset and a response once it commits or is +event delivery's attempt is recorded inside the transaction that leases it, before its request +leaves. If that lease transaction does not commit, no request leaves and the attempt is answered +`worker_interrupted`, so an attempt entry does not prove a request was sent. An attempt's terminal +outcome is recorded only once the delivery's state has committed; an operator replay is recorded as a request before the reset and a response once it commits or is refused. A crash or a killed process can also tear the file's final line itself, leaving it without its @@ -373,10 +377,12 @@ Choose a fresh audit path when changing from another audit format, and preserve outside the new writer's retention directory. ::: -{/* Evidence: crates/registry-breg/src/audit.rs; +{/* Evidence: crates/registry-breg/src/audit.rs, unfinished_record; crates/registry-breg/src/runtime_config.rs, AuditConfig; crates/registry-platform-audit/src/writer.rs, AuditRequest; - crates/registry-breg/src/request_retention.rs, ErasureUnaudited; + crates/registry-breg/src/request_retention.rs, ErasureUnaudited, retention_outcome_record, + and pendingExternalDeletions; + crates/registry-platform-hooks/src/delivery/service.rs, WorkerInterrupted; crates/registry-breg/src/action_evidence_maintenance.rs, EVIDENCE_RETENTION_AUDIT_SCHEMA; crates/registry-platform-hooks/src/delivery/seams.rs, record_audit; crates/registry-breg/src/postgres/schema.rs; diff --git a/docs/site/src/content/docs/operate/registry-render.mdx b/docs/site/src/content/docs/operate/registry-render.mdx index f01db47cc..a35dc3e6e 100644 --- a/docs/site/src/content/docs/operate/registry-render.mdx +++ b/docs/site/src/content/docs/operate/registry-render.mdx @@ -197,6 +197,8 @@ audit file resolves under a persistent root, as a container preflight, and refus destination. {/* Evidence: crates/registry-render/src/audit.rs, AUDIT_SCHEMA, RenderAuditEvent, and correlation; + crates/registry-render/src/audit.rs, RenderAudit::request writes unfinished when dropped; + crates/registry-render/src/server.rs, the_request_entry_is_accepted_before_the_render_starts; crates/registry-render/src/cli.rs, require_audit_under; crates/registry-platform-audit/src/writer.rs, DEFAULT_AUDIT_ROTATE_BYTES and DEFAULT_AUDIT_RETAIN_DAYS. */}