diff --git a/crates/registry-breg/src/action_evidence_maintenance.rs b/crates/registry-breg/src/action_evidence_maintenance.rs index 31d85286a5..67efa7f8c1 100644 --- a/crates/registry-breg/src/action_evidence_maintenance.rs +++ b/crates/registry-breg/src/action_evidence_maintenance.rs @@ -3,6 +3,11 @@ use std::{path::Path, time::Duration}; +use registry_platform_audit::AuditEntry; +use serde_json::{json, Value}; +use uuid::Uuid; + +use crate::audit::RegistryAudit; use crate::mutation::{erase_expired_action_evidence, MutationError}; use crate::postgres::{ verify_catalog_identity_for_catalog, verify_migration_role, ConnectionConfig, @@ -10,6 +15,10 @@ use crate::postgres::{ }; use crate::runtime_config::load_runtime_config; +/// Schema of the request and response entries of one expired-Evidence +/// erasure. +pub const EVIDENCE_RETENTION_AUDIT_SCHEMA: &str = "breg-evidence-retention-audit/v1"; + /// Package-bound authority for erasing expired protected Evidence material. /// The migration identity, actual target catalog and registry interlock are /// checked again in the same transaction that deletes the retained material. @@ -22,6 +31,7 @@ pub struct ActionEvidenceRetentionOperatorService { runtime_role: SqlIdentifier, lock_timeout: Duration, statement_timeout: Duration, + audit: RegistryAudit, } impl ActionEvidenceRetentionOperatorService { @@ -56,6 +66,9 @@ impl ActionEvidenceRetentionOperatorService { runtime_role: config.database().roles().runtime().clone(), lock_timeout: config.operational_timeouts().migration_lock, statement_timeout: config.operational_timeouts().migration_statement, + audit: RegistryAudit::open_companion(&config) + .await + .map_err(|_| MutationError::Unavailable)?, }) } @@ -68,6 +81,7 @@ impl ActionEvidenceRetentionOperatorService { migration_connection: ConnectionConfig, migration_role: SqlIdentifier, runtime_role: SqlIdentifier, + audit: RegistryAudit, ) -> Self { Self { expected, @@ -78,9 +92,16 @@ impl ActionEvidenceRetentionOperatorService { runtime_role, lock_timeout: Duration::from_secs(5), statement_timeout: Duration::from_secs(10), + audit, } } + /// Erase the retained Evidence material whose expiry is before `before`. + /// + /// The request entry, naming the threshold, is accepted before the + /// erasure transaction opens, so an audit outage erases nothing. Its + /// response records the erased count once the transaction commits, or + /// the failure when it does not. pub async fn erase_expired( &self, before: chrono::DateTime, @@ -88,6 +109,108 @@ impl ActionEvidenceRetentionOperatorService { if before > chrono::Utc::now() { return Err(MutationError::InvalidRequest); } + let correlation = Uuid::new_v4().to_string(); + let record = |phase: &str, outcome: &str| -> Value { + json!({ + "kind": "evidenceRetention", + "phase": phase, + "outcome": outcome, + "packageRevision": self.expected.package_revision, + "actor": "breg:evidence-retention-operator", + "before": before.to_rfc3339_opts(chrono::SecondsFormat::Secs, true), + "correlation": correlation, + }) + }; + let mut attempt = self + .audit + .begin( + AuditEntry::request( + EVIDENCE_RETENTION_AUDIT_SCHEMA, + correlation.clone(), + record("attempt", "started"), + ), + record("terminal", "unfinished"), + ) + .await + .map_err(|_| MutationError::Unavailable)?; + let erased = match self.erase_in_transaction(before).await { + Ok(Erasure::Committed(erased)) => Ok(erased), + // A commit that returned an error may still have committed, so + // the outcome recorded for this destructive operation is the one + // the database holds, read on a fresh connection. + Ok(Erasure::Unacknowledged { erased, cutoff }) => { + match self.expired_evidence_remains(cutoff).await { + Some(false) => Ok(erased), + resolved => { + let outcome = if resolved == Some(true) { + "failed" + } else { + "unfinished" + }; + if attempt.respond(record("terminal", outcome)).await.is_err() { + tracing::error!( + "the unacknowledged Evidence retention's response audit entry was not recorded" + ); + } + return Err(MutationError::Unavailable); + } + } + } + Err(error) => Err(error), + }; + match erased { + Ok(erased) => { + let mut response = record("terminal", "erased"); + response["erased"] = json!(erased); + // The erasure committed; a refused entry reports the command + // unavailable, and the writer then refuses every later entry. + attempt + .respond(response) + .await + .map_err(|_| MutationError::Unavailable)?; + Ok(erased) + } + Err(error) => { + if attempt.respond(record("terminal", "failed")).await.is_err() { + tracing::error!( + "the failed Evidence retention's response audit entry was not recorded" + ); + } + Err(error) + } + } + } + + /// Whether retained Evidence expiring at or before `cutoff` remains, read + /// on a fresh connection after an erasure commit returned an error. + /// `None` when it cannot be read. + async fn expired_evidence_remains( + &self, + cutoff: chrono::DateTime, + ) -> Option { + let pool = self.migration_connection.build_pool().ok()?; + let client = pool.get().await.ok()?; + let row = client + .query_one( + "SELECT EXISTS (SELECT 1 FROM registry_internal.registry_action_evidence_uses + WHERE expires_at <= $1) + OR EXISTS (SELECT 1 FROM registry_internal.registry_request_evidence_uses + WHERE expires_at <= $1)", + &[&cutoff], + ) + .await + .ok()?; + row.try_get(0).ok() + } + + /// Erase the expired material in one transaction. A commit that returned + /// an error, which does not prove the transaction rolled back, is + /// reported with the count it would have erased and the cutoff it erased + /// through; every earlier error is returned as one. + async fn erase_in_transaction( + &self, + before: chrono::DateTime, + ) -> Result { let pool = self .migration_connection .build_pool() @@ -127,15 +250,29 @@ impl ActionEvidenceRetentionOperatorService { if !ready { return Err(MutationError::Unavailable); } - let erased = erase_expired_action_evidence(&transaction, before).await?; - transaction - .commit() + // The cutoff the deletion applies, fixed by the transaction's start. + let cutoff: chrono::DateTime = transaction + .query_one("SELECT LEAST($1, CURRENT_TIMESTAMP)", &[&before]) .await - .map_err(|_| MutationError::Unavailable)?; - Ok(erased) + .map_err(|_| MutationError::Unavailable)? + .get(0); + let erased = erase_expired_action_evidence(&transaction, before).await?; + if transaction.commit().await.is_err() { + return Ok(Erasure::Unacknowledged { erased, cutoff }); + } + Ok(Erasure::Committed(erased)) } } +/// How an erasure transaction ended once every statement in it succeeded. +enum Erasure { + Committed(u64), + Unacknowledged { + erased: u64, + cutoff: chrono::DateTime, + }, +} + /// Erase only material whose declared expiry has passed. Diagnostics contain /// no connection, selector, assertion or provider values. pub async fn erase_expired(path: &Path, before: &str) -> Result { diff --git a/crates/registry-breg/src/api/ingestion.rs b/crates/registry-breg/src/api/ingestion.rs index 42e788a659..46675d0c2f 100644 --- a/crates/registry-breg/src/api/ingestion.rs +++ b/crates/registry-breg/src/api/ingestion.rs @@ -177,15 +177,17 @@ async fn create_run( .await { Ok(run) => ingestion_response(StatusCode::CREATED, json!({ "run": run })), - // A refusal after the request entry was accepted owes the journal - // its response entry, as a refused chunk submission does. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -395,15 +397,17 @@ async fn cancel_run( .await { Ok(run) => ingestion_response(StatusCode::OK, json!({ "run": run })), - // A refusal after the request entry was accepted owes the journal - // its response entry, as a refused chunk submission does. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -545,16 +549,17 @@ async fn submit_chunk( .await { Ok(answer) => ingestion_response(StatusCode::OK, answer), - // A submission the run refuses after parsing owes the journal the - // same durable refusal envelope pre-parse failures write, so an audit - // outage gates the refusal instead of passing silently. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await @@ -631,16 +636,17 @@ async fn chunk_receipt( .await { Ok(receipt) => ingestion_response(StatusCode::OK, receipt), - // A recovery the run refuses after its request entry was accepted - // owes the journal the same durable refusal envelope a refused - // cancellation or chunk submission owes. - Err(error) => { + // A refusal after the ingestion request entry is already answered in + // the ingestion schema; any other refusal owes the journal its + // single refusal entry here, so an audit outage gates it. + Err(refusal) if refusal.answered => ingestion_problem(refusal.error), + Err(refusal) => { audited_mutation_refusal( mutations, &binding.base, &surface.context, None, - ingestion_problem(error), + ingestion_problem(refusal.error), &correlation, ) .await diff --git a/crates/registry-breg/src/attachment_verification_worker.rs b/crates/registry-breg/src/attachment_verification_worker.rs index b2f0fb8554..0d1db164c9 100644 --- a/crates/registry-breg/src/attachment_verification_worker.rs +++ b/crates/registry-breg/src/attachment_verification_worker.rs @@ -113,7 +113,9 @@ impl AttachmentVerificationWorker { }; transaction.commit().await.map_err(unavailable)?; drop(client); - self.audit_job(&job, "attempt", "started").await?; + // Held until the terminal entry answers it; a run that ends first, + // including one its time budget cancels, answers it as unfinished. + let _attempt = self.begin_job(&job).await?; let verdict = match self.content(&job).await { Ok(bytes) => verifier @@ -240,10 +242,42 @@ impl AttachmentVerificationWorker { Ok(transaction) } + /// Append the attempt `request` entry of one leased job and return the + /// handle that owes its terminal `response`. + async fn begin_job( + &self, + job: &VerificationJob, + ) -> Result { + let (reference, record) = self.job_record(job, "attempt", "started")?; + let (_, unfinished) = self.job_record(job, "terminal", "unfinished")?; + self.audit + .begin( + AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record), + unfinished, + ) + .await + .map_err(unavailable) + } + /// Append one verification entry. The attempt is the `request` entry and /// the terminal outcome is the `response` entry; both are correlated by /// the keyed verification reference of the leased job. async fn audit_job(&self, job: &VerificationJob, phase: &str, outcome: &str) -> Result<()> { + let (reference, record) = self.job_record(job, phase, outcome)?; + let entry = if phase == "attempt" { + AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) + } else { + AuditEntry::response(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) + }; + self.audit.append(entry).await.map_err(unavailable) + } + + fn job_record( + &self, + job: &VerificationJob, + phase: &str, + outcome: &str, + ) -> Result<(String, serde_json::Value)> { let hasher = self.audit.profile().key_hasher(); let reference = hasher .audit_reference_hash( @@ -257,12 +291,7 @@ impl AttachmentVerificationWorker { "packageRevision": self.expected.package_revision, "actor": "breg:attachment-verifier", "verificationReference": reference, }); - let entry = if phase == "attempt" { - AuditEntry::request(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) - } else { - AuditEntry::response(ATTACHMENT_VERIFICATION_AUDIT_SCHEMA, reference, record) - }; - self.audit.append(entry).await.map_err(unavailable) + Ok((reference, record)) } } diff --git a/crates/registry-breg/src/audit.rs b/crates/registry-breg/src/audit.rs index 4cad4e569f..023a44b052 100644 --- a/crates/registry-breg/src/audit.rs +++ b/crates/registry-breg/src/audit.rs @@ -6,10 +6,15 @@ //! `response` entry with its outcome, both through the one platform //! [`AuditWriter`] the process opened at startup and both correlated by the //! request id Base Registry Engine minted. A refusal is one `response` entry. +//! A request that goes on to protected I/O holds its attempt as an +//! [`AuditRequest`] until its terminal or refusal entry answers it; one that +//! ends first writes an `unfinished` response, so no attempt stays unpaired. //! Entries carry keyed references and closed-vocabulary terms, never a raw //! principal, record id, selector, token, or free text. -use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile, AuditWriter}; +use registry_platform_audit::{ + AuditEntry, AuditKeyHasher, AuditProfile, AuditRequest, AuditWriter, +}; use serde_json::{json, Value}; use uuid::Uuid; @@ -82,6 +87,28 @@ impl RegistryAudit { Ok(Self::new(profile, writer)) } + /// Append a `request` entry and return the handle that owes its + /// `response`. Dropped unanswered, the handle writes `unfinished` as the + /// response under the entry's schema and correlation. + pub(crate) async fn begin( + &self, + entry: AuditEntry, + unfinished: Value, + ) -> Result { + if entry.phase() != registry_platform_audit::AuditPhase::Request { + return Err(RegistryAuditError::InvalidContext); + } + self.writer + .begin( + entry.schema(), + entry.correlation(), + entry.record().clone(), + unfinished, + ) + .await + .map_err(|_| RegistryAuditError::Unavailable) + } + /// Append one entry. A refused append is the audit-unavailable refusal: /// the caller performs no protected I/O and releases no disclosure. pub async fn append(&self, entry: AuditEntry) -> Result<(), RegistryAuditError> { @@ -258,6 +285,8 @@ pub(crate) enum WebhookAuditOutcome { PayloadExpired, WorkerInterrupted, ReplayRequested, + ReplayCommitted, + ReplayRefused, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -281,17 +310,50 @@ pub(crate) struct WebhookAudit<'a> { pub disposition: WebhookAuditDisposition, } -/// Append one minimized attempt or refusal before protected record I/O. -/// -/// An attempt is the `request` entry of the request it names and must be -/// accepted before any protected read or write starts. A refusal is the single -/// `response` entry of a request that performs no protected I/O. +/// Append one minimized refusal before protected record I/O: the single +/// `response` entry of a request that performs no protected I/O, or the one +/// that answers the attempt a request holds. An attempt goes through +/// [`begin_pre_io_audit`], which owes its response, so this refuses one. pub async fn record_pre_io_audit( audit: &RegistryAudit, expected: &ExpectedRegistryIdentity, claims: &ClaimContext, event: PreIoAudit<'_>, ) -> Result<(), RegistryAuditError> { + if event.kind != PreIoAuditKind::Refusal { + return Err(RegistryAuditError::InvalidContext); + } + let record = pre_io_record(audit, expected, claims, &event)?; + audit + .append(pre_io_entry(event.kind, event.correlation, record)) + .await +} + +/// Append the attempt `request` entry of a request that goes on to protected +/// I/O, and return the handle that owes its `response`. The terminal or +/// refusal entry the request writes under its request id answers it; a +/// request that returns, fails, or is canceled first writes an `unfinished` +/// response naming only the operation and the request when the handle is +/// dropped, so no attempt is left unpaired. +pub(crate) async fn begin_pre_io_audit( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ClaimContext, + event: PreIoAudit<'_>, +) -> Result { + if event.kind != PreIoAuditKind::Attempt { + return Err(RegistryAuditError::InvalidContext); + } + let record = pre_io_record(audit, expected, claims, &event)?; + begin_attempt(audit, event.correlation, record).await +} + +fn pre_io_record( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ClaimContext, + event: &PreIoAudit<'_>, +) -> Result { let profile = audit.profile(); if event.operation_id.is_empty() || !profile_is_keyed(profile) { return Err(RegistryAuditError::InvalidContext); @@ -341,17 +403,45 @@ pub async fn record_pre_io_audit( } } insert_refusal_reason(&mut record, event.refusal_reason); + Ok(record) +} + +/// [`record_pre_io_audit`] for a governed action request. +pub(crate) async fn record_action_pre_io_audit( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ActionClaimContext, + event: PreIoAudit<'_>, +) -> Result<(), RegistryAuditError> { + if event.kind != PreIoAuditKind::Refusal { + return Err(RegistryAuditError::InvalidContext); + } + let record = action_pre_io_record(audit, expected, claims, &event)?; audit .append(pre_io_entry(event.kind, event.correlation, record)) .await } -pub(crate) async fn record_action_pre_io_audit( +/// [`begin_pre_io_audit`] for a governed action request. +pub(crate) async fn begin_action_pre_io_audit( audit: &RegistryAudit, expected: &ExpectedRegistryIdentity, claims: &ActionClaimContext, event: PreIoAudit<'_>, -) -> Result<(), RegistryAuditError> { +) -> Result { + if event.kind != PreIoAuditKind::Attempt { + return Err(RegistryAuditError::InvalidContext); + } + let record = action_pre_io_record(audit, expected, claims, &event)?; + begin_attempt(audit, event.correlation, record).await +} + +fn action_pre_io_record( + audit: &RegistryAudit, + expected: &ExpectedRegistryIdentity, + claims: &ActionClaimContext, + event: &PreIoAudit<'_>, +) -> Result { let profile = audit.profile(); if event.operation_id.is_empty() || event.target_record.is_some() @@ -384,9 +474,47 @@ pub(crate) async fn record_action_pre_io_audit( "actionId": claims.action_id(), }); insert_refusal_reason(&mut record, event.refusal_reason); + Ok(record) +} + +/// Append `record` as the attempt `request` entry of `correlation`. +async fn begin_attempt( + audit: &RegistryAudit, + correlation: &RequestCorrelation, + record: Value, +) -> Result { + let unfinished = unfinished_record(&record); audit - .append(pre_io_entry(event.kind, event.correlation, record)) + .writer() + .begin( + AUDIT_SCHEMA, + correlation.request_id().to_string(), + record, + unfinished, + ) .await + .map_err(|_| RegistryAuditError::Unavailable) +} + +/// The `response` an attempt writes when its request ends without a +/// terminal or refusal entry: the operation and request it answers, and +/// nothing the request read or was about to write. +fn unfinished_record(attempt: &Value) -> Value { + let mut record = serde_json::Map::new(); + record.insert("phase".to_owned(), json!("unfinished")); + for field in [ + "method", + "operationId", + "requestId", + "traceId", + "packageRevision", + "actionId", + ] { + if let Some(value) = attempt.get(field) { + record.insert(field.to_owned(), value.clone()); + } + } + Value::Object(record) } fn pre_io_phase_name(kind: PreIoAuditKind) -> &'static str { @@ -680,8 +808,13 @@ pub(crate) fn webhook_entry( ) => event.attempt >= 0, ( WebhookAuditPhase::Replay, - WebhookAuditOutcome::ReplayRequested, + WebhookAuditOutcome::ReplayRequested | WebhookAuditOutcome::ReplayCommitted, WebhookAuditDisposition::ReplayPending, + ) + | ( + WebhookAuditPhase::Replay, + WebhookAuditOutcome::ReplayRefused, + WebhookAuditDisposition::DeadLettered, ) => event.attempt == 0, _ => false, }; @@ -723,11 +856,15 @@ pub(crate) fn webhook_entry( "generation": event.generation, "attempt": event.attempt, }); - Ok(match event.phase { - WebhookAuditPhase::Attempt => { + // An attempt's start and an operator's replay request are requests; + // the terminal disposition and the replay's committed or refused reset + // answer them under the same correlation. + Ok(match (event.phase, event.outcome) { + (WebhookAuditPhase::Attempt, _) + | (WebhookAuditPhase::Replay, WebhookAuditOutcome::ReplayRequested) => { AuditEntry::request(WEBHOOK_AUDIT_SCHEMA, correlation, record) } - WebhookAuditPhase::Terminal | WebhookAuditPhase::Replay => { + (WebhookAuditPhase::Terminal | WebhookAuditPhase::Replay, _) => { AuditEntry::response(WEBHOOK_AUDIT_SCHEMA, correlation, record) } }) @@ -761,6 +898,8 @@ fn webhook_outcome_name(outcome: WebhookAuditOutcome) -> &'static str { WebhookAuditOutcome::PayloadExpired => "payload_expired", WebhookAuditOutcome::WorkerInterrupted => "worker_interrupted", WebhookAuditOutcome::ReplayRequested => "replay_requested", + WebhookAuditOutcome::ReplayCommitted => "replay_committed", + WebhookAuditOutcome::ReplayRefused => "replay_refused", } } @@ -939,6 +1078,9 @@ pub mod test_support { /// The envelope schema and record phase of the first entry refused /// regardless of `remaining`. refuse: Option<(String, String)>, + /// Every writer recording here, so a read can wait for the entries + /// they write when a request handle is dropped. + writers: Vec, } /// The entries one [`RegistryAudit`] accepted, in append order. @@ -988,6 +1130,10 @@ pub mod test_support { /// Every accepted entry, parsed. #[must_use] pub fn entries(&self) -> Vec { + let writers = self.0.lock().expect("audit capture lock").writers.clone(); + for writer in &writers { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture lock"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -1023,6 +1169,11 @@ pub mod test_support { #[must_use] pub fn audit(&self, profile: AuditProfile) -> RegistryAudit { let writer = AuditWriter::from_line_sink(Box::new(CaptureSink(Arc::clone(&self.0)))); + self.0 + .lock() + .expect("audit capture lock") + .writers + .push(writer.clone()); RegistryAudit::new(profile, writer) } } diff --git a/crates/registry-breg/src/field_encryption_backfill.rs b/crates/registry-breg/src/field_encryption_backfill.rs index a0ca647f33..73a1663da7 100644 --- a/crates/registry-breg/src/field_encryption_backfill.rs +++ b/crates/registry-breg/src/field_encryption_backfill.rs @@ -34,8 +34,8 @@ use crate::history_erasure::{ RecordHistoryErasureTarget, MAX_ERASURE_REVISIONS, }; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceTimeouts, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceTimeouts, }; use crate::history_rebaseline::{ history_rebaseline_entry, rebaseline_history_coverage_in_transaction, HistoryRebaselineError, @@ -588,8 +588,11 @@ pub async fn erase_field_encryption_history( // copies that still carry plaintext for a recorded erase-and-rebaseline // flip. The transaction also marks coverage incomplete, which is the // existing durable retry signal if this lifecycle stops before rebaseline. + // The lifecycle's request entry, when this run is one, is held until the + // run ends, so a run that stops before its terminal still answers it. + let mut lifecycle_attempt = None; let (_, _, needs_rebaseline, lifecycle_reference) = - scrub_plaintext_request_snapshots(client, &request).await?; + scrub_plaintext_request_snapshots(client, &request, &mut lifecycle_attempt).await?; let mut erased_any = false; loop { let targets = pending_erase_targets(client, &request).await?; @@ -702,6 +705,7 @@ pub async fn erase_field_encryption_history( async fn scrub_plaintext_request_snapshots( client: &mut Client, request: &FieldEncryptionHistoryErasureRequest<'_>, + lifecycle_attempt: &mut Option, ) -> Result<(u64, u64, bool, String), FieldEncryptionHistoryErasureError> { let transaction = client .transaction() @@ -754,11 +758,13 @@ async fn scrub_plaintext_request_snapshots( // A recorded erase-and-rebaseline flip makes this a lifecycle run: its // request entry must be accepted before any request snapshot, record // history, or coverage state is read for erasure or changed. - append_maintenance_entries( - request.audit, - vec![lifecycle_request_entry(request, &lifecycle_reference)?], - ) - .await?; + *lifecycle_attempt = Some( + begin_maintenance_request( + request.audit, + lifecycle_request_entry(request, &lifecycle_reference)?, + ) + .await?, + ); let (terminal_exists, correlated_progress_exists) = lifecycle_progress_state(&transaction, &lifecycle_reference).await?; let unresolved_provenance: bool = transaction diff --git a/crates/registry-breg/src/history_erasure.rs b/crates/registry-breg/src/history_erasure.rs index 859d6d2b73..a2d7698694 100644 --- a/crates/registry-breg/src/history_erasure.rs +++ b/crates/registry-breg/src/history_erasure.rs @@ -23,8 +23,8 @@ use uuid::Uuid; use crate::audit::RegistryAudit; use crate::history_commit::{lock_history_head, HistoryCommitError}; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceError, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceError, }; use crate::idempotency::{tombstone_erased_cached_responses, IdempotencyError}; use crate::postgres::{ @@ -203,13 +203,13 @@ async fn erase_record_history_scoped( // A standalone erasure's request entry is accepted before its // transaction opens, so an audit outage erases nothing. A lifecycle // erasure runs under the request entry its parent lifecycle appended. - if lifecycle_reference.is_none() { - append_maintenance_entries( - request.audit, - vec![history_erasure_request_entry(&request)?], - ) - .await?; - } + let _attempt = match lifecycle_reference { + None => Some( + begin_maintenance_request(request.audit, history_erasure_request_entry(&request)?) + .await?, + ), + Some(_) => None, + }; let transaction = client .transaction() diff --git a/crates/registry-breg/src/history_maintenance.rs b/crates/registry-breg/src/history_maintenance.rs index a9a6d78586..41db26f4a5 100644 --- a/crates/registry-breg/src/history_maintenance.rs +++ b/crates/registry-breg/src/history_maintenance.rs @@ -15,7 +15,7 @@ use std::time::Duration; -use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile}; +use registry_platform_audit::{AuditEntry, AuditKeyHasher, AuditProfile, AuditRequest}; use crate::audit::RegistryAudit; use crate::postgres::{ExpectedRegistryIdentity, PostgresKernelError}; @@ -136,6 +136,26 @@ pub(crate) fn profile_is_keyed(profile: &AuditProfile) -> bool { matches!(profile.key_hasher(), AuditKeyHasher::Keyed(_)) } +/// Append a maintenance `request` entry before its transaction opens and +/// return the handle that owes its response. The terminal entry appended +/// after commit, under the same schema and correlation, answers it; a +/// maintenance run that returns first writes the request's fields with the +/// `unfinished` outcome when the handle is dropped. +pub(crate) async fn begin_maintenance_request( + audit: &RegistryAudit, + entry: AuditEntry, +) -> Result { + let mut unfinished = entry.record().clone(); + if let Some(fields) = unfinished.as_object_mut() { + fields.insert("phase".to_owned(), "terminal".into()); + fields.insert("outcome".to_owned(), "unfinished".into()); + } + audit + .begin(entry, unfinished) + .await + .map_err(|_| HistoryMaintenanceError::Unavailable) +} + /// Append maintenance entries after the transaction that made their change /// durable has committed. A refused entry reports the maintenance path /// unavailable even though its change already committed: the operator sees diff --git a/crates/registry-breg/src/history_rebaseline.rs b/crates/registry-breg/src/history_rebaseline.rs index 616042094e..d74fd404d5 100644 --- a/crates/registry-breg/src/history_rebaseline.rs +++ b/crates/registry-breg/src/history_rebaseline.rs @@ -24,8 +24,8 @@ use crate::history_commit::{ allocate_coverage_baseline_commit, lock_history_head, HistoryCommitError, }; use crate::history_maintenance::{ - append_maintenance_entries, profile_is_keyed, set_local_timeouts, verify_ready_identity, - HistoryMaintenanceError, + append_maintenance_entries, begin_maintenance_request, profile_is_keyed, set_local_timeouts, + verify_ready_identity, HistoryMaintenanceError, }; use crate::history_migration::{verify_live_rows_match_journal_heads, HistoryMigrationError}; use crate::model::CompiledRegistry; @@ -156,12 +156,9 @@ pub async fn rebaseline_history_coverage( // The baseline position is known only once the transaction allocates it, // so one invocation correlates its two entries by a fresh identifier. let correlation = Uuid::new_v4().to_string(); - append_maintenance_entries( + let _attempt = begin_maintenance_request( request.audit, - vec![history_rebaseline_request_entry( - &request, - correlation.clone(), - )?], + history_rebaseline_request_entry(&request, correlation.clone())?, ) .await?; diff --git a/crates/registry-breg/src/ingestion_store.rs b/crates/registry-breg/src/ingestion_store.rs index 3a6c8d2dac..179702ed5e 100644 --- a/crates/registry-breg/src/ingestion_store.rs +++ b/crates/registry-breg/src/ingestion_store.rs @@ -1221,16 +1221,54 @@ pub(crate) struct RunRequest<'a> { pub(crate) correlation: &'a str, } +/// The accepted `request` entry of one run transition, which owes its +/// `response` in the ingestion schema. +pub(crate) struct RunAttempt { + request: registry_platform_audit::AuditRequest, + record: Value, +} + +impl RunAttempt { + /// Whether the transition's `response` entry was accepted. + pub(crate) fn is_answered(&self) -> bool { + self.request.is_answered() + } + + /// Answer the request with the refusal the transition ended in, + /// reporting whether the destination accepted it. + pub(crate) async fn refuse(mut self) -> bool { + let record = outcome_record(&self.record, "refused"); + self.request.respond(record).await.is_ok() + } + + /// Answer the request as unfinished: the transition's commit returned + /// an error, which does not prove it rolled back, so its outcome is + /// unknown. Reports whether the destination accepted the answer. + pub(crate) async fn abandon(mut self) -> bool { + let record = outcome_record(&self.record, "unfinished"); + self.request.respond(record).await.is_ok() + } +} + +fn outcome_record(request: &Value, outcome: &str) -> Value { + let mut record = request.clone(); + record["phase"] = json!("terminal"); + record["outcome"] = json!(outcome); + record +} + /// Append the value-free `request` entry of one run transition, correlated by -/// the request that drives it. Callers append it before the transition's -/// first protected read or write and perform neither unless the append is -/// accepted; the transition's `response` entry, appended after commit, shares -/// the correlation. The entry names only what that response entry already +/// the request that drives it, and return the handle that owes its +/// `response`. Callers append it before the transition's first protected +/// write and perform none unless it is accepted; the transition's `response` +/// entry, appended after commit, shares the correlation and answers it. A +/// transition that ends without one writes the `unfinished` outcome when the +/// handle is dropped. The entry names only what that response entry already /// records. -pub(crate) async fn append_run_request( +pub(crate) async fn begin_run_request( audit: &crate::audit::RegistryAudit, request: RunRequest<'_>, -) -> Result<(), IngestionStoreError> { +) -> Result { if !crate::audit::profile_is_keyed(audit.profile()) { return Err(IngestionStoreError::Unavailable); } @@ -1250,14 +1288,18 @@ pub(crate) async fn append_run_request( if let Some(chunk_index) = request.chunk_index { record["chunkIndex"] = json!(chunk_index); } - audit - .append(AuditEntry::request( - INGESTION_AUDIT_SCHEMA, - request.correlation.to_owned(), - record, - )) + let request = audit + .begin( + AuditEntry::request( + INGESTION_AUDIT_SCHEMA, + request.correlation.to_owned(), + record.clone(), + ), + outcome_record(&record, "unfinished"), + ) .await - .map_err(|_| IngestionStoreError::Unavailable) + .map_err(|_| IngestionStoreError::Unavailable)?; + Ok(RunAttempt { request, record }) } /// The canonical lifecycle record for one run transition. diff --git a/crates/registry-breg/src/migration_reconcile.rs b/crates/registry-breg/src/migration_reconcile.rs index e587f535e0..80ffb3a432 100644 --- a/crates/registry-breg/src/migration_reconcile.rs +++ b/crates/registry-breg/src/migration_reconcile.rs @@ -25,8 +25,8 @@ use std::time::Duration; -use registry_platform_audit::AuditEntry; -use serde_json::json; +use registry_platform_audit::{AuditEntry, AuditRequest}; +use serde_json::{json, Value}; use crate::audit::RegistryAudit; use crate::history_maintenance::profile_is_keyed; @@ -344,12 +344,8 @@ async fn reconcile_under_lock( match report.outcome { ReconcileOutcome::Completable => { let entry = audit_entry(request, target, ledger, "completed", &report)?; - append_request( - request.audit, - request_entry(request, target, ledger, "completed")?, - ) - .await?; - connection + let mut attempt = begin_request(request, target, ledger, "completed").await?; + let transition = connection .activate_verified_package( Some(request.current), target, @@ -360,17 +356,22 @@ async fn reconcile_under_lock( runtime_role: request.runtime_role, }, ) - .await?; + .await; + if let Err(error) = transition { + let landed = + transition_landed(connection, |snapshot| snapshot.identity == *target).await; + if landed != Some(true) { + respond_unlanded(&mut attempt, request, target, ledger, "completed", landed) + .await; + return Err(error.into()); + } + } append_after_commit(request.audit, entry).await?; } ReconcileOutcome::Revertible => { let entry = audit_entry(request, target, ledger, "reverted", &report)?; - append_request( - request.audit, - request_entry(request, target, ledger, "reverted")?, - ) - .await?; - connection + let mut attempt = begin_request(request, target, ledger, "reverted").await?; + let transition = connection .revert_failed_package( request.current, &target.package_revision, @@ -381,7 +382,17 @@ async fn reconcile_under_lock( runtime_role: request.runtime_role, }, ) - .await?; + .await; + if let Err(error) = transition { + let landed = + transition_landed(connection, |snapshot| snapshot.identity == *request.current) + .await; + if landed != Some(true) { + respond_unlanded(&mut attempt, request, target, ledger, "reverted", landed) + .await; + return Err(error.into()); + } + } append_after_commit(request.audit, entry).await?; } outcome @ (ReconcileOutcome::Ready @@ -427,16 +438,88 @@ fn unresolvable_reason(progress: Option) -> &'static } } -/// Append the reconciliation's request entry before its transition runs. A -/// refused entry reports the reconciliation unavailable and leaves the pinned -/// target exactly as the assessment found it. -async fn append_request(audit: &RegistryAudit, entry: AuditEntry) -> Result<(), ReconcileError> { - audit - .append(entry) +/// Append the reconciliation's request entry before its transition runs and +/// return the handle that owes its response. A refused entry reports the +/// reconciliation unavailable and leaves the pinned target exactly as the +/// assessment found it. A reconciliation that ends without a response +/// writes the `unfinished` outcome when the handle is dropped. +async fn begin_request( + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, +) -> Result { + request + .audit + .begin( + request_entry(request, target, ledger, action)?, + outcome_record(request, target, ledger, action, "unfinished")?, + ) .await .map_err(|_| ReconcileError::Unavailable) } +/// Whether a transition that returned an error nonetheless landed: the +/// maintenance target is cleared and the active identity is the one +/// `landed` expects. An error does not prove the transaction rolled back, so +/// the durable state decides; `None` when it cannot be read. +async fn transition_landed( + connection: &mut VerifiedPackageApplyConnection, + landed: impl FnOnce(&MaintenanceSnapshot) -> bool, +) -> Option { + let snapshot = connection.maintenance_snapshot().await.ok()?; + Some(snapshot.maintenance_target_revision.is_none() && landed(&snapshot)) +} + +/// Answer the request entry of a transition that did not land: `failed` +/// when the durable state shows it did not, `unfinished` when that state +/// could not be read. The reconciliation already failed, so a refused entry +/// is only logged; the held request then writes its `unfinished` outcome +/// instead. +async fn respond_unlanded( + attempt: &mut AuditRequest, + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, + landed: Option, +) { + let outcome = if landed == Some(false) { + "failed" + } else { + "unfinished" + }; + let recorded = match outcome_record(request, target, ledger, action, outcome) { + Ok(record) => attempt.respond(record).await.is_ok(), + Err(_) => false, + }; + if !recorded { + tracing::error!("the failed reconciliation's response audit entry was not recorded"); + } +} + +/// The `response` of a transition that did not commit: the request's +/// identities and plan shape with the outcome, and no count or finding. +fn outcome_record( + request: &ReconcileRequest<'_>, + target: &ExpectedRegistryIdentity, + ledger: &MigrationLedgerEntry, + action: &'static str, + outcome: &'static str, +) -> Result { + Ok(json!({ + "phase": "terminal", + "outcome": outcome, + "operationId": AUDIT_OPERATION_ID, + "action": action, + "packageRevision": request.current.package_revision, + "targetPackageRevision": target.package_revision, + "packageSequence": target.package_sequence, + "planKind": ledger.plan_kind.as_str(), + "operatorReference": operator_reference(request)?, + })) +} + /// Append the reconciliation's response entry after its transition committed. /// A refused entry reports the reconciliation unavailable even though the /// transition is durable; a rerun then finds the Registry ready. diff --git a/crates/registry-breg/src/mutation.rs b/crates/registry-breg/src/mutation.rs index bf9b9cdc72..2e28e9db4d 100644 --- a/crates/registry-breg/src/mutation.rs +++ b/crates/registry-breg/src/mutation.rs @@ -14,7 +14,7 @@ use std::sync::Arc; use std::time::Duration; use deadpool_postgres::Client; -use registry_platform_audit::{AuditEntry, AuditProfile}; +use registry_platform_audit::{AuditEntry, AuditProfile, AuditRequest}; use registry_platform_canonical_json::canonicalize_json; use registry_platform_crypto::field_encryption::{ envelope_member_json, FieldCryptoError, MAX_FIELD_PLAINTEXT_BYTES, @@ -29,9 +29,9 @@ use uuid::Uuid; use crate::artifacts::event_data_schema_binding; use crate::audit::{ - action_terminal_entry, profile_is_keyed, record_action_pre_io_audit, record_pre_io_audit, - terminal_entry, PreIoAudit, PreIoAuditKind, RegistryAudit, RegistryAuditError, TerminalAudit, - TerminalAuditOutcome, + action_terminal_entry, begin_action_pre_io_audit, begin_pre_io_audit, profile_is_keyed, + record_action_pre_io_audit, record_pre_io_audit, terminal_entry, PreIoAudit, PreIoAuditKind, + RegistryAudit, RegistryAuditError, TerminalAudit, TerminalAuditOutcome, }; use crate::compiler::{ WEBHOOK_ATTEMPT_TIMEOUT_MS, WEBHOOK_BACKOFF_MULTIPLIER, WEBHOOK_INITIAL_BACKOFF_MS, @@ -1134,8 +1134,7 @@ impl MutationCoordinator { .await?; return Err(error); } - self.record_boundary_audit(&request, PreIoAuditKind::Attempt) - .await?; + let _attempt = self.begin_boundary_audit(&request).await?; if let Err(error) = self.stage_attachment(client, &request).await { self.record_boundary_audit(&request, PreIoAuditKind::Refusal) .await?; @@ -1187,8 +1186,7 @@ impl MutationCoordinator { .await?; return Err(error); } - self.record_batch_boundary_audit(&request, PreIoAuditKind::Attempt) - .await?; + let _attempt = self.begin_batch_boundary_audit(&request).await?; let result = self .execute_batch_after_attempt(client, &request, fault) .await; @@ -1202,6 +1200,50 @@ impl MutationCoordinator { }) } + /// Append the batch's attempt and hold it until its terminal or refusal + /// entry answers it. + async fn begin_batch_boundary_audit( + &self, + request: &BatchMutationRequest<'_>, + ) -> Result { + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + request.claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: request.plan.route.method, + operation_id: &request.plan.route.id, + target_record: None, + refusal_reason: None, + correlation: &request.correlation, + }, + ) + .await?) + } + + /// Append the mutation's attempt and hold it until its terminal or + /// refusal entry answers it. + async fn begin_boundary_audit( + &self, + request: &MutationRequest<'_>, + ) -> Result { + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + request.claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: request.plan.route.method, + operation_id: &request.plan.route.id, + target_record: request.record_id, + refusal_reason: None, + correlation: &request.correlation, + }, + ) + .await?) + } + async fn record_batch_boundary_audit( &self, request: &BatchMutationRequest<'_>, diff --git a/crates/registry-breg/src/mutation/action.rs b/crates/registry-breg/src/mutation/action.rs index 1108ce2e5a..1495b8f7eb 100644 --- a/crates/registry-breg/src/mutation/action.rs +++ b/crates/registry-breg/src/mutation/action.rs @@ -173,13 +173,9 @@ impl MutationCoordinator { validate_action_claims(action, claims, Operation::Invoke)?; let normalized_input = validate_action_input(action, input.input)?; validate_precondition_set(action, &input.preconditions)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + let _attempt = self + .begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?; let request_digest = canonical_action_request_digest(action, &normalized_input, &input.preconditions)?; let binding = resolve_action_binding( @@ -712,14 +708,10 @@ impl MutationCoordinator { }; let application_id = Uuid::new_v4(); let correlation = RequestCorrelation::breg_created(); - self.record_action_boundary_audit( - &claims, - &route_id, - &correlation, - PreIoAuditKind::Attempt, - ) - .await - .map_err(|_| UncertainApply)?; + let _attempt = self + .begin_action_boundary_audit(&claims, &route_id, &correlation) + .await + .map_err(|_| UncertainApply)?; let deadline = tokio::time::Instant::now() + HOOK_PROPOSAL_APPLY_BUDGET; let fault = FaultControl::Disabled; @@ -840,13 +832,9 @@ impl MutationCoordinator { )?; validate_action_claims(action, claims, Operation::Invoke)?; let refs = validate_condition_inputs(action, input.input)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + let _attempt = self + .begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?; let result = self .action_target_conditions_after_attempt( client, @@ -871,6 +859,30 @@ impl MutationCoordinator { result } + /// Append the action's attempt and hold it until its terminal or refusal + /// entry answers it. + pub(crate) async fn begin_action_boundary_audit( + &self, + claims: &ActionClaimContext, + operation_id: &str, + correlation: &RequestCorrelation, + ) -> Result { + Ok(begin_action_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: HttpMethod::Post, + operation_id, + target_record: None, + refusal_reason: None, + correlation, + }, + ) + .await?) + } + pub(crate) async fn record_action_boundary_audit( &self, claims: &ActionClaimContext, @@ -2683,9 +2695,16 @@ pub(crate) struct PreparedEvidenceAction { binding: crate::idempotency::ResolvedIdempotencyBinding, reserved_creates: BTreeMap, application_id: Uuid, + /// The action's attempt, held across Evidence evaluation until the + /// finalize terminal or refusal answers it. Dropping the admission + /// before that answers the attempt as unfinished. + attempt: Option, } impl MutationCoordinator { + /// Admit an Evidence action, recording its refusal when admission fails + /// before the deadline. A successful admission carries the held attempt + /// to [`Self::finalize_evidence_action`]; a recovered receipt answers it. #[allow(clippy::too_many_arguments)] pub(crate) async fn preflight_evidence_action( &self, @@ -2695,6 +2714,51 @@ impl MutationCoordinator { claims: &ActionClaimContext, target_authority: &BTreeMap>, deadline: tokio::time::Instant, + ) -> Result, MutationError> { + let route_id = input.route_id; + let correlation = input.correlation; + let mut attempt = None; + let result = self + .preflight_evidence_action_holding( + client, + registry, + input, + claims, + target_authority, + deadline, + &mut attempt, + ) + .await; + if result.is_err() && tokio::time::Instant::now() < deadline { + self.record_action_boundary_audit( + claims, + route_id, + correlation, + PreIoAuditKind::Refusal, + ) + .await?; + } + Ok(match result? { + Ok(mut prepared) => { + prepared.attempt = attempt; + Ok(prepared) + } + Err(receipt) => Err(receipt), + }) + } + + /// Admission itself. The attempt it begins is left in `attempt`, so the + /// caller answers it on every path. + #[allow(clippy::too_many_arguments)] + async fn preflight_evidence_action_holding( + &self, + client: &mut Client, + registry: &CompiledRegistry, + input: ImmediateActionInput<'_>, + claims: &ActionClaimContext, + target_authority: &BTreeMap>, + deadline: tokio::time::Instant, + attempt: &mut Option, ) -> Result, MutationError> { if !profile_is_keyed(self.audit.profile()) { return Err(MutationError::Unavailable); @@ -2708,13 +2772,10 @@ impl MutationCoordinator { validate_action_claims(action, claims, Operation::Invoke)?; let normalized = validate_action_input(action, input.input)?; validate_precondition_set(action, &input.preconditions)?; - self.record_action_boundary_audit( - claims, - input.route_id, - input.correlation, - PreIoAuditKind::Attempt, - ) - .await?; + *attempt = Some( + self.begin_action_boundary_audit(claims, input.route_id, input.correlation) + .await?, + ); let binding = resolve_action_binding( self.audit.profile(), &ActionIdempotencyBinding { @@ -2811,6 +2872,7 @@ impl MutationCoordinator { binding, reserved_creates: reserve_action_create_ids(action)?, application_id, + attempt: None, })) } diff --git a/crates/registry-breg/src/mutation/request.rs b/crates/registry-breg/src/mutation/request.rs index 9110e6d72b..73b9c8bdc9 100644 --- a/crates/registry-breg/src/mutation/request.rs +++ b/crates/registry-breg/src/mutation/request.rs @@ -285,6 +285,38 @@ impl MutationCoordinator { }) } + /// Append the attempt of one request action and return the handle that + /// owes its response. An orchestrating caller that runs protected reads + /// before the action itself, such as a reviewed apply's receipt + /// preflight, holds it across all of them. + pub(crate) async fn begin_request_action_audit( + &self, + registry: &CompiledRegistry, + input: &RequestActionInput<'_>, + claims: &ClaimContext, + ) -> Result { + let route = registry + .routes() + .routes + .iter() + .find(|route| route.id == input.route_id) + .ok_or(MutationError::InvalidRequest)?; + Ok(begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: route.method, + operation_id: &route.id, + target_record: Some(input.record_id), + refusal_reason: None, + correlation: input.correlation, + }, + ) + .await?) + } + pub(crate) async fn record_request_boundary_refusal( &self, registry: &CompiledRegistry, @@ -321,6 +353,7 @@ impl MutationCoordinator { input: &RequestActionInput<'_>, claims: &ClaimContext, deadline: tokio::time::Instant, + attempt: &mut Option, ) -> Result { let route = registry .routes() @@ -365,20 +398,26 @@ impl MutationCoordinator { { return Err(MutationError::InvalidRequest); } - record_pre_io_audit( - &self.audit, - &self.expected, - claims, - PreIoAudit { - kind: PreIoAuditKind::Attempt, - method: route.method, - operation_id: &route.id, - target_record: Some(input.record_id), - refusal_reason: None, - correlation: input.correlation, - }, - ) - .await?; + // The caller holds the attempt across the remote Evidence I/O and the + // action transaction that follow this preflight. + if attempt.is_none() { + *attempt = Some( + begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + PreIoAudit { + kind: PreIoAuditKind::Attempt, + method: route.method, + operation_id: &route.id, + target_record: Some(input.record_id), + refusal_reason: None, + correlation: input.correlation, + }, + ) + .await?, + ); + } let binding = resolve_binding( self.audit.profile(), &IdempotencyBinding { @@ -668,15 +707,20 @@ impl MutationCoordinator { refusal_reason, correlation: input.correlation, }; - if !attempt_recorded { - record_pre_io_audit( - &self.audit, - &self.expected, - claims, - audit(PreIoAuditKind::Attempt, None), + // A caller that recorded the attempt holds it until this returns. + let _attempt = if attempt_recorded { + None + } else { + Some( + begin_pre_io_audit( + &self.audit, + &self.expected, + claims, + audit(PreIoAuditKind::Attempt, None), + ) + .await?, ) - .await?; - } + }; let deadline = tokio::time::Instant::now() + REQUEST_ACTION_TIMEOUT; // Capture the exact intake under request RLS, close that transaction, // then run the bounded planner exactly once outside retry and target diff --git a/crates/registry-breg/src/postgres/history_read.rs b/crates/registry-breg/src/postgres/history_read.rs index 8d48dfd246..2d9f33181d 100644 --- a/crates/registry-breg/src/postgres/history_read.rs +++ b/crates/registry-breg/src/postgres/history_read.rs @@ -20,8 +20,8 @@ use crate::api::{ SnapshotReadRequest, SnapshotReadService, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation}; use crate::cursor::{ @@ -154,7 +154,7 @@ impl PostgresSnapshotReadService { } }; - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -174,25 +174,18 @@ impl PostgresSnapshotReadService { let materialized = match materialized { Ok(materialized) => materialized, Err(error) => { - let _ = self - .record_terminal( - &claims, - &request, - &plan, - None, - TerminalAuditOutcome::Refused, - 0, - ) - .await; - return Err(error); + return Err(self.refused(&claims, &request, &plan, error).await); } }; - let held = SnapshotReadResult::from_materialized( + let held = match SnapshotReadResult::from_materialized( &self.registry, &plan.entity, request.plan.cursor_binding.representation, materialized, - )?; + ) { + Ok(held) => held, + Err(error) => return Err(self.refused(&claims, &request, &plan, error).await), + }; self.fault .fail_at(SnapshotReadFaultPoint::BeforeTerminalAudit)?; let outcome = if held.result_count == 0 { @@ -213,6 +206,34 @@ impl PostgresSnapshotReadService { Ok(held) } + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused( + &self, + claims: &ClaimContext, + request: &SnapshotReadRequest, + plan: &SnapshotReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + if self + .record_terminal( + claims, + request, + plan, + None, + TerminalAuditOutcome::Refused, + 0, + ) + .await + .is_err() + { + tracing::error!("the refused history read's terminal audit entry was not recorded"); + } + error + } + async fn read_rows( &self, client: &mut deadpool_postgres::Client, diff --git a/crates/registry-breg/src/postgres/mutation.rs b/crates/registry-breg/src/postgres/mutation.rs index 31a36a69f2..619c7c7902 100644 --- a/crates/registry-breg/src/postgres/mutation.rs +++ b/crates/registry-breg/src/postgres/mutation.rs @@ -87,6 +87,36 @@ pub struct IngestionRunListQuery { pub limit: i64, } +/// A refused ingestion-run call. `answered` is true when the call's refusal +/// is already on record as the `response` of the ingestion `request` entry it +/// wrote, so the caller must not record it again in another schema. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub struct IngestionRefusal { + pub error: IngestionServiceError, + pub answered: bool, +} + +/// What one ingestion-run call has recorded so far: the ingestion `request` +/// entry it wrote, or whether it handed the chunk to the batch mutation, +/// which records its own attempt and refusal. +#[derive(Default)] +struct IngestionAudit { + run: Option, + batch: bool, + /// The transition's commit returned an error, so whether it committed + /// is unknown and the request is answered unfinished, never refused. + commit_unknown: bool, +} + +impl IngestionAudit { + /// Mark the transition's commit outcome unknown and refuse the call as + /// an outage. + fn commit_failed(&mut self) -> IngestionServiceError { + self.commit_unknown = true; + IngestionServiceError::Unavailable + } +} + /// The closed refusal vocabulary of the ingestion-run service. It is bounded /// and value-free: no chunk bytes, row values, or bearer material appear. #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -267,16 +297,6 @@ impl PostgresRecordMutationService { deadline, ) .await; - if result.is_err() && tokio::time::Instant::now() < deadline { - self.coordinator - .record_action_boundary_audit( - claims, - route_id, - correlation, - crate::audit::PreIoAuditKind::Refusal, - ) - .await?; - } guard.disarm(); match result? { Ok(prepared) => prepared, @@ -385,7 +405,9 @@ impl PostgresRecordMutationService { .await; } if is_evidence_apply { - return self.request_evidence_apply(input, &claims, None).await; + return self + .request_evidence_apply(input, &claims, None, None) + .await; } let client = self .pool @@ -429,7 +451,12 @@ impl PostgresRecordMutationService { input: crate::api::RequestActionInput<'_>, claims: &ClaimContext, review_evidence: Option<&crate::review_integration::AcceptedReviewEvidence>, + attempt: Option, ) -> Result { + // The request's attempt, held until the action answers it: the + // reviewed apply that routes here has already recorded it, and the + // preflight records it otherwise. + let mut attempt = attempt; let evaluator = self .evidence_evaluator .as_ref() @@ -456,6 +483,7 @@ impl PostgresRecordMutationService { &input, claims, deadline, + &mut attempt, ) .await; if result.is_err() && tokio::time::Instant::now() < deadline { @@ -543,6 +571,15 @@ impl PostgresRecordMutationService { MutationFaultControl::At(point) => crate::mutation::FaultControl::At(point), _ => crate::mutation::FaultControl::Disabled, }; + // The attempt precedes the receipt preflight's reads, the review + // authority, and the action transaction, and is held across them. + let attempt = tokio::time::timeout_at( + deadline, + self.coordinator + .begin_request_action_audit(&self.registry, &input, claims), + ) + .await + .map_err(|_| MutationError::Unavailable)??; let receipt_preflight = { let client = self .pool @@ -561,9 +598,23 @@ impl PostgresRecordMutationService { ) .await; match result { - Ok(result) => { + Ok(Ok(preflight)) => { guard.disarm(); - result? + preflight + } + Ok(Err(error)) => { + guard.disarm(); + tokio::time::timeout_at( + deadline, + self.coordinator.record_request_boundary_refusal( + &self.registry, + &input, + claims, + ), + ) + .await + .map_err(|_| MutationError::Unavailable)??; + return Err(error); } Err(_) => { guard.cancel_and_discard().await; @@ -592,7 +643,7 @@ impl PostgresRecordMutationService { fault, None, None, - false, + true, ), ) .await; @@ -648,8 +699,8 @@ impl PostgresRecordMutationService { } .await; if let Err(error) = task_authority { - // The refusal happens before any journaled attempt, so it is - // recorded here, the same as a refused evidence-apply preflight. + // The refusal answers the held attempt, the same as a refused + // evidence-apply preflight. tokio::time::timeout_at( deadline, self.coordinator @@ -666,7 +717,7 @@ impl PostgresRecordMutationService { let review_evidence = source.approved_evidence(&authority, &accepted).await?; if needs_action_evidence { return self - .request_evidence_apply(input, claims, Some(&review_evidence)) + .request_evidence_apply(input, claims, Some(&review_evidence), Some(attempt)) .await; } let client = self @@ -685,7 +736,7 @@ impl PostgresRecordMutationService { fault, Some(&review_evidence), None, - false, + true, ), ) .await; @@ -1024,14 +1075,102 @@ impl PostgresRecordMutationService { run.response_json(active.0, active.1) } + /// Open one ingestion run. + pub async fn create_ingestion_run( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + input: IngestionRunCreateInput, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .create_ingestion_run_in(context, correlation, input, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Cancel one open ingestion run. + pub async fn cancel_ingestion_run( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + entity_id: &str, + run_id: Uuid, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .cancel_ingestion_run_in(context, correlation, entity_id, run_id, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Submit one chunk of an open ingestion run. + pub async fn submit_ingestion_chunk( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + input: IngestionChunkSubmitInput, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .submit_ingestion_chunk_in(context, correlation, input, &mut attempt) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Recover the retained receipt of one committed chunk. + pub async fn ingestion_chunk_receipt( + &self, + context: &AuthorizedRequestContext, + correlation: &RequestCorrelation, + entity_id: &str, + run_id: Uuid, + chunk_index: i64, + ) -> Result { + let mut attempt = IngestionAudit::default(); + let result = self + .ingestion_chunk_receipt_in( + context, + correlation, + entity_id, + run_id, + chunk_index, + &mut attempt, + ) + .await; + Self::settle_ingestion(result, attempt).await + } + + /// Answer the ingestion `request` entry a refused call wrote, in the + /// ingestion schema, so the refusal is never recorded only in another + /// schema. A call refused before it wrote one leaves the refusal to its + /// caller. + async fn settle_ingestion( + result: Result, + audit: IngestionAudit, + ) -> Result { + let error = match result { + Ok(value) => return Ok(value), + Err(error) => error, + }; + let answered = match audit.run { + Some(attempt) if attempt.is_answered() => true, + Some(attempt) if audit.commit_unknown => attempt.abandon().await, + Some(attempt) => attempt.refuse().await, + None => audit.batch, + }; + Err(IngestionRefusal { error, answered }) + } + /// Create a durable ingestion run bound to the active package revision, /// schema fingerprint, entity, profile, operation, input digest, chunking /// algorithm, and announced counts. - pub async fn create_ingestion_run( + async fn create_ingestion_run_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, input: IngestionRunCreateInput, + attempt: &mut IngestionAudit, ) -> Result { if !crate::audit::profile_is_keyed(self.audit.profile()) { return Err(IngestionServiceError::Unavailable); @@ -1107,21 +1246,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the run is opened: an audit // outage refuses the creation instead of opening a run nobody // recorded asking for. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "create", - run_id: None, - chunk_index: None, - package_revision: &self.expected.package_revision, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &run.created_principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "create", + run_id: None, + chunk_index: None, + package_revision: &self.expected.package_revision, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &run.created_principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut client = self.client().await?; // The run binding must name the package the database still holds // active, so creation takes the same guarded transaction ordinary @@ -1150,7 +1291,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; // The run exists once the transaction commits; its answer leaves only // after the audit entry is accepted. ingestion_store::append_run_audit(&self.audit, audit_record) @@ -1316,12 +1457,13 @@ impl PostgresRecordMutationService { /// Cancel an open or blocked run, preserving the committed prefix, the /// counts, and the audit trail. - pub async fn cancel_ingestion_run( + async fn cancel_ingestion_run_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, entity_id: &str, run_id: Uuid, + attempt: &mut IngestionAudit, ) -> Result { if !crate::audit::profile_is_keyed(self.audit.profile()) { return Err(IngestionServiceError::Unavailable); @@ -1334,21 +1476,23 @@ impl PostgresRecordMutationService { let request_correlation = correlation.request_id().to_string(); // The request entry is accepted before the run is read or closed: an // audit outage leaves the run open and resumable. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "cancel", - run_id: Some(run_id), - chunk_index: None, - package_revision: &self.expected.package_revision, - entity_id, - profile_id: claims.access_profile(), - principal_reference: &self.ingestion_principal_reference(principal)?, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "cancel", + run_id: Some(run_id), + chunk_index: None, + package_revision: &self.expected.package_revision, + entity_id, + profile_id: claims.access_profile(), + principal_reference: &self.ingestion_principal_reference(principal)?, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut client = self.client().await?; let run = self .visible_run(&**client, context, entity_id, run_id) @@ -1392,7 +1536,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, audit_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; @@ -1411,11 +1555,12 @@ impl PostgresRecordMutationService { /// Submit the next exact chunk of one run. The server derives the /// idempotency key from the run binding, so an interrupted submission /// replays the original receipt without a duplicate mutation. - pub async fn submit_ingestion_chunk( + async fn submit_ingestion_chunk_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, input: IngestionChunkSubmitInput, + attempt: &mut IngestionAudit, ) -> Result { let claims = strict_claim_context(&self.registry, context, &input.entity_id) .map_err(|_| IngestionServiceError::RequestInvalid)?; @@ -1505,21 +1650,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the release transaction // opens: an audit outage moves no attempt marker and releases // nothing. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "submitChunk", - run_id: Some(run.run_id), - chunk_index: Some(input.chunk_index), - package_revision: &self.expected.package_revision, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "submitChunk", + run_id: Some(run.run_id), + chunk_index: Some(input.chunk_index), + package_revision: &self.expected.package_revision, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut disclosure_writer = self.client().await?; let disclosure_transaction = begin_record_transaction( &mut disclosure_writer, @@ -1611,29 +1758,33 @@ impl PostgresRecordMutationService { &principal_reference, Some(&request_correlation), ); - disclosure_transaction - .commit() - .await - .map_err(|_| IngestionServiceError::Unavailable)?; - ingestion_store::append_run_audit(&self.audit, disclosure_record) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; // The attempt row above moved the run's last-attempt marker, so // the answer describes the run as it now stands, not as this - // request found it. The guarded transaction proved the durable - // binding equals this process's identity, so the replayed run - // renders under it. - let run = ingestion_store::load_run(&**client, run.run_id) + // request found it. It is read inside the release transaction, + // which sees that row, and the answer is built before the + // disclosure entry: once that entry is accepted, nothing fallible + // stands between it and the caller. The guarded transaction + // proved the durable binding equals this process's identity, so + // the replayed run renders under it. + let current = ingestion_store::load_run(tx, run.run_id) .await .map_err(|_| IngestionServiceError::Unavailable)? .ok_or(IngestionServiceError::Unavailable)?; - return Ok(json!({ + let answer = json!({ "run": Self::run_response( - &run, + ¤t, (&self.expected.package_revision, &self.expected.schema_fingerprint), ), "receipt": receipt_json(input.chunk_index, &input.digest, true, false, batch), - })); + }); + disclosure_transaction + .commit() + .await + .map_err(|_| attempt.commit_failed())?; + ingestion_store::append_run_audit(&self.audit, disclosure_record) + .await + .map_err(|_| IngestionServiceError::Unavailable)?; + return Ok(answer); } // A terminal run stays terminal when the active package later // changes: the blocking transition belongs to open runs alone, so a @@ -1650,21 +1801,23 @@ impl PostgresRecordMutationService { let request_correlation = correlation.request_id().to_string(); // The request entry is accepted before the blocking transition // opens: an audit outage leaves the run open. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "submitChunk", - run_id: Some(run.run_id), - chunk_index: Some(input.chunk_index), - package_revision: &active.0, - entity_id: &run.entity_id, - profile_id: &run.profile_id, - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "submitChunk", + run_id: Some(run.run_id), + chunk_index: Some(input.chunk_index), + package_revision: &active.0, + entity_id: &run.entity_id, + profile_id: &run.profile_id, + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let mut writer = self.client().await?; let transaction = writer .transaction() @@ -1723,7 +1876,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, blocked_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; @@ -1789,6 +1942,9 @@ impl PostgresRecordMutationService { ingestion: Some(&chunk_binding), }; let mut writer = self.client().await?; + // From here the batch mutation records the chunk's attempt and its + // refusal under this request's correlation. + attempt.batch = true; #[cfg(feature = "postgres-test")] if let MutationFaultControl::At(fault) = self.fault { return self @@ -2010,13 +2166,14 @@ impl PostgresRecordMutationService { /// Recover the stored receipt of one committed chunk after a lost /// response. The receipt is erased with the record history it describes. - pub async fn ingestion_chunk_receipt( + async fn ingestion_chunk_receipt_in( &self, context: &AuthorizedRequestContext, correlation: &RequestCorrelation, entity_id: &str, run_id: Uuid, chunk_index: i64, + attempt: &mut IngestionAudit, ) -> Result { if chunk_index < 0 { return Err(IngestionServiceError::RequestInvalid); @@ -2034,21 +2191,23 @@ impl PostgresRecordMutationService { // The request entry is accepted before the run or the stored chunk is // read: an audit outage refuses the recovery instead of releasing a // receipt nobody recorded asking for. - ingestion_store::append_run_request( - &self.audit, - ingestion_store::RunRequest { - transition: "chunkReceipt", - run_id: Some(run_id), - chunk_index: Some(chunk_index), - package_revision: &self.expected.package_revision, - entity_id, - profile_id: claims.access_profile(), - principal_reference: &principal_reference, - correlation: &request_correlation, - }, - ) - .await - .map_err(|_| IngestionServiceError::Unavailable)?; + attempt.run = Some( + ingestion_store::begin_run_request( + &self.audit, + ingestion_store::RunRequest { + transition: "chunkReceipt", + run_id: Some(run_id), + chunk_index: Some(chunk_index), + package_revision: &self.expected.package_revision, + entity_id, + profile_id: claims.access_profile(), + principal_reference: &principal_reference, + correlation: &request_correlation, + }, + ) + .await + .map_err(|_| IngestionServiceError::Unavailable)?, + ); let client = self.client().await?; let run = self .visible_run(&**client, context, entity_id, run_id) @@ -2133,7 +2292,7 @@ impl PostgresRecordMutationService { transaction .commit() .await - .map_err(|_| IngestionServiceError::Unavailable)?; + .map_err(|_| attempt.commit_failed())?; ingestion_store::append_run_audit(&self.audit, disclosure_record) .await .map_err(|_| IngestionServiceError::Unavailable)?; diff --git a/crates/registry-breg/src/postgres/read.rs b/crates/registry-breg/src/postgres/read.rs index cbfc4a3bac..5adb321d86 100644 --- a/crates/registry-breg/src/postgres/read.rs +++ b/crates/registry-breg/src/postgres/read.rs @@ -23,8 +23,8 @@ use crate::api::{ RowBoundaryOperator as ApiRowBoundaryOperator, ServiceFuture, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation}; use crate::cursor::{ @@ -238,7 +238,7 @@ impl PostgresRecordReadService { return Ok(ReadResult::empty_get()); } - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -257,22 +257,7 @@ impl PostgresRecordReadService { let materialized = self.read_rows(&mut client, &request, &claims, &plan).await; let materialized = match materialized { Ok(materialized) => materialized, - Err(error) => { - let _ = self - .record_read_terminal_audit( - &request, - self.terminal( - &request, - &claims, - &plan, - TerminalAuditOutcome::Refused, - 0, - None, - )?, - ) - .await; - return Err(error); - } + Err(error) => return Err(self.refused_read(&request, &claims, &plan, error).await), }; let attachment_verification = materialized.rows.first().and_then(|record| { crate::mutation::attachment_verification_etag_fields(&plan.entity, &record.data) @@ -287,47 +272,21 @@ impl PostgresRecordReadService { .and_then(|result| result.enforce_spatial_response_budget(&request)) { Ok(held) => held, - Err(error) => { - let _ = self - .record_read_terminal_audit( - &request, - self.terminal( - &request, - &claims, - &plan, - TerminalAuditOutcome::Refused, - 0, - None, - )?, - ) - .await; - return Err(error); - } + Err(error) => return Err(self.refused_read(&request, &claims, &plan, error).await), }; if plan.operation == Operation::Get && request.representation != CursorRepresentation::GeoJson && held.response.is_some() { - let response = held.response.take().ok_or(ReadServiceError::Unavailable)?; - let record_id = target_record.ok_or(ReadServiceError::Unavailable)?; - let record_revision = held.record_revision.ok_or(ReadServiceError::Unavailable)?; - let representation = match request.representation { - CursorRepresentation::Json => RecordRepresentation::Json, - CursorRepresentation::JsonLd => RecordRepresentation::JsonLd, - CursorRepresentation::GeoJson => return Err(ReadServiceError::Unavailable), - }; - let etag = strong_record_etag_for_representation( - self.audit.profile(), + if let Err(error) = self.attach_strong_etag( + &mut held, + &request, &claims, - &self.expected.package_revision, - record_id, - record_revision, - &request.selected_fields, - representation, + target_record, attachment_verification.as_ref(), - ) - .map_err(|_| ReadServiceError::Unavailable)?; - held.response = Some(response.with_strong_etag(etag)); + ) { + return Err(self.refused_read(&request, &claims, &plan, error).await); + } } self.fault.fail_at(ReadFaultPoint::BeforeTerminalAudit)?; let outcome = match (plan.operation, held.result_count) { @@ -388,28 +347,27 @@ impl PostgresRecordReadService { let mut request = request; request.operation_id = crate::attachment::operation_id(&request.operation_id, &slot_id, request.method); - record_pre_io_audit( - &self.audit, - &self.expected, - &claims, - PreIoAudit { - kind: if valid { - PreIoAuditKind::Attempt - } else { - PreIoAuditKind::Refusal - }, - method: request.method, - operation_id: &request.operation_id, - target_record: target_record(&request.kind), - refusal_reason: None, - correlation: &request.correlation, + let event = PreIoAudit { + kind: if valid { + PreIoAuditKind::Attempt + } else { + PreIoAuditKind::Refusal }, - ) - .await - .map_err(|_| ReadServiceError::Unavailable)?; + method: request.method, + operation_id: &request.operation_id, + target_record: target_record(&request.kind), + refusal_reason: None, + correlation: &request.correlation, + }; if !valid { + record_pre_io_audit(&self.audit, &self.expected, &claims, event) + .await + .map_err(|_| ReadServiceError::Unavailable)?; return Ok(None); } + let _attempt = begin_pre_io_audit(&self.audit, &self.expected, &claims, event) + .await + .map_err(|_| ReadServiceError::Unavailable)?; let plan = plan.map_err(|_| ReadServiceError::Unavailable)?; let transaction = begin_record_transaction( &mut client, @@ -590,6 +548,70 @@ impl PostgresRecordReadService { /// Append the read's `response` entry. The caller releases the result /// only after this returns `Ok`. + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused_read( + &self, + request: &RecordReadRequest, + claims: &ClaimContext, + plan: &ReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + let recorded = match self.terminal( + request, + claims, + plan, + TerminalAuditOutcome::Refused, + 0, + None, + ) { + Ok(terminal) => self + .record_read_terminal_audit(request, terminal) + .await + .map_err(|_| ()), + Err(_) => Err(()), + }; + if recorded.is_err() { + tracing::error!("the refused read's terminal audit entry was not recorded"); + } + error + } + + /// Bind the strong entity tag of a single-record read to its response. + fn attach_strong_etag( + &self, + held: &mut ReadResult, + request: &RecordReadRequest, + claims: &ClaimContext, + target_record: Option<&str>, + attachment_verification: Option<&Value>, + ) -> Result<(), ReadServiceError> { + self.fault.fail_at(ReadFaultPoint::StrongEtag)?; + let response = held.response.take().ok_or(ReadServiceError::Unavailable)?; + let record_id = target_record.ok_or(ReadServiceError::Unavailable)?; + let record_revision = held.record_revision.ok_or(ReadServiceError::Unavailable)?; + let representation = match request.representation { + CursorRepresentation::Json => RecordRepresentation::Json, + CursorRepresentation::JsonLd => RecordRepresentation::JsonLd, + CursorRepresentation::GeoJson => return Err(ReadServiceError::Unavailable), + }; + let etag = strong_record_etag_for_representation( + self.audit.profile(), + claims, + &self.expected.package_revision, + record_id, + record_revision, + &request.selected_fields, + representation, + attachment_verification, + ) + .map_err(|_| ReadServiceError::Unavailable)?; + held.response = Some(response.with_strong_etag(etag)); + Ok(()) + } + async fn record_read_terminal_audit( &self, request: &RecordReadRequest, @@ -4090,12 +4112,16 @@ mod tests { #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum ReadFaultPoint { BeforeTerminalAudit, + /// The strong entity tag of a single-record read cannot be bound, after + /// the rows were read. + StrongEtag, } #[cfg(not(feature = "postgres-test"))] #[derive(Clone, Copy, Debug, Eq, PartialEq)] enum ReadFaultPoint { BeforeTerminalAudit, + StrongEtag, } #[derive(Clone, Copy)] diff --git a/crates/registry-breg/src/postgres/revision_read.rs b/crates/registry-breg/src/postgres/revision_read.rs index bfd2055cae..4e6b6f2792 100644 --- a/crates/registry-breg/src/postgres/revision_read.rs +++ b/crates/registry-breg/src/postgres/revision_read.rs @@ -19,8 +19,8 @@ use crate::api::{ RevisionReadService, RowBoundaryOperator as ApiRowBoundaryOperator, ServiceFuture, }; use crate::audit::{ - profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, PreIoAuditKind, - ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, + begin_pre_io_audit, profile_is_keyed, read_terminal_entry, record_pre_io_audit, PreIoAudit, + PreIoAuditKind, ReadTerminalAudit, RegistryAudit, TerminalAudit, TerminalAuditOutcome, }; use crate::contract::{FieldTypeSource, Operation, ProvenanceFieldSource}; use crate::cursor::CursorRepresentation; @@ -135,7 +135,7 @@ impl PostgresRevisionReadService { } }; - record_pre_io_audit( + let _attempt = begin_pre_io_audit( &self.audit, &self.expected, &claims, @@ -155,26 +155,19 @@ impl PostgresRevisionReadService { let materialized = match materialized { Ok(materialized) => materialized, Err(error) => { - let _ = self - .record_terminal( - &claims, - &request, - &plan, - TerminalAuditOutcome::Refused, - 0, - &[], - ) - .await; - return Err(error); + return Err(self.refused(&claims, &request, &plan, error).await); } }; - let held = RevisionReadResult::from_rows( + let held = match RevisionReadResult::from_rows( &self.registry, &plan.entity, request.representation, plan.kind, materialized, - )?; + ) { + Ok(held) => held, + Err(error) => return Err(self.refused(&claims, &request, &plan, error).await), + }; self.fault .fail_at(RevisionReadFaultPoint::BeforeTerminalAudit)?; let outcome = if held.result_count == 0 { @@ -195,6 +188,27 @@ impl PostgresRevisionReadService { Ok(held) } + /// Record the Refused terminal of a read that failed after its attempt, + /// then hand back the failure. A terminal the destination refuses is + /// logged: the read already fails, and its held attempt then writes the + /// unfinished response instead. + async fn refused( + &self, + claims: &ClaimContext, + request: &RevisionReadRequest, + plan: &RevisionReadPlan, + error: ReadServiceError, + ) -> ReadServiceError { + if self + .record_terminal(claims, request, plan, TerminalAuditOutcome::Refused, 0, &[]) + .await + .is_err() + { + tracing::error!("the refused revision read's terminal audit entry was not recorded"); + } + error + } + async fn read_rows( &self, client: &mut deadpool_postgres::Client, diff --git a/crates/registry-breg/src/request_retention.rs b/crates/registry-breg/src/request_retention.rs index c6e0fabc03..f5b8fd2cce 100644 --- a/crates/registry-breg/src/request_retention.rs +++ b/crates/registry-breg/src/request_retention.rs @@ -8,6 +8,7 @@ use std::time::Duration; use registry_platform_audit::{AuditEntry, AuditProfile}; use serde::Serialize; +use serde_json::{json, Value}; use tokio_postgres::GenericClient; use uuid::Uuid; @@ -48,6 +49,12 @@ pub enum RequestRetentionError { AttachmentStorageBindingMismatch, #[error("request retention state is unavailable")] Unavailable, + /// The erasure committed, but the audit destination refused the entry + /// recording it. The erased detail is gone; restore the destination, + /// then reconcile the erasure against the database before relying on + /// the audit journal for it. + #[error("the request detail erasure committed but its audit entry was not recorded; restore the audit destination")] + ErasureUnaudited, } pub type Result = std::result::Result; @@ -451,19 +458,116 @@ impl RequestRetentionOperatorService { .await .map_err(|_| RequestRetentionError::Unavailable)?; // The request entry is accepted before the erasure transaction opens, - // so an audit outage erases nothing; the committed response shares - // its correlation. + // so an audit outage erases nothing; the response shares its + // correlation. An erasure that ends without one writes the + // unfinished outcome when the held request is dropped. let correlation = RequestCorrelation::breg_created(); - self.audit - .append(retention_request_entry( - self.audit.profile(), - &self.expected, - scope.clone(), - &correlation, - )?) + let request_entry = retention_request_entry( + self.audit.profile(), + &self.expected, + scope.clone(), + &correlation, + )?; + let unfinished = retention_outcome_record(&request_entry, "unfinished"); + let request_record = request_entry.clone(); + let mut attempt = self + .audit + .begin(request_entry, unfinished) .await .map_err(|_| RequestRetentionError::Unavailable)?; - let transaction = self.begin_verified_transaction(&mut client).await?; + let erased = self + .erase_in_transaction(&mut client, scope.clone(), correlation) + .await; + let (plan, erasure, entry) = match erased { + Ok((plan, erasure, entry, true)) => (plan, erasure, entry), + Ok((plan, erasure, entry, false)) => { + // A commit that returned an error may still have committed, + // so the outcome recorded for this destructive operation is + // the one the database holds, read on a fresh connection. + match self.erasure_committed(&pool, scope.clone()).await { + Some(true) => (plan, erasure, entry), + resolved => { + let outcome = if resolved == Some(false) { + "failed" + } else { + "unfinished" + }; + let answer = retention_outcome_record(&request_record, outcome); + if attempt.respond(answer).await.is_err() { + tracing::error!( + "the unacknowledged erasure's response audit entry was not recorded" + ); + } + return Err(RequestRetentionError::Unavailable); + } + } + } + Err(error) => { + // The erasure failed before its commit, so nothing + // committed: answer the request with the refusal or the + // failure. + let outcome = if error == RequestRetentionError::Unavailable { + "failed" + } else { + "refused" + }; + let answer = retention_outcome_record(&request_record, outcome); + if attempt.respond(answer).await.is_err() { + tracing::error!("the failed erasure's response audit entry was not recorded"); + } + return Err(error); + } + }; + // The committed erasure is recorded before external objects are + // retried, so a slow or failing backend cannot hold its response + // back. The result, not the entry, reports the registry-wide + // external deletions still pending. + attempt + .respond(entry.record().clone()) + .await + .map_err(|_| RequestRetentionError::ErasureUnaudited)?; + let (pending_external_deletions, external_deletion_tombstones) = + self.retry_external_deletions(&mut client).await?; + Ok(RequestRetentionErase { + request_entity_id: scope.request_entity_id.to_owned(), + request_id: scope.request_id.to_string(), + proposal_version: scope.proposal_version, + request_state: plan.current_state, + retention_mode: retention_mode_name(plan.retention_mode), + erasure, + pending_external_deletions, + external_deletion_tombstones, + }) + } + + /// Whether the detail `scope` names is erased, read on a fresh + /// connection after an erasure commit returned an error. `None` when the + /// state cannot be read. + async fn erasure_committed( + &self, + pool: &crate::postgres::RuntimePool, + scope: RequestDetailErasureScope<'_>, + ) -> Option { + let mut client = pool.get().await.ok()?; + let transaction = self.begin_verified_transaction(&mut client).await.ok()?; + let plan = load_erasure_plan(&transaction, &self.registry, scope, false) + .await + .ok()?; + transaction.commit().await.ok()?; + Some(plan.detail_erased) + } + + /// Erase one request's detail in one transaction and build the terminal + /// entry that records it. The flag is false when the commit returned an + /// error, which does not prove the transaction rolled back; every + /// earlier error is returned as one. + async fn erase_in_transaction( + &self, + client: &mut deadpool_postgres::Client, + scope: RequestDetailErasureScope<'_>, + correlation: RequestCorrelation, + ) -> Result<(RequestErasurePlan, RequestDetailErasure, AuditEntry, bool)> { + let transaction = self.begin_verified_transaction(client).await?; let plan = load_erasure_plan(&transaction, &self.registry, scope.clone(), true).await?; let (erasure, current_revision) = erase_request_detail_in_transaction(&transaction, &self.registry, scope.clone(), &plan) @@ -496,26 +600,8 @@ impl RequestRetentionOperatorService { erasure, correlation, )?; - transaction - .commit() - .await - .map_err(|_| RequestRetentionError::Unavailable)?; - self.audit - .append(entry) - .await - .map_err(|_| RequestRetentionError::Unavailable)?; - let (pending_external_deletions, external_deletion_tombstones) = - self.retry_external_deletions(&mut client).await?; - Ok(RequestRetentionErase { - request_entity_id: scope.request_entity_id.to_owned(), - request_id: scope.request_id.to_string(), - proposal_version: scope.proposal_version, - request_state: plan.current_state, - retention_mode: retention_mode_name(plan.retention_mode), - erasure, - pending_external_deletions, - external_deletion_tombstones, - }) + let acknowledged = transaction.commit().await.is_ok(); + Ok((plan, erasure, entry, acknowledged)) } /// Retry orphaned external objects even when every request is active or @@ -535,36 +621,50 @@ impl RequestRetentionOperatorService { // audit state. let transaction = self.begin_verified_transaction(&mut client).await?; transaction.commit().await.map_err(map_retention_error)?; - self.audit - .append(AuditEntry::request( - ATTACHMENT_CLEANUP_AUDIT_SCHEMA, - correlation.clone(), - serde_json::json!({ - "kind":"attachmentCleanup", "phase":"attempt", "outcome":"started", - "packageRevision":self.expected.package_revision, - "actor":"breg:request-retention-operator", "correlation":correlation, - }), - )) + let record = |phase: &str, outcome: &str| { + serde_json::json!({ + "kind":"attachmentCleanup", "phase":phase, "outcome":outcome, + "packageRevision":self.expected.package_revision, + "actor":"breg:request-retention-operator", "correlation":correlation, + }) + }; + // A cleanup that ends before its response writes the unfinished + // outcome when the held request is dropped. + let mut attempt = self + .audit + .begin( + AuditEntry::request( + ATTACHMENT_CLEANUP_AUDIT_SCHEMA, + correlation.clone(), + record("attempt", "started"), + ), + record("terminal", "unfinished"), + ) .await .map_err(|_| RequestRetentionError::Unavailable)?; - let (pending_external_deletions, external_deletion_tombstones) = - self.retry_external_deletions(&mut client).await?; + let (pending_external_deletions, external_deletion_tombstones) = match self + .retry_external_deletions(&mut client) + .await + { + Ok(counts) => counts, + Err(error) => { + if attempt.respond(record("terminal", "failed")).await.is_err() { + tracing::error!("the failed cleanup's response audit entry was not recorded"); + } + return Err(error); + } + }; let result = AttachmentCleanup { pending_external_deletions, external_deletion_tombstones, }; - self.audit - .append(AuditEntry::response( - ATTACHMENT_CLEANUP_AUDIT_SCHEMA, - correlation.clone(), - serde_json::json!({ - "kind":"attachmentCleanup", "phase":"terminal", "outcome":"completed", - "packageRevision":self.expected.package_revision, - "actor":"breg:request-retention-operator", "correlation":correlation, - "pendingExternalDeletions":result.pending_external_deletions, - "externalDeletionTombstones":result.external_deletion_tombstones, - }), - )) + let mut completed = record("terminal", "completed"); + completed["pendingExternalDeletions"] = + serde_json::json!(result.pending_external_deletions); + completed["externalDeletionTombstones"] = + serde_json::json!(result.external_deletion_tombstones); + attempt + .respond(completed) .await .map_err(|_| RequestRetentionError::Unavailable)?; Ok(result) @@ -1466,6 +1566,17 @@ fn retention_request_entry( )) } +/// The `response` of an erasure that did not record its committed terminal: +/// the request's identities with `outcome`, and no count. +fn retention_outcome_record(request: &AuditEntry, outcome: &str) -> Value { + let mut record = request.record().clone(); + if let Some(fields) = record.as_object_mut() { + fields.insert("phase".to_owned(), json!("terminal")); + fields.insert("outcome".to_owned(), json!(outcome)); + } + record +} + fn retention_terminal_entry( profile: &AuditProfile, expected: &ExpectedRegistryIdentity, diff --git a/crates/registry-breg/src/webhook.rs b/crates/registry-breg/src/webhook.rs index ae39b2e7f1..1e5aca2b83 100644 --- a/crates/registry-breg/src/webhook.rs +++ b/crates/registry-breg/src/webhook.rs @@ -403,17 +403,12 @@ impl DeliverySeams for BregDeliverySeams { self.handlers.handler(binding) } - /// The platform worker calls this seam only after the guarded transition - /// the entry reports has already succeeded, immediately before it commits - /// the transaction: a refused append still rolls that transition back, - /// but a failed commit after an accepted append leaves an entry for a - /// transition that did not happen, since this durable append cannot be - /// rolled back with the transaction. - async fn record_audit( - &self, - _transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { + /// The platform worker records an attempt's start before its lease + /// commits, answering it with a worker interruption if that commit + /// fails, and records a terminal disposition, an expiry, and a replay's + /// outcome only after the transition commits, so an entry never stands + /// for a transition that rolled back. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { let entry = webhook_entry( self.audit.profile(), WebhookAudit { @@ -696,6 +691,8 @@ fn audit_outcome(outcome: DeliveryAuditOutcome) -> WebhookAuditOutcome { DeliveryAuditOutcome::PayloadExpired => WebhookAuditOutcome::PayloadExpired, DeliveryAuditOutcome::WorkerInterrupted => WebhookAuditOutcome::WorkerInterrupted, DeliveryAuditOutcome::ReplayRequested => WebhookAuditOutcome::ReplayRequested, + DeliveryAuditOutcome::ReplayCommitted => WebhookAuditOutcome::ReplayCommitted, + DeliveryAuditOutcome::ReplayRefused => WebhookAuditOutcome::ReplayRefused, DeliveryAuditOutcome::HandlerBindingRefused => WebhookAuditOutcome::HandlerBindingRefused, DeliveryAuditOutcome::HandlerDeadline => WebhookAuditOutcome::HandlerDeadline, DeliveryAuditOutcome::HandlerResource => WebhookAuditOutcome::HandlerResource, diff --git a/crates/registry-breg/tests/postgres_action_evidence.rs b/crates/registry-breg/tests/postgres_action_evidence.rs index 164a58ffcf..c67f536654 100644 --- a/crates/registry-breg/tests/postgres_action_evidence.rs +++ b/crates/registry-breg/tests/postgres_action_evidence.rs @@ -162,10 +162,8 @@ fn app_with_client( fault: Option, ) -> (axum::Router, RuntimePool) { let pool = database.runtime_config.build_pool().unwrap(); - let audit = registry_breg::audit::test_support::capturing( - AuditProfile::production_from_secret_bytes(vec![0x42; 32].into()).unwrap(), - ) - .0; + let audit = + database.audit(AuditProfile::production_from_secret_bytes(vec![0x42; 32].into()).unwrap()); let lock = RegistryLockKey::derive(PACKAGE).unwrap(); let cursors = Arc::new( CursorCodec::new(Zeroizing::new(vec![0x63; 32]), Duration::from_secs(300)).unwrap(), @@ -570,6 +568,7 @@ async fn signed_evidence_actions_release_postgres_and_commit_atomic_transcripts( committed, "retention failure rolls back all operation material" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -596,6 +595,7 @@ async fn caught_failed_helper_cannot_commit_or_make_another_disclosure() { ); assert_eq!(counts(&database, ®istry).await, vec![0, 0, 0, 0, 0, 0]); assert!(!failed.1.to_string().contains("FR-12345")); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -682,6 +682,7 @@ async fn concurrent_receipt_overrides_failed_acquisition_and_recovers_ambiguous_ "ambiguous commit recovery uses retained receipt" ); assert_eq!(counts(&database, ®istry).await, committed); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -734,6 +735,7 @@ async fn verified_acquisition_expiring_during_sql_wait_cannot_commit() { 2, "only a later caller attempt obtains fresh evidence" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } @@ -879,5 +881,6 @@ async fn real_evidence_service_resolves_exact_selector_and_commits_verified_post ); assert_eq!(provider.requests().len(), calls + 1); assert_eq!(counts(&database, ®istry).await, committed); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } diff --git a/crates/registry-breg/tests/postgres_action_evidence_retention.rs b/crates/registry-breg/tests/postgres_action_evidence_retention.rs index 85886443d5..5891fa5e5a 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_retention.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_retention.rs @@ -10,6 +10,7 @@ use registry_breg::{ action_evidence_maintenance::ActionEvidenceRetentionOperatorService, compiler::{compile_project_with_assets, CompileProfile}, contract::{parse_project_yaml, ModuleAssetSource}, + mutation::MutationError, postgres::{ initialize_registry_state_for_catalog_test, install_compiled_schema, ConnectionConfig, ExpectedManagedCatalog, ExpectedRegistryIdentity, RegistryLockKey, @@ -84,9 +85,40 @@ fn service( connection, database.migration_role.clone(), database.runtime_role.clone(), + database.audit( + registry_platform_audit::AuditProfile::production_from_secret_bytes( + vec![0x5e; 32].into(), + ) + .unwrap(), + ), )) } +/// Every erasure is one request entry naming its threshold, answered under +/// its correlation: `outcomes` lists each answer's outcome and erased count. +fn assert_retention_audited(database: &TestDatabase, outcomes: &[(&str, Option)]) { + let entries = database + .audit_entries() + .into_iter() + .filter(|entry| entry["schema"] == "breg-evidence-retention-audit/v1") + .collect::>(); + assert_eq!(entries.len(), outcomes.len() * 2, "{entries:?}"); + for (pair, (outcome, erased)) in entries.chunks(2).zip(outcomes) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); + assert!(pair[0]["record"]["before"].is_string()); + assert_eq!(pair[1]["record"]["before"], pair[0]["record"]["before"]); + assert_eq!(pair[1]["record"]["outcome"], *outcome); + assert_eq!( + pair[1]["record"] + .get("erased") + .and_then(serde_json::Value::as_u64), + *erased + ); + } +} + async fn sentinel(database: &TestDatabase) { database.admin.batch_execute("INSERT INTO registry_internal.registry_idempotency (key_reference,binding_reference,result_kind,result_count,response_status,response_body,response_headers) @@ -154,6 +186,35 @@ async fn expired_request_evidence_erases_only_retained_uses() { &expected, database.migration_config.clone(), ); + // The erasure's commit is refused after every statement succeeded, so + // its outcome is read back from the database: nothing was erased. + database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_evidence_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'test refuses this erasure commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_evidence_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_evidence_commit + AFTER DELETE ON registry_internal.registry_request_evidence_uses + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_evidence_commit();", + ) + .await + .unwrap(); + assert!(matches!( + operator.erase_expired(cutoff()).await, + Err(MutationError::Unavailable) + )); + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_evidence_commit + ON registry_internal.registry_request_evidence_uses; + DROP FUNCTION public.test_refuse_evidence_commit();", + ) + .await + .unwrap(); assert_eq!(operator.erase_expired(cutoff()).await.unwrap(), 1); let remaining = database .admin @@ -168,7 +229,12 @@ async fn expired_request_evidence_erases_only_retained_uses() { assert_eq!(remaining.get::<_, i64>(0), 1); assert_eq!(remaining.get::<_, i64>(1), 1); assert_eq!(operator.erase_expired(cutoff()).await.unwrap(), 0); + assert_retention_audited( + &database, + &[("failed", None), ("erased", Some(1)), ("erased", Some(0))], + ); drop(operator); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } fn cutoff() -> chrono::DateTime { @@ -236,6 +302,7 @@ async fn retention_refuses_misbound_database_with_identical_roles_and_catalog_dr let wrong = service(&original, ®istry, &expected, other_connection.clone()); assert!(wrong.erase_expired(cutoff()).await.is_err(), "a verified runtime identity cannot authorize deletion in another database sharing its role names"); assert_eq!(count(&other).await, 1); + assert_retention_audited(&original, &[("failed", None)]); let correct = service(&original, ®istry, &other_expected, other_connection); other .admin @@ -261,7 +328,9 @@ async fn retention_refuses_misbound_database_with_identical_roles_and_catalog_dr assert_eq!(correct.erase_expired(cutoff()).await.unwrap(), 1); assert_eq!(count(&other).await, 0); drop((wrong, correct)); + other.assert_every_audit_request_answered_once(); other.cleanup().await; + original.assert_every_audit_request_answered_once(); original.cleanup().await; } @@ -354,5 +423,6 @@ async fn retention_serializes_activation_and_holds_identity_lock_through_deletio assert_eq!(erase.await.unwrap().unwrap(), 1); assert_eq!(count(&database).await, 0); drop(operator); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } diff --git a/crates/registry-breg/tests/postgres_action_evidence_targets.rs b/crates/registry-breg/tests/postgres_action_evidence_targets.rs index 9acc1fd316..76b26acef3 100644 --- a/crates/registry-breg/tests/postgres_action_evidence_targets.rs +++ b/crates/registry-breg/tests/postgres_action_evidence_targets.rs @@ -348,6 +348,7 @@ async fn final_local_target_checks_refuse_changes_during_external_wait_even_for_ !row.get::<_, bool>(0), "the script omitted the declared patch slot" ); + database.assert_every_audit_request_answered_once(); database.cleanup().await; } } diff --git a/crates/registry-breg/tests/postgres_change_requests.rs b/crates/registry-breg/tests/postgres_change_requests.rs index 10732e1879..3a787a8360 100644 --- a/crates/registry-breg/tests/postgres_change_requests.rs +++ b/crates/registry-breg/tests/postgres_change_requests.rs @@ -209,6 +209,7 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt .await; let apply = action(&before.body, "apply_request", None); + let entries_before = database.audit_entries().len(); let unavailable = send_action( &app, &apply, @@ -218,6 +219,9 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt ) .await; assert_eq!(unavailable.status, StatusCode::SERVICE_UNAVAILABLE); + // The attempt precedes the receipt preflight's reads and the review + // authority, so even this refusal is a request answered in order. + assert_requested_then_answered(&database.audit_entries()[entries_before..]); assert_eq!(application_result_count(&database).await, 0); assert_eq!(authority_state.lookups.load(Ordering::SeqCst), 1); @@ -248,6 +252,7 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt assert_eq!(authority_state.lookups.load(Ordering::SeqCst), 3); authority_state.mode.store(0, Ordering::SeqCst); + let entries_before = database.audit_entries().len(); let replay = send_action( &app, &apply, @@ -257,6 +262,8 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt ) .await; assert_eq!(replay.status, StatusCode::OK, "{}", replay.body); + // The receipt branch records its attempt once, before its preflight. + assert_requested_then_answered(&database.audit_entries()[entries_before..]); assert_eq!(replay.body, applied.body); assert_eq!( authority_state.lookups.load(Ordering::SeqCst), @@ -268,6 +275,22 @@ async fn cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt database.cleanup().await; } +/// One call's general audit entries: its single attempt request entry first, +/// then at least one response under the same correlation. +fn assert_requested_then_answered(entries: &[serde_json::Value]) { + let entries = entries + .iter() + .filter(|entry| entry["schema"] == "breg-audit/v2") + .collect::>(); + assert!(entries.len() >= 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request", "{entries:?}"); + assert_eq!(entries[0]["record"]["phase"], "attempt"); + for entry in &entries[1..] { + assert_eq!(entry["phase"], "response", "{entries:?}"); + assert_eq!(entry["correlation"], entries[0]["correlation"]); + } +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn review_submissions_bind_the_subject_to_the_registrys_request_entity() { let database = TestDatabase::create(8).await; diff --git a/crates/registry-breg/tests/postgres_history_erasure.rs b/crates/registry-breg/tests/postgres_history_erasure.rs index fba965239c..19990fea9c 100644 --- a/crates/registry-breg/tests/postgres_history_erasure.rs +++ b/crates/registry-breg/tests/postgres_history_erasure.rs @@ -1092,6 +1092,14 @@ async fn field_encryption_erasure_uses_flip_provenance_for_structured_plaintext( .expect("completed lifecycle replay state resolves"); assert!(replay_state.get::<_, bool>(0)); assert_eq!(field_encryption_terminal_entries(&database).len(), 1); + // The refused replay still answers the lifecycle request it wrote. + let last = database + .audit_entries() + .pop() + .expect("the replay's entries"); + assert_eq!(last["schema"], FIELD_ENCRYPTION_AUDIT_SCHEMA); + assert_eq!(last["phase"], "response"); + assert_eq!(last["record"]["outcome"], "unfinished"); migration_task.abort(); database.cleanup().await; @@ -2366,6 +2374,7 @@ fn field_encryption_terminal_entries(database: &TestDatabase) -> Vec>(); assert_eq!( rebaseline_entries, - ["request"], - "a refused rebaseline records its request and no committed response" + ["request", "response"], + "a refused rebaseline answers its request without a committed response" ); + let answer = database + .audit_entries() + .into_iter() + .rfind(|entry| entry["schema"] == HISTORY_REBASELINE_AUDIT_SCHEMA) + .expect("the refused rebaseline's answer"); + assert_eq!(answer["record"]["outcome"], "unfinished"); migration_task.abort(); database.cleanup().await; @@ -852,8 +858,8 @@ fn assert_rebaseline_audit_is_minimized(database: &TestDatabase) { }) .collect::>(); // The erasure, the completed rebaseline, and the second rebaseline that - // had nothing to do each record their request; the two that committed - // record their response under the same correlation. + // had nothing to do each record their request and a response under the + // same correlation; the one that had nothing to do answers unfinished. assert_eq!( shape, [ @@ -862,10 +868,13 @@ fn assert_rebaseline_audit_is_minimized(database: &TestDatabase) { (HISTORY_REBASELINE_AUDIT_SCHEMA, "request"), (HISTORY_REBASELINE_AUDIT_SCHEMA, "response"), (HISTORY_REBASELINE_AUDIT_SCHEMA, "request"), + (HISTORY_REBASELINE_AUDIT_SCHEMA, "response"), ] ); assert_eq!(entries[2]["correlation"], entries[3]["correlation"]); assert_ne!(entries[2]["correlation"], entries[4]["correlation"]); + assert_eq!(entries[4]["correlation"], entries[5]["correlation"]); + assert_eq!(entries[5]["record"]["outcome"], "unfinished"); let audit_text = serde_json::Value::Array(entries[2..].to_vec()).to_string(); assert!(audit_text.contains("history-rebaseline-maintenance")); assert!(!audit_text.contains(OPERATOR_CANARY)); diff --git a/crates/registry-breg/tests/postgres_ingestion_runs.rs b/crates/registry-breg/tests/postgres_ingestion_runs.rs index 485fa5f263..72569b6efa 100644 --- a/crates/registry-breg/tests/postgres_ingestion_runs.rs +++ b/crates/registry-breg/tests/postgres_ingestion_runs.rs @@ -710,6 +710,196 @@ async fn cancel_closes_the_run_and_preserves_the_committed_prefix() { assert_eq!(body_json(second).await["code"], "ingestion.run_not_open"); } +/// A run creation refused after its ingestion request entry is answered in +/// the ingestion schema under the same correlation, and a chunk the batch +/// mutation refuses is answered by that mutation's own refusal; neither is +/// recorded a second time as a general refusal. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema() { + let harness = IngestionHarness::create().await; + let claims = operator_claims(PRINCIPAL, "zone-a"); + let items = announce_items("refused-after-request", 4); + let chunks = plan_chunks(&items, 2); + + refuse_inserts(&harness, "registry_ingestion_runs").await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + "/v1/records/widgets/ingestion-runs", + &claims, + harness.run_body("create", &chunks), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_inserts(&harness, "registry_ingestion_runs").await; + assert_answered_in_the_ingestion_schema(&harness.database.audit_entries()[before..], "create"); + + let run_id = harness.create_run(&claims, &chunks).await; + refuse_inserts(&harness, "registry_ingestion_run_chunks").await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + &format!("/v1/records/widgets/ingestion-runs/{run_id}/chunks"), + &claims, + chunk_body(&chunks, 0), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_inserts(&harness, "registry_ingestion_run_chunks").await; + let entries = &harness.database.audit_entries()[before..]; + assert_eq!( + entries + .iter() + .map(|entry| ( + entry["schema"].as_str().expect("schema"), + entry["phase"].as_str().expect("phase") + )) + .collect::>(), + [("breg-audit/v2", "request"), ("breg-audit/v2", "response")], + "{entries:?}" + ); + assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); +} + +/// A run transition whose commit returns an error may still have committed, +/// so its request entry is answered unfinished rather than refused. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_transition_whose_commit_fails_is_answered_unfinished() { + let harness = IngestionHarness::create().await; + let claims = operator_claims(PRINCIPAL, "zone-a"); + let items = announce_items("unacknowledged-commit", 4); + let chunks = plan_chunks(&items, 2); + + refuse_run_commits(&harness).await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_json( + "/v1/records/widgets/ingestion-runs", + &claims, + harness.run_body("create", &chunks), + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_run_commits(&harness).await; + assert_unfinished_in_the_ingestion_schema( + &harness.database.audit_entries()[before..], + "create", + ); + + let run_id = harness.create_run(&claims, &chunks).await; + refuse_run_commits(&harness).await; + let before = harness.database.audit_entries().len(); + let refused = harness + .post_empty( + &format!("/v1/records/widgets/ingestion-runs/{run_id}/cancel"), + &claims, + ) + .await; + assert_eq!(refused.status(), StatusCode::SERVICE_UNAVAILABLE); + allow_run_commits(&harness).await; + assert_unfinished_in_the_ingestion_schema( + &harness.database.audit_entries()[before..], + "cancel", + ); + harness.database.assert_every_audit_request_answered_once(); +} + +fn assert_unfinished_in_the_ingestion_schema(entries: &[Value], transition: &str) { + let ingestion = entries + .iter() + .filter(|entry| entry["record"]["transition"] == transition) + .collect::>(); + assert_eq!(ingestion.len(), 2, "{entries:?}"); + assert_eq!(ingestion[0]["phase"], "request"); + assert_eq!(ingestion[1]["phase"], "response"); + assert_eq!( + ingestion[1]["record"]["outcome"], "unfinished", + "{entries:?}" + ); + assert_eq!(ingestion[0]["correlation"], ingestion[1]["correlation"]); +} + +/// Refuse every commit that wrote a run row, after all its statements ran. +async fn refuse_run_commits(harness: &IngestionHarness) { + harness + .database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_run_commit() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN RAISE EXCEPTION 'test refuses this commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_run_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_run_commit + AFTER INSERT OR UPDATE ON registry_internal.registry_ingestion_runs + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_run_commit();", + ) + .await + .expect("administrator installs the commit refusal"); +} + +async fn allow_run_commits(harness: &IngestionHarness) { + harness + .database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_run_commit ON registry_internal.registry_ingestion_runs; + DROP FUNCTION public.test_refuse_run_commit();", + ) + .await + .expect("administrator removes the commit refusal"); +} + +/// The refused call wrote one ingestion request entry and one response +/// entry answering it, both in the ingestion schema, and nothing else. +fn assert_answered_in_the_ingestion_schema(entries: &[Value], transition: &str) { + let ingestion = entries + .iter() + .filter(|entry| entry["record"]["transition"] == transition) + .collect::>(); + assert_eq!(ingestion.len(), 2, "{entries:?}"); + for entry in &ingestion { + assert_eq!(entry["schema"], "breg-ingestion-audit/v1"); + } + assert_eq!(ingestion[0]["phase"], "request"); + assert_eq!(ingestion[1]["phase"], "response"); + assert_eq!(ingestion[1]["record"]["outcome"], "refused"); + assert_eq!(ingestion[0]["correlation"], ingestion[1]["correlation"]); + assert!( + entries + .iter() + .all(|entry| entry["correlation"] != ingestion[0]["correlation"] + || entry["schema"] == "breg-ingestion-audit/v1"), + "the refusal is not recorded again in another schema: {entries:?}" + ); +} + +async fn refuse_inserts(harness: &IngestionHarness, table: &str) { + harness + .database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_insert() RETURNS trigger + LANGUAGE plpgsql AS $$ BEGIN RAISE EXCEPTION 'test refuses this insert'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_insert() TO PUBLIC; + CREATE TRIGGER test_refuse_insert BEFORE INSERT ON registry_internal.{table} + FOR EACH ROW EXECUTE FUNCTION public.test_refuse_insert();" + )) + .await + .expect("administrator installs the insert refusal"); +} + +async fn allow_inserts(harness: &IngestionHarness, table: &str) { + harness + .database + .admin + .batch_execute(&format!( + "DROP TRIGGER test_refuse_insert ON registry_internal.{table}; + DROP FUNCTION public.test_refuse_insert();" + )) + .await + .expect("administrator removes the insert refusal"); +} + /// Cancellation is itself the run's last attempt, and it is not a chunk /// attempt: the metadata renders the refused outcome with no chunk index, /// so a cancelled run never reports an earlier chunk's index as its last. diff --git a/crates/registry-breg/tests/postgres_migration.rs b/crates/registry-breg/tests/postgres_migration.rs index b524ace528..0cf50cbead 100644 --- a/crates/registry-breg/tests/postgres_migration.rs +++ b/crates/registry-breg/tests/postgres_migration.rs @@ -1091,6 +1091,39 @@ async fn reconciliation_completes_a_target_the_catalog_already_reached() { assert_eq!(durable_snapshot(&database).await, before_refusal); database.audit_capture().restore(); + // A transition that fails after its request entry answers that entry + // with a failed response under the same correlation. + database + .admin + .batch_execute( + "BEGIN; SELECT 1 FROM registry_internal.registry_state WHERE singleton FOR UPDATE", + ) + .await + .expect("administrator holds the Registry state row"); + let entries_before = database.audit_entries().len(); + reconcile(&database, &package, &active, &base, true) + .await + .expect_err("the activation cannot take the held Registry state row"); + database + .admin + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the Registry state row"); + let failed = database.audit_entries().split_off(entries_before); + assert_eq!( + failed + .iter() + .map(|entry| ( + entry["phase"].as_str().expect("phase"), + entry["record"]["outcome"].as_str().expect("outcome") + )) + .collect::>(), + [("request", "started"), ("response", "failed")], + "{failed:?}" + ); + assert_eq!(failed[0]["correlation"], failed[1]["correlation"]); + assert!(!failed[1].to_string().contains(RECONCILE_OPERATOR_CANARY)); + let completed = reconcile(&database, &package, &active, &base, true) .await .expect("the missing activation transition completes"); @@ -4061,14 +4094,15 @@ async fn assert_reconcile_audit_is_minimized(database: &TestDatabase, action: &s matched.push(entry); } } - assert_eq!( - matched - .iter() - .map(|entry| entry["phase"].as_str().expect("phase is a string")) - .collect::>(), - ["request", "response"] - ); - assert_eq!(matched[0]["correlation"], matched[1]["correlation"]); + // Every execution is a request answered by one response under its + // correlation; the last one committed. + assert!(!matched.is_empty() && matched.len() % 2 == 0, "{matched:?}"); + for pair in matched.chunks(2) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); + } + assert_eq!(matched[matched.len() - 1]["record"]["outcome"], "committed"); } fn assert_value_free(actual: Option, expected: MigrationError) { diff --git a/crates/registry-breg/tests/postgres_mutation.rs b/crates/registry-breg/tests/postgres_mutation.rs index 4a34c7345c..ddc0bd727b 100644 --- a/crates/registry-breg/tests/postgres_mutation.rs +++ b/crates/registry-breg/tests/postgres_mutation.rs @@ -304,10 +304,10 @@ async fn real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable() assert_eq!( durable_counts(&database, table).await, DurableCounts { - audit: before.audit + 1, + audit: before.audit + 2, ..before }, - "fault {fault:?} retains only its unavoidable durable attempt" + "fault {fault:?} retains only its attempt and the unfinished answer to it" ); } @@ -2257,10 +2257,11 @@ async fn real_postgres_http_mutations_are_guarded_and_exactly_replayable() { assert_eq!( durable_counts(&database, &table).await, DurableCounts { - audit: before_fault.audit + 1, + audit: before_fault.audit + 2, ..before_fault }, - "terminal audit failure releases no success bytes and commits no mutation packet" + "terminal audit failure releases no success bytes, commits no mutation packet, and \ + answers its attempt as unfinished" ); assert_journals_are_minimized_and_paired(&database).await; @@ -3111,6 +3112,7 @@ async fn assert_patch_preserved_omitted_field( } async fn assert_journals_are_minimized_and_paired(database: &TestDatabase) { + database.assert_every_audit_request_answered_once(); let ordered = database.audit_entries(); for entry in &ordered { let expected = if entry["record"]["phase"] == "attempt" { diff --git a/crates/registry-breg/tests/postgres_read.rs b/crates/registry-breg/tests/postgres_read.rs index 2c854dd5c8..edca9aecc0 100644 --- a/crates/registry-breg/tests/postgres_read.rs +++ b/crates/registry-breg/tests/postgres_read.rs @@ -570,10 +570,35 @@ async fn real_postgres_read_is_authorized_bounded_minimized_and_audit_gated() { assert!(!faulted_body.to_string().contains("label-001")); assert_eq!( audit_count(&database).await, - before_fault + 1, - "a terminal audit fault releases no protected data and commits only the prior attempt" + before_fault + 2, + "a read ended before its terminal releases no protected data and answers its attempt \ + as unfinished" ); + // A failure after the rows were read, binding the strong entity tag, + // answers the attempt with the Refused terminal and releases nothing. + let before_etag_fault = audit_count(&database).await; + let etag_faulting_app = read_router( + pool.clone(), + compiled.clone(), + identity.clone(), + lock_key, + profile.clone(), + Some(ReadFaultPoint::StrongEtag), + ); + let etag_faulted = send( + &etag_faulting_app, + &format!("/v1/records/widgets/{VISIBLE_RECORD}?$select=label"), + Some(read_claims(["zone-a"])), + ) + .await; + assert_eq!(etag_faulted.status(), StatusCode::SERVICE_UNAVAILABLE); + assert!(!body_json(etag_faulted) + .await + .to_string() + .contains("label-001")); + assert_eq!(audit_count(&database).await, before_etag_fault + 2); + assert_read_audit_is_ordered_paired_and_minimized(&database, &compiled); // A restarted process over the recovered destination. A writer that // refused an append stays failed, so recovery is a fresh writer. @@ -1642,7 +1667,10 @@ fn assert_read_audit_is_ordered_paired_and_minimized( assert_eq!(entry["schema"], registry_breg::audit::AUDIT_SCHEMA); } for pair in entries.windows(2) { - if pair[0]["phase"] == "request" && pair[1]["record"]["phase"] == "terminal" { + if pair[0]["phase"] == "request" + && (pair[1]["record"]["phase"] == "terminal" + || pair[1]["record"]["phase"] == "unfinished") + { assert_eq!(pair[0]["correlation"], pair[1]["correlation"]); } } @@ -1688,6 +1716,9 @@ fn assert_read_audit_is_ordered_paired_and_minimized( ("terminal", Some("empty")), ("refusal", None), ("attempt", None), + ("unfinished", None), + ("attempt", None), + ("terminal", Some("refused")), ], "durable read audit records bracket release in order" ); diff --git a/crates/registry-breg/tests/postgres_request_read_retention.rs b/crates/registry-breg/tests/postgres_request_read_retention.rs index dc29bdbc1a..080ae5a8a3 100644 --- a/crates/registry-breg/tests/postgres_request_read_retention.rs +++ b/crates/registry-breg/tests/postgres_request_read_retention.rs @@ -762,6 +762,7 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it assert_eq!(phases, ["request", "response"]); assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); assert_eq!(entries[0]["record"]["phase"], "attempt"); + assert_eq!(entries[1]["record"]["outcome"], "committed"); assert_eq!( entries[0]["record"]["recordReference"], entries[1]["record"]["recordReference"] @@ -773,6 +774,275 @@ async fn request_detail_erasure_changes_nothing_when_the_audit_writer_refuses_it database.cleanup().await; } +/// A committed request-detail erasure is recorded as soon as its commit is +/// confirmed. The external-deletion retry that follows cannot hold the +/// committed erasure's response back: the retry here waits for the registry +/// lock while the response is already on record. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn request_detail_erasure_records_its_commit_before_retrying_external_deletions() { + let database = TestDatabase::create(6).await; + let registry = Arc::new(compiled_registry()); + let identity = install_registry(&database, ®istry).await; + let app = request_router(&database, registry.clone(), identity.clone()); + let operator = claims("operator", "operator-principal"); + let request_id = applied_correction_request(&app, operator, "recorded-erasure").await; + let request_uuid = Uuid::parse_str(&request_id).expect("request id parses"); + let scope = RequestDetailErasureScope { + request_entity_id: "correction-request", + request_id: request_uuid, + proposal_version: 1, + }; + let (audit, capture) = registry_breg::audit::test_support::capturing( + AuditProfile::production_from_secret_bytes(vec![0x8e; 32].into()) + .expect("test audit profile is keyed"), + ); + let lock_key = RegistryLockKey::derive(PACKAGE_ID).expect("lock key derives"); + let retention = RequestRetentionOperatorService::new_for_test( + registry.as_ref().clone(), + identity.clone(), + ExpectedManagedCatalog::compiled(®istry), + lock_key, + database.migration_config.clone(), + database.migration_role.clone(), + database.runtime_role.clone(), + audit, + ); + // Other tests share the cluster, so only this database's sessions count. + let waiting = |wait_event: &'static str| { + let admin = &database.admin; + async move { + tokio::time::timeout(Duration::from_secs(4), async { + loop { + let waiting: i64 = admin + .query_one( + "SELECT count(*) FROM pg_catalog.pg_stat_activity + WHERE datname = current_database() + AND wait_event_type = 'Lock' AND wait_event = $1", + &[&wait_event], + ) + .await + .expect("administrator reads lock waits") + .get(0); + if waiting > 0 { + return; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("a session waits for the held lock"); + } + }; + + // The erasure holds the registry lock while it waits for the request + // row. A second session queues for the registry lock behind it, so it + // takes the lock the moment the erasure commits, and the + // external-deletion retry that follows waits for it. + let (holder, holder_task) = database.connect_admin().await; + holder + .batch_execute(&format!( + "BEGIN; SELECT 1 FROM registry_internal.registry_request_state + WHERE request_id = '{request_uuid}' FOR UPDATE" + )) + .await + .expect("administrator holds the request row"); + let erasure = tokio::spawn(async move { retention.erase(scope).await }); + waiting("transactionid").await; + let (queued, queued_task) = database.connect_admin().await; + let queued = Arc::new(queued); + let registry_lock = tokio::spawn({ + let queued = Arc::clone(&queued); + async move { + queued + .execute("SELECT pg_catalog.pg_advisory_lock($1)", &[&lock_key.get()]) + .await + .expect("the queued session takes the registry lock"); + } + }); + waiting("advisory").await; + holder + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the request row"); + + let recorded = tokio::time::timeout(Duration::from_secs(4), async { + loop { + let entries = capture.entries(); + if entries.len() >= 2 { + return entries; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await; + let still_retrying = !erasure.is_finished(); + registry_lock + .await + .expect("the queued session holds the registry lock"); + queued + .execute( + "SELECT pg_catalog.pg_advisory_unlock($1)", + &[&lock_key.get()], + ) + .await + .expect("the queued session releases the registry lock"); + let entries = recorded.expect("the committed erasure is recorded while the retry waits"); + assert!(still_retrying, "the retry was still waiting for the lock"); + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["record"]["outcome"], "committed"); + assert_eq!(entries[0]["correlation"], entries[1]["correlation"]); + let erased = erasure + .await + .expect("the erasure task completes") + .expect("the erasure succeeds once the retry proceeds"); + assert_eq!(erased.pending_external_deletions, 0); + assert_eq!(erased.external_deletion_tombstones, 0); + assert_eq!(capture.entries().len(), 2); + + drop(holder); + drop(queued); + holder_task.abort(); + queued_task.abort(); + database.cleanup().await; +} + +/// An erasure whose transaction fails after its request entry answers that +/// entry with a failed response and erases nothing. One whose committed +/// erasure the destination refuses to record reports that distinctly: the +/// detail is gone, and the journal holds only the request. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn request_detail_erasure_pairs_its_request_entry_on_every_outcome() { + let database = TestDatabase::create(6).await; + let registry = Arc::new(compiled_registry()); + let identity = install_registry(&database, ®istry).await; + let app = request_router(&database, registry.clone(), identity.clone()); + let operator = claims("operator", "operator-principal"); + let request_id = applied_correction_request(&app, operator, "paired-erasure").await; + let request_uuid = Uuid::parse_str(&request_id).expect("request id parses"); + let scope = RequestDetailErasureScope { + request_entity_id: "correction-request", + request_id: request_uuid, + proposal_version: 1, + }; + let profile = || { + AuditProfile::production_from_secret_bytes(vec![0x8d; 32].into()) + .expect("test audit profile is keyed") + }; + let retention_with = |audit| { + RequestRetentionOperatorService::new_for_test( + registry.as_ref().clone(), + identity.clone(), + ExpectedManagedCatalog::compiled(®istry), + RegistryLockKey::derive(PACKAGE_ID).expect("lock key derives"), + database.migration_config.clone(), + database.migration_role.clone(), + database.runtime_role.clone(), + audit, + ) + }; + let (audit, capture) = registry_breg::audit::test_support::capturing(profile()); + + // The request row is held, so the erasure transaction cannot lock it. + database + .admin + .batch_execute(&format!( + "BEGIN; SELECT 1 FROM registry_internal.registry_request_state + WHERE request_id = '{request_uuid}' FOR UPDATE" + )) + .await + .expect("administrator holds the request row"); + retention_with(audit) + .erase(scope.clone()) + .await + .expect_err("the erasure cannot lock the held request"); + database + .admin + .batch_execute("ROLLBACK") + .await + .expect("administrator releases the request row"); + let failed = capture.entries(); + assert_eq!(failed.len(), 2, "{failed:?}"); + assert_eq!(failed[0]["phase"], "request"); + assert_eq!(failed[1]["phase"], "response"); + assert_eq!(failed[1]["record"]["outcome"], "failed"); + assert_eq!(failed[0]["correlation"], failed[1]["correlation"]); + let retained = retention_with(capture.audit(profile())); + assert!( + !retained + .dry_run(scope.clone()) + .await + .expect("the detail still plans") + .detail_erased, + "the failed erasure erased nothing" + ); + + // The erasure's commit is refused after every statement succeeded, so + // the outcome is read back from the database: nothing was erased. + database + .admin + .batch_execute( + "CREATE OR REPLACE FUNCTION public.test_refuse_erasure_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'test refuses this erasure commit'; END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_erasure_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_erasure_commit + AFTER UPDATE ON registry_internal.registry_request_proposals + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_erasure_commit();", + ) + .await + .expect("administrator installs the erasure commit refusal"); + assert_eq!( + retained.erase(scope.clone()).await, + Err(RequestRetentionError::Unavailable) + ); + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_erasure_commit + ON registry_internal.registry_request_proposals; + DROP FUNCTION public.test_refuse_erasure_commit();", + ) + .await + .expect("administrator removes the erasure commit refusal"); + let unacknowledged = &capture.entries()[failed.len()..]; + assert_eq!(unacknowledged.len(), 2, "{unacknowledged:?}"); + assert_eq!(unacknowledged[1]["record"]["outcome"], "failed"); + assert_eq!( + unacknowledged[0]["correlation"], + unacknowledged[1]["correlation"] + ); + assert!( + !retention_with(capture.audit(profile())) + .dry_run(scope.clone()) + .await + .expect("the detail still plans") + .detail_erased, + "the refused commit erased nothing" + ); + let failed = capture.entries(); + + // The destination accepts the request entry and refuses the response. + capture.fail_after(1); + assert_eq!( + retained.erase(scope.clone()).await, + Err(RequestRetentionError::ErasureUnaudited) + ); + capture.restore(); + assert!( + retention_with(capture.audit(profile())) + .dry_run(scope.clone()) + .await + .expect("the erased detail still plans") + .detail_erased, + "the unaudited erasure committed" + ); + assert_eq!(capture.entries().len(), failed.len() + 1); + + database.cleanup().await; +} + /// Create, submit, and apply one correction request, returning its id. async fn applied_correction_request( app: &axum::Router, diff --git a/crates/registry-breg/tests/postgres_request_upgrade_retention.rs b/crates/registry-breg/tests/postgres_request_upgrade_retention.rs index 840849a478..b1885d5aaf 100644 --- a/crates/registry-breg/tests/postgres_request_upgrade_retention.rs +++ b/crates/registry-breg/tests/postgres_request_upgrade_retention.rs @@ -1426,8 +1426,8 @@ async fn operator_retention_service_counts_pages_erases_under_forced_rls_and_aud let audit_entries = database.audit_entries(); assert_eq!( audit_entries.len(), - 5, - "the refused erasure records its request, and cleanup appends a request and a \ + 6, + "the refused erasure answers its request, and cleanup appends a request and a \ response entry independently of erasure eligibility" ); assert_eq!(audit_entries[2]["phase"], "request"); @@ -1435,14 +1435,20 @@ async fn operator_retention_service_counts_pages_erases_under_forced_rls_and_aud audit_entries[2]["correlation"], audit_entries[1]["correlation"], "each erasure invocation has its own correlation" ); - assert_eq!(audit_entries[3]["phase"], "request"); - assert_eq!(audit_entries[4]["phase"], "response"); + assert_eq!(audit_entries[3]["phase"], "response"); + assert_eq!(audit_entries[3]["record"]["outcome"], "refused"); assert_eq!( - audit_entries[3]["correlation"], audit_entries[4]["correlation"], + audit_entries[2]["correlation"], audit_entries[3]["correlation"], + "the refusal answers the erasure's request" + ); + assert_eq!(audit_entries[4]["phase"], "request"); + assert_eq!(audit_entries[5]["phase"], "response"); + assert_eq!( + audit_entries[4]["correlation"], audit_entries[5]["correlation"], "the cleanup response shares its request's correlation" ); assert_eq!( - audit_entries[3]["schema"], + audit_entries[4]["schema"], "breg-attachment-cleanup-audit/v1" ); diff --git a/crates/registry-breg/tests/postgres_revision_http.rs b/crates/registry-breg/tests/postgres_revision_http.rs index a05d395ff9..ffa1d87fcd 100644 --- a/crates/registry-breg/tests/postgres_revision_http.rs +++ b/crates/registry-breg/tests/postgres_revision_http.rs @@ -301,8 +301,9 @@ async fn real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gate assert_eq!(body_json(faulted).await["code"], "source.unavailable"); assert_eq!( audit_count(&database).await, - before_fault + 1, - "terminal audit gate failure releases no held revision and leaves only the attempt" + before_fault + 2, + "terminal audit gate failure releases no held revision and answers the attempt as \ + unfinished" ); let unkeyed = revision_router( @@ -779,7 +780,10 @@ async fn assert_revision_audit_is_ordered_and_minimized( && window[1]["phase"] == "terminal" && window[1]["outcome"] == "returned" })); - assert_eq!(records.last().expect("fault attempt")["phase"], "attempt"); + // The faulted read's attempt is answered as unfinished. + let fault_answer = records.len() - 1; + assert_eq!(records[fault_answer]["phase"], "unfinished"); + assert_eq!(records[fault_answer - 1]["phase"], "attempt"); assert!(records.iter().any(|record| record["phase"] == "refusal")); assert!(records.iter().any(|record| { record["phase"] == "terminal" diff --git a/crates/registry-breg/tests/postgres_spatial_read.rs b/crates/registry-breg/tests/postgres_spatial_read.rs index 7ccf1c1b11..80fa66f512 100644 --- a/crates/registry-breg/tests/postgres_spatial_read.rs +++ b/crates/registry-breg/tests/postgres_spatial_read.rs @@ -613,8 +613,9 @@ async fn real_postgres_spatial_bbox_reads_preserve_authority_and_geojson_audit() assert!(!faulted.to_string().contains("edge-west")); assert_eq!( audit_count(&harness.database).await, - before_fault + 1, - "terminal audit failure releases no held GeoJSON bytes and commits only the attempt" + before_fault + 2, + "terminal audit failure releases no held GeoJSON bytes and answers the attempt as \ + unfinished" ); let before_adapter_fault = audit_count(&harness.database).await; @@ -631,8 +632,9 @@ async fn real_postgres_spatial_bbox_reads_preserve_authority_and_geojson_audit() assert!(!adapter_faulted.to_string().contains("zero-area")); assert_eq!( audit_count(&harness.database).await, - before_adapter_fault + 1, - "GIS adapter terminal audit failure releases no held GeoJSON bytes" + before_adapter_fault + 2, + "GIS adapter terminal audit failure releases no held GeoJSON bytes and answers the \ + attempt as unfinished" ); assert_pool_context_clean(&harness.pool, &harness.database.runtime_role).await; diff --git a/crates/registry-breg/tests/postgres_tombstone_revision.rs b/crates/registry-breg/tests/postgres_tombstone_revision.rs index 14801cfa96..a010bd38d4 100644 --- a/crates/registry-breg/tests/postgres_tombstone_revision.rs +++ b/crates/registry-breg/tests/postgres_tombstone_revision.rs @@ -337,7 +337,8 @@ async fn tombstone_refusals_faults_and_concurrency_have_no_duplicate_effects() { assert_eq!( durable_counts(&fixture.database, &fixture.table).await, DurableCounts { - audit: before.audit + 1, + // The attempt and the unfinished answer to it. + audit: before.audit + 2, ..before } ); diff --git a/crates/registry-breg/tests/postgres_webhook_delivery.rs b/crates/registry-breg/tests/postgres_webhook_delivery.rs index 5aa91bc6fb..c564eddc8b 100644 --- a/crates/registry-breg/tests/postgres_webhook_delivery.rs +++ b/crates/registry-breg/tests/postgres_webhook_delivery.rs @@ -353,14 +353,38 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun ) .await; - service - .replay( - timeout_event.event_id, - &timeout_event.compiled_delivery_id, - 1, + // The operator stops waiting while the reset's commit is in flight: + // the replay still runs to its response, so its accepted request is + // answered once the reset commits. + slow_delivery_state_commit(&database, "pending").await; + assert!( + tokio::time::timeout( + Duration::from_millis(300), + service.replay( + timeout_event.event_id, + &timeout_event.compiled_delivery_id, + 1, + ), ) .await - .expect("compiled operator replay resets one terminal generation"); + .is_err(), + "the caller leaves before the reset commits" + ); + allow_delivery_state_commit(&database).await; + let mut answered = Vec::new(); + for _ in 0..50 { + answered = audit_outcomes(&database, &audit_profile, &timeout_event, 2, 0, "replay").await; + if answered.len() == 2 { + break; + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + assert_eq!( + answered, + ["replay_requested", "replay_committed"], + "a replay whose caller left is still answered" + ); + assert_eq!(delivery_state(&database, &timeout_event).await.0, 2); assert_eq!( service .replay( @@ -384,16 +408,13 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun header(&replay_request, "idempotency-key"), "operator replay changes the deterministic generation binding" ); - assert_exact_audit_outcome( - &database, - &audit_profile, - &timeout_event, - 2, - 0, - "replay", - "replay_requested", - ) - .await; + // The replay is a request before its reset and a response once the + // reset commits, under one correlation. + assert_eq!( + audit_outcomes(&database, &audit_profile, &timeout_event, 2, 0, "replay").await, + ["replay_requested", "replay_committed"], + "an operator replay is answered once its reset commits" + ); assert_exact_audit_outcome( &database, &audit_profile, @@ -799,6 +820,100 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun "delivered", ) .await; + // A disposition whose commit fails is never recorded as done: the + // journal keeps only the attempt, and expiry recovery answers it. + let commit_egress_before = receiver.count().await; + receiver.enqueue(ResponsePlan::Status(204)).await; + let commit_refused = create_event( + &database, + &coordinator, + &mut mutation_client, + &plan, + &claims, + "delivery-terminal-commit-refused", + "terminal-commit", + ) + .await; + refuse_delivery_state_commit(&database, "delivered").await; + assert_eq!( + service.deliver_once().await, + Err(WebhookDeliveryError::Unavailable) + ); + allow_delivery_state_commit(&database).await; + receiver.wait_for_count(commit_egress_before + 1).await; + assert_eq!( + delivery_state(&database, &commit_refused).await, + (1, "leased".to_owned(), 1), + "the rolled-back disposition leaves the lease for expiry recovery" + ); + assert_no_audit_outcome(&database, &audit_profile, &commit_refused, 1, 1, "terminal").await; + expire_lease(&database, &commit_refused).await; + service + .deliver_once() + .await + .expect("expiry recovery answers the interrupted attempt"); + assert_exact_audit_outcome( + &database, + &audit_profile, + &commit_refused, + 1, + 1, + "terminal", + "worker_interrupted", + ) + .await; + + // A lease whose commit fails after its attempt was recorded sends + // nothing, and its attempt is answered as interrupted. + let lease_egress_before = receiver.count().await; + let lease_refused = create_event( + &database, + &coordinator, + &mut mutation_client, + &plan, + &claims, + "delivery-lease-commit-refused", + "lease-commit", + ) + .await; + refuse_delivery_state_commit(&database, "leased").await; + assert_eq!( + service.deliver_once().await, + Err(WebhookDeliveryError::Unavailable) + ); + allow_delivery_state_commit(&database).await; + assert_eq!(receiver.count().await, lease_egress_before, "no egress"); + assert_eq!( + delivery_state(&database, &lease_refused).await, + (1, "pending".to_owned(), 0) + ); + assert_exact_audit_outcome( + &database, + &audit_profile, + &lease_refused, + 1, + 1, + "attempt", + "attempt_started", + ) + .await; + assert_exact_audit_outcome( + &database, + &audit_profile, + &lease_refused, + 1, + 1, + "terminal", + "worker_interrupted", + ) + .await; + receiver.enqueue(ResponsePlan::Status(204)).await; + assert_eq!( + service.deliver_once().await, + Ok(WebhookWorkOutcome::Delivered), + "the refused lease is claimed again once the commit succeeds" + ); + let terminal_egress_before = receiver.count().await; let terminal_response_release = Arc::new(Notify::new()); @@ -828,10 +943,13 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun Err(WebhookDeliveryError::Unavailable) ); database.audit_capture().restore(); + // The terminal is recorded only after its disposition commits, so the + // refused entry leaves the committed disposition without it: the writer + // then refuses every later entry until the destination is repaired. assert_eq!( delivery_state(&database, &terminal_audit_refused).await, - (1, "leased".to_owned(), 1), - "terminal audit refusal leaves the committed lease for expiry recovery" + (1, "delivered".to_owned(), 1), + "the terminal entry follows the committed disposition" ); assert_no_audit_outcome( &database, @@ -852,6 +970,82 @@ async fn real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_boun database.cleanup().await; } +/// Make the commit of any delivery transition into `state` fail, after +/// every statement in its transaction succeeded. +async fn refuse_delivery_state_commit(database: &TestDatabase, state: &str) { + database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_delivery_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.state = '{state}' THEN + RAISE EXCEPTION 'test refuses this delivery commit'; + END IF; + RETURN NEW; + END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_delivery_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_delivery_commit + AFTER UPDATE ON registry_internal.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_delivery_commit();" + )) + .await + .expect("administrator installs the commit refusal"); +} + +/// Hold the commit of any delivery transition into `state` for a second, +/// after every statement in its transaction succeeded. +async fn slow_delivery_state_commit(database: &TestDatabase, state: &str) { + database + .admin + .batch_execute(&format!( + "CREATE OR REPLACE FUNCTION public.test_refuse_delivery_commit() + RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.state = '{state}' THEN + PERFORM pg_sleep(1); + END IF; + RETURN NEW; + END $$; + GRANT EXECUTE ON FUNCTION public.test_refuse_delivery_commit() TO PUBLIC; + CREATE CONSTRAINT TRIGGER test_refuse_delivery_commit + AFTER UPDATE ON registry_internal.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW + EXECUTE FUNCTION public.test_refuse_delivery_commit();" + )) + .await + .expect("administrator installs the commit delay"); +} + +async fn allow_delivery_state_commit(database: &TestDatabase) { + database + .admin + .batch_execute( + "DROP TRIGGER test_refuse_delivery_commit + ON registry_internal.registry_webhook_delivery_state; + DROP FUNCTION public.test_refuse_delivery_commit();", + ) + .await + .expect("administrator removes the commit refusal"); +} + +async fn expire_lease(database: &TestDatabase, event: &CapturedEvent) { + database + .admin + .execute( + // Move the whole lease into the past, keeping its captured length. + "UPDATE registry_internal.registry_webhook_delivery_state + SET attempt_started_at = attempt_started_at + - (lease_expires_at - attempt_started_at) - interval '1 second', + lease_expires_at = attempt_started_at - interval '1 second' + WHERE event_id = $1 AND state = 'leased'", + &[&event.event_id], + ) + .await + .expect("administrator expires the lease"); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn real_postgres_webhook_delivery_finishes_prior_package_work_after_compatible_upgrade() { let receiver = HttpsReceiver::start().await; diff --git a/crates/registry-breg/tests/support/postgres_harness.rs b/crates/registry-breg/tests/support/postgres_harness.rs index a65ba967e6..75c45dbb22 100644 --- a/crates/registry-breg/tests/support/postgres_harness.rs +++ b/crates/registry-breg/tests/support/postgres_harness.rs @@ -139,6 +139,37 @@ impl TestDatabase { self.audit.entries() } + /// Assert that every audit request accepted so far has exactly one + /// response, written after it under the same schema and correlation. + /// The only response allowed without a request is a refusal for a + /// correlation that never opened one: a request refused before its + /// attempt was recorded. + #[allow(dead_code)] // Not every integration target reads audit entries. + pub fn assert_every_audit_request_answered_once(&self) { + let entries = self.audit.entries(); + let mut open = std::collections::BTreeMap::<(String, String), usize>::new(); + for entry in &entries { + let key = ( + entry["schema"].as_str().unwrap_or_default().to_owned(), + entry["correlation"].as_str().unwrap_or_default().to_owned(), + ); + match entry["phase"].as_str() { + Some("request") => *open.entry(key).or_default() += 1, + Some("response") => match open.get_mut(&key) { + Some(pending) if *pending > 0 => *pending -= 1, + None if entry["record"]["phase"] == "refusal" => {} + _ => panic!("a response answers no open request: {entry}\n{entries:#?}"), + }, + other => panic!("an audit entry has no pairing phase {other:?}: {entry}"), + } + } + let unanswered: Vec<_> = open.iter().filter(|(_, pending)| **pending > 0).collect(); + assert!( + unanswered.is_empty(), + "audit requests without a response: {unanswered:?}\n{entries:#?}" + ); + } + /// The record of every audit entry accepted so far. #[allow(dead_code)] // Not every integration target reads audit entries. pub fn audit_records(&self) -> Vec { diff --git a/crates/registry-bregctl/src/lib.rs b/crates/registry-bregctl/src/lib.rs index 51860a4c5f..c8e2b900aa 100644 --- a/crates/registry-bregctl/src/lib.rs +++ b/crates/registry-bregctl/src/lib.rs @@ -2160,6 +2160,10 @@ fn request_retention_failure( "request_retention.mode.retain", "the request retention policy does not permit operator erasure", ), + RequestRetentionCliError::ErasureUnaudited => ( + "request_retention.erasure.unaudited", + "the erasure committed but its audit entry was not recorded; restore the audit destination, then reconcile the erased request against the database", + ), RequestRetentionCliError::AttachmentStorageBindingMismatch => ( "request_retention.attachment_storage.binding_mismatch", "restore the original attachment storage binding and verification policy before retrying; the registry pin, retained content, or deletion tombstones still require them", diff --git a/crates/registry-bregctl/src/request_retention.rs b/crates/registry-bregctl/src/request_retention.rs index 3adb4d312b..f6a906cd86 100644 --- a/crates/registry-bregctl/src/request_retention.rs +++ b/crates/registry-bregctl/src/request_retention.rs @@ -22,6 +22,8 @@ pub(crate) enum RequestRetentionCliError { ActiveDetailPinned, RetainMode, AttachmentStorageBindingMismatch, + /// The erasure committed without its audit entry. + ErasureUnaudited, } #[derive(Clone, Debug, Eq, PartialEq, Serialize)] @@ -151,6 +153,7 @@ fn map_error(error: RequestRetentionError) -> RequestRetentionCliError { RequestRetentionError::AttachmentStorageBindingMismatch => { RequestRetentionCliError::AttachmentStorageBindingMismatch } + RequestRetentionError::ErasureUnaudited => RequestRetentionCliError::ErasureUnaudited, RequestRetentionError::ActiveProposalRequiresRebase | RequestRetentionError::Unavailable => RequestRetentionCliError::Operator, } @@ -166,6 +169,14 @@ fn operator_runtime() -> Result, } impl std::fmt::Debug for CaseworkAudit { @@ -45,6 +56,8 @@ impl CaseworkAudit { Self { writer, identifiers, + #[cfg(any(test, feature = "postgres-test"))] + lose_acknowledgment: std::sync::Arc::default(), } } @@ -62,21 +75,23 @@ impl CaseworkAudit { .and_then(Value::as_str) .ok_or(StoreError::Corrupt)? .to_owned(); - let correlation = Uuid::new_v4().to_string(); - self.writer - .append(AuditEntry::request( + let unfinished = self.minimized(json!({"event": event, "outcome": "unfinished"}))?; + let request = self + .writer + .begin( CASEWORK_AUDIT_SCHEMA, - correlation.clone(), + Uuid::new_v4().to_string(), record, - )) + unfinished, + ) .await .map_err(|_| StoreError::AuditUnavailable)?; Ok(AuditOperation { audit: self.clone(), - correlation, - request_event: Some(event), + pairing: Pairing::Requested { request, event }, outcome: None, responses: Vec::new(), + read_back: None, }) } @@ -90,10 +105,12 @@ impl CaseworkAudit { } Ok(AuditOperation { audit: self.clone(), - correlation: Uuid::new_v4().to_string(), - request_event: None, + pairing: Pairing::Background { + correlation: Uuid::new_v4().to_string(), + }, outcome: None, responses: Vec::new(), + read_back: None, }) } @@ -106,11 +123,25 @@ impl CaseworkAudit { fn minimized(&self, record: Value) -> Result { published_audit_record(record, &self.identifiers).map_err(|()| StoreError::Corrupt) } + + /// Fail a commit that took effect, as a connection lost before its + /// acknowledgment arrived does, when the test switch asks for it. + #[cfg(any(test, feature = "postgres-test"))] + fn acknowledged(&self) -> Result<(), StoreError> { + if self + .lose_acknowledgment + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err(StoreError::Unavailable); + } + Ok(()) + } } /// `identifiers` is a JSON object of the identifiers the request names, keyed /// by the record field that carries them (`itemId`, `grantId`, `teamId`, -/// `queueId`); minimization keeps only their keyed pseudonyms. +/// `queueId`, `reviewRequestId`); minimization keeps only their keyed +/// pseudonyms. /// /// The `request` fields an audited operation names before it opens its /// transaction: the event it performs, the caller's profile and pseudonymized @@ -161,19 +192,47 @@ impl AuditOutcome { /// One audited operation whose `request` entry was accepted. It collects the /// minimized `response` records its transaction produces and appends them -/// once the transaction commits. +/// once the transaction commits. Dropped before then, a caller-requested +/// operation appends its `unfinished` response entry. #[must_use = "an audited operation appends its response entries only when completed"] pub(crate) struct AuditOperation { audit: CaseworkAudit, - correlation: String, - /// The event of the accepted `request` entry; absent for background work, - /// which writes no `request` entry. - request_event: Option, + pairing: Pairing, outcome: Option, responses: Vec<(Uuid, Value)>, + /// The pool a commit whose acknowledgment never arrived reads its + /// transaction's status back through. + read_back: Option, +} + +/// What reading back an unacknowledged commit found. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum ReadBack { + Committed, + RolledBack, + Unknown, +} + +/// How an operation's `response` entries are correlated. +enum Pairing { + /// A caller-requested operation: its accepted `request` entry, which owes + /// the `response` entries, and the event that entry names. + Requested { + request: AuditRequest, + event: String, + }, + /// Background work no caller requested, which writes no `request` entry. + Background { correlation: String }, } impl AuditOperation { + /// Read the outcome of a commit whose acknowledgment never arrived + /// through `pool`, on a connection of its own. + pub(crate) fn with_read_back(mut self, pool: deadpool_postgres::Pool) -> Self { + self.read_back = Some(pool); + self + } + /// Name the terminal outcome of a caller-requested operation that may /// record no domain event. It is appended as the operation's `response` /// entry only when no domain event was recorded; background work ignores @@ -186,7 +245,10 @@ impl AuditOperation { /// nor a terminal outcome to append, since its result would leave without /// an accepted `response` entry. fn ensure_terminal(&self) -> Result<(), StoreError> { - if self.request_event.is_some() && self.responses.is_empty() && self.outcome.is_none() { + if matches!(self.pairing, Pairing::Requested { .. }) + && self.responses.is_empty() + && self.outcome.is_none() + { tracing::error!( "a Casework audited operation has no response entry to append; its result is withheld" ); @@ -249,47 +311,107 @@ impl AuditOperation { /// append one `response` entry per recorded event, or the terminal /// outcome when none was recorded. A caller-requested operation with /// neither is refused before `transaction` commits. + /// + /// A commit that fails may still have taken effect, as when the + /// connection is lost after `COMMIT` reached the database. The + /// transaction's status is then read back on another connection: a + /// committed change is answered and appended like any other, and one + /// that rolled back or whose status cannot be read returns the commit + /// error, so the dropped operation appends its unfinished response. pub(crate) async fn commit( mut self, transaction: deadpool_postgres::Transaction<'_>, ) -> Result<(), StoreError> { self.collect_task_invalidations(&transaction).await?; self.ensure_terminal()?; - transaction.commit().await?; + let transaction_id: String = transaction + .query_one("SELECT pg_current_xact_id()::text", &[]) + .await? + .get(0); + let committed = transaction.commit().await.map_err(StoreError::from); + #[cfg(any(test, feature = "postgres-test"))] + let committed = committed.and_then(|()| self.audit.acknowledged()); + if let Err(error) = committed { + match self.read_back(&transaction_id).await { + ReadBack::Committed => tracing::warn!( + "a Casework commit was not acknowledged but took effect; its response entries are appended" + ), + ReadBack::RolledBack => return Err(error), + ReadBack::Unknown => { + tracing::error!( + "a Casework commit was not acknowledged and its outcome could not be read back; its response entry is unfinished" + ); + return Err(error); + } + } + } self.complete().await } + /// Read whether the transaction `transaction_id` committed, on a + /// connection outside any transaction. It writes nothing. + async fn read_back(&self, transaction_id: &str) -> ReadBack { + let Some(pool) = &self.read_back else { + return ReadBack::Unknown; + }; + let Ok(client) = pool.get().await else { + return ReadBack::Unknown; + }; + let status = client + .query_one("SELECT pg_xact_status($1::text::xid8)", &[&transaction_id]) + .await + .map(|row| row.get::<_, Option>(0)); + match status.as_ref().map(|status| status.as_deref()) { + Ok(Some("committed")) => ReadBack::Committed, + Ok(Some("aborted")) => ReadBack::RolledBack, + // Still in progress, too old to report, or unreadable. + _ => ReadBack::Unknown, + } + } + /// Append one `response` entry per recorded event, or one naming the /// terminal outcome of a caller-requested operation that recorded none. /// The caller's transaction has committed, so a refusal here reports the /// destination unavailable while the committed change stays in place. + /// The entries are written in one task that outlives a canceled caller, + /// so a caller that stops waiting cannot leave part of them unwritten. pub(crate) async fn complete(self) -> Result<(), StoreError> { self.ensure_terminal()?; - let mut records: Vec = self - .responses - .into_iter() - .map(|(_, record)| record) - .collect(); + let Self { + audit, + mut pairing, + outcome, + responses, + .. + } = self; + let mut records: Vec = responses.into_iter().map(|(_, record)| record).collect(); if records.is_empty() { - if let (Some(event), Some(outcome)) = (&self.request_event, self.outcome) { - records.push( - self.audit - .minimized(json!({"event": event, "outcome": outcome.as_str()}))?, - ); + if let (Pairing::Requested { event, .. }, Some(outcome)) = (&pairing, outcome) { + records + .push(audit.minimized(json!({"event": event, "outcome": outcome.as_str()}))?); } } - for record in records { - self.audit - .writer - .append(AuditEntry::response( - CASEWORK_AUDIT_SCHEMA, - self.correlation.clone(), - record, - )) - .await + let writer = audit.writer; + tokio::spawn(async move { + for record in records { + match &mut pairing { + Pairing::Requested { request, .. } => request.respond(record).await, + Pairing::Background { correlation } => { + writer + .append(AuditEntry::response( + CASEWORK_AUDIT_SCHEMA, + correlation.clone(), + record, + )) + .await + } + } .map_err(|_| StoreError::AuditUnavailable)?; - } - Ok(()) + } + Ok(()) + }) + .await + .map_err(|_| StoreError::AuditUnavailable)? } } @@ -317,6 +439,7 @@ fn published_audit_record(record: Value, identifiers: &AuditKeyHasher) -> Result ("grantId", "grantPseudonym"), ("teamId", "teamPseudonym"), ("queueId", "queuePseudonym"), + ("reviewRequestId", "reviewRequestPseudonym"), ] { if let Some(value) = raw.get(field) { let value = value.as_str().ok_or(())?; @@ -372,7 +495,8 @@ fn audit_record_with_event_id(event_id: Uuid, mut record: Value) -> Result, accepted_lines: Option, + /// The writer recording here, so a read can wait for the entries it + /// writes when a request handle is dropped. + writer: Option, + /// The switch the audit recording here reads before it treats a + /// commit as acknowledged. + lose_acknowledgment: Arc, } - /// The lines a test audit destination accepted, and a switch that makes - /// it refuse every line past a count. + /// A switch that holds the write of one line until it is released, kept + /// apart from the accepted lines so a held write blocks no reader. + #[derive(Default)] + struct Gate { + state: Mutex, + changed: Condvar, + } + + #[derive(Default)] + struct GateState { + /// The index of the line whose write is held. + hold_at: Option, + /// Whether that write has started and is waiting. + holding: bool, + } + + /// The lines a test audit destination accepted, a switch that makes it + /// refuse every line past a count, and one that holds a line's write. #[derive(Clone, Default)] - pub struct AuditCapture(Arc>); + pub struct AuditCapture(Arc>, Arc); impl AuditCapture { /// Every accepted entry, parsed, in write order. #[must_use] pub fn entries(&self) -> Vec { + let writer = self.0.lock().expect("audit capture").writer.clone(); + if let Some(writer) = writer { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -435,10 +585,66 @@ mod capture { pub fn refuse_after(&self, lines: usize) { self.0.lock().expect("audit capture").accepted_lines = Some(lines); } + + /// Hold the write of the line at index `line` until [`Self::release`]. + pub fn hold_line(&self, line: usize) { + self.1.state.lock().expect("audit gate").hold_at = Some(line); + } + + /// Wait until the held line's write has started, and report whether + /// it did within `timeout`. + #[must_use] + pub fn wait_until_held(&self, timeout: Duration) -> bool { + let state = self.1.state.lock().expect("audit gate"); + let (state, _) = self + .1 + .changed + .wait_timeout_while(state, timeout, |state| !state.holding) + .expect("audit gate"); + state.holding + } + + /// Report the next commit of an audited operation as unacknowledged + /// after it took effect, as a connection lost during `COMMIT` does. + pub fn lose_next_commit_acknowledgment(&self) { + self.0 + .lock() + .expect("audit capture") + .lose_acknowledgment + .store(true, std::sync::atomic::Ordering::SeqCst); + } + + /// Let a held write proceed. + pub fn release(&self) { + let mut state = self.1.state.lock().expect("audit gate"); + state.hold_at = None; + state.holding = false; + self.1.changed.notify_all(); + } + + /// Wait while a gate holds the write of line `index`. + fn pass_gate(&self, index: usize) { + let mut state = self.1.state.lock().expect("audit gate"); + if state.hold_at != Some(index) { + return; + } + state.holding = true; + self.1.changed.notify_all(); + let _released = self + .1 + .changed + .wait_while(state, |state| state.hold_at == Some(index)) + .expect("audit gate"); + } } impl Write for AuditCapture { fn write(&mut self, bytes: &[u8]) -> io::Result { + let index = { + let state = self.0.lock().expect("audit capture"); + state.bytes.iter().filter(|byte| **byte == b'\n').count() + }; + self.pass_gate(index); let mut state = self.0.lock().expect("audit capture"); let written = state.bytes.iter().filter(|byte| **byte == b'\n').count(); if state.accepted_lines.is_some_and(|limit| written >= limit) { @@ -459,10 +665,13 @@ mod capture { pub fn capture() -> (Self, AuditCapture) { let capture = AuditCapture::default(); let writer = AuditWriter::from_line_sink(Box::new(capture.clone())); - ( - Self::new(writer, AuditKeyHasher::unkeyed_dev_only()), - capture, - ) + let audit = Self::new(writer.clone(), AuditKeyHasher::unkeyed_dev_only()); + { + let mut state = capture.0.lock().expect("audit capture"); + state.writer = Some(writer); + state.lose_acknowledgment = Arc::clone(&audit.lose_acknowledgment); + } + (audit, capture) } } } @@ -472,6 +681,8 @@ pub use capture::AuditCapture; #[cfg(test)] mod tests { + use std::time::Duration; + use registry_casework_core::{ActorContext, CaseworkRole, IssuerPrincipal}; use super::*; @@ -545,6 +756,55 @@ mod tests { } } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_canceled_completion_still_writes_every_response_entry() { + let (audit, capture) = CaseworkAudit::capture(); + let mut operation = audit + .begin(request_record("claimed", None, "officer", json!({}))) + .await + .unwrap(); + for event in [ + "casework.claimed", + "casework.task_invalidated", + "casework.task_invalidated", + ] { + operation + .record( + Uuid::new_v4(), + json!({"event": event, "profileId": "officer"}), + ) + .unwrap(); + } + // Hold the first response entry's write, then cancel the caller + // while it waits. + capture.hold_line(1); + let completion = tokio::spawn(operation.complete()); + assert!(capture.wait_until_held(Duration::from_secs(10))); + completion.abort(); + assert!(completion.await.unwrap_err().is_cancelled()); + capture.release(); + + let deadline = std::time::Instant::now() + Duration::from_secs(5); + let mut entries = capture.entries(); + while entries.len() < 4 && std::time::Instant::now() < deadline { + tokio::time::sleep(Duration::from_millis(10)).await; + entries = capture.entries(); + } + let phases: Vec<_> = entries.iter().map(|entry| entry["phase"].clone()).collect(); + assert_eq!( + phases, + ["request", "response", "response", "response"], + "{entries:?}" + ); + let correlation = &entries[0]["correlation"]; + assert!(entries + .iter() + .all(|entry| &entry["correlation"] == correlation)); + assert!(entries + .iter() + .all(|entry| entry["record"].get("outcome").is_none())); + } + #[tokio::test] async fn a_refused_request_entry_starts_no_operation() { let (audit, capture) = CaseworkAudit::capture(); @@ -590,9 +850,17 @@ mod tests { operation.complete().await, Err(StoreError::AuditUnavailable) )); + // The result is withheld, and the request entry is still paired: the + // operation writes its unfinished outcome as the response. let entries = capture.entries(); - assert_eq!(entries.len(), 1, "only the request entry was written"); + assert_eq!(entries.len(), 2, "{entries:?}"); assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + json!({"event": "casework.task_claimed", "outcome": "unfinished"}) + ); } #[tokio::test] diff --git a/crates/registry-casework/src/review.rs b/crates/registry-casework/src/review.rs index 7850557d49..5d2bcd93a2 100644 --- a/crates/registry-casework/src/review.rs +++ b/crates/registry-casework/src/review.rs @@ -3742,6 +3742,14 @@ impl PostgresStore { request: ReviewNoteRequest, idempotency_key: &str, ) -> Result { + let mut audit = self + .begin_audit(crate::audit::request_record( + "review_note_added", + Some(actor), + &actor.profile_id, + json!({"reviewRequestId": request_id}), + )) + .await?; if request.note.trim().is_empty() || request.note.len() > 2_000 || request.note.chars().any(char::is_control) @@ -3775,7 +3783,8 @@ impl PostgresStore { ) .await? { - transaction.commit().await?; + audit.record_outcome(crate::audit::AuditOutcome::Replayed); + audit.commit(transaction).await?; return serde_json::from_value(response).map_err(ReviewRuntimeError::from); } let entry = ReviewHistoryEntry { @@ -3819,7 +3828,21 @@ impl PostgresStore { &serde_json::to_value(&entry)?, ) .await?; - transaction.commit().await?; + // The note's text and audience stay in the review history; the audit + // record names only who added a note and which history event it is. + audit.record( + entry.event_id, + json!({ + "event": "casework.review_note_added", + "eventId": entry.event_id, + "actor": { + "issuer": actor.principal.issuer, + "subject": actor.principal.subject, + }, + "profileId": actor.profile_id, + }), + )?; + audit.commit(transaction).await?; Ok(entry) } diff --git a/crates/registry-casework/src/store.rs b/crates/registry-casework/src/store.rs index 7da7d32675..7478c59123 100644 --- a/crates/registry-casework/src/store.rs +++ b/crates/registry-casework/src/store.rs @@ -319,6 +319,7 @@ impl PostgresStore { .ok_or(StoreError::AuditUnavailable)? .begin_background() .await + .map(|operation| operation.with_read_back(self.pool.clone())) } /// Append the `request` entry of one audited operation. Call it before @@ -332,6 +333,7 @@ impl PostgresStore { .ok_or(StoreError::AuditUnavailable)? .begin(request) .await + .map(|operation| operation.with_read_back(self.pool.clone())) } /// Start an audited operation that a caller requested when `actor` names diff --git a/crates/registry-casework/tests/postgres_transactions.rs b/crates/registry-casework/tests/postgres_transactions.rs index 3b2383fb8d..a82224e0cd 100644 --- a/crates/registry-casework/tests/postgres_transactions.rs +++ b/crates/registry-casework/tests/postgres_transactions.rs @@ -1743,6 +1743,27 @@ async fn a_resubmitted_proposal_supersedes_the_earlier_application_item() { assert_eq!(current.state, OccurrenceState::WaitingApplication); } +/// The correlations of request entries no response entry answers. +fn unpaired_requests(entries: &[serde_json::Value]) -> Vec { + entries + .iter() + .filter(|entry| entry["phase"] == "request") + .filter(|request| { + !entries.iter().any(|entry| { + entry["phase"] == "response" + && entry["schema"] == request["schema"] + && entry["correlation"] == request["correlation"] + }) + }) + .map(|request| { + request["correlation"] + .as_str() + .unwrap_or_default() + .to_owned() + }) + .collect() +} + const SETTLEMENT_REASON: &str = "The source refused the saved evidence version; the registrar confirmed no change was made."; const SETTLEMENT_DECIDED_BY: &str = "Registrar duty officer, ticket OPS-4411"; @@ -1762,8 +1783,20 @@ struct SettlementFixture { binding_reference: String, } -async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFixture { - let (store, client, _schema) = isolated_schema(prefix).await; +/// One open item in a schema of its own, audited into a capture, and the +/// staff member who may claim it. +struct OpenItemFixture { + store: PostgresStore, + audit: registry_casework::AuditCapture, + client: tokio_postgres::Client, + schema: String, + holder: ActorContext, + item_id: uuid::Uuid, + revision: i64, +} + +async fn open_item_fixture(prefix: &str) -> OpenItemFixture { + let (store, client, schema) = isolated_schema(prefix).await; let (audit, audit_capture) = registry_casework::CaseworkAudit::capture(); let store = store.with_audit(audit); store.migrate().await.expect("migrate"); @@ -1806,8 +1839,29 @@ async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFix .await .expect("initial observation") .expect("item opened"); + OpenItemFixture { + store, + audit: audit_capture, + client, + schema, + holder, + item_id: item.item_id, + revision: item.revision, + } +} + +async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFixture { + let OpenItemFixture { + store, + audit, + client, + holder, + item_id, + revision, + .. + } = open_item_fixture(prefix).await; let claimed = store - .claim(&holder, item.item_id, item.revision, "claim-settlement") + .claim(&holder, item_id, revision, "claim-settlement") .await .expect("holder claims the item"); let prepared = PreparedSourceAttempt { @@ -1838,7 +1892,7 @@ async fn settlement_fixture(prefix: &str, mark_uncertain: bool) -> SettlementFix } SettlementFixture { store, - audit: audit_capture, + audit, client, holder, item_id: claimed.item_id, @@ -1890,12 +1944,22 @@ impl SettlementFixture { .await .expect("settlement snapshot") .get(0); + // A refusal pairs its request entry with an `unfinished` response; + // only a response recording a committed change counts as a write. snapshot["auditResponses"] = serde_json::json!(self .audit .entries() .iter() - .filter(|entry| entry["phase"] == "response") + .filter(|entry| { + entry["phase"] == "response" && entry["record"]["outcome"] != "unfinished" + }) .count()); + let unpaired = unpaired_requests(&self.audit.entries()); + assert!( + unpaired.is_empty(), + "unpaired request entries: {unpaired:?}" + ); + snapshot["unpairedRequests"] = serde_json::json!(unpaired); snapshot } @@ -2155,6 +2219,190 @@ async fn a_refused_audit_response_reports_unavailable_after_the_settlement_commi ); } +#[tokio::test] +async fn a_refusal_after_the_request_entry_pairs_it_with_an_unfinished_response() { + let fixture = settlement_fixture("refusal_pairs_request", true).await; + let written = fixture.audit.entries().len(); + let missing = fixture + .store + .claim(&fixture.holder, uuid::Uuid::new_v4(), 1, "claim-missing") + .await; + assert!(matches!(missing, Err(StoreError::NotFound)), "{missing:?}"); + let item = fixture.store.item(fixture.item_id).await.expect("item"); + let again = fixture + .store + .claim(&fixture.holder, item.item_id, item.revision, "claim-again") + .await; + assert!( + matches!(again, Err(StoreError::AlreadyClaimed)), + "{again:?}" + ); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 4, "{entries:#?}"); + for pair in entries.chunks(2) { + assert_eq!(pair[0]["phase"], "request"); + assert_eq!(pair[1]["phase"], "response"); + assert_eq!(pair[1]["schema"], pair[0]["schema"]); + assert_eq!(pair[1]["correlation"], pair[0]["correlation"]); + assert_eq!( + pair[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); + } +} + +/// A claim whose `COMMIT` took effect but whose acknowledgment never arrived +/// is read back as committed: the caller gets the claim, and its response +/// entry records the claim rather than an unfinished outcome. +#[tokio::test] +async fn a_claim_whose_commit_acknowledgment_is_lost_is_read_back_as_committed() { + let fixture = open_item_fixture("claim_lost_ack").await; + let written = fixture.audit.entries().len(); + fixture.audit.lose_next_commit_acknowledgment(); + let claimed = fixture + .store + .claim( + &fixture.holder, + fixture.item_id, + fixture.revision, + "claim-lost-ack", + ) + .await + .expect("the committed claim is answered"); + assert_eq!(claimed.revision, fixture.revision + 1); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!(current.revision, claimed.revision); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!(entries[1]["record"]["event"], "casework.claimed"); + assert!( + entries[1]["record"].get("outcome").is_none(), + "{entries:#?}" + ); + assert!(unpaired_requests(&fixture.audit.entries()).is_empty()); +} + +/// A claim whose `COMMIT` itself is refused rolls back, and the read-back +/// finds it rolled back: the caller gets the error and the request entry is +/// paired with an unfinished response, never with the claim. +#[tokio::test] +async fn a_claim_refused_at_commit_is_read_back_as_not_committed() { + let fixture = open_item_fixture("claim_refused_at_commit").await; + fixture + .client + .batch_execute( + "CREATE FUNCTION refuse_at_commit() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'refused at commit'; END $$; + CREATE CONSTRAINT TRIGGER refuse_claim_at_commit AFTER UPDATE ON casework_items + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION refuse_at_commit();", + ) + .await + .expect("install a trigger that refuses the claim at COMMIT"); + let written = fixture.audit.entries().len(); + let refused = fixture + .store + .claim( + &fixture.holder, + fixture.item_id, + fixture.revision, + "claim-refused-at-commit", + ) + .await; + assert!( + matches!(refused, Err(StoreError::Postgres(_))), + "{refused:?}" + ); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!(current.revision, fixture.revision, "the claim rolled back"); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); +} + +/// A claim whose future is dropped while it waits inside its transaction +/// pairs its request entry with exactly one unfinished response and changes +/// nothing. +#[tokio::test] +async fn a_claim_dropped_inside_its_transaction_writes_one_unfinished_response() { + let fixture = open_item_fixture("claim_dropped").await; + let written = fixture.audit.entries().len(); + let mut locker = connect_scoped(&fixture.schema).await; + let lock = locker.transaction().await.expect("lock transaction"); + lock.execute( + "SELECT 1 FROM casework_items WHERE item_id=$1 FOR UPDATE", + &[&fixture.item_id], + ) + .await + .expect("hold the item row"); + let locker_pid: i32 = lock + .query_one("SELECT pg_backend_pid()", &[]) + .await + .expect("lock holder pid") + .get(0); + + let store = fixture.store.clone(); + let holder = fixture.holder.clone(); + let (item_id, revision) = (fixture.item_id, fixture.revision); + let claim = tokio::spawn(async move { + store + .claim(&holder, item_id, revision, "claim-dropped") + .await + }); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + let waiting: i64 = fixture + .client + .query_one( + "SELECT count(*) FROM pg_stat_activity WHERE $1=ANY(pg_blocking_pids(pid))", + &[&locker_pid], + ) + .await + .expect("read lock waits") + .get(0); + if waiting > 0 { + break; + } + assert!( + std::time::Instant::now() < deadline, + "the claim never waited on the item row" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + claim.abort(); + assert!(claim + .await + .expect_err("the claim was dropped") + .is_cancelled()); + lock.rollback().await.expect("release the item row"); + + let entries = fixture.audit.entries()[written..].to_vec(); + assert_eq!(entries.len(), 2, "{entries:#?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + serde_json::json!({"event": "casework.claimed", "outcome": "unfinished"}) + ); + let current = fixture.store.item(fixture.item_id).await.expect("item"); + assert_eq!( + current.revision, fixture.revision, + "the dropped claim changed nothing" + ); +} + #[tokio::test] async fn a_live_execution_lease_refuses_settlement_and_writes_nothing() { let fixture = settlement_fixture("settle_live_lease", true).await; diff --git a/crates/registry-casework/tests/review_postgres.rs b/crates/registry-casework/tests/review_postgres.rs index 9840bcd04c..cf7d059eab 100644 --- a/crates/registry-casework/tests/review_postgres.rs +++ b/crates/registry-casework/tests/review_postgres.rs @@ -3318,6 +3318,98 @@ async fn review_cancellation_is_audited_after_the_cancel_commits() { assert!(!audit_text.contains(&created.accepted.request_id.to_string())); } +#[tokio::test] +async fn review_notes_are_audited_without_their_text() { + let fixture = fixture().await; + let (service, audit) = service_with_audit(&fixture, project("1")); + let created = service + .create_review_request( + &fixture.producer, + request("note-audit", "producer-ref-note-audit"), + "create-note-audit", + ) + .await + .expect("create note audit review"); + let request_id = created.accepted.request_id; + let note = |text: &str| ReviewNoteRequest { + audience: ReviewHistoryAudience::Requester, + note: text.to_owned(), + }; + let added = service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note("a note that stays out of the audit"), + "note-audit", + ) + .await + .expect("add note"); + let record = audited_response( + &audit, + "review_note_added", + "eventId", + &added.event_id.to_string(), + ); + let audit_text = serde_json::to_string(&audit.entries()).expect("audit JSON"); + assert!(!audit_text.contains("stays out of the audit")); + assert!(!audit_text.contains(&request_id.to_string())); + assert!(record.get("principalPseudonym").is_some()); + // The request entry names the review request only by its pseudonym. + let requested: Vec<_> = audit + .entries() + .into_iter() + .filter(|entry| { + entry["phase"] == "request" && entry["record"]["event"] == "casework.review_note_added" + }) + .collect(); + assert_eq!(requested.len(), 1, "{requested:?}"); + assert_eq!( + requested[0]["record"]["reviewRequestPseudonym"], + audit.reference("reviewRequestId", &request_id.to_string()) + ); + + let (replay_service, replay_audit) = service_with_audit(&fixture, project("1")); + replay_service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note("a note that stays out of the audit"), + "note-audit", + ) + .await + .expect("replay note"); + assert_replay_audited(&replay_audit, "review_note_added"); + + // A refused note still pairs the request entry it wrote. + let (refused_service, refused_audit) = service_with_audit(&fixture, project("1")); + assert!(matches!( + refused_service + .add_review_note( + &fixture.producer, + request_id, + None, + "producer-token", + note(" "), + "note-audit-refused", + ) + .await, + Err(ReviewRuntimeError::Invalid) + )); + let entries = refused_audit.entries(); + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!( + entries[1]["record"], + json!({"event": "casework.review_note_added", "outcome": "unfinished"}) + ); +} + #[tokio::test] async fn review_task_ownership_transitions_are_audited() { let fixture = fixture().await; diff --git a/crates/registry-platform-audit/src/lib.rs b/crates/registry-platform-audit/src/lib.rs index 6b6c87708c..a5f4a04c38 100644 --- a/crates/registry-platform-audit/src/lib.rs +++ b/crates/registry-platform-audit/src/lib.rs @@ -5,6 +5,9 @@ //! file or to stdout: a `request` entry before protected I/O and a //! `response` entry with the outcome, sharing one correlation. It fails //! closed: an entry the destination does not accept is an error. +//! [`AuditWriter::begin`] writes the `request` entry and returns an +//! [`AuditRequest`] that owes the `response`: dropped unanswered, it writes +//! the product's `unfinished` record, so no request entry stays unpaired. //! - [`AuditProfile`] and [`AuditKeyHasher`] derive keyed, domain-separated //! references so audit records never carry raw identifiers. //! - [`redact`] minimizes query strings, email addresses, and phone numbers. @@ -25,7 +28,7 @@ mod writer; #[cfg(unix)] pub use writer::{ AuditDestination, AuditDestinationError, AuditDestinationKind, AuditEntry, AuditPhase, - AuditUnavailable, AuditUnavailableReason, AuditWriter, FileDestination, + AuditRequest, AuditUnavailable, AuditUnavailableReason, AuditWriter, FileDestination, DEFAULT_AUDIT_RETAIN_DAYS, DEFAULT_AUDIT_ROTATE_BYTES, MAX_AUDIT_RETAIN_DAYS, MIN_AUDIT_ROTATE_BYTES, }; diff --git a/crates/registry-platform-audit/src/writer.rs b/crates/registry-platform-audit/src/writer.rs index 6f984aa6ae..88843ff19c 100644 --- a/crates/registry-platform-audit/src/writer.rs +++ b/crates/registry-platform-audit/src/writer.rs @@ -17,7 +17,7 @@ use std::{ os::unix::fs::{DirBuilderExt, FileExt, MetadataExt, OpenOptionsExt, PermissionsExt}, path::{Path, PathBuf}, sync::{ - atomic::{AtomicBool, AtomicU64, Ordering}, + atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}, Arc, Mutex as StdMutex, }, time::{Duration, SystemTime}, @@ -44,6 +44,7 @@ const MAX_ENTRY_BYTES: usize = 1024 * 1024; const MAX_SCHEMA_BYTES: usize = 128; const MAX_CORRELATION_BYTES: usize = 256; const SEGMENT_SEQUENCE_DIGITS: usize = 8; +const DETACHED_LOCK_ATTEMPTS: usize = 1024; const SECONDS_PER_DAY: u64 = 24 * 60 * 60; const TIME_FORMAT: &[FormatItem<'static>] = format_description!("[year]-[month]-[day]T[hour]:[minute]:[second].[subsecond digits:3]Z"); @@ -525,11 +526,146 @@ impl AuditDestination { #[derive(Clone)] pub struct AuditWriter { inner: Arc, + open: Arc, +} + +/// The schema and correlation a request entry and its responses share. +type RequestKey = (String, String); + +/// One request entry still owed a response: whether one was accepted, and +/// the state its handle shares with the responses being written. +#[derive(Clone)] +struct OpenRequest { + answered: Arc, + state: Arc>, +} + +/// The request entries whose [`AuditRequest`] still owes a response, by +/// schema and correlation, oldest first. +#[derive(Default)] +struct OpenRequests(StdMutex>>); + +impl OpenRequests { + fn open(&self, key: RequestKey, request: OpenRequest) { + match self.0.lock() { + Ok(mut open) => open.entry(key).or_default().push(request), + // The request is still owed by its handle, which writes its + // unfinished response; only a response appended elsewhere can + // no longer answer it. + Err(_) => tracing::error!( + "the open audit requests are poisoned; a response appended elsewhere will not answer this request" + ), + } + } + + /// Claim the oldest open request under `key` that no appended response + /// has claimed yet, counting that response as in flight on it, so its + /// handle dropped meanwhile leaves the request to that response. + fn claim(&self, key: &RequestKey) -> Option { + let open = self.0.lock().ok()?; + open.get(key)?.iter().find_map(|request| { + let mut state = request.state.lock().ok()?; + if state.claimed { + return None; + } + state.claimed = true; + state.in_flight += 1; + Some(request.clone()) + }) + } + + /// Close `answered` under `key` once a response to it was accepted. + fn answer(&self, key: &RequestKey, answered: &Arc) { + if let Ok(mut open) = self.0.lock() { + Self::remove(&mut open, key, answered); + } + } + + /// Close `answered` under `key` for its unfinished response, reporting + /// whether it is still owed one: not when a response was accepted, and + /// not when one is in flight on it, since that response settles the + /// request itself. Checked and removed under the lock [`Self::claim`] + /// takes, so a response cannot be claimed after the owner decided it had + /// none. + fn close_unanswered( + &self, + key: &RequestKey, + answered: &Arc, + state: &StdMutex, + ) -> bool { + let Ok(mut open) = self.0.lock() else { + return !answered.load(Ordering::Acquire); + }; + if answered.load(Ordering::Acquire) { + return false; + } + if state.lock().is_ok_and(|state| state.in_flight > 0) { + return false; + } + Self::remove(&mut open, key, answered); + true + } + + fn remove( + open: &mut std::collections::HashMap>, + key: &RequestKey, + answered: &Arc, + ) { + if let Some(waiting) = open.get_mut(key) { + waiting.retain(|candidate| !Arc::ptr_eq(&candidate.answered, answered)); + if waiting.is_empty() { + open.remove(key); + } + } + } +} + +/// One response counted in flight on the request it answers. Dropping it +/// settles that count, whether its write finished or its task was dropped +/// before it could, such as by a runtime shutting down, so the request's +/// handle is never left waiting on a response that will not come. +struct InFlightResponse { + writer: AuditWriter, + key: RequestKey, + answered: Arc, + state: Arc>, + /// The response was appended through [`AuditWriter::append`] and holds + /// the request's claim. + claimed: bool, +} + +impl InFlightResponse { + /// Record that the response was accepted, answering this request and + /// not an older one open under its correlation. + fn accepted(&self) { + self.writer.open.answer(&self.key, &self.answered); + self.answered.store(true, Ordering::Release); + } +} + +impl Drop for InFlightResponse { + fn drop(&mut self) { + let owes_unfinished = self.state.lock().is_ok_and(|mut state| { + if self.claimed { + state.claimed = false; + } + state.in_flight -= 1; + state.in_flight == 0 && state.dropped + }); + if owes_unfinished { + settle_unanswered( + &self.writer, + std::mem::take(&mut self.key), + &self.answered, + &self.state, + ); + } + } } enum WriterInner { File(Arc), - Stream(Arc), + Stream(Arc, DetachedLines), } impl std::fmt::Debug for AuditWriter { @@ -539,7 +675,7 @@ impl std::fmt::Debug for AuditWriter { WriterInner::File(file) => debug .field("destination", &"file") .field("path", &file.file.path), - WriterInner::Stream(_) => debug.field("destination", &"stdout"), + WriterInner::Stream(..) => debug.field("destination", &"stdout"), }; debug.finish_non_exhaustive() } @@ -556,15 +692,12 @@ impl AuditWriter { .map_err(|error| AuditError::Io(io::Error::other(error)))??; WriterInner::File(Arc::new(GroupCommitFile::new(opened))) } - AuditDestination::Stdout => { - WriterInner::Stream(Arc::new(LineStream::new(Box::new(io::stdout())))) - } - AuditDestination::Stderr => { - WriterInner::Stream(Arc::new(LineStream::new(Box::new(io::stderr())))) - } + AuditDestination::Stdout => WriterInner::stream(Box::new(io::stdout())), + AuditDestination::Stderr => WriterInner::stream(Box::new(io::stderr())), }; Ok(Self { inner: Arc::new(inner), + open: Arc::default(), }) } @@ -573,14 +706,50 @@ impl AuditWriter { #[must_use] pub fn from_line_sink(sink: Box) -> Self { Self { - inner: Arc::new(WriterInner::Stream(Arc::new(LineStream::new(sink)))), + inner: Arc::new(WriterInner::stream(sink)), + open: Arc::default(), } } /// Append one entry. For the file destination this returns only after the /// entry's bytes are durable. Canceling the caller does not cancel an /// enqueued file write or the other entries in its group commit. + /// + /// An accepted `response` entry answers the oldest [`AuditRequest`] still + /// open under the same schema and correlation. pub async fn append(&self, entry: AuditEntry) -> Result<(), AuditUnavailable> { + // The write and the bookkeeping it implies run in one task that + // outlives a canceled caller, so an accepted response always answers + // its request, whether or not the caller is still waiting. The + // request is claimed before that task starts, so a handle dropped + // while the response is written leaves the request to it. + let claimed = if entry.phase == AuditPhase::Response { + let key = (entry.schema.clone(), entry.correlation.clone()); + self.open.claim(&key).map(|request| InFlightResponse { + writer: self.clone(), + key, + answered: request.answered, + state: request.state, + claimed: true, + }) + } else { + None + }; + let writer = self.clone(); + tokio::spawn(async move { + let result = writer.write(&entry).await; + if let Some(claimed) = claimed { + if result.is_ok() { + claimed.accepted(); + } + } + result + }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? + } + + async fn write(&self, entry: &AuditEntry) -> Result<(), AuditUnavailable> { let line = entry.to_line()?; match self.inner.as_ref() { WriterInner::File(file) => { @@ -589,7 +758,7 @@ impl AuditWriter { .await .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? } - WriterInner::Stream(stream) => { + WriterInner::Stream(stream, _) => { let stream = Arc::clone(stream); tokio::task::spawn_blocking(move || stream.append(&line)) .await @@ -598,12 +767,109 @@ impl AuditWriter { } } + /// Append the `request` entry of one audited operation and return the + /// [`AuditRequest`] that owes its `response` entry. + /// + /// `request` and `unfinished` are the product's minimized records. The + /// handle writes `unfinished` as the `response` entry if it is dropped + /// before any response is accepted, so an early return, an error, a + /// panic, or a canceled future still pairs the request entry. Every + /// `response` the handle writes carries the request's schema and + /// correlation. A response appended through [`Self::append`] under the + /// same schema and correlation answers it too, so an operation whose + /// outcome is written elsewhere only holds the handle until it returns. + /// + /// That pairing holds while the process runs. A process killed or + /// exited without unwinding, or a runtime shut down while an entry is + /// being written, can leave a request entry without its response. + pub async fn begin( + &self, + schema: impl Into, + correlation: impl Into, + request: Value, + unfinished: Value, + ) -> Result { + let schema = schema.into(); + let correlation = correlation.into(); + // The unfinished response must be writable before the request is: + // a request accepted with a response it could never write would stay + // unpaired. + AuditEntry::response(schema.clone(), correlation.clone(), unfinished.clone()).to_line()?; + // The request is written and its handle registered in one task that + // outlives a canceled caller. A caller that stops waiting drops the + // finished handle with the task's output, which writes the + // unfinished response. + let writer = self.clone(); + tokio::spawn(async move { + writer + .write(&AuditEntry::request( + schema.clone(), + correlation.clone(), + request, + )) + .await?; + let answered = Arc::new(AtomicBool::new(false)); + let state = Arc::new(StdMutex::new(RequestState { + unfinished: Some(unfinished), + in_flight: 0, + claimed: false, + dropped: false, + })); + writer.open.open( + (schema.clone(), correlation.clone()), + OpenRequest { + answered: Arc::clone(&answered), + state: Arc::clone(&state), + }, + ); + Ok(AuditRequest { + writer, + schema, + correlation, + answered, + state, + }) + }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? + } + + /// Block until every response entry a dropped [`AuditRequest`] handed + /// to a stream destination has been written. Those entries are written + /// on a dedicated thread so a drop never blocks the runtime; this is for + /// tests and shutdown paths that read the stream right after a drop. A + /// file destination queues its entries into the group commit instead, + /// and this returns at once for it. + pub fn wait_for_detached_entries(&self) { + if let WriterInner::Stream(_, detached) = self.inner.as_ref() { + detached.wait(); + } + } + + /// Write `entry` without waiting for the destination to accept it. + fn append_detached(&self, entry: &AuditEntry) { + let line = match entry.to_line() { + Ok(line) => line, + Err(_) => { + tracing::error!("an unfinished response entry is malformed and was not written"); + return; + } + }; + match self.inner.as_ref() { + WriterInner::File(file) => file.enqueue_detached(line), + // A dedicated thread writes it, so a drop never blocks a runtime + // thread on a slow stream, and the thread is joined when the + // writer is dropped, so shutdown does not lose it. + WriterInner::Stream(_, detached) => detached.send(line), + } + } + /// Report whether the writer can still accept entries. For the file /// destination this also confirms the writer still owns the active file. pub async fn ready(&self) -> bool { match self.inner.as_ref() { WriterInner::File(file) => file.ready().await, - WriterInner::Stream(stream) => stream.healthy(), + WriterInner::Stream(stream, _) => stream.healthy(), } } @@ -611,7 +877,7 @@ impl AuditWriter { pub fn kind(&self) -> AuditDestinationKind { match self.inner.as_ref() { WriterInner::File(_) => AuditDestinationKind::File, - WriterInner::Stream(_) => AuditDestinationKind::Stdout, + WriterInner::Stream(..) => AuditDestinationKind::Stdout, } } @@ -620,7 +886,7 @@ impl AuditWriter { pub fn path(&self) -> Option<&Path> { match self.inner.as_ref() { WriterInner::File(file) => Some(&file.file.path), - WriterInner::Stream(_) => None, + WriterInner::Stream(..) => None, } } @@ -630,8 +896,241 @@ impl AuditWriter { pub fn durable_writes(&self) -> u64 { match self.inner.as_ref() { WriterInner::File(file) => file.durable_writes.load(Ordering::Relaxed), - WriterInner::Stream(_) => 0, + WriterInner::Stream(..) => 0, + } + } +} + +/// An accepted `request` entry that still owes its `response` entry. +/// +/// [`AuditRequest::respond`] appends a `response` entry and waits for the +/// destination to accept it; an operation may respond more than once. A +/// handle dropped before any response was accepted writes the `unfinished` +/// record given to [`AuditWriter::begin`] as its `response` entry. That write +/// cannot be awaited, so an operation with a known outcome, a refusal +/// included, responds with it instead of relying on the drop. +#[must_use = "an audit request writes its unfinished response entry when dropped"] +pub struct AuditRequest { + writer: AuditWriter, + schema: String, + correlation: String, + /// Set once a response entry under this schema and correlation was + /// accepted. + answered: Arc, + /// The unfinished record and the responses still being written, shared + /// with those writes so the last one to settle owns the drop's duty. + state: Arc>, +} + +struct RequestState { + /// The record written if the request ends unanswered. + unfinished: Option, + /// Responses whose write has started and not yet settled. + in_flight: usize, + /// A response appended through [`AuditWriter::append`] claimed this + /// request and has not settled. + claimed: bool, + /// The handle was dropped while a response was in flight. + dropped: bool, +} + +impl std::fmt::Debug for AuditRequest { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter + .debug_struct("AuditRequest") + .field("schema", &self.schema) + .field("answered", &self.is_answered()) + .finish_non_exhaustive() + } +} + +impl AuditRequest { + /// The correlation its request and response entries share. + #[must_use] + pub fn correlation(&self) -> &str { + &self.correlation + } + + /// Whether a response entry was accepted. + #[must_use] + pub fn is_answered(&self) -> bool { + self.answered.load(Ordering::Acquire) + } + + /// Append one `response` entry. A refused entry leaves the request + /// unanswered, so a later drop still writes the unfinished record. The + /// write and its bookkeeping outlive a canceled caller: a response + /// accepted after the caller stopped waiting still answers the request, + /// and a drop while it is in flight writes the unfinished record only if + /// that response is refused. + pub async fn respond(&mut self, record: Value) -> Result<(), AuditUnavailable> { + if let Ok(mut state) = self.state.lock() { + state.in_flight += 1; } + // A poisoned state skips both the count and its release. + let in_flight = InFlightResponse { + writer: self.writer.clone(), + key: (self.schema.clone(), self.correlation.clone()), + answered: Arc::clone(&self.answered), + state: Arc::clone(&self.state), + claimed: false, + }; + let entry = AuditEntry::response(self.schema.clone(), self.correlation.clone(), record); + tokio::spawn(async move { + let result = in_flight.writer.write(&entry).await; + if result.is_ok() { + in_flight.accepted(); + } + result + }) + .await + .map_err(|_| AuditUnavailable::new(AuditUnavailableReason::Stopped))? + } + + /// Append `record` as the only `response` entry and release the handle. + pub async fn finish(mut self, record: Value) -> Result<(), AuditUnavailable> { + self.respond(record).await + } +} + +/// Write the unfinished record of a request still owed a response. +fn settle_unanswered( + writer: &AuditWriter, + key: RequestKey, + answered: &Arc, + state: &StdMutex, +) { + if !writer.open.close_unanswered(&key, answered, state) { + return; + } + let unfinished = state + .lock() + .ok() + .and_then(|mut state| state.unfinished.take()); + if let Some(unfinished) = unfinished { + writer.append_detached(&AuditEntry::response(key.0, key.1, unfinished)); + } +} + +impl Drop for AuditRequest { + fn drop(&mut self) { + // A response still being written settles the request itself. + let in_flight = self.state.lock().is_ok_and(|mut state| { + state.dropped = true; + state.in_flight > 0 + }); + if in_flight { + return; + } + let key = ( + std::mem::take(&mut self.schema), + std::mem::take(&mut self.correlation), + ); + settle_unanswered(&self.writer, key, &self.answered, &self.state); + } +} + +/// The unfinished response entries dropped requests hand to a stream +/// destination, written in order on one dedicated thread. +struct DetachedLines { + stream: Arc, + sender: StdMutex>>, + thread: StdMutex>>, + /// Lines handed over and not yet written, with a signal on each write. + pending: Arc<(StdMutex, std::sync::Condvar)>, +} + +impl DetachedLines { + fn new(stream: Arc) -> Self { + Self { + stream, + sender: StdMutex::new(None), + thread: StdMutex::new(None), + pending: Arc::new((StdMutex::new(0), std::sync::Condvar::new())), + } + } + + fn send(&self, line: String) { + let Ok(mut sender) = self.sender.lock() else { + tracing::error!("an unfinished response entry was not written"); + return; + }; + if sender.is_none() { + let (lines, received) = std::sync::mpsc::channel::(); + let stream = Arc::clone(&self.stream); + let pending = Arc::clone(&self.pending); + let spawned = std::thread::Builder::new() + .name("audit-detached".to_owned()) + .spawn(move || { + for line in received { + if stream.append(&line).is_err() { + tracing::error!("an unfinished response entry was not accepted"); + } + let (count, written) = &*pending; + if let Ok(mut count) = count.lock() { + *count = count.saturating_sub(1); + } + written.notify_all(); + } + }); + match spawned { + Ok(thread) => { + if let Ok(mut slot) = self.thread.lock() { + *slot = Some(thread); + } + *sender = Some(lines); + } + Err(error) => { + tracing::error!(%error, "an unfinished response entry was not written"); + return; + } + } + } + if let Ok(mut count) = self.pending.0.lock() { + *count += 1; + } + if sender + .as_ref() + .is_some_and(|lines| lines.send(line).is_err()) + { + tracing::error!("an unfinished response entry was not written"); + if let Ok(mut count) = self.pending.0.lock() { + *count = count.saturating_sub(1); + } + } + } + + fn wait(&self) { + let (count, written) = &*self.pending; + let Ok(mut count) = count.lock() else { + return; + }; + while *count > 0 { + count = match written.wait(count) { + Ok(count) => count, + Err(_) => return, + }; + } + } +} + +impl Drop for DetachedLines { + /// Close the queue and let the thread write what is left. + fn drop(&mut self) { + if let Ok(mut sender) = self.sender.lock() { + sender.take(); + } + let thread = self.thread.lock().ok().and_then(|mut thread| thread.take()); + if let Some(thread) = thread { + let _ = thread.join(); + } + } +} + +impl WriterInner { + fn stream(out: Box) -> Self { + let stream = Arc::new(LineStream::new(out)); + Self::Stream(Arc::clone(&stream), DetachedLines::new(stream)) } } @@ -724,6 +1223,12 @@ impl GroupCommitFile { state.enqueued = state.enqueued.saturating_add(1); state.enqueued }; + self.wait_durable(position).await + } + + /// Wait until the line queued at `position` is durable, flushing the + /// queue when no other append is. + async fn wait_durable(&self, position: u64) -> Result<(), AuditUnavailable> { loop { if self.durable.load(Ordering::Acquire) >= position { return Ok(()); @@ -767,6 +1272,107 @@ impl GroupCommitFile { } self.file.ready().await } + + /// Queue `line` without waiting for it to be durable, then flush it on + /// the current runtime. A line queued when no runtime can flush it, or + /// whose flush is canceled at shutdown, is written by the next append or + /// when the last reference to the file is dropped. + fn enqueue_detached(self: &Arc, line: String) { + let runtime = tokio::runtime::Handle::try_current().ok(); + let mut line = Some(line); + // Every holder of the state lock releases it without awaiting, so a + // short wait is enough unless the state is poisoned by a stop. + for _ in 0..DETACHED_LOCK_ATTEMPTS { + if let Ok(mut state) = self.state.try_lock() { + if state.stopped { + tracing::error!( + "audit writer stopped; an unfinished response entry was not written" + ); + return; + } + state.pending.extend(line.take()); + state.enqueued = state.enqueued.saturating_add(1); + break; + } + std::thread::yield_now(); + } + let Some(runtime) = runtime else { + // Outside a runtime nothing can flush the line later, so it is + // queued under a blocking wait for the lock, which no holder + // keeps across an await; the file's drop or the next group + // commit writes it. + if let Some(line) = line { + let mut state = self.state.blocking_lock(); + if state.stopped { + tracing::error!( + "audit writer stopped; an unfinished response entry was not written" + ); + return; + } + state.pending.push(line); + state.enqueued = state.enqueued.saturating_add(1); + } + return; + }; + let file = Arc::clone(self); + // A line not queued yet is lost if the runtime drops this task + // before it runs, such as at shutdown; that loss is reported. + let unqueued = line.map(|line| UnqueuedLine(Some(line))); + runtime.spawn(async move { + let result = match unqueued { + Some(mut unqueued) => { + let position = { + let mut state = file.state.lock().await; + if state.stopped { + Err(AuditUnavailable::new(AuditUnavailableReason::Stopped)) + } else { + state.pending.extend(unqueued.0.take()); + state.enqueued = state.enqueued.saturating_add(1); + Ok(state.enqueued) + } + }; + match position { + Ok(position) => file.wait_durable(position).await, + Err(error) => Err(error), + } + } + None => { + let _writer = file.flush.lock().await; + file.flush_once().await + } + }; + if result.is_err() { + tracing::error!("an unfinished response entry was not accepted"); + } + }); + } +} + +/// A detached line not yet handed to the group commit, which reports its +/// loss if dropped still holding it. +struct UnqueuedLine(Option); + +impl Drop for UnqueuedLine { + fn drop(&mut self) { + if self.0.is_some() { + tracing::error!("an unfinished response entry was dropped before it was written"); + } + } +} + +impl Drop for GroupCommitFile { + /// Write the lines still queued, such as an unfinished response whose + /// flush was canceled when the runtime shut down. + fn drop(&mut self) { + let state = self.state.get_mut(); + if state.stopped || state.pending.is_empty() { + return; + } + let lines = std::mem::take(&mut state.pending); + if let Err(error) = self.file.write_lines_blocking(lines) { + tracing::error!(%error, "queued audit entries were not written at shutdown"); + } + } } /// A single-writer JSON Lines file with online size rotation and age-based @@ -783,6 +1389,7 @@ struct SegmentedFile { retain: Duration, state: tokio::sync::Mutex, healthy: AtomicBool, + in_flight: Arc, lock_fingerprint: FileFingerprint, writer_lock: File, #[cfg(test)] @@ -864,6 +1471,7 @@ impl SegmentedFile { next_sequence, }), healthy: AtomicBool::new(true), + in_flight: Arc::new(AtomicUsize::new(0)), lock_fingerprint, writer_lock, #[cfg(test)] @@ -906,7 +1514,49 @@ impl SegmentedFile { ))); } let mut state = self.state.lock().await; - let request = AppendRequest { + let request = self.append_request(&state, lines)?; + let in_flight = InFlightWrite::start(&self.in_flight); + let outcome = tokio::task::spawn_blocking(move || { + let _in_flight = in_flight; + request.run() + }) + .await + .map_err(|error| AuditError::Io(io::Error::other(error))) + .and_then(|result| result); + self.settle(&mut state, outcome) + } + + /// Write `lines` on the calling thread, outside any runtime. The owner of + /// the last reference calls it, so no other write can hold the state. + fn write_lines_blocking(&self, lines: Vec) -> Result<(), AuditError> { + if !self.healthy.load(Ordering::Acquire) { + return Err(AuditError::Io(io::Error::other( + "audit writer stopped after a failed write", + ))); + } + // A canceled caller can leave its blocking write running; writing + // beside it could interleave two runs of lines in the active file. + if self.in_flight.load(Ordering::Acquire) != 0 { + return Err(AuditError::Io(io::Error::other( + "an earlier audit write is still in flight", + ))); + } + let mut state = self + .state + .try_lock() + .map_err(|_| AuditError::Io(io::Error::other("audit file state is held")))?; + let outcome = self + .append_request(&state, lines) + .and_then(AppendRequest::run); + self.settle(&mut state, outcome) + } + + fn append_request( + &self, + state: &FileState, + lines: Vec, + ) -> Result { + Ok(AppendRequest { path: self.path.clone(), rotate_bytes: self.rotate_bytes, retain: self.retain, @@ -919,11 +1569,14 @@ impl SegmentedFile { next_sequence: state.next_sequence, #[cfg(test)] sync_hook: self.sync_hook.clone(), - }; - let outcome = tokio::task::spawn_blocking(move || request.run()) - .await - .map_err(|error| AuditError::Io(io::Error::other(error))) - .and_then(|result| result); + }) + } + + fn settle( + &self, + state: &mut FileState, + outcome: Result, + ) -> Result<(), AuditError> { match outcome { Ok(result) => { if let Some(active) = result.replacement { @@ -941,6 +1594,23 @@ impl SegmentedFile { } } +/// Counts one blocking write from its start until its thread finishes it, +/// even when the caller that awaited it was canceled. +struct InFlightWrite(Arc); + +impl InFlightWrite { + fn start(counter: &Arc) -> Self { + counter.fetch_add(1, Ordering::AcqRel); + Self(Arc::clone(counter)) + } +} + +impl Drop for InFlightWrite { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::AcqRel); + } +} + struct AppendRequest { path: PathBuf, rotate_bytes: u64, @@ -1399,6 +2069,7 @@ mod tests { file.sync_hook = Some(hook); AuditWriter { inner: Arc::new(WriterInner::File(Arc::new(GroupCommitFile::new(file)))), + open: Arc::default(), } } @@ -2711,4 +3382,732 @@ mod tests { .await .expect_err("leftover hash-chained journal"); } + + const SCHEMA: &str = "registry.test.audit/v2"; + + fn unfinished() -> Value { + json!({"operationId": "read", "outcome": "unfinished"}) + } + + fn buffered() -> (AuditWriter, SharedBuffer) { + let buffer = SharedBuffer::default(); + ( + AuditWriter::from_line_sink(Box::new(buffer.clone())), + buffer, + ) + } + + fn buffered_lines(buffer: &SharedBuffer) -> Vec { + String::from_utf8(buffer.0.lock().expect("buffer").clone()) + .expect("utf-8") + .lines() + .map(|line| serde_json::from_str(line).expect("json line")) + .collect() + } + + /// The lines accepted once every detached unfinished response is + /// written. + fn settled_lines(writer: &AuditWriter, buffer: &SharedBuffer) -> Vec { + writer.wait_for_detached_entries(); + buffered_lines(buffer) + } + + fn assert_paired(entries: &[Value], outcome: &str) { + assert_eq!(entries.len(), 2, "{entries:?}"); + assert_eq!(entries[0]["phase"], "request"); + assert_eq!(entries[1]["phase"], "response"); + assert_eq!(entries[1]["schema"], entries[0]["schema"]); + assert_eq!(entries[1]["correlation"], entries[0]["correlation"]); + assert_eq!(entries[1]["record"]["outcome"], outcome); + } + + #[tokio::test] + async fn an_answered_request_writes_only_its_responses() { + let (writer, buffer) = buffered(); + let mut request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + assert_eq!(request.correlation(), "req-1"); + assert!(!request.is_answered()); + request + .respond(json!({"outcome": "returned"})) + .await + .expect("response"); + assert!(request.is_answered()); + request + .respond(json!({"outcome": "returned"})) + .await + .expect("second response"); + drop(request); + let entries = settled_lines(&writer, &buffer); + assert_eq!(entries.len(), 3); + assert!(entries[1..] + .iter() + .all(|entry| entry["record"]["outcome"] == "returned" + && entry["correlation"] == "req-1" + && entry["schema"] == SCHEMA)); + } + + #[tokio::test] + async fn a_response_appended_elsewhere_answers_the_open_request() { + let (writer, buffer) = buffered(); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + .expect("response"); + assert!(request.is_answered()); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "returned"); + } + + #[tokio::test] + async fn a_response_in_another_schema_does_not_answer_the_request() { + let (writer, buffer) = buffered(); + let request = writer + .begin(SCHEMA, "req-1", json!({"operationId": "run"}), unfinished()) + .await + .expect("request"); + writer + .append(AuditEntry::response( + "registry.test.other/v1", + "req-1", + json!({"outcome": "refused"}), + )) + .await + .expect("other schema"); + assert!(!request.is_answered()); + drop(request); + let entries = settled_lines(&writer, &buffer); + assert_eq!(entries.len(), 3); + assert_eq!(entries[2]["schema"], SCHEMA); + assert_eq!(entries[2]["correlation"], "req-1"); + assert_eq!(entries[2]["record"]["outcome"], "unfinished"); + } + + #[tokio::test] + async fn one_response_answers_one_of_two_requests_sharing_a_correlation() { + let (writer, buffer) = buffered(); + let first = writer + .begin(SCHEMA, "shared", json!({"n": 1}), unfinished()) + .await + .expect("first"); + let mut second = writer + .begin(SCHEMA, "shared", json!({"n": 2}), unfinished()) + .await + .expect("second"); + second + .respond(json!({"outcome": "returned"})) + .await + .expect("second answers itself"); + assert!(!first.is_answered(), "the second's response is its own"); + drop(second); + drop(first); + let outcomes: Vec<_> = settled_lines(&writer, &buffer) + .iter() + .filter(|entry| entry["phase"] == "response") + .map(|entry| entry["record"]["outcome"].clone()) + .collect(); + assert_eq!(outcomes, [json!("returned"), json!("unfinished")]); + } + + #[tokio::test] + async fn a_request_dropped_unanswered_writes_its_unfinished_response() { + let (writer, buffer) = buffered(); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[tokio::test] + async fn an_early_error_return_pairs_the_request() { + async fn refused(writer: &AuditWriter) -> Result<(), &'static str> { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .map_err(|_| "audit")?; + Err("not found")?; + Ok(()) + } + let (writer, buffer) = buffered(); + assert_eq!(refused(&writer).await, Err("not found")); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[tokio::test] + async fn a_refused_response_leaves_the_request_owed() { + let (writer, buffer) = buffered(); + let mut request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + let refused = request.respond(json!(["not", "an", "object"])).await; + assert_eq!( + refused.map_err(|error| error.reason()), + Err(AuditUnavailableReason::InvalidEntry) + ); + assert!(!request.is_answered()); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[tokio::test] + async fn an_unfinished_record_that_is_not_an_object_writes_no_request() { + let (writer, buffer) = buffered(); + let refused = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + json!("gone"), + ) + .await; + assert!(refused.is_err()); + assert!(settled_lines(&writer, &buffer).is_empty()); + } + + #[tokio::test] + async fn a_canceled_operation_pairs_its_request_in_the_file() { + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let (started, begun) = tokio::sync::oneshot::channel(); + let operation = tokio::spawn({ + let writer = writer.clone(); + async move { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + started.send(()).expect("signal"); + std::future::pending::<()>().await; + } + }); + begun.await.expect("begun"); + operation.abort(); + assert!(operation.await.expect_err("aborted").is_cancelled()); + // A later append is ordered after the queued unfinished response. + writer + .append(AuditEntry::response(SCHEMA, "other", json!({}))) + .await + .expect("later entry"); + let entries = lines(&path); + assert_paired(&entries[..2], "unfinished"); + assert_eq!(entries[2]["correlation"], "other"); + } + + #[tokio::test] + async fn a_panicking_operation_pairs_its_request() { + let (writer, buffer) = buffered(); + let operation = tokio::spawn({ + let writer = writer.clone(); + async move { + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + panic!("handler failed"); + } + }); + assert!(operation.await.expect_err("panicked").is_panic()); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[test] + fn a_command_that_exits_after_an_early_return_pairs_its_request() { + // An operator command runs on a current-thread runtime and exits as + // soon as its future returns, before a spawned flush can run. + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let result: Result<(), &str> = runtime.block_on(async move { + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let _request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "erase"}), + unfinished(), + ) + .await + .expect("request"); + Err("database unavailable") + }); + assert_eq!(result, Err("database unavailable")); + drop(runtime); + assert_paired(&lines(&path), "unfinished"); + } + + #[test] + fn a_request_dropped_outside_a_runtime_while_the_file_is_busy_still_pairs() { + let directory = directory(); + let destination = file_destination(&directory); + let path = destination.path().to_path_buf(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let (writer, request) = runtime.block_on(async move { + let writer = AuditWriter::open(AuditDestination::File(destination)) + .await + .expect("open"); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "erase"}), + unfinished(), + ) + .await + .expect("request"); + (writer, request) + }); + drop(runtime); + let WriterInner::File(file) = writer.inner.as_ref() else { + panic!("a file destination"); + }; + let file = Arc::clone(file); + let (held, holding) = std::sync::mpsc::channel(); + // Another thread holds the file state for longer than any bounded + // wait while the handle is dropped outside a runtime. + let holder = std::thread::spawn(move || { + let _state = file.state.blocking_lock(); + held.send(()).expect("signal"); + std::thread::sleep(Duration::from_millis(200)); + }); + holding.recv().expect("the state is held"); + drop(request); + holder.join().expect("holder"); + drop(writer); + assert_paired(&lines(&path), "unfinished"); + } + + #[tokio::test] + async fn a_stopped_writer_refuses_the_request_and_writes_nothing() { + let writer = AuditWriter::from_line_sink(Box::new(FailingSink)); + assert!(writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished() + ) + .await + .is_err()); + assert!(!writer.ready().await); + } + + /// A line sink that holds each write until the test releases it. + #[derive(Clone)] + struct GatedSink { + buffer: SharedBuffer, + open: Arc<(Mutex, std::sync::Condvar)>, + entered: Arc, + } + + impl GatedSink { + fn new() -> Self { + Self { + buffer: SharedBuffer::default(), + open: Arc::new((Mutex::new(false), std::sync::Condvar::new())), + entered: Arc::default(), + } + } + + fn release(&self) { + *self.open.0.lock().expect("gate") = true; + self.open.1.notify_all(); + } + + async fn wait_entered(&self, writes: usize) { + for _ in 0..500 { + if self.entered.load(Ordering::SeqCst) >= writes { + return; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + panic!("the sink never received write {writes}"); + } + } + + impl Write for GatedSink { + fn write(&mut self, bytes: &[u8]) -> io::Result { + self.entered.fetch_add(1, Ordering::SeqCst); + let mut open = self.open.0.lock().expect("gate"); + while !*open { + open = self.open.1.wait(open).expect("gate"); + } + drop(open); + self.buffer.write(bytes) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + + /// The accepted lines once `count` of them arrived. + async fn lines_eventually(buffer: &SharedBuffer, count: usize) -> Vec { + for _ in 0..500 { + let lines = buffered_lines(buffer); + if lines.len() >= count { + return lines; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + buffered_lines(buffer) + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_request_canceled_while_its_entry_is_written_is_still_paired() { + let sink = GatedSink::new(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let begun = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + } + }); + sink.wait_entered(1).await; + // The caller goes away while its request entry is being written. + begun.abort(); + sink.release(); + assert_paired(&lines_eventually(&sink.buffer, 2).await, "unfinished"); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_accepted_after_its_caller_left_is_the_only_answer() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let responding = tokio::spawn(async move { + let mut request = request; + request.respond(json!({"outcome": "returned"})).await + }); + sink.wait_entered(2).await; + // The caller and its handle go away while the response is written. + responding.abort(); + sink.release(); + let lines = lines_eventually(&sink.buffer, 2).await; + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + assert_eq!(buffered_lines(&sink.buffer).len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_appended_after_its_caller_left_still_answers_the_request() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The caller goes away while the response is written. + appending.abort(); + sink.release(); + lines_eventually(&sink.buffer, 2).await; + for _ in 0..500 { + if request.is_answered() { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + drop(request); + writer.wait_for_detached_entries(); + assert_paired(&buffered_lines(&sink.buffer), "returned"); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_request_dropped_while_an_appended_response_is_written_is_answered_once() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The caller and its handle go away while the appended response is + // written: that response answers the request, and the drop writes + // nothing more. + appending.abort(); + drop(request); + sink.release(); + lines_eventually(&sink.buffer, 2).await; + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + let lines = buffered_lines(&sink.buffer); + assert_eq!(lines.len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + + #[tokio::test] + async fn an_unfinished_record_too_large_to_write_refuses_the_request() { + let (writer, buffer) = buffered(); + let oversized = json!({"outcome": "unfinished", "padding": "x".repeat(MAX_ENTRY_BYTES)}); + assert!(writer + .begin(SCHEMA, "req-1", json!({"operationId": "read"}), oversized) + .await + .is_err()); + assert!(buffered_lines(&buffer).is_empty()); + assert!(writer.ready().await, "a refused request stops nothing"); + } + + #[test] + fn dropping_a_request_never_waits_on_a_stalled_stream() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + *sink.open.0.lock().expect("gate") = false; + // The stream stalls; the drop must still return to the runtime. + let started = std::time::Instant::now(); + runtime.block_on(async move { drop(request) }); + assert!(started.elapsed() < Duration::from_secs(1)); + sink.release(); + writer.wait_for_detached_entries(); + assert_paired(&buffered_lines(&sink.buffer), "unfinished"); + } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_response_claimed_while_its_request_is_dropped_is_the_only_answer() { + let sink = GatedSink::new(); + sink.release(); + let writer = AuditWriter::from_line_sink(Box::new(sink.clone())); + let request = writer + .begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + ) + .await + .expect("request"); + // The drop's first half: it marks the handle dropped and finds no + // response in flight. + let in_flight = request.state.lock().is_ok_and(|mut state| { + state.dropped = true; + state.in_flight > 0 + }); + assert!(!in_flight); + // A response appended elsewhere claims the request before the drop + // settles it. + *sink.open.0.lock().expect("gate") = false; + let appending = tokio::spawn({ + let writer = writer.clone(); + async move { + writer + .append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) + .await + } + }); + sink.wait_entered(2).await; + // The drop's second half. + settle_unanswered( + &writer, + (SCHEMA.to_owned(), "req-1".to_owned()), + &request.answered, + &request.state, + ); + sink.release(); + appending + .await + .expect("append task") + .expect("the claimed response"); + drop(request); + tokio::time::sleep(Duration::from_millis(100)).await; + writer.wait_for_detached_entries(); + let lines = buffered_lines(&sink.buffer); + assert_eq!(lines.len(), 2, "{lines:?}"); + assert_paired(&lines, "returned"); + } + + #[test] + fn a_response_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle() { + let (writer, buffer) = buffered(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let mut request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + // The response's task is spawned and never runs: the runtime shuts + // down first and drops it. + runtime.block_on(async { + tokio::select! { + biased; + _ = request.respond(json!({"outcome": "returned"})) => { + panic!("the response task never ran") + } + () = std::future::ready(()) => {} + } + }); + drop(runtime); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } + + #[test] + fn an_append_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle() { + let (writer, buffer) = buffered(); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime"); + let request = runtime + .block_on(writer.begin( + SCHEMA, + "req-1", + json!({"operationId": "read"}), + unfinished(), + )) + .expect("request"); + // The appended response claims the request, and its task is dropped + // by the runtime's shutdown before it runs. + runtime.block_on(async { + tokio::select! { + biased; + _ = writer.append(AuditEntry::response( + SCHEMA, + "req-1", + json!({"outcome": "returned"}), + )) => panic!("the append task never ran"), + () = std::future::ready(()) => {} + } + }); + drop(runtime); + drop(request); + assert_paired(&settled_lines(&writer, &buffer), "unfinished"); + } } diff --git a/crates/registry-platform-hooks/src/delivery/seams.rs b/crates/registry-platform-hooks/src/delivery/seams.rs index a0db3f3935..031a034f0f 100644 --- a/crates/registry-platform-hooks/src/delivery/seams.rs +++ b/crates/registry-platform-hooks/src/delivery/seams.rs @@ -138,17 +138,27 @@ pub trait DeliverySeams: Send + Sync + 'static { ) -> Result, DeliveryError>; /// Record one neutral delivery-audit event in the product's audit - /// journal. The worker calls this only after every guarded transition the - /// event reports has already succeeded, immediately before it commits the - /// transaction: a failed commit after an accepted append still leaves an - /// entry for a transition that did not happen, since a durable append - /// cannot be rolled back with the transaction. Every audited occurrence - /// and every audited field of the moved worker arrives here. - async fn record_audit( - &self, - transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError>; + /// journal. Every audited occurrence and every audited field of the + /// moved worker arrives here, in an order that keeps the journal from + /// claiming more than the database committed: + /// + /// - An attempt's start is its request, recorded inside the lease + /// transaction immediately before that transaction commits, so it is + /// on record before the request can leave the process and a refusal + /// rolls the lease back. If that commit then fails, the worker records + /// the attempt's `WorkerInterrupted` terminal, so the request is + /// answered and no egress follows. + /// - A terminal disposition and a payload expiry are recorded only after + /// the transaction that made them commits, so an entry never stands + /// for a transition that rolled back. + /// - An operator replay is a request, `ReplayRequested`, recorded before + /// the reset, and a response recorded after it: `ReplayCommitted` once + /// the reset commits, `ReplayRefused` when it does not. + /// + /// A refused append after a commit leaves that committed transition + /// without its entry; the writer then refuses every later entry until + /// the operator repairs the destination, which readiness reports. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError>; /// Report one operational event through the product's vocabulary. fn operational_event(&self, event: DeliveryOperationalEvent); @@ -418,7 +428,13 @@ pub enum DeliveryAuditOutcome { PayloadRefused, PayloadExpired, WorkerInterrupted, + /// An operator asked for a dead-lettered delivery to be replayed: the + /// request of a replay. ReplayRequested, + /// The replay's reset committed, and the delivery is pending again. + ReplayCommitted, + /// The replay's reset did not commit; the delivery stays dead-lettered. + ReplayRefused, } impl DeliveryAuditOutcome { diff --git a/crates/registry-platform-hooks/src/delivery/service.rs b/crates/registry-platform-hooks/src/delivery/service.rs index c5f15aa580..e0c2195381 100644 --- a/crates/registry-platform-hooks/src/delivery/service.rs +++ b/crates/registry-platform-hooks/src/delivery/service.rs @@ -40,6 +40,10 @@ impl From for DeliveryError { const LEASE_FINALIZATION_ALLOWANCE: Duration = Duration::from_secs(5); const WORKER_POLL_INTERVAL: Duration = Duration::from_millis(100); +/// The waits between the read-backs of a transition whose commit returned an +/// error: one fewer than the read-backs, and short, since a caller is failing +/// while they run. +const READ_BACK_BACKOFF: [Duration; 2] = [Duration::from_millis(50), Duration::from_millis(100)]; pub const MAX_DELIVERY_STATUS_RESULTS: u16 = 100; /// The product-supplied constants the worker cannot derive. @@ -61,6 +65,35 @@ pub struct DeliveryConfig { pub delivery_source: String, } +/// A delivery-audit event held by the worker until it is recorded, such as +/// one whose transition must commit first. +#[derive(Clone)] +struct PendingAudit { + event_id: Uuid, + compiled_delivery_id: String, + package_revision: String, + generation: i64, + attempt: i16, + phase: DeliveryAuditPhase, + outcome: DeliveryAuditOutcome, + disposition: DeliveryAuditDisposition, +} + +impl PendingAudit { + fn record(&self) -> DeliveryAuditRecord<'_> { + DeliveryAuditRecord { + event_id: self.event_id, + compiled_delivery_id: &self.compiled_delivery_id, + package_revision: &self.package_revision, + generation: self.generation, + attempt: self.attempt, + phase: self.phase, + outcome: self.outcome, + disposition: self.disposition, + } + } +} + /// The delivery worker over one product's seams. #[derive(Clone)] pub struct DeliveryService { @@ -246,6 +279,29 @@ impl DeliveryService { event_id: Uuid, compiled_delivery_id: &str, expected_generation: i64, + ) -> Result + where + S: Clone, + { + // The replay runs to its response in a task of its own: once its + // request entry is accepted, a caller that stops waiting (a timeout + // or a disconnect) cannot leave that request unanswered. + let service = self.clone(); + let compiled_delivery_id = compiled_delivery_id.to_owned(); + tokio::spawn(async move { + service + .replay_in(event_id, &compiled_delivery_id, expected_generation) + .await + }) + .await + .map_err(|_| DeliveryError::Unavailable)? + } + + async fn replay_in( + &self, + event_id: Uuid, + compiled_delivery_id: &str, + expected_generation: i64, ) -> Result { if compiled_delivery_id.is_empty() || compiled_delivery_id.len() > 256 @@ -323,6 +379,86 @@ impl DeliveryService { { return Err(DeliveryError::Unavailable); } + let next_generation = generation + .checked_add(1) + .ok_or(DeliveryError::Unavailable)?; + // The replay's request is on record before the reset, and its + // response records whether the reset committed. + let replay = PendingAudit { + event_id, + compiled_delivery_id: compiled_delivery_id.to_owned(), + package_revision: package_revision.clone(), + generation: next_generation, + attempt: 0, + phase: DeliveryAuditPhase::Replay, + outcome: DeliveryAuditOutcome::ReplayRequested, + disposition: DeliveryAuditDisposition::ReplayPending, + }; + self.seams.record_audit(replay.record()).await?; + let reset = match self + .reset_for_replay(&transaction, event_id, compiled_delivery_id, generation) + .await + { + // A reset that failed or changed no row did not commit. + Err(error) => Err(error), + Ok(()) => match transaction.commit().await { + Ok(()) => Ok(()), + Err(error) => { + // The commit's acknowledgement may be all that was lost. + if self + .transition_committed( + &PendingAudit { + outcome: DeliveryAuditOutcome::ReplayCommitted, + ..replay.clone() + }, + None, + ) + .await + == Some(true) + { + Ok(()) + } else { + Err(error.into()) + } + } + }, + }; + let (outcome, disposition) = if reset.is_ok() { + ( + DeliveryAuditOutcome::ReplayCommitted, + DeliveryAuditDisposition::ReplayPending, + ) + } else { + ( + DeliveryAuditOutcome::ReplayRefused, + DeliveryAuditDisposition::DeadLettered, + ) + }; + let recorded = self + .seams + .record_audit( + PendingAudit { + outcome, + disposition, + ..replay + } + .record(), + ) + .await; + reset?; + recorded?; + Ok(next_generation) + } + + /// Reset one dead-lettered delivery to pending under its next generation, + /// leaving the commit to the caller. + async fn reset_for_replay( + &self, + transaction: &Transaction<'_>, + event_id: Uuid, + compiled_delivery_id: &str, + generation: i64, + ) -> Result<(), DeliveryError> { let next_generation = generation .checked_add(1) .ok_or(DeliveryError::Unavailable)?; @@ -357,23 +493,7 @@ impl DeliveryService { if changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id, - package_revision: &package_revision, - generation: next_generation, - attempt: 0, - phase: DeliveryAuditPhase::Replay, - outcome: DeliveryAuditOutcome::ReplayRequested, - disposition: DeliveryAuditDisposition::ReplayPending, - }, - ) - .await?; - transaction.commit().await?; - Ok(next_generation) + Ok(()) } async fn claim(&self) -> Result, DeliveryError> { @@ -383,11 +503,19 @@ impl DeliveryService { self.refused(DeliveryTransitionCode::ClaimIdentityRefused); return Err(DeliveryError::Unavailable); } - if self.reap_expired_leases(&transaction).await.is_err() { + // Recovered and expired deliveries are recorded once this + // transaction commits. + let mut committed = Vec::new(); + if self + .reap_expired_leases(&transaction, &mut committed) + .await + .is_err() + { self.refused(DeliveryTransitionCode::ClaimRecoveryFailed); return Err(DeliveryError::Unavailable); } - self.expire_retained_payload(&transaction).await?; + self.expire_retained_payload(&transaction, &mut committed) + .await?; let row = transaction .query_opt( &self.sql( @@ -421,7 +549,11 @@ impl DeliveryService { DeliveryError::Unavailable })?; let Some(row) = row else { - transaction.commit().await?; + if let Err(error) = transaction.commit().await { + self.record_resolved(committed).await; + return Err(error.into()); + } + self.record_committed(committed).await?; return Ok(None); }; let event_id = row.try_get::<_, Uuid>(0)?; @@ -498,31 +630,45 @@ impl DeliveryService { .query_one("SELECT transaction_timestamp()", &[]) .await? .try_get::<_, SystemTime>(0)?; - if self - .seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Attempt, - outcome: DeliveryAuditOutcome::AttemptStarted, - disposition: DeliveryAuditDisposition::Leased, - }, - ) - .await - .is_err() - { + let started = PendingAudit { + event_id, + compiled_delivery_id: compiled_delivery_id.clone(), + package_revision: package_revision.clone(), + generation, + attempt, + phase: DeliveryAuditPhase::Attempt, + outcome: DeliveryAuditOutcome::AttemptStarted, + disposition: DeliveryAuditDisposition::Leased, + }; + if self.seams.record_audit(started.record()).await.is_err() { self.refused(DeliveryTransitionCode::ClaimAuditFailed); return Err(DeliveryError::Unavailable); } - transaction.commit().await.map_err(|_| { + if transaction.commit().await.is_err() { self.refused(DeliveryTransitionCode::ClaimCommitFailed); - DeliveryError::Unavailable - })?; + // A failed commit acknowledgement does not prove a rollback, so + // read what the database holds. Each recovered or expired + // delivery is recorded if it reads back as durable, whatever the + // lease's own fate. A lease that did commit sends nothing from + // here and is answered when it expires; one that rolled back, or + // one whose fate cannot be read, is answered now, since a second + // interrupted answer is harmless and none is not. + let lease_committed = self.transition_committed(&started, Some(lease_token)).await; + self.record_resolved(committed).await; + if lease_committed != Some(true) { + let interrupted = PendingAudit { + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: DeliveryAuditDisposition::RetryPending, + ..started + }; + // A refused entry has already stopped the product's writer, + // which reports it; the claim fails either way. + let _ = self.seams.record_audit(interrupted.record()).await; + } + return Err(DeliveryError::Unavailable); + } + self.record_committed(committed).await?; Ok(Some(DeliveryClaim { event_id, compiled_delivery_id, @@ -540,6 +686,7 @@ impl DeliveryService { async fn reap_expired_leases( &self, transaction: &Transaction<'_>, + committed: &mut Vec, ) -> Result<(), DeliveryError> { let row = transaction .query_opt( @@ -672,31 +819,27 @@ impl DeliveryService { if changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Terminal, - outcome: DeliveryAuditOutcome::WorkerInterrupted, - disposition: if dead_lettered { - DeliveryAuditDisposition::DeadLettered - } else { - DeliveryAuditDisposition::RetryPending - }, - }, - ) - .await?; + committed.push(PendingAudit { + event_id, + compiled_delivery_id, + package_revision, + generation, + attempt, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: if dead_lettered { + DeliveryAuditDisposition::DeadLettered + } else { + DeliveryAuditDisposition::RetryPending + }, + }); Ok(()) } async fn expire_retained_payload( &self, transaction: &Transaction<'_>, + committed: &mut Vec, ) -> Result<(), DeliveryError> { let row = transaction .query_opt( @@ -762,21 +905,16 @@ impl DeliveryService { if state_changed != 1 || payload_changed != 1 { return Err(DeliveryError::Unavailable); } - self.seams - .record_audit( - transaction, - DeliveryAuditRecord { - event_id, - compiled_delivery_id: &compiled_delivery_id, - package_revision: &package_revision, - generation, - attempt, - phase: DeliveryAuditPhase::Terminal, - outcome: DeliveryAuditOutcome::PayloadExpired, - disposition: DeliveryAuditDisposition::Expired, - }, - ) - .await?; + committed.push(PendingAudit { + event_id, + compiled_delivery_id, + package_revision, + generation, + attempt, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::PayloadExpired, + disposition: DeliveryAuditDisposition::Expired, + }); Ok(()) } @@ -1277,25 +1415,127 @@ impl DeliveryService { return Err(DeliveryError::Unavailable); } } - self.seams - .record_audit( - &transaction, - DeliveryAuditRecord { - event_id: claim.event_id, - compiled_delivery_id: &claim.compiled_delivery_id, - package_revision: &claim.package_revision, - generation: claim.generation, - attempt: claim.attempt, - phase: DeliveryAuditPhase::Terminal, - outcome, - disposition, - }, - ) - .await?; - transaction.commit().await?; + let terminal = PendingAudit { + event_id: claim.event_id, + compiled_delivery_id: claim.compiled_delivery_id.clone(), + package_revision: claim.package_revision.clone(), + generation: claim.generation, + attempt: claim.attempt, + phase: DeliveryAuditPhase::Terminal, + outcome, + disposition, + }; + if let Err(error) = transaction.commit().await { + // A lost acknowledgement does not prove a rollback, and a + // terminal row is never reaped, so a disposition that did commit + // is recorded here or never. One that rolled back leaves the + // lease for expiry recovery, which answers the attempt. + match self.transition_committed(&terminal, None).await { + Some(true) => {} + Some(false) => return Err(error.into()), + None => { + // The disposition's fate cannot be read, and one that + // did commit is never answered by expiry recovery, so the + // attempt is answered now as interrupted: a second + // interrupted answer is harmless and none is not. + let interrupted = PendingAudit { + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition: DeliveryAuditDisposition::RetryPending, + ..terminal + }; + // A refused entry has already stopped the product's + // writer, which reports it; finalize fails either way. + let _ = self.seams.record_audit(interrupted.record()).await; + return Err(error.into()); + } + } + } + // Recorded once the disposition committed, so the journal never + // names a disposition the database does not hold. + self.seams.record_audit(terminal.record()).await?; Ok(work_outcome) } + /// Record the events of a transaction whose commit returned an error, + /// each only if the transition it names is durable. + async fn record_resolved(&self, events: Vec) { + for event in events { + // A refused entry has already stopped the product's writer, + // which reports it; the caller is failing either way. + if self.transition_committed(&event, None).await == Some(true) { + let _ = self.seams.record_audit(event.record()).await; + } + } + } + + /// Whether the transition `event` records is durable, read on a fresh + /// connection after its commit returned an error, since a failed commit + /// acknowledgement does not prove the transaction rolled back. `None` + /// when the state cannot be read. An attempt's lease is matched on the + /// token this worker wrote. + async fn transition_committed( + &self, + event: &PendingAudit, + lease_token: Option, + ) -> Option { + for read_back in 0..=READ_BACK_BACKOFF.len() { + if let Some(wait) = read_back + .checked_sub(1) + .and_then(|index| READ_BACK_BACKOFF.get(index)) + { + tokio::time::sleep(*wait).await; + } + let Ok(client) = self.seams.connection().await else { + continue; + }; + let Ok(row) = client + .query_opt( + &self.sql( + "SELECT state, generation, attempt, lease_token, expired_at IS NOT NULL + FROM {schema}.registry_webhook_delivery_state + WHERE event_id = $1 AND compiled_delivery_id = $2", + ), + &[&event.event_id, &event.compiled_delivery_id], + ) + .await + else { + continue; + }; + let Some(row) = row else { + return Some(false); + }; + let (Ok(state), Ok(generation), Ok(attempt), Ok(token), Ok(expired)) = ( + row.try_get::<_, String>(0), + row.try_get::<_, i64>(1), + row.try_get::<_, i16>(2), + row.try_get::<_, Option>(3), + row.try_get::<_, bool>(4), + ) else { + return None; + }; + return Some(transition_holds( + event, + &ObservedDelivery { + state: &state, + generation, + attempt, + lease_token: token, + expired, + }, + lease_token, + )); + } + None + } + + /// Record the events of a transaction that has committed. + async fn record_committed(&self, committed: Vec) -> Result<(), DeliveryError> { + for event in committed { + self.seams.record_audit(event.record()).await?; + } + Ok(()) + } + async fn update_terminal_state( &self, transaction: &Transaction<'_>, @@ -1372,6 +1612,62 @@ impl DeliveryService { } } +/// One delivery row as read back after a commit returned an error. +struct ObservedDelivery<'a> { + state: &'a str, + generation: i64, + attempt: i16, + lease_token: Option, + expired: bool, +} + +/// Whether `observed` shows the transition `event` records as durable. An +/// attempt's lease is matched on the token this worker wrote. +fn transition_holds( + event: &PendingAudit, + observed: &ObservedDelivery<'_>, + lease_token: Option, +) -> bool { + let current = observed.generation == event.generation; + let at_attempt = current && observed.attempt == event.attempt; + let state = observed.state; + match (event.phase, event.disposition) { + (DeliveryAuditPhase::Attempt, _) => { + at_attempt + && state == "leased" + && observed.lease_token.is_some() + && observed.lease_token == lease_token + } + // Only a replay's reset writes a generation, and generations only + // grow, so the replacement generation or a later one proves the reset + // committed whatever state the worker or a later replay has since + // moved it to. + (DeliveryAuditPhase::Replay, _) => observed.generation >= event.generation, + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Expired) => { + current && observed.expired + } + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::Delivered) => { + at_attempt && state == "delivered" + } + // A dead letter leaves that state only through a replay, which + // writes a later generation. + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::DeadLettered) => { + (at_attempt && state == "dead_lettered") || observed.generation > event.generation + } + // A scheduled retry is claimed again under a later attempt of the + // same generation, which a rolled-back retry reaches only after its + // lease expires and the reaper answers the attempt itself. + (DeliveryAuditPhase::Terminal, DeliveryAuditDisposition::RetryPending) => { + (at_attempt && state == "pending" && observed.lease_token.is_none()) + || (current && observed.attempt > event.attempt) + } + ( + DeliveryAuditPhase::Terminal, + DeliveryAuditDisposition::Leased | DeliveryAuditDisposition::ReplayPending, + ) => false, + } +} + /// The post-commit delivery loop: claim, send, finalize, until shutdown. pub struct DeliveryWorker { service: DeliveryService, @@ -1926,7 +2222,6 @@ mod tests { async fn record_audit( &self, - _transaction: &Transaction<'_>, _record: DeliveryAuditRecord<'_>, ) -> Result<(), DeliveryError> { Err(DeliveryError::Unavailable) @@ -1993,6 +2288,98 @@ mod tests { } } + fn replay_of(generation: i64) -> PendingAudit { + PendingAudit { + event_id: Uuid::nil(), + compiled_delivery_id: "delivery".to_owned(), + package_revision: "revision".to_owned(), + generation, + attempt: 0, + phase: DeliveryAuditPhase::Replay, + outcome: DeliveryAuditOutcome::ReplayCommitted, + disposition: DeliveryAuditDisposition::ReplayPending, + } + } + + fn observed(state: &str, generation: i64, attempt: i16) -> ObservedDelivery<'_> { + ObservedDelivery { + state, + generation, + attempt, + lease_token: None, + expired: false, + } + } + + fn terminal_of(disposition: DeliveryAuditDisposition) -> PendingAudit { + PendingAudit { + attempt: 1, + phase: DeliveryAuditPhase::Terminal, + outcome: DeliveryAuditOutcome::WorkerInterrupted, + disposition, + ..replay_of(1) + } + } + + #[test] + fn a_committed_retry_holds_after_its_next_attempt_is_claimed() { + let retry = terminal_of(DeliveryAuditDisposition::RetryPending); + assert!(transition_holds(&retry, &observed("pending", 1, 1), None)); + assert!( + transition_holds(&retry, &observed("leased", 1, 2), None), + "the next attempt proves the retry committed" + ); + assert!( + !transition_holds( + &retry, + &ObservedDelivery { + lease_token: Some(Uuid::nil()), + ..observed("leased", 1, 1) + }, + None + ), + "a lease still at the attempt proves the retry rolled back" + ); + } + + #[test] + fn a_committed_dead_letter_holds_after_it_is_replayed() { + let dead_letter = terminal_of(DeliveryAuditDisposition::DeadLettered); + assert!(transition_holds( + &dead_letter, + &observed("dead_lettered", 1, 1), + None + )); + assert!( + transition_holds(&dead_letter, &observed("pending", 2, 0), None), + "a replay generation proves the dead letter committed" + ); + assert!(!transition_holds( + &dead_letter, + &observed("leased", 1, 1), + None + )); + } + + #[test] + fn a_replay_whose_reset_committed_holds_after_the_worker_moves_it() { + let replay = replay_of(2); + for state in ["pending", "leased", "delivered", "dead_lettered"] { + assert!( + transition_holds(&replay, &observed(state, 2, 1), None), + "the replacement generation proves the reset in state {state}" + ); + } + assert!( + transition_holds(&replay, &observed("pending", 3, 0), None), + "a later replay's generation proves this reset committed before it" + ); + assert!( + !transition_holds(&replay, &observed("dead_lettered", 1, 2), None), + "the prior generation proves the reset rolled back" + ); + } + #[test] fn the_worker_accepts_the_stored_envelope_it_will_deliver_unchanged() { let body = stored_envelope() @@ -2406,7 +2793,6 @@ mod tests { async fn record_audit( &self, - _transaction: &Transaction<'_>, _record: DeliveryAuditRecord<'_>, ) -> Result<(), DeliveryError> { Err(DeliveryError::Unavailable) @@ -2734,23 +3120,89 @@ mod tests { client } + type RecordedAudit = ( + DeliveryAuditPhase, + DeliveryAuditOutcome, + DeliveryAuditDisposition, + ); + + /// A local handler that only answers to its reviewed digest, so a replay + /// can find the binding its row was written against. + struct DigestHandler(String); + + #[async_trait::async_trait] + impl HookHandler for DigestHandler { + fn handler_digest(&self) -> &str { + &self.0 + } + + fn attempt_timeout(&self) -> Duration { + Duration::ZERO + } + + fn maximum_attempts(&self) -> u8 { + 1 + } + + async fn run( + &self, + _envelope: &[u8], + _remaining: Duration, + ) -> Result, crate::delivery::HandlerRunFailure> { + unreachable!("no real-database test runs a handler") + } + } + /// A seam set that opens its own connection against a real PostgreSQL - /// test database and counts every audit record it is given, so a test can - /// prove no record was written for a transition that never happened. + /// test database and records every audit record it is given, so a test can + /// prove which transitions were written. Connections are numbered from + /// one in the order they are opened: a test may refuse chosen ones, and + /// may run one statement on a chosen one before the service uses it, to + /// stand for another session acting at that moment. struct RealDbSeams { url: String, - audit_calls: Arc>, + audit: Arc>>, + connections: Mutex, + refused_connections: Vec, + interleave: Option<(u32, String)>, + handler_digest: Option, + } + + impl RealDbSeams { + fn new(url: &str, audit: &Arc>>) -> Self { + Self { + url: url.to_owned(), + audit: Arc::clone(audit), + connections: Mutex::new(0), + refused_connections: Vec::new(), + interleave: None, + handler_digest: None, + } + } } #[async_trait::async_trait] impl DeliverySeams for RealDbSeams { type Destination = UnusedDestination; - type Handler = UnusedHandler; + type Handler = DigestHandler; async fn connection(&self) -> Result { - Ok(Box::new(DirectClient( - connect_test_database(&self.url).await, - ))) + let number = { + let mut connections = self.connections.lock().expect("connections lock"); + *connections += 1; + *connections + }; + if self.refused_connections.contains(&number) { + return Err(DeliveryError::Unavailable); + } + let client = connect_test_database(&self.url).await; + if let Some((_, statement)) = self.interleave.as_ref().filter(|(at, _)| *at == number) { + client + .batch_execute(statement) + .await + .expect("the interleaved statement runs"); + } + Ok(Box::new(DirectClient(client))) } async fn verify_transaction( @@ -2764,16 +3216,19 @@ mod tests { None } - fn handler(&self, _binding: HookHandlerBinding<'_>) -> Option { - None + fn handler(&self, binding: HookHandlerBinding<'_>) -> Option { + self.handler_digest + .as_deref() + .filter(|digest| *digest == binding.handler_digest) + .map(|digest| DigestHandler(digest.to_owned())) } - async fn record_audit( - &self, - _transaction: &Transaction<'_>, - _record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { - *self.audit_calls.lock().expect("audit calls lock") += 1; + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { + self.audit.lock().expect("audit lock").push(( + record.phase, + record.outcome, + record.disposition, + )); Ok(()) } @@ -2909,12 +3364,9 @@ mod tests { .expect("simulate an active lease"); assert_eq!(changed, 1, "the inserted delivery state row exists"); - let audit_calls = Arc::new(Mutex::new(0u32)); + let audit = Arc::new(Mutex::new(Vec::new())); let service = DeliveryService::new( - RealDbSeams { - url: url.clone(), - audit_calls: Arc::clone(&audit_calls), - }, + RealDbSeams::new(&url, &audit), DeliveryConfig { schema: schema.to_owned(), idempotency_domain: b"hooks-delivery-lease-test-v1".to_vec(), @@ -2945,9 +3397,367 @@ mod tests { "a lease-guarded update that changes zero rows must fail closed" ); assert_eq!( - *audit_calls.lock().expect("audit calls lock"), - 0, + *audit.lock().expect("audit lock"), + Vec::::new(), "no audit entry for a transition that did not happen" ); } + + const REAL_DELIVERY_ID: &str = "events.permit.granted.webhook"; + const REAL_PACKAGE_REVISION: &str = + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + const REAL_HANDLER_DIGEST: &str = + "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; + const REAL_SCHEMA_FINGERPRINT: &str = + "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc"; + const OTHER_EVENT_ID: &str = "9b1c3a5e-2d4f-4e6a-8b0c-1d3e5f7a9b2c"; + + fn real_database_url() -> String { + std::env::var("HOOKS_TEST_DATABASE_URL") + .expect("HOOKS_TEST_DATABASE_URL is required for the real PostgreSQL tests") + } + + /// Install the delivery schema afresh under `schema` and return an + /// administrative connection to the test database. + async fn fresh_delivery_schema(url: &str, schema: &str) -> tokio_postgres::Client { + let client = connect_test_database(url).await; + client + .batch_execute(&format!( + "DROP SCHEMA IF EXISTS {schema} CASCADE; CREATE SCHEMA {schema};" + )) + .await + .expect("reset the test schema"); + delivery_schema::install(&client, schema) + .await + .expect("install the delivery schema"); + client + } + + /// Insert one pending local-handler delivery of `event_id` whose stored + /// payload expires after `payload_lifetime`, a PostgreSQL interval. + async fn insert_real_delivery( + client: &mut tokio_postgres::Client, + schema: &str, + event_id: Uuid, + payload_lifetime: &str, + operator_replay: bool, + ) { + client + .execute( + &format!( + "INSERT INTO {schema}.registry_outbox + (event_id, event_type, trigger, entity_id, record_reference, + record_revision, package_revision, schema_fingerprint, payload, + payload_expires_at) + VALUES ($1, 'permit.granted', 'test', 'permit', $2, 3, $3, $4, $5, + transaction_timestamp() + interval '{payload_lifetime}')", + ), + &[ + &event_id, + &"8f14e45fceea467a9cc18b2a4b9e2a1103b41d5ad4c88c8f2a0a9ac4f4c0d2b7", + &REAL_PACKAGE_REVISION, + &REAL_SCHEMA_FINGERPRINT, + &b"{}".as_slice(), + ], + ) + .await + .expect("insert the outbox row"); + let transaction = client.transaction().await.expect("insert transaction"); + insert_delivery( + &transaction, + schema, + event_id, + DeliveryCapture { + compiled_delivery_id: REAL_DELIVERY_ID, + handler_kind: HookHandlerKind::Rhai, + logical_destination_id: None, + destination_binding_digest: REAL_HANDLER_DIGEST, + package_revision: REAL_PACKAGE_REVISION, + schema_fingerprint: REAL_SCHEMA_FINGERPRINT, + data_schema: STORED_DATA_SCHEMA, + classification_ceiling: "public", + authentication_profile: "hmac_sha256_v1", + delivery_mode: "after_commit", + attempt_timeout_ms: 5_000, + initial_backoff_ms: 1_000, + maximum_backoff_ms: 60_000, + exponential_backoff_multiplier: 2, + maximum_attempts: 3, + retry_delays_ms: &[1_000, 2_000], + maximum_payload_bytes: 1_024, + payload: b"{}", + deployed_attempt_timeout_ms: 5_000, + deployed_maximum_attempts: 3, + dead_letter: "required", + operator_replay, + }, + ) + .await + .expect("insert the delivery row"); + transaction.commit().await.expect("commit the insert"); + } + + fn real_service(seams: RealDbSeams, schema: &str) -> DeliveryService { + DeliveryService::new( + seams, + DeliveryConfig { + schema: schema.to_owned(), + idempotency_domain: b"hooks-delivery-real-test-v1".to_vec(), + delivery_source: STORED_SOURCE.to_owned(), + }, + ) + } + + /// A replay whose reset changed no row did not commit, so its response + /// is a refusal even when the row reads back under the generation the + /// replay would have written, here because another replay wrote it. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn a_replay_whose_reset_changed_no_row_is_refused() { + let url = real_database_url(); + let schema = "hooks_delivery_replay_refused_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let event_id = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, event_id, "1 day", true).await; + client + .batch_execute(&format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'dead_lettered', attempt = 3, next_attempt_at = NULL, + dead_lettered_at = transaction_timestamp(); + CREATE FUNCTION {schema}.skip_reset() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + IF current_setting('hooks_test.allow_reset', true) = 'on' THEN + RETURN NEW; + END IF; + RETURN NULL; + END $$; + CREATE TRIGGER skip_reset + BEFORE UPDATE ON {schema}.registry_webhook_delivery_state + FOR EACH ROW WHEN (NEW.generation > OLD.generation) + EXECUTE FUNCTION {schema}.skip_reset();" + )) + .await + .expect("make the reset change no row"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + handler_digest: Some(REAL_HANDLER_DIGEST.to_owned()), + // The first connection after the replay's own stands for a + // concurrent replay that commits the next generation. + interleave: Some(( + 2, + format!( + "SET hooks_test.allow_reset = 'on'; + UPDATE {schema}.registry_webhook_delivery_state + SET generation = 2, state = 'pending', attempt = 0, + next_attempt_at = transaction_timestamp(), + dead_lettered_at = NULL" + ), + )), + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + + let result = service.replay_in(event_id, REAL_DELIVERY_ID, 1).await; + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a reset that changed no row fails: {result:?}" + ); + assert_eq!( + *audit.lock().expect("audit lock"), + vec![ + ( + DeliveryAuditPhase::Replay, + DeliveryAuditOutcome::ReplayRequested, + DeliveryAuditDisposition::ReplayPending, + ), + ( + DeliveryAuditPhase::Replay, + DeliveryAuditOutcome::ReplayRefused, + DeliveryAuditDisposition::DeadLettered, + ), + ] + ); + } + + /// A claim whose lease commit failed still records every transition of + /// that transaction that reads back as durable, even when the lease's + /// own fate cannot be read. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn a_claim_whose_lease_commit_failed_records_its_durable_transitions() { + let url = real_database_url(); + let schema = "hooks_delivery_claim_commit_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let expiring = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + let claimable = Uuid::parse_str(OTHER_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, expiring, "-1 second", false).await; + insert_real_delivery(&mut client, schema, claimable, "1 day", false).await; + client + .batch_execute(&format!( + "CREATE FUNCTION {schema}.refuse_lease() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + RAISE EXCEPTION 'lease commit refused'; + END $$; + CREATE CONSTRAINT TRIGGER refuse_lease + AFTER UPDATE ON {schema}.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED + FOR EACH ROW WHEN (NEW.state = 'leased') + EXECUTE FUNCTION {schema}.refuse_lease();" + )) + .await + .expect("make the lease commit fail"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + // Every read-back of the lease fails, so its fate is unknown. + refused_connections: vec![2, 3, 4], + // The next connection reads back the payload expiry, which + // this stands for having committed. + interleave: Some(( + 5, + format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'expired', next_attempt_at = NULL, + expired_at = transaction_timestamp() + WHERE event_id = '{expiring}'; + UPDATE {schema}.registry_outbox SET payload = NULL + WHERE event_id = '{expiring}'" + ), + )), + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + + let result = service.claim().await; + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a claim whose commit failed fails" + ); + let recorded = audit.lock().expect("audit lock").clone(); + assert_eq!( + recorded.first(), + Some(&( + DeliveryAuditPhase::Attempt, + DeliveryAuditOutcome::AttemptStarted, + DeliveryAuditDisposition::Leased, + )) + ); + assert!( + recorded.contains(&( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::PayloadExpired, + DeliveryAuditDisposition::Expired, + )), + "the durable expiry is recorded: {recorded:?}" + ); + assert!( + recorded.contains(&( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::WorkerInterrupted, + DeliveryAuditDisposition::RetryPending, + )), + "the attempt of unknown fate is answered: {recorded:?}" + ); + assert_eq!(recorded.len(), 3, "{recorded:?}"); + } + + /// A terminal transition whose commit failed and whose fate cannot be + /// read back still answers the attempt, after a bounded backoff between + /// the read-backs. + #[tokio::test] + #[ignore = "requires a local PostgreSQL test database named by HOOKS_TEST_DATABASE_URL"] + async fn finalize_answers_an_attempt_whose_commit_cannot_be_read_back() { + let url = real_database_url(); + let schema = "hooks_delivery_finalize_unknown_test"; + let mut client = fresh_delivery_schema(&url, schema).await; + let event_id = Uuid::parse_str(STORED_EVENT_ID).expect("event id"); + insert_real_delivery(&mut client, schema, event_id, "1 day", false).await; + let lease_token = Uuid::new_v4(); + let changed = client + .execute( + &format!( + "UPDATE {schema}.registry_webhook_delivery_state + SET state = 'leased', + attempt = 1, + next_attempt_at = NULL, + attempt_started_at = transaction_timestamp(), + lease_expires_at = transaction_timestamp() + interval '30 seconds', + lease_token = $1 + WHERE event_id = $2", + ), + &[&lease_token, &event_id], + ) + .await + .expect("lease the delivery"); + assert_eq!(changed, 1); + client + .batch_execute(&format!( + "CREATE FUNCTION {schema}.refuse_delivered() RETURNS trigger + LANGUAGE plpgsql AS $$ + BEGIN + RAISE EXCEPTION 'terminal commit refused'; + END $$; + CREATE CONSTRAINT TRIGGER refuse_delivered + AFTER UPDATE ON {schema}.registry_webhook_delivery_state + DEFERRABLE INITIALLY DEFERRED + FOR EACH ROW WHEN (NEW.state = 'delivered') + EXECUTE FUNCTION {schema}.refuse_delivered();" + )) + .await + .expect("make the terminal commit fail"); + + let audit = Arc::new(Mutex::new(Vec::new())); + let service = real_service( + RealDbSeams { + refused_connections: vec![2, 3, 4], + ..RealDbSeams::new(&url, &audit) + }, + schema, + ); + let claim = DeliveryClaim { + event_id, + compiled_delivery_id: REAL_DELIVERY_ID.to_owned(), + generation: 1, + attempt: 1, + attempt_started_at: SystemTime::now(), + lease_token, + deployed_maximum_attempts: 3, + retry_delays_ms: vec![1_000, 2_000], + package_revision: REAL_PACKAGE_REVISION.to_owned(), + handler_kind: HookHandlerKind::Rhai, + }; + + let started = Instant::now(); + let result = service + .finalize(&claim, AttemptResult::from(DeliveryAuditOutcome::Delivered)) + .await; + let elapsed = started.elapsed(); + + assert!( + matches!(result, Err(DeliveryError::Unavailable)), + "a terminal commit that failed fails: {result:?}" + ); + assert_eq!( + *audit.lock().expect("audit lock"), + vec![( + DeliveryAuditPhase::Terminal, + DeliveryAuditOutcome::WorkerInterrupted, + DeliveryAuditDisposition::RetryPending, + )] + ); + let backoff: Duration = READ_BACK_BACKOFF.iter().sum(); + assert!( + elapsed >= backoff, + "the read-backs wait {backoff:?} between them, took {elapsed:?}" + ); + } } diff --git a/crates/registry-render/src/audit.rs b/crates/registry-render/src/audit.rs index 7e594e0023..6f30009098 100644 --- a/crates/registry-render/src/audit.rs +++ b/crates/registry-render/src/audit.rs @@ -3,14 +3,18 @@ //! and a `response` entry carrying the outcome before the document leaves; //! both fail closed. A refusal decided before any render (validation, 401, //! 413) is one `response` entry, so the log distinguishes "no attempt" from -//! a refused one. The log carries no hash chain or signature. +//! a refused one. A render whose call ends before its outcome is written, a +//! caller that disconnects included, writes an `unfinished` response, so no +//! request entry stays unpaired. The pairing correlation is drawn by the +//! server for every call; the caller's `Idempotency-Key` is only echoed in +//! the record as `correlationId`. The log carries no hash chain or signature. //! //! Events carry no data values and no asset bytes: identifiers, versions, //! hashes, outcomes, caller, trace and correlation ids only. use serde::Serialize; -use registry_platform_audit::{AuditDestination, AuditEntry, AuditWriter}; +use registry_platform_audit::{AuditDestination, AuditEntry, AuditRequest, AuditWriter}; use crate::problem::{ProblemKind, RenderProblem}; @@ -31,8 +35,9 @@ pub struct RenderAuditEvent { pub document_version: u32, pub bundle_version: u32, pub bundle_hash: String, - /// "rendered" or "refused"; absent on the request entry written before - /// the render starts. + /// "rendered", "refused", or "unfinished" for a call that ended before + /// its outcome; absent on the request entry written before the render + /// starts. #[serde(skip_serializing_if = "Option::is_none")] pub outcome: Option<&'static str>, /// Problem slug for refusals. @@ -116,15 +121,11 @@ impl RenderAuditEvent { } } -/// The envelope correlation for one request: the caller's idempotency key -/// when it is a usable correlation, otherwise a fresh random id. The key is -/// already bounded to 128 characters; one carrying a control character is -/// not a valid correlation, so it gets a drawn id instead. -pub fn correlation(correlation_id: Option<&str>) -> String { - match correlation_id { - Some(id) if !id.is_empty() && !id.chars().any(char::is_control) => id.to_owned(), - _ => uuid::Uuid::new_v4().to_string(), - } +/// The envelope correlation for one call: a fresh random id the server +/// draws. The caller's `Idempotency-Key` is not unique to one call, so it is +/// never the value that pairs a call's entries. +pub fn correlation() -> String { + uuid::Uuid::new_v4().to_string() } impl RenderAudit { @@ -144,16 +145,29 @@ impl RenderAudit { Self { writer } } - /// Append the request entry. It must be accepted before the render - /// starts. + /// Append the request entry and return the handle that owes its + /// response. It must be accepted before the render starts. A call that + /// ends before it responds writes `unfinished` as the response. pub async fn request( &self, correlation: &str, event: RenderAuditEvent, - ) -> Result<(), RenderProblem> { + ) -> Result { + let mut unfinished = record(RenderAuditEvent { + outcome: Some("unfinished"), + ..event.clone() + })?; + if let Some(fields) = unfinished.as_object_mut() { + fields.remove("pdfSha256"); + fields.remove("dataSha256"); + } let record = record(event)?; - self.append(AuditEntry::request(AUDIT_SCHEMA, correlation, record)) + self.writer + .begin(AUDIT_SCHEMA, correlation, record, unfinished) .await + .map_err(|err| { + RenderProblem::new(ProblemKind::AuditFailed, format!("audit append: {err}")) + }) } /// Append the response entry. It must be accepted before the caller @@ -174,6 +188,13 @@ impl RenderAudit { }) } + /// Wait for the unfinished response entries dropped calls handed to a + /// stream destination. + #[cfg(test)] + pub(crate) fn wait_for_detached_entries(&self) { + self.writer.wait_for_detached_entries(); + } + /// Readiness: the destination still accepts entries. pub async fn ready(&self) -> bool { self.writer.ready().await diff --git a/crates/registry-render/src/server.rs b/crates/registry-render/src/server.rs index 51d0d36d81..ca088b1ddf 100644 --- a/crates/registry-render/src/server.rs +++ b/crates/registry-render/src/server.rs @@ -265,7 +265,7 @@ async fn require_bearer( correlation.as_deref(), trace_id(request.headers()).as_deref(), ); - let envelope = crate::audit::correlation(correlation.as_deref()); + let envelope = crate::audit::correlation(); if let Err(audit_problem) = service.audit.response(&envelope, event).await { tracing::error!(problem = %audit_problem, "401 audit append failed"); } @@ -322,7 +322,7 @@ async fn refuse_oversized_bodies( correlation.as_deref(), trace_id(request.headers()).as_deref(), ); - let envelope = crate::audit::correlation(correlation.as_deref()); + let envelope = crate::audit::correlation(); if let Err(audit_problem) = service.audit.response(&envelope, event).await { tracing::error!(problem = %audit_problem, "413 audit append failed"); } @@ -582,8 +582,9 @@ async fn render_route( let document_type = sanitize_document_type(&document_type); let trace = trace_id(&headers); let correlation = correlation_id(&headers); - // One envelope correlation joins this call's request and response entries. - let envelope = crate::audit::correlation(correlation.as_deref()); + // One server-drawn envelope correlation joins this call's request and + // response entries; the caller's key is only echoed in the record. + let envelope = crate::audit::correlation(); let caller = service.caller_fingerprint.clone(); if method != Method::POST { return refuse( @@ -641,28 +642,37 @@ async fn render_route( trace.as_deref(), ); // The request entry is accepted before the render starts; a refused one - // means the render never starts and no response entry follows. - if let Err(audit_problem) = service.audit.request(&envelope, started.clone()).await { - return problem_response(&audit_problem); - } - let outcome = run_render(&service, worker_request).await; - let event = match &outcome { - Ok(rendered) => RenderAuditEvent { - document_id: document_type.clone(), - document_version: rendered.document_version, - bundle_version: rendered.bundle_version, - bundle_hash: rendered.bundle_hash.clone(), - outcome: Some("rendered"), - problem: None, - pdf_sha256: Some(rendered.pdf_sha256.clone()), - data_sha256: Some(rendered.data_sha256.clone()), - caller, - correlation_id: correlation.clone(), - trace_id: trace.clone(), - renderer_version: crate::display_version(), - typst_pin: crate::TYPST_PIN.to_owned(), - }, - Err(problem) => started.refused_after_start(problem), + // means the render never starts and no response entry follows. A call + // dropped from here on writes its unfinished response. + let _request = match service.audit.request(&envelope, started.clone()).await { + Ok(request) => request, + Err(audit_problem) => return problem_response(&audit_problem), + }; + // The answer is built before its response entry, so the entry records + // the outcome the caller actually receives. + let outcome = run_render(&service, worker_request) + .await + .and_then(|rendered| { + let event = RenderAuditEvent { + document_id: document_type.clone(), + document_version: rendered.document_version, + bundle_version: rendered.bundle_version, + bundle_hash: rendered.bundle_hash.clone(), + outcome: Some("rendered"), + problem: None, + pdf_sha256: Some(rendered.pdf_sha256.clone()), + data_sha256: Some(rendered.data_sha256.clone()), + caller, + correlation_id: correlation.clone(), + trace_id: trace.clone(), + renderer_version: crate::display_version(), + typst_pin: crate::TYPST_PIN.to_owned(), + }; + Ok((event, respond_rendered(&headers, rendered)?)) + }); + let (event, outcome) = match outcome { + Ok((event, response)) => (event, Ok(response)), + Err(problem) => (started.refused_after_start(&problem), Err(problem)), }; // The response entry is accepted before anything leaves: audit failure // fails closed and withholds the document. @@ -670,20 +680,17 @@ async fn render_route( return problem_response(&audit_problem); } match outcome { - Ok(rendered) => match respond_rendered(&headers, rendered) { - Ok(mut response) => { - if let Some(correlation) = correlation - .as_ref() - .and_then(|c| header::HeaderValue::from_str(c).ok()) - { - response - .headers_mut() - .insert("idempotency-key", correlation); - } + Ok(mut response) => { + if let Some(correlation) = correlation + .as_ref() + .and_then(|c| header::HeaderValue::from_str(c).ok()) + { response + .headers_mut() + .insert("idempotency-key", correlation); } - Err(problem) => problem_response(&problem), - }, + response + } Err(problem) => problem_response(&problem), } } @@ -1027,12 +1034,20 @@ mod tests { pending.is_err(), "with no render slot free the request must be waiting at the render step" ); + service.audit.wait_for_detached_entries(); let accepted = lines.accepted(); - assert_eq!(accepted.len(), 1, "only the request entry: {accepted:?}"); + // The call dropped while it waited for a render slot, as a canceled + // request is: its request entry is paired with an unfinished + // response, and nothing else was written. + assert_eq!(accepted.len(), 2, "{accepted:?}"); let entry = &accepted[0]; assert_eq!(entry["schema"], crate::audit::AUDIT_SCHEMA); assert_eq!(entry["phase"], "request"); - assert_eq!(entry["correlation"], "effect-1234"); + let correlation = entry["correlation"].as_str().expect("correlation"); + assert_eq!(correlation.len(), 36, "a drawn correlation: {correlation}"); + assert_eq!(accepted[1]["phase"], "response"); + assert_eq!(accepted[1]["correlation"], correlation); + assert_eq!(accepted[1]["record"]["outcome"], "unfinished"); let record = &entry["record"]; assert!( record.get("outcome").is_none(), @@ -1095,7 +1110,8 @@ mod tests { let accepted = lines.accepted(); assert_eq!(accepted.len(), 1, "{accepted:?}"); assert_eq!(accepted[0]["phase"], "response"); - assert_eq!(accepted[0]["correlation"], "effect-5678"); + assert_ne!(accepted[0]["correlation"], "effect-5678"); + assert_eq!(accepted[0]["record"]["correlationId"], "effect-5678"); assert_eq!(accepted[0]["record"]["outcome"], "refused"); assert_eq!(accepted[0]["record"]["problem"], "issued-at-missing"); } @@ -1115,8 +1131,9 @@ mod tests { assert_eq!(accepted.len(), 2, "{accepted:?}"); let entry = &accepted[1]; assert_eq!(entry["phase"], "response"); - assert_eq!(entry["correlation"], "effect-9012"); + assert_eq!(entry["correlation"], accepted[0]["correlation"]); let record = &entry["record"]; + assert_eq!(record["correlationId"], "effect-9012"); assert_eq!(record["outcome"], "refused"); assert_eq!(record["problem"], "internal"); assert_eq!(record["documentId"], "receipt"); @@ -1129,16 +1146,49 @@ mod tests { } #[test] - fn the_envelope_correlation_reuses_the_idempotency_key_or_draws_one() { + fn the_envelope_correlation_is_always_drawn_by_the_server() { + let drawn = crate::audit::correlation(); + assert_eq!(drawn.len(), 36, "a random UUID: {drawn}"); + assert_ne!(drawn, crate::audit::correlation()); + } + + /// Two calls that carry the same caller key still pair unambiguously: + /// the key is the caller's own reference, never the journal's join. + #[tokio::test] + async fn concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries() { + let lines = AuditLines::new(None); + let service = service(&lines, 2); + let (first, second) = tokio::join!( + router(Arc::clone(&service)).oneshot(render_request(Some("shared-key"))), + router(Arc::clone(&service)).oneshot(render_request(Some("shared-key"))), + ); + // Whatever each call's outcome, both reach the render step. assert_eq!( - crate::audit::correlation(Some("effect-1234")), - "effect-1234" + first.expect("router").status(), + second.expect("router").status() ); - let drawn = crate::audit::correlation(None); - assert_eq!(drawn.len(), 36, "a random UUID: {drawn}"); - assert_ne!(drawn, crate::audit::correlation(None)); - // A key the envelope cannot carry verbatim gets a drawn value too; - // the record keeps the caller's key as before. - assert_eq!(crate::audit::correlation(Some("tab\there")).len(), 36); + let accepted = lines.accepted(); + assert_eq!(accepted.len(), 4, "{accepted:?}"); + let mut correlations = std::collections::BTreeMap::>::new(); + for entry in &accepted { + assert_eq!(entry["record"]["correlationId"], "shared-key"); + correlations + .entry( + entry["correlation"] + .as_str() + .expect("correlation") + .to_owned(), + ) + .or_default() + .push(entry["phase"].as_str().expect("phase").to_owned()); + } + assert_eq!( + correlations.len(), + 2, + "one correlation per call: {correlations:?}" + ); + for phases in correlations.values() { + assert_eq!(phases, &["request", "response"]); + } } } diff --git a/crates/registry-render/tests/serve.rs b/crates/registry-render/tests/serve.rs index 84b5d4af39..5ab10bf3ca 100644 --- a/crates/registry-render/tests/serve.rs +++ b/crates/registry-render/tests/serve.rs @@ -747,7 +747,10 @@ fn a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation() { ); } } - assert_eq!(entries[0]["correlation"], "effect-1234"); + // The pairing correlation is always drawn by the server; the caller's + // key stays in the record for the caller's own cross-referencing. + let drawn = entries[0]["correlation"].as_str().expect("correlation"); + assert_eq!(drawn.len(), 36, "a drawn random id: {drawn}"); assert_eq!(entries[0]["record"]["correlationId"], "effect-1234"); let drawn = entries[2]["correlation"].as_str().expect("correlation"); assert_eq!(drawn.len(), 36, "a drawn random id: {drawn}"); diff --git a/crates/registry-scheduling/src/audit.rs b/crates/registry-scheduling/src/audit.rs index cc50f46a7f..d215e98e00 100644 --- a/crates/registry-scheduling/src/audit.rs +++ b/crates/registry-scheduling/src/audit.rs @@ -5,10 +5,20 @@ //! opens and one `response` entry once the decision is known: after commit //! for an allowed commitment, after rollback for a refused one. Both share a //! correlation, which is also the `eventId` the response record carries. +//! A commitment nothing decided, because its transaction rolled back on a +//! failure, the environment records moved under it, or its idempotency key +//! was refused, still answers its request entry with an `unfinished` +//! response naming the reason. A capacity commit that is not acknowledged is +//! read back from the database before it is recorded: one that took effect is +//! answered and recorded as committed, one that rolled back as +//! `commitment.failed`, and one whose status cannot be read is recorded as +//! `commitment.unfinished`, never as failed, because it may have taken +//! effect. One that returns or is canceled without answering writes the +//! `commitment.unfinished` response when its request handle is dropped. //! Entries carry only pseudonymized references and closed codes, never a raw //! principal, grant, claim identifier, or free-text reason. -use registry_platform_audit::{AuditEntry, AuditUnavailable, AuditWriter}; +use registry_platform_audit::{AuditEntry, AuditRequest, AuditUnavailable, AuditWriter}; use serde_json::Value; use uuid::Uuid; @@ -50,6 +60,25 @@ impl SchedulingAudit { .await } + /// Append the `request` entry of one audited operation and return the + /// handle that owes its `response`. Dropped unanswered, the handle writes + /// `unfinished` as the response under the same correlation. + pub async fn begin( + &self, + correlation: Uuid, + record: Value, + unfinished: Value, + ) -> Result { + self.writer + .begin( + SCHEDULING_AUDIT_SCHEMA, + correlation.to_string(), + record, + unfinished, + ) + .await + } + /// Append the `response` entry of one audited operation. pub async fn response(&self, correlation: Uuid, record: Value) -> Result<(), AuditUnavailable> { self.writer @@ -73,6 +102,18 @@ pub fn request_record(mut record: Value) -> Value { record } +/// The `response` form of a commitment nothing decided: the request's fields +/// with the `unfinished` outcome and the closed `reason` it did not finish. +#[must_use] +pub fn unfinished_record(record: Value, reason: &str) -> Value { + let mut record = request_record(record); + if let Some(fields) = record.as_object_mut() { + fields.insert("outcome".to_owned(), Value::String("unfinished".to_owned())); + fields.insert("reason".to_owned(), Value::String(reason.to_owned())); + } + record +} + /// Stamp a `response` record with its audit identity, refusing a record that /// already carries a different one. #[must_use] @@ -103,6 +144,9 @@ mod capture { struct CaptureState { bytes: Vec, accepted_lines: Option, + /// The writer recording here, so a read can wait for the entries it + /// writes when a request handle is dropped. + writer: Option, } /// The lines a test audit destination accepted, and a switch that makes @@ -114,6 +158,10 @@ mod capture { /// Every accepted entry, parsed, in write order. #[must_use] pub fn entries(&self) -> Vec { + let writer = self.0.lock().expect("audit capture").writer.clone(); + if let Some(writer) = writer { + writer.wait_for_detached_entries(); + } let state = self.0.lock().expect("audit capture"); String::from_utf8(state.bytes.clone()) .expect("audit lines are UTF-8") @@ -161,6 +209,7 @@ mod capture { pub fn capture() -> (Self, AuditCapture) { let capture = AuditCapture::default(); let writer = AuditWriter::from_line_sink(Box::new(capture.clone())); + capture.0.lock().expect("audit capture").writer = Some(writer.clone()); (Self::new(writer), capture) } } diff --git a/crates/registry-scheduling/src/hooks.rs b/crates/registry-scheduling/src/hooks.rs index d8233b90c4..a18db6ab41 100644 --- a/crates/registry-scheduling/src/hooks.rs +++ b/crates/registry-scheduling/src/hooks.rs @@ -760,23 +760,20 @@ impl DeliverySeams for SchedulingDeliverySeams { Ok(None) } - /// The platform worker calls this inside the transaction it is about to - /// commit, so the entry is written to the audit destination before that - /// commit: an attempt is on record before its request can leave the - /// process, and a destination that refuses it fails the transition, which - /// rolls back without egress. A transaction that rolls back after the - /// entry was accepted leaves an entry naming a transition that did not - /// commit, and a later pass that takes the transition writes its own. + /// The platform worker records an attempt's start inside the lease + /// transaction before it commits, so an attempt is on record before its + /// request can leave the process and a destination that refuses it rolls + /// the lease back without egress; a lease whose commit then fails is + /// answered with a worker interruption. A terminal disposition, an + /// expiry, and a replay's outcome are recorded only after the transition + /// commits, so no entry names a transition that rolled back. /// - /// An attempt's start is its `request` entry; its terminal disposition - /// and an operator replay are `response` entries. One attempt's entries - /// share a correlation built from the hook event, the compiled delivery, - /// the generation, and the attempt. - async fn record_audit( - &self, - _transaction: &Transaction<'_>, - record: DeliveryAuditRecord<'_>, - ) -> Result<(), DeliveryError> { + /// An attempt's start and an operator's replay request are `request` + /// entries; the terminal disposition and the replay's committed or + /// refused reset are `response` entries. One attempt's entries share a + /// correlation built from the hook event, the compiled delivery, the + /// generation, and the attempt. + async fn record_audit(&self, record: DeliveryAuditRecord<'_>) -> Result<(), DeliveryError> { let audit = json!({ "event": "scheduling.hook-delivery", "hookEventId": record.event_id, @@ -792,11 +789,12 @@ impl DeliverySeams for SchedulingDeliverySeams { "{}/{}/{}/{}", record.event_id, record.compiled_delivery_id, record.generation, record.attempt ); - let entry = match record.phase { - DeliveryAuditPhase::Attempt => { + let entry = match (record.phase, record.outcome) { + (DeliveryAuditPhase::Attempt, _) + | (DeliveryAuditPhase::Replay, DeliveryAuditOutcome::ReplayRequested) => { AuditEntry::request(SCHEDULING_AUDIT_SCHEMA, correlation, audit) } - DeliveryAuditPhase::Terminal | DeliveryAuditPhase::Replay => { + (DeliveryAuditPhase::Terminal | DeliveryAuditPhase::Replay, _) => { AuditEntry::response(SCHEDULING_AUDIT_SCHEMA, correlation, audit) } }; @@ -1089,6 +1087,8 @@ fn audit_outcome(value: DeliveryAuditOutcome) -> &'static str { DeliveryAuditOutcome::PayloadExpired => "payload_expired", DeliveryAuditOutcome::WorkerInterrupted => "worker_interrupted", DeliveryAuditOutcome::ReplayRequested => "replay_requested", + DeliveryAuditOutcome::ReplayCommitted => "replay_committed", + DeliveryAuditOutcome::ReplayRefused => "replay_refused", } } diff --git a/crates/registry-scheduling/src/service.rs b/crates/registry-scheduling/src/service.rs index 80e9f852f5..cb46a6e0df 100644 --- a/crates/registry-scheduling/src/service.rs +++ b/crates/registry-scheduling/src/service.rs @@ -16,7 +16,9 @@ //! a replayed idempotency key answers as it first did. use chrono::{DateTime, TimeDelta, Utc}; -use registry_platform_audit::{AuditKeyHasher, AuthorizationAuditEvent, AuthorizationOutcome}; +use registry_platform_audit::{ + AuditKeyHasher, AuditRequest, AuthorizationAuditEvent, AuthorizationOutcome, +}; use registry_platform_calendar::CalendarInterval; use registry_platform_canonical_json::canonicalize_json; use registry_platform_oidc::GrantClaims; @@ -36,7 +38,7 @@ use sha2::{Digest as _, Sha256}; use std::borrow::Cow; use uuid::Uuid; -use crate::audit::{request_record, with_event_id, SchedulingAudit}; +use crate::audit::{request_record, unfinished_record, with_event_id, SchedulingAudit}; use crate::cursors::{ bind_stored, cursor_expiry, decode_cursor, encode_cursor, CursorError, ListingPosition, StoredCursor, @@ -595,7 +597,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, HOLD_CREATE_ACTION) .await?; let outcome = self @@ -655,7 +657,7 @@ impl SchedulingService { // closing a hold cannot name a resource. let commitment = self.commitment(caller, &actor, &grant, &release_key, &request_hash, now, 0)?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, HOLD_RELEASE_ACTION) .await?; let outcome = self.store.release_hold(hold_id, commitment).await; @@ -736,7 +738,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CREATE_ACTION) .await?; let outcome = self @@ -782,7 +784,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CREATE_ACTION) .await?; let outcome = self @@ -862,7 +864,7 @@ impl SchedulingService { now, facts_revision, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_RESCHEDULE_ACTION) .await?; let outcome = self @@ -928,7 +930,7 @@ impl SchedulingService { now, 0, )?; - let correlation = self + let (correlation, _request) = self .audit_request(caller, &grant, APPOINTMENT_CANCEL_ACTION) .await?; let outcome = self @@ -1141,7 +1143,8 @@ impl SchedulingService { /// the refusal. Readable availability is not authority to book: the /// permission must name all three. /// - /// Both refusals are audited. A refusal decided here never opens the + /// Both refusals are audited, and answered only once their entry is + /// accepted. A refusal decided here never opens the /// capacity transaction, so nothing further in the request would record /// that it happened, and a caller probing which services and locations its /// grant reaches would leave no audit entry. Each refusal is one `response` @@ -1159,7 +1162,7 @@ impl SchedulingService { Uuid::new_v4(), grantless_refusal_record(&self.hasher, &self.scheduling_id, caller, action), ) - .await; + .await?; return Err(ServiceError::Problem(ProblemCode::OperationNotAuthorized)); }; let allowed = grant @@ -1190,43 +1193,47 @@ impl SchedulingService { "authorization.refused", ), ) - .await; + .await?; Err(ServiceError::Problem(ProblemCode::OperationNotAuthorized)) } } - /// Write one authorization refusal as the `response` entry of - /// `correlation`, whose identity it also carries as `eventId`. A - /// destination that refuses it must not change the caller's answer: the - /// decision is already made and the caller is refused either way, so the - /// failure is logged loudly and the refusal stands. - async fn record_refusal(&self, correlation: Uuid, record: Result) { - match record.map(|record| with_event_id(correlation, record)) { - Ok(Some(record)) => { - if let Err(failure) = self.audit.response(correlation, record).await { - tracing::error!(%failure, "the refusal audit entry could not be recorded"); - } - } - Ok(None) => { - tracing::error!("the refusal audit entry carries another identity"); - } - Err(refused) => { - tracing::error!(error = %refused, "the refusal audit entry could not be built"); - } - } + /// Write one refusal, or one commitment nothing decided, as the + /// `response` entry of `correlation`, whose identity it also carries as + /// `eventId`. It fails closed like an allowed response: a refusal whose + /// entry the destination does not accept is answered + /// `service.unavailable`, so no refusal is answered without its audit. + async fn record_refusal( + &self, + correlation: Uuid, + record: Result, + ) -> Result<(), ServiceError> { + let record = with_event_id(correlation, record?).ok_or_else(|| { + ServiceError::internal("the refusal audit record carries another identity") + })?; + self.audit + .response(correlation, record) + .await + .map_err(|failure| { + tracing::error!(%failure, "the refusal audit entry was refused"); + ServiceError::Problem(ProblemCode::ServiceUnavailable) + }) } /// Append the `request` entry of one commitment and return the - /// correlation its `response` entry will carry. It names the caller, the - /// grant, and the operation the capacity transaction will decide, and no - /// outcome. A destination that refuses it refuses the commitment: the - /// capacity transaction does not open. + /// correlation its `response` entry will carry, with the handle that owes + /// it. It names the caller, the grant, and the operation the capacity + /// transaction will decide, and no outcome. A destination that refuses it + /// refuses the commitment: the capacity transaction does not open. The + /// caller holds the handle until it answers: the response written under + /// the correlation answers it, and a commitment that returns or is + /// canceled before then writes the `commitment.unfinished` response. async fn audit_request( &self, caller: &Caller, grant: &GrantClaims, operation: &str, - ) -> Result { + ) -> Result<(Uuid, AuditRequest), ServiceError> { let record = audit_record( &self.hasher, &self.scheduling_id, @@ -1237,14 +1244,20 @@ impl SchedulingService { "authorization.allowed", )?; let correlation = Uuid::new_v4(); - self.audit - .request(correlation, request_record(record)) + let unfinished = with_event_id( + correlation, + unfinished_record(record.clone(), "commitment.unfinished"), + ) + .ok_or_else(|| ServiceError::internal("the commitment audit record carries an identity"))?; + let request = self + .audit + .begin(correlation, request_record(record), unfinished) .await .map_err(|failure| { tracing::error!(%failure, "the commitment request audit entry was refused"); ServiceError::Problem(ProblemCode::ServiceUnavailable) })?; - Ok(correlation) + Ok((correlation, request)) } /// Append the `response` entry that gates an answer: the allowed entry @@ -1398,9 +1411,13 @@ impl SchedulingService { /// accepted, and a replayed receipt, including one a concurrent identical /// request won, only once the `response` entry recording its decision /// is. A refusal writes its receipt under the caller's idempotency key so - /// a replay of that key answers the same, and writes its denied - /// `response` entry when the refusal was an authorization decision. Both - /// entries carry the `correlation` of the commitment's `request` entry. + /// a replay of that key answers the same, and is answered only once its + /// `response` entry is accepted: denied for a decision the ledger took, + /// `unfinished` with its reason for a failed transaction, a replaced + /// environment, a refused idempotency key, or a commit whose outcome + /// could not be read back. A commit that took effect though its + /// acknowledgment was lost reaches here as minted. Every entry carries the + /// `correlation` of the commitment's `request` entry. #[allow(clippy::too_many_arguments)] async fn commitment_outcome( &self, @@ -1439,32 +1456,63 @@ impl SchedulingService { Ok(CommitmentAnswer::Minted(T::from_minted(minted))) } Err(error) => { - if matches!( + let mut problem = problem_of(&error); + // Every commitment answers its request entry, refused as much + // as allowed, and one nothing decided as well. The match is + // exhaustive on purpose: a new variant must state which side + // it falls on rather than inherit silence from a wildcard. + let mut response = match &error { + // The transaction failed and the caller sees no detail. + CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) => { + tracing::error!(%error, "the Scheduling store failed mid-commitment"); + Unanswered::Unfinished("commitment.failed") + } + // The commit was not acknowledged and its read-back + // failed, so it may have taken effect: this records the + // outcome as unknown, never as failed. + CommitError::Unacknowledged => { + tracing::error!(%error, "a Scheduling commitment's outcome is unknown"); + Unanswered::Unfinished("commitment.unfinished") + } + // A records replacement moved under this request, and the + // caller retries; the swap is an expected operator act, so + // this is a warning, not a failure. + CommitError::FactsStale => { + tracing::warn!( + %error, + "the environment records were replaced while a commitment was in flight" + ); + Unanswered::Unfinished("commitment.facts-stale") + } + // The idempotency layer refused the key, not the + // commitment, and the key says nothing about what a grant + // reaches, so this is no authorization decision. + CommitError::KeyReused | CommitError::KeyExpired => { + Unanswered::Unfinished(key_refusal_reason(&error)) + } + CommitError::Unauthorized => Unanswered::Denied("authorization.refused"), + CommitError::Refused(_) + | CommitError::HoldCeiling + | CommitError::RevisionMismatch + | CommitError::CutoffPassed => Unanswered::Denied("authorization.profile"), + }; + // A refusal writes its receipt so a replay of the key answers + // the same. A failed transaction and a replaced environment + // decided nothing, an unacknowledged commit may already hold + // the key's receipt, and an expired receipt cannot be + // recreated. A reused key still passes through the + // insert-or-replay path: a concurrent identical winner is + // replayed, while a different request hash remains key-reused. + let receipted = !matches!( error, - CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) - ) { - // Nothing was decided: no receipt, no response entry, and - // the caller sees no detail. - tracing::error!(%error, "the Scheduling store failed mid-commitment"); - return Err(ServiceError::Problem(ProblemCode::ServiceUnavailable)); - } - if matches!(error, CommitError::FactsStale) { - // A records replacement moved under this request. Nothing - // was decided and the caller retries; the swap is an - // expected operator act, so this is a warning, not a - // failure. - tracing::warn!( - %error, - "the environment records were replaced while a commitment was in flight" - ); - return Err(ServiceError::Problem(ProblemCode::ServiceUnavailable)); - } - let problem = problem_of(&error); - // An expired receipt cannot be recreated. A reused key still - // passes through the insert-or-replay path: a concurrent - // identical winner is replayed, while a different request - // hash remains key-reused. - if !matches!(error, CommitError::KeyExpired) { + CommitError::Store(_) + | CommitError::Query(_) + | CommitError::Hooks(_) + | CommitError::FactsStale + | CommitError::KeyExpired + | CommitError::Unacknowledged + ); + if receipted { let receipt = self.commitment( caller, actor, @@ -1506,7 +1554,8 @@ impl SchedulingService { Err( failure @ (CommitError::KeyReused | CommitError::KeyExpired), ) => { - return Err(ServiceError::Problem(problem_of(&failure))); + problem = problem_of(&failure); + response = Unanswered::Unfinished(key_refusal_reason(&failure)); } Err(failure) => { tracing::error!(%failure, "the refused attempt receipt could not be recorded"); @@ -1518,43 +1567,28 @@ impl SchedulingService { } } } - // Every commitment the ledger decides is attributable, refused - // as much as allowed. The match is exhaustive on purpose: a - // new variant must state which side it falls on rather than - // inherit silence from a wildcard. - let audited = match error { - // Nothing was decided. The transaction failed, or the - // environment moved and the caller retries against the - // current records, so there is no verdict to attribute. - CommitError::Store(_) - | CommitError::Query(_) - | CommitError::Hooks(_) - | CommitError::FactsStale => None, - // The idempotency layer refused the key, not the - // commitment. The attempt receipt above already records - // it, and the key says nothing about what a grant reaches. - CommitError::KeyReused | CommitError::KeyExpired => None, - CommitError::Unauthorized => Some("authorization.refused"), - CommitError::Refused(_) - | CommitError::HoldCeiling - | CommitError::RevisionMismatch - | CommitError::CutoffPassed => Some("authorization.profile"), - }; - if let Some(reason) = audited { - self.record_refusal( - correlation, - audit_record( - &self.hasher, - &self.scheduling_id, - caller, - grant, - operation, - AuthorizationOutcome::Denied, - reason, - ), + let record = match response { + Unanswered::Denied(reason) => audit_record( + &self.hasher, + &self.scheduling_id, + caller, + grant, + operation, + AuthorizationOutcome::Denied, + reason, + ), + Unanswered::Unfinished(reason) => audit_record( + &self.hasher, + &self.scheduling_id, + caller, + grant, + operation, + AuthorizationOutcome::Allowed, + "authorization.allowed", ) - .await; - } + .map(|record| unfinished_record(record, reason)), + }; + self.record_refusal(correlation, record).await?; Err(ServiceError::Problem(problem)) } } @@ -2110,6 +2144,24 @@ impl ClaimRow { } } +/// The `response` a commitment that is not answered by a success or a replay +/// writes: a refusal the ledger or the permission check decided, or the +/// closed reason a commitment nothing decided did not finish. +#[derive(Clone, Copy)] +enum Unanswered { + Denied(&'static str), + Unfinished(&'static str), +} + +/// The closed reason an idempotency key refusal records. +fn key_refusal_reason(error: &CommitError) -> &'static str { + if matches!(error, CommitError::KeyExpired) { + "idempotency.expired" + } else { + "idempotency.key-reused" + } +} + fn problem_of(error: &CommitError) -> ProblemCode { match error { CommitError::Store(_) | CommitError::Query(_) | CommitError::Hooks(_) => { @@ -2122,9 +2174,11 @@ fn problem_of(error: &CommitError) -> ProblemCode { CommitError::Unauthorized => ProblemCode::OperationNotAuthorized, CommitError::RevisionMismatch => ProblemCode::RevisionMismatch, CommitError::CutoffPassed => ProblemCode::CancellationCutoffPassed, - // Never reached: the outcome handler intercepts a stale-facts - // refusal before it projects, because nothing was decided. + // Nothing was decided: the caller retries against the current records. CommitError::FactsStale => ProblemCode::ServiceUnavailable, + // The caller retries under the same key, which replays the receipt + // if the commit took effect. + CommitError::Unacknowledged => ProblemCode::ServiceUnavailable, } } diff --git a/crates/registry-scheduling/src/store.rs b/crates/registry-scheduling/src/store.rs index 75e094c5e5..b862524aa2 100644 --- a/crates/registry-scheduling/src/store.rs +++ b/crates/registry-scheduling/src/store.rs @@ -34,6 +34,10 @@ use std::collections::HashMap; use std::str::FromStr; +#[cfg(feature = "postgres-test")] +use std::sync::atomic::{AtomicBool, Ordering}; +#[cfg(feature = "postgres-test")] +use std::sync::Arc; use std::time::Duration; use chrono::{DateTime, TimeDelta, Utc}; @@ -177,6 +181,12 @@ pub enum StoreError { pub struct PostgresStore { pool: Pool, clock: SharedClock, + /// Test switches that report the next capacity commit that took effect + /// as unacknowledged, and the next read-back of one as unreadable. + #[cfg(feature = "postgres-test")] + lose_acknowledgment: Arc, + #[cfg(feature = "postgres-test")] + fail_read_back: Arc, } impl PostgresStore { @@ -210,6 +220,117 @@ impl PostgresStore { .write() .expect("the store clock is never held across a panic") = clock; } + + /// Report the next capacity commit as unacknowledged after it took + /// effect, as a connection lost during `COMMIT` does. Test-only, and + /// shared by every handle cloned from this store. + #[cfg(feature = "postgres-test")] + pub fn lose_next_commit_acknowledgment(&self) { + self.lose_acknowledgment.store(true, Ordering::SeqCst); + } + + /// Make the next read-back of an unacknowledged commit unreadable, as a + /// database that cannot be reached again does. Test-only. + #[cfg(feature = "postgres-test")] + pub fn fail_next_read_back(&self) { + self.fail_read_back.store(true, Ordering::SeqCst); + } + + /// Whether the test switch reports this commit as unacknowledged. Always + /// false outside tests. + fn acknowledgment_lost(&self) -> bool { + #[cfg(feature = "postgres-test")] + { + self.lose_acknowledgment.swap(false, Ordering::SeqCst) + } + #[cfg(not(feature = "postgres-test"))] + { + false + } + } + + /// Whether the test switch makes this read-back unreadable. Always false + /// outside tests. + fn read_back_fails(&self) -> bool { + #[cfg(feature = "postgres-test")] + { + self.fail_read_back.swap(false, Ordering::SeqCst) + } + #[cfg(not(feature = "postgres-test"))] + { + false + } + } + + /// Commit a capacity transaction whose decision is already written. + /// + /// A `COMMIT` whose acknowledgment never arrives may still have taken + /// effect, as when the connection is lost after it reached the database. + /// Its transaction's status is then read back on another connection, + /// outside any transaction: the read writes nothing and takes no + /// capacity lock. A commit that took effect is answered as committed, + /// one that rolled back as the query failure it was, and one whose + /// status cannot be read as [`CommitError::Unacknowledged`]. + async fn commit_capacity( + &self, + transaction: deadpool_postgres::Transaction<'_>, + ) -> Result<(), CommitError> { + let transaction_id: String = transaction + .query_one("SELECT pg_current_xact_id()::text", &[]) + .await? + .get(0); + let failure = match transaction.commit().await { + Ok(()) if !self.acknowledgment_lost() => return Ok(()), + Ok(()) => None, + Err(error) => Some(error), + }; + match self.read_back(&transaction_id).await { + ReadBack::Committed => { + tracing::warn!( + "a Scheduling capacity commit was not acknowledged but took effect; it is answered as committed" + ); + Ok(()) + } + ReadBack::RolledBack => { + Err(failure.map_or(CommitError::Unacknowledged, CommitError::Query)) + } + ReadBack::Unknown => { + tracing::error!( + "a Scheduling capacity commit was not acknowledged and its outcome could not be read back" + ); + Err(CommitError::Unacknowledged) + } + } + } + + /// Read whether the transaction `transaction_id` committed, on a pooled + /// connection outside any transaction. + async fn read_back(&self, transaction_id: &str) -> ReadBack { + if self.read_back_fails() { + return ReadBack::Unknown; + } + let Ok(client) = self.client().await else { + return ReadBack::Unknown; + }; + let status = client + .query_one("SELECT pg_xact_status($1::text::xid8)", &[&transaction_id]) + .await + .map(|row| row.get::<_, Option>(0)); + match status.as_ref().map(|status| status.as_deref()) { + Ok(Some("committed")) => ReadBack::Committed, + Ok(Some("aborted")) => ReadBack::RolledBack, + // Still in progress, too old to report, or unreadable. + _ => ReadBack::Unknown, + } + } +} + +/// What reading back an unacknowledged capacity commit found. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum ReadBack { + Committed, + RolledBack, + Unknown, } impl std::fmt::Debug for PostgresStore { @@ -410,6 +531,11 @@ pub enum CommitError { /// current records. #[error("the environment records were replaced while the request was in flight")] FactsStale, + /// The capacity transaction's `COMMIT` was not acknowledged and reading + /// its status back failed, so whether it took effect is unknown. A retry + /// under the same idempotency key replays its receipt if it did. + #[error("the capacity commit was not acknowledged and its outcome is unknown")] + Unacknowledged, } /// One due or delivered outbox intent. @@ -512,6 +638,10 @@ impl PostgresStore { Ok(Self { pool, clock: system_clock(), + #[cfg(feature = "postgres-test")] + lose_acknowledgment: Arc::default(), + #[cfg(feature = "postgres-test")] + fail_read_back: Arc::default(), }) } @@ -1366,7 +1496,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Hold(claim)) } @@ -1474,7 +1604,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(claim)) } @@ -1620,7 +1750,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(claim)) } @@ -1687,7 +1817,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Released) } @@ -1845,7 +1975,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Booking(moved)) } @@ -1971,7 +2101,7 @@ impl PostgresStore { ) .await?; self.recheck_grant(&commitment)?; - transaction.commit().await?; + self.commit_capacity(transaction).await?; Ok(CommitOutcome::Cancelled(cancelled)) } diff --git a/crates/registry-scheduling/tests/postgres_commitments.rs b/crates/registry-scheduling/tests/postgres_commitments.rs index 133319c83b..88dbc16b93 100644 --- a/crates/registry-scheduling/tests/postgres_commitments.rs +++ b/crates/registry-scheduling/tests/postgres_commitments.rs @@ -6550,3 +6550,400 @@ async fn migration_refuses_to_drop_unpublished_audit_and_drops_a_drained_outbox( .await .expect("the schema is at this release"); } + +/// The single `response` entry that answers the commitment's `request` +/// entry among `appended`, after checking the pair shares one correlation +/// and records a commitment nothing decided. +fn one_unfinished_response(appended: &[Value], reason: &str) -> Value { + let response = one_request_and_one_response(appended); + assert_eq!(response["outcome"], "unfinished", "{response}"); + assert_eq!(response["reason"], reason, "{response}"); + response +} + +/// SCHEDULING-SEC-14: a capacity transaction that fails decided nothing, and +/// its request entry is still answered: one response entry records that the +/// commitment did not finish. +#[tokio::test] +async fn a_failed_capacity_transaction_pairs_its_request_entry() { + let fx = hook_fixture().await; + fx.admin + .batch_execute( + "ALTER TABLE registry_outbox + ADD CONSTRAINT test_refuse_hook_capture + CHECK (event_type <> 'confirmed-observer')", + ) + .await + .expect("install the transaction failure seam"); + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "failed-transaction", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + let response = + one_unfinished_response(&fx.capture.entries().split_off(before), "commitment.failed"); + assert_eq!(response["operation"], "appointment.create"); +} + +/// SCHEDULING-SEC-14: a capacity commit that took effect though its +/// acknowledgment was lost is read back as committed. The caller gets the +/// appointment, its response entry records it allowed, and a retry under the +/// same key replays the same appointment. +#[tokio::test] +async fn a_commitment_whose_commit_acknowledgment_is_lost_is_read_back_as_committed() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let before = fx.capture.entries().len(); + fx.store.lose_next_commit_acknowledgment(); + let (status, appointment) = fx + .post("/v1/appointments", &fx.agent, "lost-ack", body.clone()) + .await; + assert_eq!(status, StatusCode::CREATED, "{appointment}"); + let response = one_request_and_one_response(&fx.capture.entries().split_off(before)); + assert_eq!(response["operation"], "appointment.create"); + assert_eq!(response["outcome"], "allowed"); + assert_eq!(response["reason"], "authorization.allowed"); + + let (status, replayed) = fx + .post("/v1/appointments", &fx.agent, "lost-ack", body) + .await; + assert_eq!(status, StatusCode::CREATED, "{replayed}"); + assert_eq!(replayed["appointmentId"], appointment["appointmentId"]); +} + +/// SCHEDULING-SEC-14: a capacity commit that was not acknowledged and whose +/// status cannot be read back is recorded as unfinished, never as failed, +/// because it may have taken effect. Here it did, so a retry under the same +/// key replays the appointment it committed. +#[tokio::test] +async fn a_commitment_whose_outcome_cannot_be_read_back_is_recorded_unfinished() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let before = fx.capture.entries().len(); + fx.store.lose_next_commit_acknowledgment(); + fx.store.fail_next_read_back(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "unknown-commit", + body.clone(), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); + let response = one_unfinished_response( + &fx.capture.entries().split_off(before), + "commitment.unfinished", + ); + assert_eq!(response["operation"], "appointment.create"); + + let (status, replayed) = fx + .post("/v1/appointments", &fx.agent, "unknown-commit", body) + .await; + assert_eq!(status, StatusCode::CREATED, "{replayed}"); +} + +/// SCHEDULING-SEC-14: a capacity transaction refused at `COMMIT` itself rolls +/// back, and the read-back finds it rolled back, so its request entry is +/// answered as failed. +#[tokio::test] +async fn a_commitment_refused_at_commit_is_read_back_as_failed() { + let fx = fixture().await; + fx.admin + .batch_execute( + "CREATE FUNCTION test_refuse_at_commit() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'refused at commit'; END $$; + CREATE CONSTRAINT TRIGGER test_refuse_claim_at_commit AFTER INSERT ON scheduling_claims + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION test_refuse_at_commit();", + ) + .await + .expect("install a trigger that refuses the commitment at COMMIT"); + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "refused-at-commit", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + let response = + one_unfinished_response(&fx.capture.entries().split_off(before), "commitment.failed"); + assert_eq!(response["operation"], "appointment.create"); + let claims: i64 = fx + .admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + assert_eq!(claims, 0, "the refused commitment rolled back"); +} + +/// SCHEDULING-SEC-14: a commitment whose caller goes away while its capacity +/// transaction waits for the supply lock pairs its request entry with exactly +/// one unfinished response, and commits nothing. +#[tokio::test] +async fn a_commitment_dropped_inside_its_capacity_transaction_writes_one_unfinished_response() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let (http, agent, capture) = (fx.http.clone(), fx.agent.clone(), fx.capture.clone()); + let before = capture.entries().len(); + let mut admin = fx.admin; + let claims_before: i64 = admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + let holder = admin + .transaction() + .await + .expect("the stand-in capacity transaction opens"); + holder + .execute("SELECT supply_id FROM scheduling_supply FOR UPDATE", &[]) + .await + .expect("the stand-in holds every supply anchor"); + let holder_pid: i32 = holder + .query_one("SELECT pg_backend_pid()", &[]) + .await + .expect("the stand-in's backend") + .get(0); + + let commitment = tokio::spawn(send( + http, + "POST".to_owned(), + "/v1/appointments".to_owned(), + agent, + Some("dropped-commitment".to_owned()), + Some(body), + )); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + let waiting: i64 = holder + .query_one( + "SELECT count(DISTINCT pid) FROM pg_locks WHERE $1=ANY(pg_blocking_pids(pid))", + &[&holder_pid], + ) + .await + .expect("read the waiting commitment") + .get(0); + if waiting > 0 { + break; + } + assert!( + !commitment.is_finished(), + "the commitment finished before reaching the supply lock" + ); + assert!( + std::time::Instant::now() < deadline, + "the commitment never waited on the supply lock" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + commitment.abort(); + assert!(commitment + .await + .expect_err("the commitment was dropped") + .is_cancelled()); + holder + .rollback() + .await + .expect("the stand-in releases the supply"); + + let response = one_unfinished_response( + &capture.entries().split_off(before), + "commitment.unfinished", + ); + assert_eq!(response["operation"], "appointment.create"); + let claims_after: i64 = admin + .query_one("SELECT count(*) FROM scheduling_claims", &[]) + .await + .expect("count claims") + .get(0); + assert_eq!( + claims_after, claims_before, + "the dropped commitment committed nothing" + ); +} + +/// SCHEDULING-SEC-14: an idempotency key refused as reused, and one refused +/// as expired, each answer the request entry the commitment wrote, without +/// recording the key refusal as an authorization decision. +#[tokio::test] +async fn an_idempotency_key_refusal_pairs_its_request_entry() { + let fx = fixture().await; + let (first, second) = first_overlapping_pair(&fx, OFFERING, 90, 260).await; + let (status, _) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, first)}), + ) + .await; + assert_eq!(status, StatusCode::CREATED); + + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, second)}), + ) + .await; + assert_eq!(status, StatusCode::CONFLICT, "{problem}"); + one_unfinished_response( + &fx.capture.entries().split_off(before), + "idempotency.key-reused", + ); + + fx.store + .erase_expired_attempts(Utc::now() + TimeDelta::days(8)) + .await + .expect("the retention sweep runs"); + let before = fx.capture.entries().len(); + let (status, problem) = fx + .post( + "/v1/appointments", + &fx.agent, + "idem-audit", + json!({"hold": null, "admission": admission(&fx, OFFERING, first)}), + ) + .await; + assert_eq!(status, StatusCode::GONE, "{problem}"); + one_unfinished_response( + &fx.capture.entries().split_off(before), + "idempotency.expired", + ); +} + +/// SCHEDULING-SEC-14: a records replacement that lands while a commitment +/// waits on its revision guard leaves the commitment undecided, and its +/// request entry is still answered. +#[tokio::test] +async fn a_records_swap_under_a_commitment_pairs_its_request_entry() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let body = json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}); + let (http, agent, capture) = (fx.http.clone(), fx.agent.clone(), fx.capture.clone()); + let before = capture.entries().len(); + let mut admin = fx.admin; + let swap = admin + .transaction() + .await + .expect("the stand-in records swap opens"); + swap.execute( + "UPDATE scheduling_meta SET facts_revision = facts_revision + 1 WHERE singleton", + &[], + ) + .await + .expect("the stand-in swap moves the records revision"); + + let commitment = tokio::spawn(send( + http, + "POST".to_owned(), + "/v1/appointments".to_owned(), + agent, + Some("swapped-records".to_owned()), + Some(body), + )); + // Wait until the commitment is blocked on the revision guard, which it + // reaches only after reading the records it was evaluated against. + loop { + swap.batch_execute("SELECT pg_stat_clear_snapshot()") + .await + .expect("refresh the activity snapshot"); + let waiting: i64 = swap + .query_one( + "SELECT count(*) FROM pg_stat_activity + WHERE wait_event_type = 'Lock' AND query LIKE '%FROM scheduling_meta%FOR SHARE%'", + &[], + ) + .await + .expect("read the waiting commitment") + .get(0); + if waiting > 0 { + break; + } + assert!( + !commitment.is_finished(), + "the commitment finished before reaching its revision guard" + ); + tokio::task::yield_now().await; + } + swap.commit().await.expect("the stand-in swap commits"); + + let (status, problem) = commitment.await.expect("the commitment answers"); + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + one_unfinished_response( + &capture.entries().split_off(before), + "commitment.facts-stale", + ); +} + +/// SCHEDULING-SEC-14: a refusal is answered only once its response entry is +/// accepted, like an allowed commitment. When the destination refuses the +/// response entry of a refusal the ledger decided, the caller is told the +/// service is unavailable. The permission mismatch decided before the +/// transaction is covered by the test that follows. +#[tokio::test] +async fn a_refusal_whose_response_entry_is_refused_answers_service_unavailable() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + let (status, appointment) = fx + .post( + "/v1/appointments", + &fx.agent, + "refusal-audit-create", + json!({"hold": null, "admission": admission(&fx, OFFERING, slot)}), + ) + .await; + assert_eq!(status, StatusCode::CREATED); + let appointment_id = appointment["appointmentId"].as_str().unwrap().to_owned(); + let stale = appointment["revision"].as_u64().unwrap() + 1; + + // A ledger refusal: the request entry is accepted, its response is not. + fx.capture.refuse_after(fx.capture.entries().len() + 1); + let other = first_slot(&fx, OFFERING, 480, 620).await; + let (status, problem) = fx + .post( + &format!("/v1/appointments/{appointment_id}/reschedule"), + &fx.agent, + "refusal-audit-stale", + json!({"observedRevision": stale, "admission": admission(&fx, OFFERING, other)}), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); +} + +/// SCHEDULING-SEC-14: a permission mismatch refused before the transaction +/// is answered only once its response entry is accepted. +#[tokio::test] +async fn a_permission_refusal_the_destination_refuses_answers_service_unavailable() { + let fx = fixture().await; + let slot = first_slot(&fx, OFFERING, 300, 440).await; + fx.capture.refuse_after(fx.capture.entries().len()); + let (status, problem) = fx + .post( + "/v1/holds", + &agent_token_outside_its_bounds(), + "outside-bounds-unaudited", + admission(&fx, OFFERING, slot), + ) + .await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE, "{problem}"); + assert_eq!(problem["code"], "service.unavailable"); +} diff --git a/docs/site/src/content/docs/operate/breg-retention.mdx b/docs/site/src/content/docs/operate/breg-retention.mdx index db60c26b24..f178ce7796 100644 --- a/docs/site/src/content/docs/operate/breg-retention.mdx +++ b/docs/site/src/content/docs/operate/breg-retention.mdx @@ -312,7 +312,25 @@ entry before disclosure or after a mutation commits. The entries share a correla keyed references and closed vocabulary, not raw record values. A refusal can have only a response entry. The file destination acknowledges an append after fsync; `stdout` flushes each line and provides best-effort delivery. A failed request entry blocks protected I/O, and a failed response -entry blocks disclosure. A crash can leave a request entry without a response. +entry blocks disclosure. A request that ends before its outcome is known, because it failed, timed +out, or its caller went away, or an operator command that exits early, still answers its request +entry. A runtime read or mutation answers with a response whose `phase` is `unfinished`, carrying +only the operation and request it answers; an operator command answers with a `terminal` response +whose `outcome` is `unfinished`. Only a crash, or a destination that already stopped accepting +entries, can leave a request entry without a response. + +Operator commands follow the same rule. `evidence-retention erase-expired` records the cutoff it +was given in its request entry and the number of assertions it erased in its response. +`request-retention erase` deletes the external attachment objects before it records its response, +which states how many objects still wait for deletion; when that deletion pass itself fails, the +count is `null` and the command reports the failure. If the erasure committed but its response +entry could not be written, the command reports `request_retention.erasure.unaudited`: the detail is +gone, so restore the audit destination and reconcile the erased request against the database. An +event delivery's attempt is recorded inside the transaction that leases it, before its request +leaves. If that lease transaction does not commit, no request leaves and the attempt is answered +`worker_interrupted`, so an attempt entry does not prove a request was sent. An attempt's terminal +outcome is recorded only once the delivery's state has committed; an operator replay is recorded as a request before the reset and a response once it commits or is +refused. A crash or a killed process can also tear the file's final line itself, leaving it without its closing newline. The writer refuses to open a destination in that state, so the next start of the @@ -359,9 +377,14 @@ Choose a fresh audit path when changing from another audit format, and preserve outside the new writer's retention directory. ::: -{/* Evidence: crates/registry-breg/src/audit.rs; +{/* Evidence: crates/registry-breg/src/audit.rs, unfinished_record; crates/registry-breg/src/runtime_config.rs, AuditConfig; - crates/registry-platform-audit/src/writer.rs; + crates/registry-platform-audit/src/writer.rs, AuditRequest; + crates/registry-breg/src/request_retention.rs, ErasureUnaudited, retention_outcome_record, + and pendingExternalDeletions; + crates/registry-platform-hooks/src/delivery/service.rs, WorkerInterrupted; + crates/registry-breg/src/action_evidence_maintenance.rs, EVIDENCE_RETENTION_AUDIT_SCHEMA; + crates/registry-platform-hooks/src/delivery/seams.rs, record_audit; crates/registry-breg/src/postgres/schema.rs; crates/registry-bregctl/src/dev/mod.rs. */} diff --git a/docs/site/src/content/docs/operate/registry-render.mdx b/docs/site/src/content/docs/operate/registry-render.mdx index b254310177..a35dc3e6e5 100644 --- a/docs/site/src/content/docs/operate/registry-render.mdx +++ b/docs/site/src/content/docs/operate/registry-render.mdx @@ -172,8 +172,10 @@ Each line is one entry: `schema` (`render.registrystack.org/audit/v1`), `eventId `phase` (`request` or `response`), `correlation`, and the value-free `record`: hashes, versions, the caller's key fingerprint, the outcome, and the trace and correlation ids, never data values or image bytes. The `request` entry carries no outcome. Both entries of one render share their -`correlation`: the caller's `Idempotency-Key`, or a random identifier when there is none. For -example, to list outcomes by correlation: +`correlation`, a random identifier the server draws for every call; the caller's `Idempotency-Key` +is recorded only as the record's `correlationId`, because two calls may carry the same key. A call +that ends before its outcome is written, such as one whose caller disconnected, has the outcome +`unfinished`. For example, to list outcomes by correlation: ```sh jq -c '{correlation, phase, outcome: .record.outcome}' /var/lib/registry-render/audit/render.jsonl @@ -195,6 +197,8 @@ audit file resolves under a persistent root, as a container preflight, and refus destination. {/* Evidence: crates/registry-render/src/audit.rs, AUDIT_SCHEMA, RenderAuditEvent, and correlation; + crates/registry-render/src/audit.rs, RenderAudit::request writes unfinished when dropped; + crates/registry-render/src/server.rs, the_request_entry_is_accepted_before_the_render_starts; crates/registry-render/src/cli.rs, require_audit_under; crates/registry-platform-audit/src/writer.rs, DEFAULT_AUDIT_ROTATE_BYTES and DEFAULT_AUDIT_RETAIN_DAYS. */} diff --git a/products/breg/CHANGELOG.md b/products/breg/CHANGELOG.md index 8fee4f0f59..244c0d7e44 100644 --- a/products/breg/CHANGELOG.md +++ b/products/breg/CHANGELOG.md @@ -2,6 +2,40 @@ ## Unreleased +- Answer every audited request entry. A read, mutation, action, or request + action that ends after its attempt without a terminal or refusal entry, + because it failed, timed out, or its caller went away, writes a response + with the phase `unfinished` under the same correlation. A read that fails + after its rows were read writes the Refused terminal, and a terminal the + destination refuses is logged instead of discarded. + - A reviewed change-request apply records its attempt before the receipt + preflight's reads and the review authority, and holds it through the + action. + - `migration reconcile` answers a transition that fails after its request + entry with a `failed` response. + - `request-retention erase` answers a refused or failed erasure with a + `refused` or `failed` response, deletes the external attachment objects + before it records a committed erasure, and records how many objects still + wait for deletion. An erasure that committed without its response entry + reports `request_retention.erasure.unaudited`. `request-retention + cleanup-attachments` answers a failed cleanup with a `failed` response. + - `history rebaseline`, a standalone `history erase`, and a + field-encryption erase-and-rebaseline run answer their request entry with + an `unfinished` response when they are refused or fail before their + terminal entry, as does an attachment verification job that stops early. + - `evidence-retention erase-expired` is audited under + `breg-evidence-retention-audit/v1`: a request entry naming the cutoff + before the erasure, and a response with the erased count or `failed`. + - An ingestion run creation, cancellation, chunk replay, or receipt + recovery refused after its `breg-ingestion-audit/v1` request entry is + answered in that schema with a `refused` response, and not recorded again + as a general refusal. + - Event delivery records a terminal outcome and a payload expiry only after + the delivery state commits. An attempt whose lease commit fails is + answered with `worker_interrupted`. An operator replay writes a + `replay_requested` request before the reset and a `replay_committed` or + `replay_refused` response after it. + - BREAKING: write audit through the platform audit writer instead of a hash-chained journal in PostgreSQL. Each process opens one writer at startup and writes JSON Lines entries `{schema, correlation, phase, time, diff --git a/products/breg/DEFINITION-OF-DONE.md b/products/breg/DEFINITION-OF-DONE.md index 5ab044e6cb..dd2332f728 100644 --- a/products/breg/DEFINITION-OF-DONE.md +++ b/products/breg/DEFINITION-OF-DONE.md @@ -86,10 +86,12 @@ rebaseline, or reconciliation finds the state the first run committed rather than replaying its terminal entry. Both recoveries are operational, not a second audit mechanism. -`BREG-V1-27` is `partial`: most caller-requested operations pair their -audit request entry with a response entry, but the paths its `gap` lists can -still leave a request entry unpaired or write no entry, so the row makes no -completion claim until they do. +The pairing of a request entry with its response is a property of a running +process. A request whose operation is abandoned is answered `unfinished`, and +so is an ingestion transition whose commit returned an error, since that error +does not prove a rollback. A process that is killed, exits, or shuts its +runtime down while a write is in flight may leave a request without its +response, and `BREG-V1-27` claims no more than that. The HTTP record contract is also explicit: caller-filtered and generated OpenAPI artifacts assign every record-related route to the shared single or diff --git a/products/breg/contracts/definition-of-done.yaml b/products/breg/contracts/definition-of-done.yaml index 13bd82829c..3c76ed2a2f 100644 --- a/products/breg/contracts/definition-of-done.yaml +++ b/products/breg/contracts/definition-of-done.yaml @@ -48,7 +48,7 @@ requirements: - {id: BREG-V1-24, phase: W3, state: enforced, doneWhen: "Application authorization and RLS agree for positive, negative, malformed, and pooled authority.", journeys: [BREG-J05, BREG-J11], evidence: [{path: crates/registry-breg/tests/postgres_compiled_schema.rs, name: compiled_postgres_schema_enforces_context_rls_and_exact_catalog}, {path: crates/registry-breg/tests/postgres_kernel.rs, name: real_postgres_kernel_proves_roles_rls_interlock_and_pool_isolation}, {path: crates/registry-breg/tests/postgres_pilot_acceptance.rs, name: real_postgres_five_domain_pilot_is_configured_production_closed_and_source_neutral}]} - {id: BREG-V1-25, phase: W3, state: enforced, doneWhen: "Provenance stays distinct from minimized value-free audit, logs, metrics, and traces.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_tombstone_revision.rs, name: tombstone_revisions_survive_package_upgrade_and_replay_exactly}, {path: crates/registry-breg/tests/startup_http.rs, name: operational_log_level_is_a_closed_vocabulary}, {path: crates/registry-breg/tests/startup_http.rs, name: every_operational_event_renders_exact_closed_value_free_json_fields}, {path: crates/registry-breg/tests/startup_http.rs, name: provenance_operational_logs_metrics_and_traces_are_separate_closed_and_value_free}]} - {id: BREG-V1-26, phase: W3, state: enforced, doneWhen: "Exact media-specific Registry Record bytes release or replay only after successful attempt and terminal audit gates.", journeys: [BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_http_mutations_are_guarded_and_exactly_replayable}]} - - {id: BREG-V1-27, phase: W3, state: partial, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup.", journeys: [BREG-J10, BREG-J12], gap: "Some refusal and failure paths still return after the request entry without a paired response entry: migration reconciliation (#1592), reviewed-apply preflight reads before the request entry (#1587), ingestion refusals closed in the general schema (#1597), read post-processing failures (#1593), and evidence-retention erasure, which writes no entry (#1590).", evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}]} + - {id: BREG-V1-27, phase: W3, state: enforced, doneWhen: "Every caller-requested operation writes one platform audit request entry before protected I/O and at least one response entry sharing its correlation before release, through the one audit writer its process opens at startup. The pairing holds while that process runs; a process that is killed, exits, or shuts its runtime down while a write is in flight may leave a request without its response.", journeys: [BREG-J10, BREG-J12], evidence: [{path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_is_audited_atomic_typed_and_exactly_replayable}, {path: crates/registry-breg/tests/postgres_read.rs, name: real_postgres_read_is_authorized_bounded_minimized_and_audit_gated}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: real_postgres_mutation_audit_refusals_fail_closed_around_the_commit}, {path: crates/registry-breg/tests/postgres_mutation.rs, name: two_runtimes_audit_concurrent_mutations_through_their_own_writers}, {path: crates/registry-platform-audit/src/writer.rs, name: a_request_dropped_unanswered_writes_its_unfinished_response}, {path: crates/registry-platform-audit/src/writer.rs, name: a_canceled_operation_pairs_its_request_in_the_file}, {path: crates/registry-platform-audit/src/writer.rs, name: a_command_that_exits_after_an_early_return_pairs_its_request}, {path: crates/registry-platform-audit/src/writer.rs, name: a_response_claimed_while_its_request_is_dropped_is_the_only_answer}, {path: crates/registry-platform-audit/src/writer.rs, name: a_response_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle}, {path: crates/registry-platform-audit/src/writer.rs, name: an_append_task_dropped_at_runtime_shutdown_leaves_the_request_to_its_handle}, {path: crates/registry-breg/tests/postgres_change_requests.rs, name: cached_review_result_cannot_authorize_fresh_apply_but_committed_receipt_recovers_offline}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_reconciliation_completes_reverts_or_refuses_a_pinned_target}, {path: crates/registry-breg/tests/postgres_webhook_delivery.rs, name: real_postgres_webhook_delivery_retry_dead_letter_replay_is_package_bound_audited_and_confined}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_pairs_its_request_entry_on_every_outcome}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_refusal_after_the_ingestion_request_is_answered_in_the_ingestion_schema}, {path: crates/registry-breg/tests/postgres_ingestion_runs.rs, name: a_transition_whose_commit_fails_is_answered_unfinished}, {path: crates/registry-breg/tests/postgres_request_read_retention.rs, name: request_detail_erasure_records_its_commit_before_retrying_external_deletions}, {path: crates/registry-breg/tests/postgres_action_evidence_retention.rs, name: expired_request_evidence_erases_only_retained_uses}, {path: crates/registry-breg/tests/postgres_revision_http.rs, name: real_postgres_revision_http_is_bounded_authorized_atomic_and_audit_gated}, {path: crates/registry-breg/tests/postgres_history_rebaseline.rs, name: rebaseline_refuses_while_maintenance_is_not_ready}]} - {id: BREG-V1-28, phase: W4, state: enforced, doneWhen: "Production packages capture the governed closure and sign exact canonical bytes with monotonic identity.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_builder_is_deterministic_and_local_publication_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: production_package_requires_exact_trust_anchor_threshold_and_signature}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_layout_contract_conditional_manifest_projection_is_in_projected_closure}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_omits_manifest_projection_from_signed_closure_and_loads}, {path: crates/registry-breg/tests/postgres_package.rs, name: projection_free_package_refuses_claimed_manifest_artifacts}]} - {id: BREG-V1-29, phase: W4, state: enforced, doneWhen: "Activation verifies trust, identity, inventory, filesystem safety, artifacts, and schema before readiness.", journeys: [BREG-J14], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: package_binding_refuses_wrong_environment_instance_database_sequence_and_prior}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_refuses_symlinks_and_production_writable_permissions}, {path: crates/registry-breg/tests/postgres_package.rs, name: package_manifest_refuses_ddl_checksum_path_and_canonical_json_tampering}, {path: crates/registry-breg/tests/postgres_package.rs, name: signed_schema_fingerprint_mismatch_is_durably_failed_and_never_ready}]} - {id: BREG-V1-30, phase: W4, state: enforced, doneWhen: "Apply retains the lock through migrations, catalog verification, activation, and maintenance clearing.", journeys: [BREG-J13, BREG-J15], evidence: [{path: crates/registry-breg/tests/postgres_package.rs, name: real_postgres_package_startup_apply_failure_and_old_process_are_closed}, {path: crates/registry-breg/tests/postgres_migration.rs, name: real_postgres_backfill_and_destructive_recovery_are_bounded_resumable_and_activation_closed}]} diff --git a/products/casework/CHANGELOG.md b/products/casework/CHANGELOG.md index 3e60a2d3b5..2bd0cb3aa7 100644 --- a/products/casework/CHANGELOG.md +++ b/products/casework/CHANGELOG.md @@ -2,6 +2,18 @@ ## Unreleased +- Answer every audited request entry. An operation whose change is not + known to have committed (a refusal, a failure, a canceled request, or a + commit whose acknowledgment was lost and whose outcome could not be read + back) writes `{event, outcome: "unfinished"}` as its response under the + same correlation. A commit whose acknowledgment was lost is read back + first, and one that took effect is answered and recorded like any other. + A committed operation writes all of its response entries even when its + caller disconnects while they are written. Adding a review note is audited + as `casework.review_note_added`, naming the review request by its keyed + pseudonym and the note's history event, but never the note's text or + audience. + - BREAKING: write audit through the shared platform audit writer instead of a hash-chained journal published from a PostgreSQL outbox. - The `audit` block takes `hashKeyRef`, `destination` (`file`, the diff --git a/products/platform/CHANGELOG.md b/products/platform/CHANGELOG.md index 1a20e01867..1864e979a9 100644 --- a/products/platform/CHANGELOG.md +++ b/products/platform/CHANGELOG.md @@ -2,6 +2,14 @@ ## Unreleased +- Add `AuditWriter::begin`, which appends a `request` entry and returns an + `AuditRequest` that owes its `response`. A response the handle writes, or + one appended under the same schema and correlation, answers it; a handle + dropped unanswered, by an early return, a panic, or a canceled future, + writes the product's `unfinished` record as the response, so no request + entry stays unpaired. A file destination flushes that line on the runtime, + or when the last reference to the writer is dropped at shutdown. + - Refuse to reopen an audit file ending in an incomplete JSONL entry, preserving its bytes for operator archival before starting a fresh stream. - Keep queued stream appends stopped after an earlier write fails, and finish diff --git a/products/render/ACCEPTANCE.md b/products/render/ACCEPTANCE.md index 20acde544b..8cb9e1733a 100644 --- a/products/render/ACCEPTANCE.md +++ b/products/render/ACCEPTANCE.md @@ -18,7 +18,7 @@ hardening (2026-09-19); see EVIDENCE.md for the change list. | Resource enforcement (kill at timeout, recycle, recover) | `serve.rs`: `pathological_renders_are_bounded_and_the_service_recovers` (504 problem, service healthy afterwards, both kills audited); the CLI shares the wall: `scaffold::compile_is_bounded_by_a_timeout` | | Auth (constant-time, indistinguishable 401s, audited, key-file normalization) | `serve.rs`: `unauthorized_requests_are_refused_and_audited`, `api_key_file_with_one_trailing_newline_is_trimmed`, `api_key_with_stray_whitespace_is_refused_at_startup` | | HTTP contract (headers, JSON variant, correlation, problems, 413, versions in /health) | `serve.rs`: `render_returns_pdf_with_hash_headers_matching_golden`, `json_variant_serves_openfn_clients`, `missing_issued_at_and_bad_data_are_named_problems`, `oversized_bodies_are_refused_after_auth_as_problems`, `serve_health_and_ready` | -| Audit (request entry before the render, response entry before the response, failures audited, value-free) | `serve.rs`: `a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation`, `a_refused_request_entry_prevents_the_render`, `a_refused_response_entry_withholds_the_pdf`, `audit_events_are_value_free` (canary + API-key scans), `unauthorized_requests_are_refused_and_audited`, `pathological_renders_are_bounded_and_the_service_recovers`; unit: `server::tests::the_request_entry_is_accepted_before_the_render_starts`, `server::tests::a_refused_request_entry_prevents_the_render`, `server::tests::a_refusal_before_the_render_is_one_response_entry`, `server::tests::a_refusal_after_the_render_starts_keeps_the_render_identity`, `runtime::tests::the_audit_block_takes_the_shared_destination_shape`, `runtime::tests::an_audit_block_outside_the_shared_shape_is_refused`; `scaffold.rs`: `check_proves_the_audit_file_resolves_under_the_root`, `check_refuses_to_prove_a_stdout_audit_destination` | +| Audit (request entry before the render, response entry before the response, failures audited, value-free) | `serve.rs`: `a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation`, `a_refused_request_entry_prevents_the_render`, `a_refused_response_entry_withholds_the_pdf`, `audit_events_are_value_free` (canary + API-key scans), `unauthorized_requests_are_refused_and_audited`, `pathological_renders_are_bounded_and_the_service_recovers`; unit: `server::tests::the_request_entry_is_accepted_before_the_render_starts`, `server::tests::a_refused_request_entry_prevents_the_render`, `server::tests::a_refusal_before_the_render_is_one_response_entry`, `server::tests::a_refusal_after_the_render_starts_keeps_the_render_identity`, `server::tests::concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries`, `runtime::tests::the_audit_block_takes_the_shared_destination_shape`, `runtime::tests::an_audit_block_outside_the_shared_shape_is_refused`; `scaffold.rs`: `check_proves_the_audit_file_resolves_under_the_root`, `check_refuses_to_prove_a_stdout_audit_destination` | | Sealed-bundle-only serving (startup and per render) | `serve.rs`: `tampered_bundle_refuses_to_serve` (exit code = BundleTampered), `bundle_drift_after_serve_starts_is_refused_per_render` (tamper and unseal after startup) | | Cross-machine byte stability; closure drift | `.github/workflows/render-golden.yml` runs the golden, serve, and scaffold suites on ubuntu-24.04 and macos-14 against the same `golden.json`; `golden_hashes_match` also pins each bundle's file closure and requires every non-virtual dep to be manifest-governed | | DX (scaffold compiles offline, validate dry-run, errors) | Manual smoke 2026-09-17 recorded in EVIDENCE-style: `registry-render init` + first compile offline, warning-clean after the font fix; `registry-render validate` refuses bad data with exit 9 and pointers | diff --git a/products/render/README.md b/products/render/README.md index 26bbbaa0c8..5f852679e0 100644 --- a/products/render/README.md +++ b/products/render/README.md @@ -99,9 +99,13 @@ exits 101): Stack audit writer. A render writes a `request` entry before the worker starts and a `response` entry with the outcome before the document leaves, both failing closed; a refusal before any render is one `response` entry. - Each line is the envelope `{schema, eventId, time, phase, correlation, - record}` with schema `render.registrystack.org/audit/v1`; `correlation` is - the caller's `Idempotency-Key`, or a random id when there is none. The + A call that ends before its outcome is written, such as a disconnected + caller, writes an `unfinished` response, so every `request` entry is + paired. Each line is the envelope `{schema, eventId, time, phase, + correlation, record}` with schema `render.registrystack.org/audit/v1`; + `correlation` is a random id the server draws for every call, and the + caller's `Idempotency-Key` is echoed only as the record's `correlationId`, + since two calls may carry the same key. The runtime's `audit` block names a `file` (the default, with `path` and optional `rotateBytes` and `retainDays`) or `stdout` destination. The log carries no hash chain or signature, so it is not tamper-evident on the diff --git a/products/render/SECURITY-MATRIX.md b/products/render/SECURITY-MATRIX.md index 51bdb5754a..4db6c2cbcb 100644 --- a/products/render/SECURITY-MATRIX.md +++ b/products/render/SECURITY-MATRIX.md @@ -15,7 +15,7 @@ fixed or explicitly documented as residuals below), extended by the PR | 5 | Caller without the API key reads documents or renders | Bearer authentication as a layer on `/v1/*` **before body buffering**; constant-time compare; ≥32-byte ASCII key; identical 401 bodies for missing/wrong/short keys; the key file gets exactly one trailing line ending trimmed and any other whitespace refuses startup (no silently mis-armed key with `/health` green) | `serve::unauthorized_requests_are_refused_and_audited` (two indistinguishable 401s); `serve::api_key_file_with_one_trailing_newline_is_trimmed`; `serve::api_key_with_stray_whitespace_is_refused_at_startup`; `/v1/documents` behind the same layer | | 6 | The append-only audit log is used as an unauthenticated write oracle (huge or hostile route params) | Route parameters are bounded (64 chars) and kebab-validated (`sanitize_document_type`) before any audit append, including pre-auth 401 events | code-reviewed; audit shape asserted in `serve::unauthorized_requests_are_refused_and_audited` | | 7 | Request data values or the API key leak into the audit log or logs | Audit events carry a closed, value-free field set; lifecycle logs use fixed dimensions (method class, route template, status, latency, trace id) | `serve::audit_events_are_value_free` (canary + key scans over the audit file) | -| 8 | A render starts, or a document leaves, with no accepted audit record of it | The shared audit writer accepts a `file` entry only after `fsync` (owner-only file, single-writer lock) and a `stdout` entry only after it is written and flushed, then stops accepting after any failure. The `request` entry is accepted before the worker starts, else `503 audit-failed` and no render; the `response` entry is accepted before the document leaves, else `503 audit-failed` and no PDF or hash headers; `/ready` reports a stopped writer. Host-level tampering with the log is not defended here: the log carries no hash chain or signature (residual below) | `server::tests::the_request_entry_is_accepted_before_the_render_starts`; `server::tests::a_refused_request_entry_prevents_the_render`; `serve::a_refused_request_entry_prevents_the_render`; `serve::a_refused_response_entry_withholds_the_pdf`; `serve::a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation` (file mode 0600) | +| 8 | A render starts, or a document leaves, with no accepted audit record of it | The shared audit writer accepts a `file` entry only after `fsync` (owner-only file, single-writer lock) and a `stdout` entry only after it is written and flushed, then stops accepting after any failure. The `request` entry is accepted before the worker starts, else `503 audit-failed` and no render; the `response` entry is accepted before the document leaves, else `503 audit-failed` and no PDF or hash headers; `/ready` reports a stopped writer. Each call pairs its entries under a correlation the server draws, never the caller's `Idempotency-Key` (echoed only as `correlationId`), so calls sharing a key cannot be confused, and a call that ends before its outcome is written writes an `unfinished` response. Host-level tampering with the log is not defended here: the log carries no hash chain or signature (residual below) | `server::tests::the_request_entry_is_accepted_before_the_render_starts`; `server::tests::a_refused_request_entry_prevents_the_render`; `serve::a_refused_request_entry_prevents_the_render`; `serve::a_refused_response_entry_withholds_the_pdf`; `serve::a_render_writes_a_request_entry_then_a_response_entry_sharing_correlation` (file mode 0600); `server::tests::concurrent_calls_sharing_an_idempotency_key_pair_their_own_entries` | | 9 | A render succeeds but is wrong on paper (missing glyphs) | Check-time label-script coverage over the actual label characters; render-time warnings; `--strict` fails on warnings | `scaffold::check_seal_refuses_to_seal_a_broken_bundle` (CJK label, no font → refusal, and no seal written) | | 10 | Pathological template data exhausts CPU or memory | Serves render in a supervised worker process: killed at the configured timeout, `RLIMIT_AS` on Linux, recycled on panic; bounded concurrency; request body and output caps with hard ceilings | `serve::pathological_renders_are_bounded_and_the_service_recovers` (timeout or memory wall is audited; a fresh worker immediately serves a healthy render) | | 11 | Assets smuggle executables or balloons | base64-decoded, JPEG/PNG magic sniffed, per-asset 2 MiB / per-request 8 MiB caps, exact bytes covered by `dataSha256` | `golden::wrong_media_type_and_oversize_assets_are_refused` | diff --git a/products/render/integrations/openfn/JOURNEY.md b/products/render/integrations/openfn/JOURNEY.md index 7e709ed4ce..4347fa5877 100644 --- a/products/render/integrations/openfn/JOURNEY.md +++ b/products/render/integrations/openfn/JOURNEY.md @@ -94,8 +94,9 @@ is the profile's, not the caller's. ## Audit ledger The walk predates the shared audit writer. Render now writes a `request` -and a `response` entry per render, joined by `correlation` (the job's -`eventEffectId` here), and has no `audit-verify` command or audit key; a +and a `response` entry per render, joined by a `correlation` the server +draws for each call, with the job's `eventEffectId` recorded as +`correlationId`, and has no `audit-verify` command or audit key; a re-walk reads the audit file directly. What the walk recorded: `registry-render audit-verify` (after graceful shutdown; a live writer diff --git a/products/scheduling/CHANGELOG.md b/products/scheduling/CHANGELOG.md index 766783b1d9..2a9817f889 100644 --- a/products/scheduling/CHANGELOG.md +++ b/products/scheduling/CHANGELOG.md @@ -21,6 +21,20 @@ response entry, recording the decision the receipt carries, is accepted; a refused one answers `service.unavailable` without the receipt. A permission refused before the transaction is one response entry. + - Every commitment request entry is answered. A refusal, decided by the + ledger or by the permission check, is answered only once its response + entry is accepted, and `service.unavailable` otherwise, where it was + previously answered with the write failure only logged. A transaction + rolled back on a failure, records replaced under a commitment, and a + reused or expired idempotency key now write a response with the outcome + `unfinished` and a closed reason, and a commitment that returns or is + canceled before answering writes `commitment.unfinished`. + - A capacity commit that is not acknowledged is read back by its + transaction identifier before it is recorded: one that took effect is + answered and recorded as committed, one that rolled back writes + `commitment.failed`, and one whose status cannot be read writes + `commitment.unfinished` and answers `service.unavailable`, never + `commitment.failed`. - Hook delivery writes an attempt's request entry before egress and its terminal response entry with the same correlation; a refused entry leaves the delivery pending and sends nothing. diff --git a/products/scheduling/RUNTIME-CONFIG.md b/products/scheduling/RUNTIME-CONFIG.md index ea73009c00..60a6074316 100644 --- a/products/scheduling/RUNTIME-CONFIG.md +++ b/products/scheduling/RUNTIME-CONFIG.md @@ -99,15 +99,25 @@ and its `response` entry after the transaction commits or rolls back; a permission refused before the transaction is one `response` entry. Audit fails closed: a refused `request` entry opens no transaction, and a refused `response` entry for a committed change answers `service.unavailable` with the change -committed. One case is best effort instead: a refusal the ledger itself -decides (an admission refusal, the hold ceiling, a lapsed grant, a stale -observed revision, or a cancellation past its cutoff), and a permission -mismatch refused before the transaction opens, still reach the caller when -their `response` entry cannot be written; the write failure is only logged, -and the journal is left holding a `request` entry with no paired `response`, -or, for a permission mismatch, no entry at all. `/readyz` reports unavailable -while the destination refuses writes. An expired hold writes its history entry -as `system` and no audit entry. +committed. A refusal, whether the ledger decided it (an admission refusal, +the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation +past its cutoff) or the permission check refused it before the transaction +opened, reaches the caller only once its `response` entry is accepted, and +answers `service.unavailable` otherwise. A commitment nothing decided still +answers its `request` entry: a transaction rolled back on a failure, records +replaced under it, or a reused or expired idempotency key writes a `response` +with the outcome `unfinished` and the reason `commitment.failed`, +`commitment.facts-stale`, `idempotency.key-reused`, or `idempotency.expired`, +and one that returns or is canceled before answering writes +`commitment.unfinished`. A capacity commit that is not acknowledged is read +back by its transaction identifier on a separate connection that changes +nothing: one that took effect is answered and recorded as committed, one that +rolled back writes `commitment.failed`, and one whose status cannot be read +writes `commitment.unfinished` and answers `service.unavailable`, because it +may have taken effect; a retry under the same idempotency key then replays +whatever committed. `/readyz` reports unavailable while the destination +refuses writes. An expired hold writes its history entry as `system` and no +audit entry. `destinations` is optional. `destinations.reminders` is the one place due reminder intents are delivered, as CloudEvents 1.0 events over HTTPS POST with diff --git a/products/scheduling/contracts/security-invariant-matrix.yaml b/products/scheduling/contracts/security-invariant-matrix.yaml index ded7f4ecee..e214ecdfbd 100644 --- a/products/scheduling/contracts/security-invariant-matrix.yaml +++ b/products/scheduling/contracts/security-invariant-matrix.yaml @@ -254,25 +254,30 @@ invariants: transaction opens is one response entry. A refused commitment is one the ledger decided: an admission refusal, the hold ceiling, a lapsed grant, a stale observed revision, or a cancellation past its cutoff. - Both that refusal and a permission mismatch write their response entry - best effort: the failure is only logged, the caller still receives the - refusal already decided, and the journal is left holding a request - entry with no paired response, or, for a permission mismatch, no entry - at all. A failed transaction and a replaced environment decided nothing - and write no response, and an idempotency key refusal is carried by the - attempt receipt instead. The match that selects between them is - exhaustive, so a new outcome states its side. A replayed receipt, - including one a concurrent identical request won, is answered only - after its own response entry is accepted, and that entry records the + Every request entry is answered. A commitment nothing decided, because + its transaction rolled back on a failure, the environment records were + replaced under it, or its idempotency key was refused as reused or + expired, writes one response entry with the outcome unfinished and a + closed reason, and never an authorization verdict. A capacity commit + that is not acknowledged is read back by its transaction identifier on + a separate connection that mutates nothing: one that took effect is + answered and recorded as committed, one that rolled back is recorded + commitment.failed, and one whose status cannot be read is recorded + commitment.unfinished, never failed. One that returns or is canceled + before it answers writes the commitment.unfinished response when its + request handle is dropped. The match that selects between them is exhaustive, + so a new outcome states its side. A replayed receipt, including one a + concurrent identical request won, is answered only after its own + response entry is accepted, and that entry records the decision the receipt carries: allowed for a success, denied with the original reason for a refusal. refusal: >- Record authorization.allowed or authorization.refused as an audit reason, never as a problem code, and never carry the caller's raw identifier. A refused request entry opens no transaction, and a refused response entry - for a committed commitment or a replayed receipt answers - service.unavailable without releasing the receipt, so no commitment or - replay is answered without its audit. + for a committed commitment, a replayed receipt, a refusal, or a + commitment nothing decided answers service.unavailable without releasing + the receipt, so no answer leaves without its audit. negativeTest: path: crates/registry-scheduling/tests/postgres_commitments.rs name: a_replayed_receipt_is_released_only_after_its_response_entry_is_accepted diff --git a/products/scheduling/contracts/security-test-traceability.yaml b/products/scheduling/contracts/security-test-traceability.yaml index ae796f9deb..a2eb2efe51 100644 --- a/products/scheduling/contracts/security-test-traceability.yaml +++ b/products/scheduling/contracts/security-test-traceability.yaml @@ -125,6 +125,15 @@ entries: - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_replayed_receipt_is_released_only_after_its_response_entry_is_accepted} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_replayed_refusal_is_recorded_as_the_refusal_it_replays} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_racing_replay_is_not_released_when_its_response_entry_is_refused} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_failed_capacity_transaction_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_whose_commit_acknowledgment_is_lost_is_read_back_as_committed} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_whose_outcome_cannot_be_read_back_is_recorded_unfinished} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_refused_at_commit_is_read_back_as_failed} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_commitment_dropped_inside_its_capacity_transaction_writes_one_unfinished_response} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: an_idempotency_key_refusal_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_records_swap_under_a_commitment_pairs_its_request_entry} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_refusal_whose_response_entry_is_refused_answers_service_unavailable} + - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_permission_refusal_the_destination_refuses_answers_service_unavailable} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: a_concurrent_identical_request_replays_the_winning_receipt} - {path: crates/registry-scheduling/tests/postgres_commitments.rs, name: concurrent_identical_admissible_requests_replay_one_winning_success} - {path: crates/registry-scheduling/src/service.rs, name: a_replayed_receipt_records_the_decision_it_carries}