diff --git a/.config/nextest.toml b/.config/nextest.toml new file mode 100644 index 00000000..5ea6f499 --- /dev/null +++ b/.config/nextest.toml @@ -0,0 +1,2 @@ +[profile.ci] +slow-timeout = { period = "60s", terminate-after = 2, grace-period = "10s" } diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index acd4bc4a..33f61660 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -8,6 +8,10 @@ on: permissions: contents: read +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + jobs: rust: name: Rust (${{ matrix.name }}) @@ -28,6 +32,7 @@ jobs: macos_major: "26" architecture: arm64 runs-on: ${{ matrix.runner }} + timeout-minutes: 30 steps: - name: Checkout @@ -88,7 +93,7 @@ jobs: run: shellcheck release/artifacts/recipes/common.sh release/artifacts/recipes/php/*.sh release/artifacts/recipes/composer/*.sh release/artifacts/recipes/redis/*.sh release/artifacts/recipes/mysql/*.sh release/artifacts/recipes/postgres/*.sh release/artifacts/recipes/mailpit/*.sh release/artifacts/recipes/rustfs/*.sh - name: Run tests - run: cargo nextest run --workspace --all-features --locked + run: cargo nextest run --profile ci --workspace --all-features --locked runtime-portability: name: Runtime (${{ matrix.target }}) diff --git a/crates/daemon/src/dns.rs b/crates/daemon/src/dns.rs index fe5e675e..dd14eb6b 100644 --- a/crates/daemon/src/dns.rs +++ b/crates/daemon/src/dns.rs @@ -51,7 +51,7 @@ impl RunningDnsResolver { (&mut self.task).await? } - fn signal_shutdown(&mut self) { + pub(crate) fn signal_shutdown(&mut self) { if let Some(shutdown) = self.shutdown.take() { let _ = shutdown.send(()); } diff --git a/crates/daemon/src/gateway.rs b/crates/daemon/src/gateway.rs index fe7b59e4..d5f5de68 100644 --- a/crates/daemon/src/gateway.rs +++ b/crates/daemon/src/gateway.rs @@ -1,10 +1,13 @@ use std::collections::{BTreeMap, BTreeSet, btree_map}; +use std::future::Future; use std::io; use std::net::TcpListener; +use std::pin::Pin; use std::process::{ExitStatus, Stdio}; use std::sync::{ Arc, - atomic::{AtomicU64, Ordering}, + atomic::{AtomicBool, AtomicU64, Ordering}, + mpsc, }; use std::time::{Duration, Instant}; @@ -13,7 +16,7 @@ use config::{ProjectConfig, ProjectConfigFile}; use futures_util::StreamExt; use resources::{ResourceAdapter, caddy_adapter, frankenphp_adapter}; #[cfg(target_os = "macos")] -use rustix::process::{Pid, Signal, kill_process_group}; +use rustix::process::{Pid, Signal, kill_process_group, test_kill_process_group}; use sha2::{Digest, Sha256}; use state::{ Database, ManagedResourceDesiredState, ManagedResourceTrackRecord, PortOwner, @@ -21,6 +24,8 @@ use state::{ StateError, fs, }; use tokio::io::{AsyncRead, AsyncReadExt}; +use tokio::sync::watch; +use tokio::task::JoinHandle; use tokio::time::{sleep, timeout}; use crate::gateway_config::{ @@ -51,6 +56,9 @@ type RuntimeProcessCommand = tokio::process::Command; const PHP_INI_ENVIRONMENT_KEYS: [&str; 2] = ["PHPRC", "PHP_INI_SCAN_DIR"]; const CONFIG_VALIDATION_TIMEOUT: Duration = Duration::from_secs(10); +const CONFIG_VALIDATION_CLEANUP_TIMEOUT: Duration = Duration::from_secs(1); +#[cfg(target_os = "macos")] +const CONFIG_VALIDATION_GROUP_ANCHOR: &str = "while IFS= read -r _line; do :; done\nkill -KILL 0\n"; const RUNTIME_READINESS_TIMEOUT: Duration = Duration::from_secs(60); const PF_PUBLIC_READINESS_TIMEOUT: Duration = Duration::from_secs(2); const FOREIGN_LISTENER_PROBE_TIMEOUT: Duration = Duration::from_millis(100); @@ -63,6 +71,27 @@ const RUNTIME_CONFIG_FINGERPRINT_SCHEME: &[u8] = b"pv-runtime-config:v1"; pub(crate) const CADDY_NOT_INSTALLED: &str = "Gateway runtime skipped; Caddy is not installed"; static CANDIDATE_CONFIG_DIR_COUNTER: AtomicU64 = AtomicU64::new(0); +/// Result of reconciliation while the daemon's test-only fallback shutdown is armed. +/// +/// [`Self::Cancelled`] is emitted only by the nonblocking test fallback. Ordinary daemon +/// shutdown keeps its separate task-specific drain and abort behavior. +#[derive(Debug)] +pub(crate) enum ReconciliationOutcome { + Completed(T), + Cancelled, +} + +impl ReconciliationOutcome { + pub(crate) fn into_completed(self) -> Result { + match self { + Self::Completed(value) => Ok(value), + Self::Cancelled => Err(DaemonError::UnexpectedProtocolResponse { + reason: "reconciliation was cancelled without a shutdown signal".to_owned(), + }), + } + } +} + #[derive(Clone, Debug, Eq, PartialEq)] pub struct CaddyCliCommand { executable: Utf8PathBuf, @@ -224,6 +253,23 @@ pub(crate) async fn reconcile_gateway_runtimes_with_phase_log( RUNTIME_READINESS_TIMEOUT, None, Some(phase_log), + None, + ) + .await? + .into_completed() +} + +pub(crate) async fn reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths: &PvPaths, + phase_log: &structured_log::ReconciliationPhaseLog, + fallback_shutdown: &watch::Receiver, +) -> Result, DaemonError> { + reconcile_gateway_runtimes_with_pf_state( + paths, + RUNTIME_READINESS_TIMEOUT, + None, + Some(phase_log), + Some(fallback_shutdown), ) .await } @@ -234,12 +280,31 @@ pub(crate) async fn reconcile_project_gateway_runtimes_with_phase_log( pf_routing_state: Option, phase_log: &structured_log::ReconciliationPhaseLog, ) -> Result { + reconcile_project_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + project_id, + pf_routing_state, + phase_log, + None, + ) + .await? + .into_completed() +} + +pub(crate) async fn reconcile_project_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths: &PvPaths, + project_id: &str, + pf_routing_state: Option, + phase_log: &structured_log::ReconciliationPhaseLog, + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { reconcile_project_gateway_runtimes( paths, project_id, RUNTIME_READINESS_TIMEOUT, pf_routing_state, phase_log, + fallback_shutdown, ) .await } @@ -263,8 +328,10 @@ pub async fn reconcile_project_gateway_runtimes_for_test( readiness_timeout, Some(pf_routing_state), &phase_log, + None, ) .await? + .into_completed()? { ProjectGatewayReconciliationOutcome::Reconciled { summary, .. } => Ok(summary), ProjectGatewayReconciliationOutcome::PromoteSystem => { @@ -273,8 +340,10 @@ pub async fn reconcile_project_gateway_runtimes_for_test( readiness_timeout, Some(pf_routing_state), Some(&phase_log), + None, ) - .await + .await? + .into_completed() } } } @@ -285,24 +354,39 @@ async fn reconcile_project_gateway_runtimes( readiness_timeout: Duration, pf_routing_state: Option, phase_log: &structured_log::ReconciliationPhaseLog, -) -> Result { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let Some(gateway_command) = first_installed_caddy_command(paths)? else { - let summary = reconcile_gateway_runtimes_with_pf_state( + let summary = match reconcile_gateway_runtimes_with_pf_state( paths, readiness_timeout, pf_routing_state, Some(phase_log), + shutdown, ) - .await?; + .await? + { + ReconciliationOutcome::Completed(summary) => summary, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + }; - return Ok(ProjectGatewayReconciliationOutcome::Reconciled { - summary, - gateway_evaluated: true, - }); + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::Reconciled { + summary, + gateway_evaluated: true, + }, + )); }; let mut targeted = match build_target_runtime_plan(paths, project_id) { Ok(Some(targeted)) => targeted, - Ok(None) => return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem), + Ok(None) => { + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); + } Err(error) => { return Err(record_primary_error(paths, RuntimeSubject::Gateway, error)); } @@ -316,10 +400,16 @@ async fn reconcile_project_gateway_runtimes( project_id, )? { Some(active_impact) => active_impact, - None => return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem), + None => { + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); + } }; if failed_worker_outside_target(paths, &targeted, &target_active_impact)? { - return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem); + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); } if targeted.current_runtime_key.is_none() && !target_active_impact.served @@ -344,7 +434,9 @@ async fn reconcile_project_gateway_runtimes( timeout(probe_timeout, probe_readiness_once(&readiness.check)).await, Ok(Ok(())) ) { - return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem); + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); } record_gateway_runtime_observed( paths, @@ -352,10 +444,14 @@ async fn reconcile_project_gateway_runtimes( RuntimeReadinessOutcome::Verified, )?; - return Ok(skipped_project_gateway_outcome(phase_log)); + return Ok(ReconciliationOutcome::Completed( + skipped_project_gateway_outcome(phase_log), + )); } if !complete_targeted_runtime_plan(paths, project_id, &mut targeted)? { - return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem); + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); } let active_impact = match verified_active_project_gateway_impact( paths, @@ -365,7 +461,11 @@ async fn reconcile_project_gateway_runtimes( project_id, )? { Some(active_impact) => active_impact, - None => return Ok(ProjectGatewayReconciliationOutcome::PromoteSystem), + None => { + return Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::PromoteSystem, + )); + } }; let worker_timer = phase_log.start( structured_log::ReconciliationPhase::Workers, @@ -384,16 +484,20 @@ async fn reconcile_project_gateway_runtimes( ), })?; let worker_runtime = required_installed_worker_runtime(paths, worker)?; - reconcile_planned_worker( + match reconcile_planned_worker( paths, - &supervisor, worker, &worker_runtime, readiness_timeout, active_impact.worker_fragments.get(runtime_key), None, + shutdown, ) - .await?; + .await? + { + ReconciliationOutcome::Completed(()) => {} + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + } } worker_timer.finish( @@ -418,16 +522,20 @@ async fn reconcile_project_gateway_runtimes( "target_project", ); if gateway_required { - reconcile_planned_gateway( + match reconcile_planned_gateway( paths, - &supervisor, &targeted.plan, &gateway_command, pf_routing_state, readiness_timeout, Some(&active_impact.gateway_fragments), + shutdown, ) - .await?; + .await? + { + ReconciliationOutcome::Completed(()) => {} + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + } } gateway_timer.finish( if gateway_required { @@ -456,16 +564,22 @@ async fn reconcile_project_gateway_runtimes( .find(|worker| worker.runtime_key == *previous_runtime_key) { let worker_runtime = required_installed_worker_runtime(paths, worker)?; - reconcile_planned_worker( + match reconcile_planned_worker( paths, - &supervisor, worker, &worker_runtime, readiness_timeout, active_impact.worker_fragments.get(previous_runtime_key), None, + shutdown, ) - .await?; + .await? + { + ReconciliationOutcome::Completed(()) => {} + ReconciliationOutcome::Cancelled => { + return Ok(ReconciliationOutcome::Cancelled); + } + } } else { if let Err(error) = stop_worker_if_undemanded(paths, &supervisor, previous_runtime_key).await @@ -493,10 +607,12 @@ async fn reconcile_project_gateway_runtimes( "Gateway runtime unchanged; Project has no routes".to_owned() }; - Ok(ProjectGatewayReconciliationOutcome::Reconciled { - summary, - gateway_evaluated: gateway_required, - }) + Ok(ReconciliationOutcome::Completed( + ProjectGatewayReconciliationOutcome::Reconciled { + summary, + gateway_evaluated: gateway_required, + }, + )) } pub fn probe_gateway_identity_blocking( @@ -523,7 +639,9 @@ pub async fn reconcile_gateway_runtimes_with_readiness_timeout( paths: &PvPaths, readiness_timeout: Duration, ) -> Result { - reconcile_gateway_runtimes_with_pf_state(paths, readiness_timeout, None, None).await + reconcile_gateway_runtimes_with_pf_state(paths, readiness_timeout, None, None, None) + .await? + .into_completed() } #[doc(hidden)] @@ -532,8 +650,15 @@ pub async fn reconcile_gateway_runtimes_with_pf_state_for_test( readiness_timeout: Duration, pf_routing_state: GatewayPfRoutingState, ) -> Result { - reconcile_gateway_runtimes_with_pf_state(paths, readiness_timeout, Some(pf_routing_state), None) - .await + reconcile_gateway_runtimes_with_pf_state( + paths, + readiness_timeout, + Some(pf_routing_state), + None, + None, + ) + .await? + .into_completed() } async fn reconcile_gateway_runtimes_with_pf_state( @@ -541,7 +666,11 @@ async fn reconcile_gateway_runtimes_with_pf_state( readiness_timeout: Duration, pf_routing_state: Option, phase_log: Option<&structured_log::ReconciliationPhaseLog>, -) -> Result { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let lookup_started_at = Instant::now(); let gateway_command = first_installed_caddy_command(paths).inspect_err(|_| { if let Some(phase_log) = phase_log { @@ -556,6 +685,9 @@ async fn reconcile_gateway_runtimes_with_pf_state( } })?; let Some(gateway_command) = gateway_command else { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } if let Some(phase_log) = phase_log { phase_log.report_progress(structured_log::ReconciliationPhase::Workers); phase_log.completed( @@ -586,6 +718,9 @@ async fn reconcile_gateway_runtimes_with_pf_state( ); } })?; + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } if let Some(phase_log) = phase_log { phase_log.completed( structured_log::ReconciliationPhase::Gateway, @@ -596,7 +731,9 @@ async fn reconcile_gateway_runtimes_with_pf_state( ); } - return Ok(CADDY_NOT_INSTALLED.to_owned()); + return Ok(ReconciliationOutcome::Completed( + CADDY_NOT_INSTALLED.to_owned(), + )); }; let worker_timer = phase_log.map(|phase_log| { @@ -625,19 +762,22 @@ async fn reconcile_gateway_runtimes_with_pf_state( let recording = record_runtime_error(paths, RuntimeSubject::Gateway, &error).err(); let recovery = recover_previous_gateway( paths, - &supervisor, &gateway_command, None, previous_gateway.as_ref(), pf_routing_state, readiness_timeout, + shutdown, ) - .await - .err(); + .await; let mut failures = vec![("gateway".to_owned(), error)]; if let Some(recording) = recording { failures.push(("gateway".to_owned(), recording)); } + if matches!(recovery, Ok(ReconciliationOutcome::Cancelled)) { + return cancel_or_preserve_runtime_reconciliation_errors(failures); + } + let recovery = recovery.err(); if let Some(recovery) = recovery { failures.push(("gateway".to_owned(), recovery)); } @@ -745,41 +885,50 @@ async fn reconcile_gateway_runtimes_with_pf_state( } let workers = bounded_runtime_readiness(worker_commands, |(worker, worker_runtime)| { - let supervisor = &supervisor; let retained_fragments = retained_worker_fragments.get(&worker.runtime_key); async move { let result = reconcile_planned_worker( paths, - supervisor, &worker, &worker_runtime, readiness_timeout, None, retained_fragments, + shutdown, ) .await; (worker.runtime_key.clone(), result) } }); tokio::pin!(workers); + let mut cancelled = false; while let Some((runtime_key, result)) = workers.next().await { - if let Err(error) = result { - worker_failures.push((runtime_key, error)); + match result { + Ok(ReconciliationOutcome::Completed(())) => {} + Ok(ReconciliationOutcome::Cancelled) => cancelled = true, + Err(error) => worker_failures.push((runtime_key, error)), } } + if cancelled { + return cancel_or_preserve_runtime_reconciliation_errors(worker_failures); + } if !worker_failures.is_empty() { - if let Err(error) = recover_previous_gateway( + let recovery = recover_previous_gateway( paths, - &supervisor, &gateway_command, Some(&plan), previous_gateway.as_ref(), pf_routing_state, readiness_timeout, + shutdown, ) - .await - { - worker_failures.push(("gateway".to_owned(), error)); + .await; + match recovery { + Ok(ReconciliationOutcome::Cancelled) => { + return cancel_or_preserve_runtime_reconciliation_errors(worker_failures); + } + Ok(ReconciliationOutcome::Completed(())) => {} + Err(error) => worker_failures.push(("gateway".to_owned(), error)), } return Err(combined_runtime_reconciliation_error(worker_failures)); } @@ -801,16 +950,19 @@ async fn reconcile_gateway_runtimes_with_pf_state( let gateway_timer = phase_log .map(|phase_log| phase_log.start(structured_log::ReconciliationPhase::Gateway, "gateway")); - reconcile_planned_gateway( + let gateway_outcome = reconcile_planned_gateway( paths, - &supervisor, &plan, &gateway_command, pf_routing_state, readiness_timeout, None, + shutdown, ) .await?; + if matches!(gateway_outcome, ReconciliationOutcome::Cancelled) { + return Ok(ReconciliationOutcome::Cancelled); + } if let Some(gateway_timer) = gateway_timer { gateway_timer.finish( structured_log::PhaseOutcome::Succeeded, @@ -825,28 +977,34 @@ async fn reconcile_gateway_runtimes_with_pf_state( ) }); let mut cleanup_failures: Vec<(String, DaemonError)> = Vec::new(); + let mut cleanup_cancelled = false; for worker in &plan.workers { if retained_worker_fragments.contains_key(&worker.runtime_key) { let result = match required_installed_worker_runtime(paths, worker) { Ok(worker_runtime) => { reconcile_planned_worker( paths, - &supervisor, worker, &worker_runtime, readiness_timeout, None, None, + shutdown, ) .await } Err(error) => Err(error), }; - if let Err(error) = result { - cleanup_failures.push((worker.runtime_key.clone(), error)); + match result { + Ok(ReconciliationOutcome::Completed(())) => {} + Ok(ReconciliationOutcome::Cancelled) => cleanup_cancelled = true, + Err(error) => cleanup_failures.push((worker.runtime_key.clone(), error)), } } } + if cleanup_cancelled { + return cancel_or_preserve_runtime_reconciliation_errors(cleanup_failures); + } let mut cleanup_timer = cleanup_timer; if let Err(error) = stop_stale_worker_runtimes(paths, &supervisor, &plan) .await @@ -867,51 +1025,60 @@ async fn reconcile_gateway_runtimes_with_pf_state( &[], ); } + if reconciliation_cancelled(shutdown) { + return cancel_or_preserve_runtime_reconciliation_errors(cleanup_failures); + } if !cleanup_failures.is_empty() { return Err(combined_runtime_reconciliation_error(cleanup_failures)); } - Ok(GATEWAY_RUNTIME_RECONCILED.to_owned()) + Ok(ReconciliationOutcome::Completed( + GATEWAY_RUNTIME_RECONCILED.to_owned(), + )) } async fn recover_previous_gateway( paths: &PvPaths, - supervisor: &ProcessSupervisor, gateway_command: &CaddyCliCommand, plan: Option<&RuntimePlan>, snapshot: Option<&ActiveRuntimeConfigSnapshot>, pf_routing_state: Option, readiness_timeout: Duration, -) -> Result<(), DaemonError> { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } + let supervisor = ProcessSupervisor::new(paths.clone()); let Some(snapshot) = snapshot else { // Prior bytes are unprovable: only a previously recorded Gateway that is now // definitively absent may be started from the desired plan, exactly once. A live, // unverifiable, or never-installed Gateway is left untouched so recovery never // issues a competing load or materializes a Gateway the plan never ran. let Some(plan) = plan else { - return Ok(()); + return Ok(ReconciliationOutcome::Completed(())); }; let gateway_spec = gateway_process_spec(paths, gateway_command); if supervisor .recorded_config_fingerprint(&gateway_spec)? .is_none() { - return Ok(()); + return Ok(ReconciliationOutcome::Completed(())); } match supervisor.verify_ownership(&gateway_spec) { Ok(None) => { return reconcile_planned_gateway( paths, - supervisor, plan, gateway_command, pf_routing_state, readiness_timeout, None, + shutdown, ) .await; } - Ok(Some(_)) | Err(_) => return Ok(()), + Ok(Some(_)) | Err(_) => return Ok(ReconciliationOutcome::Completed(())), } }; let active_dir = paths.gateway_projects_config_dir(); @@ -949,25 +1116,25 @@ async fn recover_previous_gateway( }; reconcile_gateway_config( paths, - supervisor, &plan, gateway_command, pf_routing_state, readiness_timeout, desired, + shutdown, ) .await } async fn reconcile_planned_gateway( paths: &PvPaths, - supervisor: &ProcessSupervisor, plan: &RuntimePlan, gateway_command: &CaddyCliCommand, pf_routing_state: Option, readiness_timeout: Duration, preserved_fragments: Option<&BTreeMap>, -) -> Result<(), DaemonError> { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { let desired_gateway_config = match desired_gateway_config(paths, plan, preserved_fragments) { Ok(desired_config) => desired_config, Err(error) => { @@ -976,25 +1143,26 @@ async fn reconcile_planned_gateway( }; reconcile_gateway_config( paths, - supervisor, plan, gateway_command, pf_routing_state, readiness_timeout, desired_gateway_config, + shutdown, ) .await } async fn reconcile_gateway_config( paths: &PvPaths, - supervisor: &ProcessSupervisor, plan: &RuntimePlan, gateway_command: &CaddyCliCommand, pf_routing_state: Option, readiness_timeout: Duration, desired_gateway_config: DesiredGatewayConfig, -) -> Result<(), DaemonError> { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + let supervisor = ProcessSupervisor::new(paths.clone()); let pf_routing_state = match pf_routing_state { Some(pf_routing_state) => pf_routing_state, None => gateway_pf_routing_state(paths, plan).await?, @@ -1007,7 +1175,7 @@ async fn reconcile_gateway_config( ); let gateway_spec = gateway_process_spec(paths, gateway_command); let readiness_outcome = if let Some(outcome) = match reconcile_unchanged_runtime( - supervisor, + &supervisor, &gateway_spec, &paths.gateway_root_config(), &gateway_readiness, @@ -1020,6 +1188,9 @@ async fn reconcile_gateway_config( return Err(record_primary_error(paths, RuntimeSubject::Gateway, error)); } } { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } outcome } else { let promoted_config = promote_runtime_config_tree( @@ -1029,65 +1200,90 @@ async fn reconcile_gateway_config( paths.gateway_root_config(), caddy_xdg_environment(paths), &desired_gateway_config.tree, + shutdown, ) .await?; - start_or_adopt_promoted_runtime( + let promoted_config = match promoted_config { + ReconciliationOutcome::Completed(promoted_config) => promoted_config, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + }; + if reconciliation_cancelled(shutdown) { + promoted_config + .rollback() + .map_err(|error| record_primary_error(paths, RuntimeSubject::Gateway, error))?; + return Ok(ReconciliationOutcome::Cancelled); + } + let runtime = prepare_promoted_runtime( paths, - supervisor, promoted_config, gateway_spec, gateway_readiness, &desired_gateway_config.tree.fingerprint, RuntimeSubject::Gateway, + shutdown, ) - .await? + .await?; + match runtime { + ReconciliationOutcome::Completed(runtime) => { + match finish_promoted_runtime(runtime.wait(shutdown).await).await? { + ReconciliationOutcome::Completed(outcome) => outcome, + ReconciliationOutcome::Cancelled => { + return Ok(ReconciliationOutcome::Cancelled); + } + } + } + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + } }; record_gateway_runtime_observed(paths, pf_routing_state, readiness_outcome)?; - Ok(()) + Ok(ReconciliationOutcome::Completed(())) } async fn reconcile_planned_worker( paths: &PvPaths, - supervisor: &ProcessSupervisor, worker: &PhpWorkerRuntimePlan, worker_runtime: &InstalledFrankenphpRuntime, readiness_timeout: Duration, preserved_fragments: Option<&BTreeMap>, retained_fragments: Option<&BTreeMap>, -) -> Result<(), DaemonError> { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { let prepared = prepare_planned_worker( paths, - supervisor, worker, worker_runtime, readiness_timeout, preserved_fragments, retained_fragments, + shutdown, ) .await?; match prepared { - PreparedWorkerReconciliation::Complete => Ok(()), + PreparedWorkerReconciliation::Complete => Ok(ReconciliationOutcome::Completed(())), + PreparedWorkerReconciliation::Cancelled => Ok(ReconciliationOutcome::Cancelled), PreparedWorkerReconciliation::Pending(worker) => { - finish_planned_worker(paths, (*worker).wait().await).await + finish_planned_worker(paths, (*worker).wait(shutdown).await).await } } } enum PreparedWorkerReconciliation { Complete, + Cancelled, Pending(Box), } async fn prepare_planned_worker( paths: &PvPaths, - supervisor: &ProcessSupervisor, worker: &PhpWorkerRuntimePlan, worker_runtime: &InstalledFrankenphpRuntime, readiness_timeout: Duration, preserved_fragments: Option<&BTreeMap>, retained_fragments: Option<&BTreeMap>, + shutdown: Option<&watch::Receiver>, ) -> Result { + let supervisor = ProcessSupervisor::new(paths.clone()); let subject = worker_runtime_subject(worker); let process_spec = match worker_process_spec( paths, @@ -1125,7 +1321,7 @@ async fn prepare_planned_worker( } }; match reconcile_unchanged_runtime( - supervisor, + &supervisor, &process_spec, &paths.worker_root_config(&worker.runtime_key), &readiness, @@ -1134,6 +1330,9 @@ async fn prepare_planned_worker( .await { Ok(Some(_outcome)) => { + if reconciliation_cancelled(shutdown) { + return Ok(PreparedWorkerReconciliation::Cancelled); + } record_runtime_observed( paths, subject, @@ -1154,28 +1353,50 @@ async fn prepare_planned_worker( paths.worker_root_config(&worker.runtime_key), private_environment, &desired_config, + shutdown, ) .await?; + let promoted_config = match promoted_config { + ReconciliationOutcome::Completed(promoted_config) => promoted_config, + ReconciliationOutcome::Cancelled => { + return Ok(PreparedWorkerReconciliation::Cancelled); + } + }; + if reconciliation_cancelled(shutdown) { + promoted_config + .rollback() + .map_err(|error| record_primary_error(paths, subject.clone(), error))?; + return Ok(PreparedWorkerReconciliation::Cancelled); + } let runtime = prepare_promoted_runtime( paths, - supervisor, promoted_config, process_spec, readiness, &desired_config.fingerprint, subject.clone(), + shutdown, ) .await?; - - Ok(PreparedWorkerReconciliation::Pending(Box::new(runtime))) + match runtime { + ReconciliationOutcome::Completed(runtime) => { + Ok(PreparedWorkerReconciliation::Pending(Box::new(runtime))) + } + ReconciliationOutcome::Cancelled => Ok(PreparedWorkerReconciliation::Cancelled), + } } async fn finish_planned_worker( paths: &PvPaths, completed: CompletedPromotedRuntime, -) -> Result<(), DaemonError> { +) -> Result, DaemonError> { let subject = completed.subject.clone(); - finish_promoted_runtime(completed).await?; + if matches!( + finish_promoted_runtime(completed).await?, + ReconciliationOutcome::Cancelled + ) { + return Ok(ReconciliationOutcome::Cancelled); + } record_runtime_observed( paths, subject, @@ -1183,7 +1404,7 @@ async fn finish_planned_worker( Some(GATEWAY_RUNTIME_RECONCILED), )?; - Ok(()) + Ok(ReconciliationOutcome::Completed(())) } fn combined_runtime_reconciliation_error(mut failures: Vec<(String, DaemonError)>) -> DaemonError { @@ -1200,6 +1421,16 @@ fn combined_runtime_reconciliation_error(mut failures: Vec<(String, DaemonError) } } +fn cancel_or_preserve_runtime_reconciliation_errors( + failures: Vec<(String, DaemonError)>, +) -> Result, DaemonError> { + if failures.is_empty() { + Ok(ReconciliationOutcome::Cancelled) + } else { + Err(combined_runtime_reconciliation_error(failures)) + } +} + fn required_installed_worker_runtime( paths: &PvPaths, worker: &PhpWorkerRuntimePlan, @@ -1642,10 +1873,33 @@ pub async fn validate_config( config_path: &Utf8Path, private_environment: &BTreeMap, ) -> Result<(), DaemonError> { - let output = run_validation_command(command, config_path, private_environment).await?; + validate_config_with_shutdown(command, config_path, private_environment, None, None) + .await? + .into_completed() +} + +async fn validate_config_with_shutdown( + command: &CaddyCliCommand, + config_path: &Utf8Path, + private_environment: &BTreeMap, + shutdown: Option<&watch::Receiver>, + cleanup_paths: Option<&PvPaths>, +) -> Result, DaemonError> { + let output = match run_validation_command( + command, + config_path, + private_environment, + shutdown, + cleanup_paths, + ) + .await? + { + ReconciliationOutcome::Completed(output) => output, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + }; if output.status.success() { - return Ok(()); + return Ok(ReconciliationOutcome::Completed(())); } let stdout = String::from_utf8_lossy(&output.stdout); @@ -1666,11 +1920,416 @@ struct ValidationOutput { stderr: Vec, } +struct ValidationProcess { + pid: u32, + child: Option, + group_pid: Option, + group_anchor: Option, + group_signal_pending: bool, + group_wait_pending: bool, + cleanup_diagnostics: Option<(PvPaths, String)>, + fallback_reaper: mpsc::Sender, +} + +struct ValidationProcessReap { + pid: u32, + child: Option, + group_pid: Option, + group_anchor: Option, + group_signal_pending: bool, + group_wait_pending: bool, + cleanup_diagnostics: Option<(PvPaths, String)>, + failures: Vec, +} + +impl ValidationProcess { + fn new( + pid: u32, + child: tokio::process::Child, + group_pid: Option, + group_anchor: Option, + cleanup_paths: Option<&PvPaths>, + runtime_label: &str, + fallback_reaper: mpsc::Sender, + ) -> Self { + Self { + pid, + child: Some(child), + group_pid, + group_anchor, + group_signal_pending: group_pid.is_some(), + group_wait_pending: group_pid.is_some(), + cleanup_diagnostics: cleanup_paths + .map(|paths| (paths.clone(), format!("{runtime_label} config validation"))), + fallback_reaper, + } + } + + fn child_mut(&mut self) -> Result<&mut tokio::process::Child, DaemonError> { + self.child + .as_mut() + .ok_or_else(|| DaemonError::MissingProcessId { + name: "config validation".to_owned(), + }) + } + + async fn wait(&mut self) -> Result { + let status = self.child_mut()?.wait().await?; + self.child = None; + + Ok(status) + } + + async fn terminate(&mut self) -> Result<(), DaemonError> { + let group_result = self.signal_group_before_reap(); + let child_result = kill_and_wait_validation_child(&mut self.child).await; + let anchor_result = if self.group_signal_pending { + Ok(()) + } else { + kill_and_wait_validation_child(&mut self.group_anchor).await + }; + let group_wait_result = if self.group_wait_pending && !self.group_signal_pending { + let result = wait_for_validation_process_group_exit(self.group_pid).await; + if result.is_ok() { + self.group_wait_pending = false; + } + result + } else { + Ok(()) + }; + + combine_validation_cleanup_results( + combine_validation_cleanup_results( + combine_validation_cleanup_results(group_result, child_result), + anchor_result, + ), + group_wait_result, + ) + } + + fn signal_group_before_reap(&mut self) -> Result<(), DaemonError> { + if !self.group_signal_pending { + return Ok(()); + } + + if self.group_anchor.is_none() { + return Err(io::Error::other( + "config validation process-group ownership was lost before signaling", + ) + .into()); + } + + signal_validation_process_group(self.group_pid)?; + self.group_signal_pending = false; + + Ok(()) + } +} + +impl Drop for ValidationProcess { + fn drop(&mut self) { + let cleanup = ValidationProcessReap { + pid: self.pid, + child: self.child.take(), + group_pid: self.group_pid, + group_anchor: self.group_anchor.take(), + group_signal_pending: self.group_signal_pending, + group_wait_pending: self.group_wait_pending, + cleanup_diagnostics: self.cleanup_diagnostics.take(), + failures: Vec::new(), + }; + let _cleanup_result = dispatch_validation_process_reap(&self.fallback_reaper, cleanup); + } +} + +impl ValidationProcessReap { + fn reap(mut self) { + let mut deadline = Instant::now() + Duration::from_secs(1); + let mut pending_signal_failure_logged = false; + loop { + if self.group_signal_pending { + match signal_validation_process_group(self.group_pid) { + Ok(()) => { + self.group_signal_pending = false; + let _kill_result = start_kill_validation_child( + &mut self.group_anchor, + "validation process-group anchor", + &mut self.failures, + ); + deadline = Instant::now() + Duration::from_secs(1); + } + Err(error) if self.failures.is_empty() => { + self.failures.push(error.to_string()); + } + Err(_error) => {} + } + } + let child_exited = validation_child_exited( + &mut self.child, + "validation process", + self.pid, + &mut self.failures, + ); + let anchor_exited = if self.group_signal_pending { + false + } else { + validation_child_exited( + &mut self.group_anchor, + "validation process-group anchor", + self.group_pid.unwrap_or_default(), + &mut self.failures, + ) + }; + let group_exited = if self.group_wait_pending && !self.group_signal_pending { + match validation_process_group_exists(self.group_pid) { + Ok(exists) => !exists, + Err(error) => { + self.failures.push(error.to_string()); + true + } + } + } else { + true + }; + if child_exited && anchor_exited && group_exited { + break; + } + if Instant::now() >= deadline { + if self.group_signal_pending { + if !pending_signal_failure_logged { + self.log_cleanup_failures(); + self.failures.clear(); + pending_signal_failure_logged = true; + } + std::thread::sleep(OWNED_READINESS_POLL_INTERVAL); + continue; + } + if !child_exited { + self.failures.push(format!( + "validation process {} was not reaped within one second", + self.pid + )); + } + if !anchor_exited { + self.failures.push(format!( + "validation process-group anchor {} was not reaped within one second", + self.group_pid.unwrap_or_default() + )); + } + if !group_exited { + self.failures.push(format!( + "validation process group {} did not exit within one second", + self.group_pid.unwrap_or_default() + )); + } + break; + } + std::thread::sleep(OWNED_READINESS_POLL_INTERVAL); + } + self.log_cleanup_failures(); + } + + fn log_cleanup_failures(&self) { + if !self.failures.is_empty() + && let Some((paths, runtime)) = &self.cleanup_diagnostics + { + structured_log::runtime_config_cleanup_failed( + paths, + runtime, + &self.failures.join("; "), + ); + } + } +} + +fn start_validation_process_reaper() -> Result, DaemonError> { + let (sender, receiver) = mpsc::channel::(); + std::thread::Builder::new() + .name("pv-config-validation-reaper".to_owned()) + .spawn(move || { + if let Ok(cleanup) = receiver.recv() { + cleanup.reap(); + } + })?; + + Ok(sender) +} + +fn dispatch_validation_process_reap( + fallback_reaper: &mpsc::Sender, + mut cleanup: ValidationProcessReap, +) -> Result<(), DaemonError> { + let group_result = if cleanup.group_signal_pending { + let result = signal_validation_process_group(cleanup.group_pid); + if result.is_ok() { + cleanup.group_signal_pending = false; + } + result + } else { + Ok(()) + }; + if let Err(error) = &group_result { + cleanup.failures.push(error.to_string()); + } + + let child_result = start_kill_validation_child( + &mut cleanup.child, + "validation process", + &mut cleanup.failures, + ); + let anchor_result = if cleanup.group_signal_pending { + Ok(()) + } else { + start_kill_validation_child( + &mut cleanup.group_anchor, + "validation process-group anchor", + &mut cleanup.failures, + ) + }; + let result = combine_validation_cleanup_results( + combine_validation_cleanup_results(group_result, child_result), + anchor_result, + ); + + if let Err(error) = fallback_reaper.send(cleanup) { + error.0.reap(); + } + + result +} + +fn validation_child_exited( + child: &mut Option, + description: &str, + pid: impl std::fmt::Display, + failures: &mut Vec, +) -> bool { + match child.as_mut().map(tokio::process::Child::try_wait) { + Some(Ok(Some(_))) | None => true, + Some(Ok(None)) => false, + Some(Err(error)) => { + failures.push(format!("failed to reap {description} {pid}: {error}")); + true + } + } +} + +fn start_kill_validation_child( + child: &mut Option, + description: &str, + failures: &mut Vec, +) -> Result<(), DaemonError> { + let result = child + .as_mut() + .map(tokio::process::Child::start_kill) + .transpose() + .map(|_result| ()) + .map_err(DaemonError::from); + if let Err(error) = &result { + failures.push(format!("failed to kill {description}: {error}")); + } + + result +} + +async fn kill_and_wait_validation_child( + child: &mut Option, +) -> Result<(), DaemonError> { + let Some(child_process) = child.as_mut() else { + return Ok(()); + }; + let kill_result = child_process.start_kill().map_err(DaemonError::from); + let wait_result = child_process + .wait() + .await + .map(|_status| ()) + .map_err(DaemonError::from); + if wait_result.is_ok() { + *child = None; + } + + combine_validation_cleanup_results(kill_result, wait_result) +} + +#[cfg(target_os = "macos")] +fn spawn_validation_process_group_anchor( + fallback_reaper: &mpsc::Sender, +) -> Result<(Option, Option), DaemonError> { + let mut command = RuntimeProcessCommand::new("/bin/sh"); + command + .args(["-c", CONFIG_VALIDATION_GROUP_ANCHOR]) + .stdin(Stdio::piped()) + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .kill_on_drop(true) + .process_group(0); + let anchor = Some(command.spawn()?); + let Some(pid) = anchor.as_ref().and_then(tokio::process::Child::id) else { + let error = DaemonError::MissingProcessId { + name: "config validation process-group anchor".to_owned(), + }; + let cleanup = dispatch_validation_process_reap( + fallback_reaper, + ValidationProcessReap { + pid: 0, + child: None, + group_pid: None, + group_anchor: anchor, + group_signal_pending: false, + group_wait_pending: false, + cleanup_diagnostics: None, + failures: Vec::new(), + }, + ); + + return Err(preserve_validation_error(error, cleanup)); + }; + let group_pid = match i32::try_from(pid) { + Ok(group_pid) => group_pid, + Err(_source) => { + let error = io::Error::new( + io::ErrorKind::InvalidInput, + format!("config validation process-group id {pid} is invalid"), + ) + .into(); + let cleanup = dispatch_validation_process_reap( + fallback_reaper, + ValidationProcessReap { + pid: 0, + child: None, + group_pid: None, + group_anchor: anchor, + group_signal_pending: false, + group_wait_pending: false, + cleanup_diagnostics: None, + failures: Vec::new(), + }, + ); + + return Err(preserve_validation_error(error, cleanup)); + } + }; + + Ok((Some(group_pid), anchor)) +} + +#[cfg(any(target_os = "linux", target_os = "windows"))] +fn spawn_validation_process_group_anchor( + _fallback_reaper: &mpsc::Sender, +) -> Result<(Option, Option), DaemonError> { + Ok((None, None)) +} + async fn run_validation_command( command: &CaddyCliCommand, config_path: &Utf8Path, private_environment: &BTreeMap, -) -> Result { + shutdown: Option<&watch::Receiver>, + cleanup_paths: Option<&PvPaths>, +) -> Result, DaemonError> { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let mut command_process = RuntimeProcessCommand::new(command.executable()); command_process .args(command.validate_arguments(config_path)) @@ -1681,35 +2340,199 @@ async fn run_validation_command( command_process.env_remove(key); } command_process.envs(private_environment); + let fallback_reaper = start_validation_process_reaper()?; + let (group_pid, group_anchor) = spawn_validation_process_group_anchor(&fallback_reaper)?; #[cfg(target_os = "macos")] - command_process.process_group(0); + if let Some(group_pid) = group_pid { + command_process.process_group(group_pid); + } + + let mut child = match command_process.spawn() { + Ok(child) => child, + Err(source) => { + let error = DaemonError::from(source); + let cleanup = dispatch_validation_process_reap( + &fallback_reaper, + ValidationProcessReap { + pid: 0, + child: None, + group_pid, + group_anchor, + group_signal_pending: group_pid.is_some(), + group_wait_pending: group_pid.is_some(), + cleanup_diagnostics: cleanup_paths.map(|paths| { + ( + paths.clone(), + format!("{} config validation", command.runtime_label()), + ) + }), + failures: Vec::new(), + }, + ); - let mut child = command_process.spawn()?; + return Err(preserve_validation_error(error, cleanup)); + } + }; let Some(pid) = child.id() else { - return Err(DaemonError::MissingProcessId { + let error = DaemonError::MissingProcessId { name: format!("{} config validation", command.runtime_label()), - }); + }; + let cleanup = dispatch_validation_process_reap( + &fallback_reaper, + ValidationProcessReap { + pid: 0, + child: Some(child), + group_pid, + group_anchor, + group_signal_pending: group_pid.is_some(), + group_wait_pending: group_pid.is_some(), + cleanup_diagnostics: cleanup_paths.map(|paths| { + ( + paths.clone(), + format!("{} config validation", command.runtime_label()), + ) + }), + failures: Vec::new(), + }, + ); + + return Err(preserve_validation_error(error, cleanup)); }; let stdout = tokio::spawn(read_child_output(child.stdout.take())); let stderr = tokio::spawn(read_child_output(child.stderr.take())); - let status = match timeout(CONFIG_VALIDATION_TIMEOUT, child.wait()).await { - Ok(result) => result?, - Err(_elapsed) => { - terminate_validation_process(pid, &mut child).await; + let mut process = ValidationProcess::new( + pid, + child, + group_pid, + group_anchor, + cleanup_paths, + command.runtime_label(), + fallback_reaper, + ); + let status = match wait_for_transaction_step( + Box::pin(timeout(CONFIG_VALIDATION_TIMEOUT, process.wait())), + shutdown, + ) + .await + { + ReconciliationOutcome::Completed(Ok(Ok(status))) => status, + ReconciliationOutcome::Completed(Ok(Err(error))) => { + let cleanup_result = cleanup_validation_process(process, stdout, stderr).await; - return Err(DaemonError::ProtocolTimedOut { + return Err(preserve_validation_error(error, cleanup_result)); + } + ReconciliationOutcome::Completed(Err(_elapsed)) => { + let error = DaemonError::ProtocolTimedOut { phase: command.validation_phase(), - }); + }; + let cleanup_result = cleanup_validation_process(process, stdout, stderr).await; + + return Err(preserve_validation_error(error, cleanup_result)); + } + ReconciliationOutcome::Cancelled => { + cleanup_validation_process(process, stdout, stderr).await?; + + return Ok(ReconciliationOutcome::Cancelled); } }; - let stdout = stdout.await.map_err(io::Error::other)??; - let stderr = stderr.await.map_err(io::Error::other)??; + let (stdout, stderr) = cleanup_validation_process(process, stdout, stderr).await?; - Ok(ValidationOutput { + Ok(ReconciliationOutcome::Completed(ValidationOutput { status, stdout, stderr, + })) +} + +async fn cleanup_validation_process( + mut process: ValidationProcess, + mut stdout: JoinHandle>>, + mut stderr: JoinHandle>>, +) -> Result<(Vec, Vec), DaemonError> { + let process_cleanup = timeout(CONFIG_VALIDATION_CLEANUP_TIMEOUT, process.terminate()) + .await + .map_err(|_| DaemonError::ProtocolTimedOut { + phase: "config validation process cleanup", + }) + .and_then(|result| result); + drop(process); + + let output_cleanup = timeout(CONFIG_VALIDATION_CLEANUP_TIMEOUT, async { + let (stdout_result, stderr_result) = tokio::join!(&mut stdout, &mut stderr); + let stdout_result = stdout_result + .map_err(io::Error::other) + .map_err(DaemonError::from) + .and_then(|result| result.map_err(DaemonError::from)); + let stderr_result = stderr_result + .map_err(io::Error::other) + .map_err(DaemonError::from) + .and_then(|result| result.map_err(DaemonError::from)); + + combine_validation_output_results(stdout_result, stderr_result) }) + .await; + let output_cleanup = match output_cleanup { + Ok(result) => result, + Err(_elapsed) => { + stdout.abort(); + stderr.abort(); + let (_stdout_result, _stderr_result) = tokio::join!(stdout, stderr); + + Err(DaemonError::ProtocolTimedOut { + phase: "config validation output cleanup", + }) + } + }; + + match (process_cleanup, output_cleanup) { + (Ok(()), Ok(output)) => Ok(output), + (Err(error), Ok(_)) | (Ok(()), Err(error)) => Err(error), + (Err(error), Err(cleanup)) => Err(runtime_cleanup_failed_error( + "config validation", + error, + cleanup, + )), + } +} + +fn combine_validation_output_results( + stdout: Result, DaemonError>, + stderr: Result, DaemonError>, +) -> Result<(Vec, Vec), DaemonError> { + match (stdout, stderr) { + (Ok(stdout), Ok(stderr)) => Ok((stdout, stderr)), + (Err(error), Ok(_)) | (Ok(_), Err(error)) => Err(error), + (Err(error), Err(cleanup)) => Err(runtime_cleanup_failed_error( + "config validation output", + error, + cleanup, + )), + } +} + +fn combine_validation_cleanup_results( + first: Result<(), DaemonError>, + second: Result<(), DaemonError>, +) -> Result<(), DaemonError> { + match (first, second) { + (Ok(()), Ok(())) => Ok(()), + (Err(error), Ok(())) | (Ok(()), Err(error)) => Err(error), + (Err(error), Err(cleanup)) => Err(runtime_cleanup_failed_error( + "config validation", + error, + cleanup, + )), + } +} + +fn preserve_validation_error( + error: DaemonError, + cleanup: Result, +) -> DaemonError { + match cleanup { + Ok(_output) => error, + Err(cleanup) => runtime_cleanup_failed_error("config validation", error, cleanup), + } } async fn read_child_output(output: Option) -> io::Result> @@ -1726,24 +2549,62 @@ where Ok(content) } -async fn terminate_validation_process(pid: u32, child: &mut tokio::process::Child) { - #[cfg(not(target_os = "macos"))] - let _ = pid; +#[cfg(target_os = "macos")] +fn signal_validation_process_group(pid: Option) -> Result<(), DaemonError> { + let process_group = validation_process_group(pid).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidInput, + "validation process id must be positive", + ) + })?; + match kill_process_group(process_group, Signal::KILL) { + Ok(()) | Err(rustix::io::Errno::SRCH) => Ok(()), + Err(error) => Err(io::Error::from(error).into()), + } +} + +#[cfg(any(target_os = "linux", target_os = "windows"))] +fn signal_validation_process_group(_pid: Option) -> Result<(), DaemonError> { + Ok(()) +} - #[cfg(target_os = "macos")] - { - if let Some(process_group) = validation_process_group(pid) { - let _result = kill_process_group(process_group, Signal::KILL); +#[cfg(target_os = "macos")] +fn validation_process_group(pid: Option) -> Option { + pid.and_then(Pid::from_raw) +} + +async fn wait_for_validation_process_group_exit(pid: Option) -> Result<(), DaemonError> { + let deadline = Instant::now() + CONFIG_VALIDATION_CLEANUP_TIMEOUT; + while validation_process_group_exists(pid)? { + if Instant::now() >= deadline { + return Err(DaemonError::ProtocolTimedOut { + phase: "config validation process-group cleanup", + }); } + sleep(OWNED_READINESS_POLL_INTERVAL).await; } - let _result = child.kill().await; - let _result = child.wait().await; + Ok(()) } #[cfg(target_os = "macos")] -fn validation_process_group(pid: u32) -> Option { - i32::try_from(pid).ok().and_then(Pid::from_raw) +fn validation_process_group_exists(pid: Option) -> Result { + let process_group = validation_process_group(pid).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidInput, + "validation process id must be positive", + ) + })?; + match test_kill_process_group(process_group) { + Ok(()) => Ok(true), + Err(rustix::io::Errno::SRCH | rustix::io::Errno::PERM) => Ok(false), + Err(error) => Err(io::Error::from(error).into()), + } +} + +#[cfg(any(target_os = "linux", target_os = "windows"))] +fn validation_process_group_exists(_pid: Option) -> Result { + Ok(false) } pub fn gateway_process_spec(paths: &PvPaths, command: &CaddyCliCommand) -> ProcessSpec { @@ -2837,17 +3698,40 @@ async fn promote_runtime_config_tree( config_path: Utf8PathBuf, private_environment: BTreeMap, desired: &DesiredRuntimeConfigTree, -) -> Result { - let result = delete_optional_dir(&desired.candidate_dir) + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + let validation_cancelled = Arc::new(AtomicBool::new(false)); + let mut candidate_dir = CandidateConfigDirGuard::new(paths, desired.candidate_dir.clone()); + let result = candidate_dir + .delete_current() .and_then(|()| write_project_config_fragments(&desired.candidate_dir, &desired.fragments)); let result = match result { Ok(()) => { + let validation_cancelled_for_command = Arc::clone(&validation_cancelled); promote_validated_config_tree_async( &config_path, &desired.candidate_content, &desired.active_content, |candidate_path| async move { - validate_config(command, &candidate_path, &private_environment).await + match validate_config_with_shutdown( + command, + &candidate_path, + &private_environment, + shutdown, + Some(paths), + ) + .await? + { + ReconciliationOutcome::Completed(()) => Ok(()), + ReconciliationOutcome::Cancelled => { + delete_optional_file(&candidate_path)?; + validation_cancelled_for_command.store(true, Ordering::SeqCst); + + Err(DaemonError::UnexpectedProtocolResponse { + reason: "config validation was cancelled".to_owned(), + }) + } + } }, || promote_config_dir(&desired.active_dir, &desired.candidate_dir), ) @@ -2855,9 +3739,18 @@ async fn promote_runtime_config_tree( } Err(error) => Err(error), }; + if validation_cancelled.load(Ordering::SeqCst) { + return match candidate_dir.cleanup() { + Ok(()) => Ok(ReconciliationOutcome::Cancelled), + Err(error) => Err(record_primary_error(paths, subject, error)), + }; + } let result = match result { - Ok(promoted_config) => Ok(promoted_config), - Err(error) => match delete_optional_dir(&desired.candidate_dir) { + Ok(promoted_config) => { + candidate_dir.disarm(); + Ok(promoted_config) + } + Err(error) => match candidate_dir.cleanup() { Ok(()) => Err(error), Err(cleanup_error) => Err(runtime_cleanup_failed_error( desired.candidate_dir.as_str(), @@ -2868,11 +3761,57 @@ async fn promote_runtime_config_tree( }; match result { - Ok(promoted_config) => Ok(promoted_config), + Ok(promoted_config) => Ok(ReconciliationOutcome::Completed(promoted_config)), Err(error) => Err(record_primary_error(paths, subject, error)), } } +struct CandidateConfigDirGuard { + paths: PvPaths, + path: Utf8PathBuf, + armed: bool, +} + +impl CandidateConfigDirGuard { + fn new(paths: &PvPaths, path: Utf8PathBuf) -> Self { + Self { + paths: paths.clone(), + path, + armed: true, + } + } + + fn delete_current(&self) -> Result<(), DaemonError> { + delete_optional_dir(&self.path) + } + + fn cleanup(&mut self) -> Result<(), DaemonError> { + delete_optional_dir(&self.path)?; + self.disarm(); + + Ok(()) + } + + fn disarm(&mut self) { + self.armed = false; + } +} + +impl Drop for CandidateConfigDirGuard { + fn drop(&mut self) { + if !self.armed { + return; + } + if let Err(error) = delete_optional_dir(&self.path) { + structured_log::runtime_config_cleanup_failed( + &self.paths, + &format!("candidate config directory `{}`", self.path), + &error.to_string(), + ); + } + } +} + async fn reconcile_unchanged_runtime( supervisor: &ProcessSupervisor, spec: &ProcessSpec, @@ -2927,29 +3866,6 @@ async fn reconcile_unchanged_runtime( Ok(Some(RuntimeReadinessOutcome::Verified)) } -async fn start_or_adopt_promoted_runtime( - paths: &PvPaths, - supervisor: &ProcessSupervisor, - promoted_config: PromotedConfigTree, - spec: ProcessSpec, - readiness: RuntimeReadinessPlan, - desired_fingerprint: &str, - subject: RuntimeSubject, -) -> Result { - let pending = prepare_promoted_runtime( - paths, - supervisor, - promoted_config, - spec, - readiness, - desired_fingerprint, - subject, - ) - .await?; - - finish_promoted_runtime(pending.wait().await).await -} - struct PendingPromotedRuntime { paths: PvPaths, promoted_config: PromotedConfigTree, @@ -2957,19 +3873,26 @@ struct PendingPromotedRuntime { readiness: RuntimeReadinessPlan, desired_fingerprint: String, subject: RuntimeSubject, - previous_fingerprint: Option, + origin: PromotedRuntimeOrigin, restoration_readiness: RuntimeReadinessPlan, started: StartedRuntimeTransaction, } +enum PromotedRuntimeOrigin { + Fresh, + Matching { + previous_fingerprint: Option, + }, +} + struct CompletedPromotedRuntime { paths: PvPaths, promoted_config: PromotedConfigTree, spec: ProcessSpec, subject: RuntimeSubject, - previous_fingerprint: Option, + origin: PromotedRuntimeOrigin, restoration_readiness: RuntimeReadinessPlan, - result: Result, + result: Result, RuntimeTransactionError>, } struct PromotedRuntimeRecovery { @@ -2982,7 +3905,7 @@ struct PromotedRuntimeRecovery { } impl PendingPromotedRuntime { - async fn wait(self) -> CompletedPromotedRuntime { + async fn wait(self, shutdown: Option<&watch::Receiver>) -> CompletedPromotedRuntime { let Self { paths, promoted_config, @@ -2990,7 +3913,7 @@ impl PendingPromotedRuntime { readiness, desired_fingerprint, subject, - previous_fingerprint, + origin, restoration_readiness, started, } = self; @@ -3002,6 +3925,7 @@ impl PendingPromotedRuntime { &readiness, &desired_fingerprint, started, + shutdown, ) .await; @@ -3010,7 +3934,7 @@ impl PendingPromotedRuntime { promoted_config, spec, subject, - previous_fingerprint, + origin, restoration_readiness, result, } @@ -3019,13 +3943,14 @@ impl PendingPromotedRuntime { async fn prepare_promoted_runtime( paths: &PvPaths, - supervisor: &ProcessSupervisor, promoted_config: PromotedConfigTree, spec: ProcessSpec, readiness: RuntimeReadinessPlan, desired_fingerprint: &str, subject: RuntimeSubject, -) -> Result { + shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + let supervisor = ProcessSupervisor::new(paths.clone()); let matching_runtime = match supervisor.verify_ownership(&spec) { Ok(Some(runtime)) => (!runtime.replacement_required() && promoted_config @@ -3044,10 +3969,13 @@ async fn prepare_promoted_runtime( return Err(record_primary_error(paths, subject.clone(), error)); } }; - let previous_fingerprint = matching_runtime + let origin = matching_runtime .as_ref() - .and_then(|runtime| runtime.applied_config_fingerprint()) - .map(str::to_owned); + .map_or(PromotedRuntimeOrigin::Fresh, |runtime| { + PromotedRuntimeOrigin::Matching { + previous_fingerprint: runtime.applied_config_fingerprint().map(str::to_owned), + } + }); let matching_runtime = matching_runtime.is_some(); let previous_readiness = if matching_runtime { match previous_runtime_readiness(&promoted_config, &readiness) { @@ -3073,16 +4001,29 @@ async fn prepare_promoted_runtime( }; let started = match begin_runtime_transaction( paths, - supervisor, + &supervisor, &spec, &readiness, desired_fingerprint, matching_runtime, + shutdown, ) .await { - Ok(started) => started, + Ok(ReconciliationOutcome::Completed(started)) => started, + Ok(ReconciliationOutcome::Cancelled) => { + promoted_config + .rollback() + .map_err(|error| record_primary_error(paths, subject, error))?; + return Ok(ReconciliationOutcome::Cancelled); + } Err(error) => { + let previous_fingerprint = match origin { + PromotedRuntimeOrigin::Fresh => None, + PromotedRuntimeOrigin::Matching { + previous_fingerprint, + } => previous_fingerprint, + }; return Err(recover_promoted_runtime( PromotedRuntimeRecovery { paths: paths.clone(), @@ -3097,34 +4038,34 @@ async fn prepare_promoted_runtime( .await); } }; - - Ok(PendingPromotedRuntime { + let pending = PendingPromotedRuntime { paths: paths.clone(), promoted_config, spec, readiness, desired_fingerprint: desired_fingerprint.to_owned(), subject, - previous_fingerprint, + origin, restoration_readiness, started, - }) + }; + Ok(ReconciliationOutcome::Completed(pending)) } async fn finish_promoted_runtime( completed: CompletedPromotedRuntime, -) -> Result { +) -> Result, DaemonError> { let CompletedPromotedRuntime { paths, promoted_config, spec, subject, - previous_fingerprint, + origin, restoration_readiness, result, } = completed; match result { - Ok(outcome) => { + Ok(ReconciliationOutcome::Completed(outcome)) => { if let Err(error) = promoted_config.cleanup() { structured_log::runtime_config_cleanup_failed( &paths, @@ -3132,20 +4073,57 @@ async fn finish_promoted_runtime( &error.to_string(), ); } - Ok(outcome) + Ok(ReconciliationOutcome::Completed(outcome)) + } + Ok(ReconciliationOutcome::Cancelled) => { + cancel_promoted_runtime(paths, promoted_config, spec, subject, origin)?; + + Ok(ReconciliationOutcome::Cancelled) + } + Err(error) => { + let previous_fingerprint = match origin { + PromotedRuntimeOrigin::Fresh => None, + PromotedRuntimeOrigin::Matching { + previous_fingerprint, + } => previous_fingerprint, + }; + Err(recover_promoted_runtime( + PromotedRuntimeRecovery { + paths, + promoted_config, + spec, + subject, + previous_fingerprint, + restoration_readiness, + }, + error, + ) + .await) + } + } +} + +fn cancel_promoted_runtime( + paths: PvPaths, + promoted_config: PromotedConfigTree, + spec: ProcessSpec, + subject: RuntimeSubject, + origin: PromotedRuntimeOrigin, +) -> Result<(), DaemonError> { + match origin { + PromotedRuntimeOrigin::Fresh => promoted_config + .rollback() + .map_err(|error| record_primary_error(&paths, subject, error)), + PromotedRuntimeOrigin::Matching { .. } => { + if let Err(error) = promoted_config.cleanup() { + structured_log::runtime_config_cleanup_failed( + &paths, + &spec.name, + &error.to_string(), + ); + } + Ok(()) } - Err(error) => Err(recover_promoted_runtime( - PromotedRuntimeRecovery { - paths, - promoted_config, - spec, - subject, - previous_fingerprint, - restoration_readiness, - }, - error, - ) - .await), } } @@ -3336,8 +4314,12 @@ async fn begin_runtime_transaction( readiness: &RuntimeReadinessPlan, desired_fingerprint: &str, matching_runtime: bool, -) -> Result { + shutdown: Option<&watch::Receiver>, +) -> Result, RuntimeTransactionError> { if matching_runtime { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } if supervisor.verify_ownership(spec)?.is_none() { return Err(RuntimeTransactionError::new( CaddyAdminError::runtime_ownership_changed(spec.name.clone()).into(), @@ -3347,18 +4329,31 @@ async fn begin_runtime_transaction( let active_content = read_config_bytes(&spec.config_path)?; let client = CaddyAdminClient::new().with_timeout(readiness.timeout); mark_runtime_config_pending(supervisor, spec, desired_fingerprint, false)?; - load_runtime_config( - paths, - spec, - client, - &readiness.admin_endpoint, - active_content, + let load_result = wait_for_transaction_step( + Box::pin(load_runtime_config( + paths, + spec, + client, + &readiness.admin_endpoint, + active_content, + )), + shutdown, ) - .await?; + .await; + match load_result { + ReconciliationOutcome::Completed(result) => result?, + ReconciliationOutcome::Cancelled => { + return Ok(ReconciliationOutcome::Completed( + StartedRuntimeTransaction::Matching, + )); + } + } verify_runtime_ownership(supervisor, spec) .map_err(RuntimeTransactionError::pending_preserve)?; - return Ok(StartedRuntimeTransaction::Matching); + return Ok(ReconciliationOutcome::Completed( + StartedRuntimeTransaction::Matching, + )); } if let Some(adopted) = supervisor.adopt_recorded(&spec.pid_path, &spec.metadata_path)? { adopted.stop(Duration::from_secs(1)).await?; @@ -3374,6 +4369,9 @@ async fn begin_runtime_transaction( } delete_optional_file(readiness.admin_endpoint.path()).map_err(RuntimeTransactionError::new)?; + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let process = supervisor.start(spec.clone()).await?; let staging_error = match supervisor.mark_replacement_required(spec, desired_fingerprint) { Ok(true) => None, @@ -3392,7 +4390,9 @@ async fn begin_runtime_transaction( .await); } - Ok(StartedRuntimeTransaction::Fresh(Box::new(process))) + Ok(ReconciliationOutcome::Completed( + StartedRuntimeTransaction::Fresh(Box::new(process)), + )) } async fn finish_runtime_transaction( @@ -3402,7 +4402,8 @@ async fn finish_runtime_transaction( readiness: &RuntimeReadinessPlan, desired_fingerprint: &str, started: StartedRuntimeTransaction, -) -> Result { + shutdown: Option<&watch::Receiver>, +) -> Result, RuntimeTransactionError> { let RuntimeReadinessPlan { check, failure_policy, @@ -3412,39 +4413,74 @@ async fn finish_runtime_transaction( } = readiness; let preserve_staged_config = *preserve_staged_config; let StartedRuntimeTransaction::Fresh(process) = started else { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let client = CaddyAdminClient::new().with_timeout(*readiness_timeout); - if let Err(error) = wait_for_owned_readiness(check.clone(), *readiness_timeout, || { - verify_runtime_ownership(supervisor, spec) - }) - .await - { + let readiness_result = wait_for_transaction_step( + Box::pin(wait_for_owned_readiness( + check.clone(), + *readiness_timeout, + || verify_runtime_ownership(supervisor, spec), + )), + shutdown, + ) + .await; + let readiness_result = match readiness_result { + ReconciliationOutcome::Completed(result) => result, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + }; + if let Err(error) = readiness_result { if *failure_policy == ReadinessFailurePolicy::PreserveRuntime && supervisor .verify_ownership(spec) .map_err(RuntimeTransactionError::pending_preserve)? .is_some() - && client - .wait_until_ready_with( + { + let admin_result = wait_for_transaction_step( + Box::pin(client.wait_until_ready_with( admin_endpoint, *readiness_timeout, runtime_ownership_verifier(paths, spec), - ) - .await - .is_ok() - { - record_applied_runtime_config(supervisor, spec, desired_fingerprint)?; - return Ok(RuntimeReadinessOutcome::Unverified); + )), + shutdown, + ) + .await; + match admin_result { + ReconciliationOutcome::Cancelled => { + return Ok(ReconciliationOutcome::Cancelled); + } + ReconciliationOutcome::Completed(Ok(())) => { + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } + record_applied_runtime_config(supervisor, spec, desired_fingerprint)?; + return Ok(ReconciliationOutcome::Completed( + RuntimeReadinessOutcome::Unverified, + )); + } + ReconciliationOutcome::Completed(Err(_error)) => {} + } } return Err(RuntimeTransactionError::pending_requiring_restore(error)); } verify_runtime_ownership(supervisor, spec) .map_err(RuntimeTransactionError::pending_preserve)?; + if reconciliation_cancelled(shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } record_applied_runtime_config(supervisor, spec, desired_fingerprint)?; - return Ok(RuntimeReadinessOutcome::Verified); + return Ok(ReconciliationOutcome::Completed( + RuntimeReadinessOutcome::Verified, + )); }; let mut process = *process; + if reconciliation_cancelled(shutdown) { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } let admin_readiness = async { CaddyAdminClient::new() .with_timeout(*readiness_timeout) @@ -3456,14 +4492,24 @@ async fn finish_runtime_transaction( .await .map_err(DaemonError::from) }; - if let Err(error) = wait_for_started_runtime_readiness( - &mut process, - &spec.name, - admin_readiness, - OWNED_READINESS_POLL_INTERVAL, + let admin_result = wait_for_transaction_step( + Box::pin(wait_for_started_runtime_readiness( + &mut process, + &spec.name, + admin_readiness, + OWNED_READINESS_POLL_INTERVAL, + )), + shutdown, ) - .await - { + .await; + let admin_result = match admin_result { + ReconciliationOutcome::Completed(result) => result, + ReconciliationOutcome::Cancelled => { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } + }; + if let Err(error) = admin_result { record_runtime_readiness_diagnostics(paths, spec, &mut process, &error); return Err(cleanup_fresh_runtime( supervisor, @@ -3478,14 +4524,24 @@ async fn finish_runtime_transaction( let service_readiness = wait_for_owned_readiness(check.clone(), *readiness_timeout, || { verify_runtime_ownership(supervisor, spec) }); - if let Err(error) = wait_for_started_runtime_readiness( - &mut process, - &spec.name, - service_readiness, - OWNED_READINESS_POLL_INTERVAL, + let service_result = wait_for_transaction_step( + Box::pin(wait_for_started_runtime_readiness( + &mut process, + &spec.name, + service_readiness, + OWNED_READINESS_POLL_INTERVAL, + )), + shutdown, ) - .await - { + .await; + let service_result = match service_result { + ReconciliationOutcome::Completed(result) => result, + ReconciliationOutcome::Cancelled => { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } + }; + if let Err(error) = service_result { record_runtime_readiness_diagnostics(paths, spec, &mut process, &error); if *failure_policy == ReadinessFailurePolicy::PreserveRuntime { let process_exited = match process.has_exited() { @@ -3505,8 +4561,18 @@ async fn finish_runtime_transaction( if !process_exited { match supervisor.verify_ownership(spec) { Ok(Some(_runtime)) => { + if reconciliation_cancelled(shutdown) { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } record_applied_runtime_config(supervisor, spec, desired_fingerprint)?; - return Ok(RuntimeReadinessOutcome::Unverified); + if reconciliation_cancelled(shutdown) { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } + return Ok(ReconciliationOutcome::Completed( + RuntimeReadinessOutcome::Unverified, + )); } Ok(None) => {} Err(error) => { @@ -3560,9 +4626,56 @@ async fn finish_runtime_transaction( ) .await); } + if reconciliation_cancelled(shutdown) { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } record_applied_runtime_config(supervisor, spec, desired_fingerprint)?; + if reconciliation_cancelled(shutdown) { + cancel_fresh_runtime_transaction(spec, process).await?; + return Ok(ReconciliationOutcome::Cancelled); + } + + Ok(ReconciliationOutcome::Completed( + RuntimeReadinessOutcome::Verified, + )) +} + +fn reconciliation_cancelled(shutdown: Option<&watch::Receiver>) -> bool { + shutdown.is_some_and(|shutdown| *shutdown.borrow()) +} - Ok(RuntimeReadinessOutcome::Verified) +async fn wait_for_transaction_step( + mut step: Pin + Send + '_>>, + shutdown: Option<&watch::Receiver>, +) -> ReconciliationOutcome { + tokio::select! { + biased; + output = step.as_mut() => ReconciliationOutcome::Completed(output), + () = wait_for_reconciliation_cancellation(shutdown) => ReconciliationOutcome::Cancelled, + } +} + +async fn wait_for_reconciliation_cancellation(shutdown: Option<&watch::Receiver>) { + let Some(shutdown) = shutdown else { + std::future::pending::<()>().await; + return; + }; + let mut shutdown = shutdown.clone(); + if shutdown.wait_for(|requested| *requested).await.is_err() { + std::future::pending::<()>().await; + } +} + +async fn cancel_fresh_runtime_transaction( + spec: &ProcessSpec, + process: ManagedProcess, +) -> Result<(), RuntimeTransactionError> { + process + .stop(Duration::from_secs(1)) + .await + .map_err(RuntimeTransactionError::pending_preserve)?; + cleanup_fresh_runtime_files(spec).map_err(RuntimeTransactionError::new) } fn record_applied_runtime_config( @@ -3713,19 +4826,27 @@ async fn restore_runtime_after_failed_load( previous_fingerprint: &str, original_error: DaemonError, ) -> DaemonError { - if let Err(error) = verify_runtime_ownership(supervisor, spec) { - return compound_runtime_restore_error(original_error, error); + match restore_runtime_after_load(paths, supervisor, spec, readiness, previous_fingerprint).await + { + Ok(()) => original_error, + Err(error) => compound_runtime_restore_error(original_error, error), } +} - let restored_content = match read_config_bytes(&spec.config_path) { - Ok(content) => content, - Err(error) => return compound_runtime_restore_error(original_error, error), - }; +async fn restore_runtime_after_load( + paths: &PvPaths, + supervisor: &ProcessSupervisor, + spec: &ProcessSpec, + readiness: &RuntimeReadinessPlan, + previous_fingerprint: &str, +) -> Result<(), DaemonError> { + verify_runtime_ownership(supervisor, spec)?; + + let restored_content = read_config_bytes(&spec.config_path)?; let client = CaddyAdminClient::new().with_timeout(readiness.timeout); - if let Err(error) = mark_runtime_config_pending(supervisor, spec, previous_fingerprint, true) { - return compound_runtime_restore_error(original_error, *error.error); - } - if let Err(error) = load_runtime_config( + mark_runtime_config_pending(supervisor, spec, previous_fingerprint, true) + .map_err(|error| *error.error)?; + load_runtime_config( paths, spec, client, @@ -3733,41 +4854,29 @@ async fn restore_runtime_after_failed_load( restored_content, ) .await - { - return compound_runtime_restore_error(original_error, *error.error); - } - if let Err(error) = verify_runtime_ownership(supervisor, spec) { - return compound_runtime_restore_error(original_error, error); - } - if let Err(error) = client + .map_err(|error| *error.error)?; + verify_runtime_ownership(supervisor, spec)?; + client .wait_until_ready_with( &readiness.admin_endpoint, readiness.timeout, runtime_ownership_verifier(paths, spec), ) .await - { - return compound_runtime_restore_error(original_error, error.into()); - } - if let Err(error) = wait_for_owned_readiness(readiness.check.clone(), readiness.timeout, || { + .map_err(DaemonError::from)?; + wait_for_owned_readiness(readiness.check.clone(), readiness.timeout, || { verify_runtime_ownership(supervisor, spec) }) - .await - { - return compound_runtime_restore_error(original_error, error); - } + .await?; match supervisor.record_restored_config(spec, previous_fingerprint) { Ok(true) => {} Ok(false) => { - return compound_runtime_restore_error( - original_error, - CaddyAdminError::runtime_ownership_changed(spec.name.clone()).into(), - ); + return Err(CaddyAdminError::runtime_ownership_changed(spec.name.clone()).into()); } - Err(error) => return compound_runtime_restore_error(original_error, error), + Err(error) => return Err(error), } - original_error + Ok(()) } fn verify_runtime_ownership( @@ -4634,6 +5743,8 @@ fn worker_config_private_environment( #[cfg(test)] mod tests { use std::collections::BTreeSet; + #[cfg(target_os = "macos")] + use std::process::Stdio; use std::sync::mpsc; use std::time::Duration; @@ -4642,18 +5753,27 @@ mod tests { use camino_tempfile::tempdir; use platform::{ActivePfRedirectInspection, PfRedirectConfig}; use state::{Database, LinkProjectInput, PvPaths}; + use tokio::sync::watch; + #[cfg(target_os = "macos")] + use tokio::time::timeout; use crate::gateway_config::GatewayProjectRoute; use crate::{DaemonError, ReadinessCheck}; use super::{ - GatewayPfRoutingState, GatewayReadinessPorts, GatewayRuntimePlan, ReadinessFailurePolicy, - RuntimePlan, build_target_runtime_plan, classify_gateway_pf_routing_state, + CaddyCliCommand, GatewayPfRoutingState, GatewayReadinessPorts, GatewayRuntimePlan, + ReadinessFailurePolicy, ReconciliationOutcome, RuntimePlan, build_target_runtime_plan, + cancel_or_preserve_runtime_reconciliation_errors, classify_gateway_pf_routing_state, combined_runtime_reconciliation_error, gateway_project_config_fragments, gateway_public_readiness_check, gateway_readiness_check_for_ports, gateway_readiness_hostname, gateway_readiness_plan, gateway_readiness_ports, - previous_runtime_readiness_from_parts, project_config_file_name, - spawn_gateway_pf_inspection, + previous_runtime_readiness_from_parts, project_config_file_name, run_validation_command, + spawn_gateway_pf_inspection, wait_for_transaction_step, + }; + #[cfg(target_os = "macos")] + use super::{ + RuntimeProcessCommand, spawn_validation_process_group_anchor, + start_validation_process_reaper, wait_for_validation_process_group_exit, }; #[test] @@ -4693,6 +5813,100 @@ mod tests { Ok(()) } + #[test] + fn runtime_reconciliation_errors_outrank_cancellation() -> Result<()> { + assert!(matches!( + cancel_or_preserve_runtime_reconciliation_errors::<()>(Vec::new()), + Ok(super::ReconciliationOutcome::Cancelled) + )); + let error = cancel_or_preserve_runtime_reconciliation_errors::<()>(vec![( + "gateway".to_owned(), + DaemonError::UnexpectedProtocolResponse { + reason: "sentinel failure".to_owned(), + }, + )]) + .err() + .ok_or_else(|| anyhow::anyhow!("cancellation replaced the sentinel failure"))?; + assert_eq!(error.to_string(), "daemon protocol error: sentinel failure"); + + Ok(()) + } + + #[tokio::test] + async fn completed_transaction_step_outranks_ready_cancellation() -> Result<()> { + let (_fallback_sender, fallback_receiver) = watch::channel(true); + let outcome = wait_for_transaction_step( + Box::pin(async { + Err::<(), DaemonError>(DaemonError::UnexpectedProtocolResponse { + reason: "transaction sentinel".to_owned(), + }) + }), + Some(&fallback_receiver), + ) + .await; + let ReconciliationOutcome::Completed(Err(error)) = outcome else { + anyhow::bail!("ready cancellation replaced a completed transaction error"); + }; + assert_eq!( + error.to_string(), + "daemon protocol error: transaction sentinel" + ); + + Ok(()) + } + + #[tokio::test] + async fn pre_signalled_fallback_does_not_spawn_config_validator() -> Result<()> { + let tempdir = tempdir()?; + let (_fallback_sender, fallback_receiver) = watch::channel(true); + + let outcome = run_validation_command( + &CaddyCliCommand::caddy(tempdir.path().join("missing-validator")), + &tempdir.path().join("Caddyfile"), + &Default::default(), + Some(&fallback_receiver), + None, + ) + .await?; + + assert!(matches!(outcome, ReconciliationOutcome::Cancelled)); + + Ok(()) + } + + #[cfg(target_os = "macos")] + #[tokio::test] + async fn validation_group_anchor_stops_group_when_owner_pipe_closes() -> Result<()> { + let fallback_reaper = start_validation_process_reaper()?; + let (group_pid, group_anchor) = spawn_validation_process_group_anchor(&fallback_reaper)?; + let Some(group_pid) = group_pid else { + anyhow::bail!("validation process-group anchor did not publish its group id"); + }; + let Some(mut group_anchor) = group_anchor else { + anyhow::bail!("validation process-group anchor was missing"); + }; + + let mut member_command = RuntimeProcessCommand::new("/bin/sleep"); + member_command + .arg("30") + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .kill_on_drop(true) + .process_group(group_pid); + let mut member = member_command.spawn()?; + let Some(owner_pipe) = group_anchor.stdin.take() else { + anyhow::bail!("validation process-group anchor owner pipe was missing"); + }; + drop(owner_pipe); + + timeout(Duration::from_secs(5), member.wait()).await??; + timeout(Duration::from_secs(5), group_anchor.wait()).await??; + wait_for_validation_process_group_exit(Some(group_pid)).await?; + + Ok(()) + } + #[tokio::test(flavor = "current_thread")] async fn pf_inspection_does_not_block_the_async_executor() -> Result<()> { let (started_sender, started_receiver) = tokio::sync::oneshot::channel(); diff --git a/crates/daemon/src/gateway_config.rs b/crates/daemon/src/gateway_config.rs index 6a87e6bb..d71bc0a9 100644 --- a/crates/daemon/src/gateway_config.rs +++ b/crates/daemon/src/gateway_config.rs @@ -209,6 +209,7 @@ where Promote: FnOnce() -> Result, { let candidate_path = candidate_path_for(path); + let mut candidate = CandidateConfigGuard::new(candidate_path.clone()); let previous_root_content = match fs::read_to_string(path) { Ok(content) => Some(content), Err(StateError::Filesystem { source, .. }) @@ -221,22 +222,23 @@ where } Err(error) => return Err(error.into()), }; - write_candidate_config(&candidate_path, candidate_content)?; + if let Err(error) = write_candidate_config(&candidate_path, candidate_content) { + return Err(candidate.preserve_error(error)); + } if let Err(error) = validate(candidate_path.clone()).await { - let _cleanup_result = remove_candidate_config(&candidate_path); - - return Err(error); + return Err(candidate.preserve_error(error)); } - write_candidate_config(&candidate_path, active_content)?; + if let Err(error) = write_candidate_config(&candidate_path, active_content) { + return Err(candidate.preserve_error(error)); + } let root = match promote_config_file(path, &candidate_path) { - Ok(root) => root, - Err(error) => { - let _cleanup_result = remove_candidate_config(&candidate_path); - - return Err(error); + Ok(root) => { + candidate.disarm(); + root } + Err(error) => return Err(candidate.preserve_error(error)), }; let fragments = match promote_fragments() { Ok(fragments) => fragments, @@ -428,15 +430,19 @@ pub(crate) fn promote_validated_config( validate: impl FnOnce(&Utf8Path) -> Result<(), DaemonError>, ) -> Result<(), DaemonError> { let candidate_path = candidate_path_for(path); - write_candidate_config(&candidate_path, content)?; + let mut candidate = CandidateConfigGuard::new(candidate_path.clone()); + if let Err(error) = write_candidate_config(&candidate_path, content) { + return Err(candidate.preserve_error(error)); + } if let Err(error) = validate(&candidate_path) { - let _cleanup_result = remove_candidate_config(&candidate_path); - - return Err(error); + return Err(candidate.preserve_error(error)); } - rename_candidate_config(&candidate_path, path)?; + if let Err(error) = rename_candidate_config(&candidate_path, path) { + return Err(candidate.preserve_error(error)); + } + candidate.disarm(); Ok(()) } @@ -521,6 +527,43 @@ fn candidate_path_for(path: &Utf8Path) -> Utf8PathBuf { path.with_file_name(format!("{file_name}.candidate.{process_id}.{counter}.tmp")) } +struct CandidateConfigGuard { + path: Utf8PathBuf, + armed: bool, +} + +impl CandidateConfigGuard { + fn new(path: Utf8PathBuf) -> Self { + Self { path, armed: true } + } + + fn disarm(&mut self) { + self.armed = false; + } + + fn preserve_error(mut self, error: DaemonError) -> DaemonError { + match delete_optional_config(&self.path) { + Ok(()) => { + self.disarm(); + error + } + Err(cleanup) => DaemonError::RuntimeCleanupFailed { + runtime: format!("candidate config `{}`", self.path), + source: Box::new(error), + cleanup: Box::new(cleanup), + }, + } + } +} + +impl Drop for CandidateConfigGuard { + fn drop(&mut self) { + if self.armed { + let _cleanup_result = delete_optional_config(&self.path); + } + } +} + fn backup_path_for(path: &Utf8Path) -> Utf8PathBuf { let file_name = path.file_name().unwrap_or("config"); let process_id = std::process::id(); @@ -619,12 +662,6 @@ fn rename_candidate_config(from: &Utf8Path, to: &Utf8Path) -> Result<(), DaemonE Ok(()) } -fn remove_candidate_config(path: &Utf8Path) -> Result<(), DaemonError> { - fs::delete_file(path)?; - - Ok(()) -} - fn delete_optional_config(path: &Utf8Path) -> Result<(), DaemonError> { match fs::delete_file(path) { Ok(()) => Ok(()), diff --git a/crates/daemon/src/health.rs b/crates/daemon/src/health.rs index f440f8d0..c38f8f8b 100644 --- a/crates/daemon/src/health.rs +++ b/crates/daemon/src/health.rs @@ -1593,7 +1593,7 @@ mod tests { ) -> anyhow::Result { let artifact_root = root.join(format!("frankenphp-{php_track}")); let runtime = artifact_root.join("bin/frankenphp"); - state::fs::copy_file_atomically(Utf8Path::new("/bin/sleep"), &runtime)?; + state::fs::write_sensitive_file(&runtime, "#!/bin/sh\nsleep \"$1\"\n")?; set_executable(&runtime)?; database.record_managed_resource_track_installed( "frankenphp", diff --git a/crates/daemon/src/jobs.rs b/crates/daemon/src/jobs.rs index 48b5c1ba..c6659b26 100644 --- a/crates/daemon/src/jobs.rs +++ b/crates/daemon/src/jobs.rs @@ -6,8 +6,10 @@ use std::sync::{Arc, Mutex, OnceLock}; use crate::DaemonError; use crate::gateway::{ CADDY_NOT_INSTALLED, GatewayPfRoutingState, ProjectGatewayReconciliationOutcome, - reconcile_gateway_runtimes, reconcile_gateway_runtimes_with_phase_log, + ReconciliationOutcome, reconcile_gateway_runtimes, reconcile_gateway_runtimes_with_phase_log, + reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown, reconcile_project_gateway_runtimes_with_phase_log, + reconcile_project_gateway_runtimes_with_phase_log_and_fallback_shutdown, }; use crate::ipc::LocalStream; use crate::managed_resources::{ @@ -16,9 +18,12 @@ use crate::managed_resources::{ reconcile_system_resources_with_catalog_and_progress, reconcile_system_resources_with_progress, stop_undemanded_system_resource_runtimes, verify_system_resource_installations, }; +#[cfg(test)] +use crate::project_env::reconcile_project_env_with_runtime_catalog_and_progress; use crate::project_env::{ - DemandedResourceTrack, ProjectApplyStage, ProjectDemand, discover_project_demand, - reconcile_project_env_with_runtime_catalog_and_progress, record_project_env_failure, + DemandedResourceTrack, ProjectApplyOptions, ProjectApplyStage, ProjectDemand, + discover_project_demand, reconcile_project_env_with_runtime_catalog_and_progress_outcome, + record_project_env_failure, }; use crate::reconciliation::{ EnqueueResult, QueuedReconciliation, ReconciliationJobTiming, ReconciliationQueue, @@ -78,8 +83,8 @@ impl FailedUpdateJob { } #[derive(Debug)] -struct StreamedJobCompletion { - result: Result, +struct StreamedJobCompletion { + result: Output, transport_is_open: bool, } @@ -326,12 +331,21 @@ pub(crate) async fn run_job( kind: &str, scope: &str, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + fallback_shutdown: &watch::Receiver, ) -> Result<(), DaemonError> { let parsed_scope = scope.parse::(); if kind == "reconcile" { return match parsed_scope { Ok(parsed_scope) => { - run_reconciliation_job(paths, queue, transport, parsed_scope, runtime_catalog).await + run_reconciliation_job( + paths, + queue, + transport, + parsed_scope, + runtime_catalog, + fallback_shutdown, + ) + .await } Err(error) => { run_invalid_reconciliation_scope_job(paths, transport, scope, error).await @@ -339,7 +353,7 @@ pub(crate) async fn run_job( }; } if kind == "update" && scope == "system" { - return run_update_job(paths, queue, transport, runtime_catalog).await; + return run_update_job(paths, queue, transport, runtime_catalog, fallback_shutdown).await; } run_started_job(paths, transport, kind, scope).await @@ -389,7 +403,7 @@ pub(crate) async fn run_background_reconciliation_job_with_origin( return Ok(()); }; - complete_queued_background_reconciliation_job(&paths, queued, runtime_catalog).await + complete_queued_background_reconciliation_job(&paths, queued, runtime_catalog, None).await } pub(crate) fn enqueue_background_reconciliation_job( @@ -418,6 +432,7 @@ pub(crate) async fn run_startup_reconciliation_job( queue: ReconciliationQueue, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, mut shutdown: oneshot::Receiver<()>, + fallback_shutdown: watch::Receiver, ) -> Result<(), BackgroundReconciliationError> { let result = loop { let enqueue_paths = paths.clone(); @@ -475,6 +490,7 @@ pub(crate) async fn run_startup_reconciliation_job( running, runtime_catalog, Some(&shutdown), + Some(&fallback_shutdown), ) .await } @@ -490,14 +506,42 @@ async fn wait_for_startup_reconciliation_turn( } } +async fn wait_for_fallback_shutdown(shutdown: &mut watch::Receiver) { + if shutdown.wait_for(|requested| *requested).await.is_err() { + std::future::pending::<()>().await; + } +} + +fn fallback_shutdown_requested(shutdown: Option<&watch::Receiver>) -> bool { + shutdown.is_some_and(|shutdown| *shutdown.borrow()) +} + pub(crate) async fn complete_queued_background_reconciliation_job( paths: &PvPaths, queued: QueuedReconciliation, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + fallback_shutdown: Option<&watch::Receiver>, ) -> Result<(), BackgroundReconciliationError> { - let running = queued.wait_for_turn().await; + let running = match fallback_shutdown { + Some(fallback_shutdown) => { + let mut fallback_shutdown = fallback_shutdown.clone(); + tokio::select! { + biased; + _ = wait_for_fallback_shutdown(&mut fallback_shutdown) => return Ok(()), + running = queued.wait_for_turn() => running, + } + } + None => queued.wait_for_turn().await, + }; - complete_running_background_reconciliation_job(paths, running, runtime_catalog, None).await + complete_running_background_reconciliation_job( + paths, + running, + runtime_catalog, + None, + fallback_shutdown, + ) + .await } pub(crate) async fn complete_running_background_reconciliation_job( @@ -505,6 +549,7 @@ pub(crate) async fn complete_running_background_reconciliation_job( running: RunningReconciliation, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, shutdown: Option<&oneshot::Receiver<()>>, + fallback_shutdown: Option<&watch::Receiver>, ) -> Result<(), BackgroundReconciliationError> { let job_id = running.job_id().to_string(); let scope = running.scope().clone(); @@ -517,24 +562,33 @@ pub(crate) async fn complete_running_background_reconciliation_job( running.timing(), ReconciliationJobOptions { discard_obsolete_project: true, + fallback_shutdown, ..ReconciliationJobOptions::default() }, shutdown, ) .await; - running.finish(); - match completion { - ReconciliationJobCompletion::Succeeded(_summary) => Ok(()), + ReconciliationJobCompletion::Succeeded(_summary) => { + running.finish(); + Ok(()) + } + ReconciliationJobCompletion::Cancelled => { + drop(running); + Ok(()) + } ReconciliationJobCompletion::Failed { error, recording_error, - } => Err(BackgroundReconciliationError::Execution { - job_id, - error, - recording_error, - }), + } => { + running.finish(); + Err(BackgroundReconciliationError::Execution { + job_id, + error, + recording_error, + }) + } } } @@ -544,6 +598,7 @@ async fn run_reconciliation_job( mut transport: DaemonTransport, scope: ReconciliationScope, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + fallback_shutdown: &watch::Receiver, ) -> Result<(), DaemonError> { let result = match enqueue_foreground_reconciliation_job(&paths, &queue, scope) { Ok(result) => result, @@ -567,26 +622,44 @@ async fn run_reconciliation_job( .await .map_err(DaemonError::from); let stream_is_open = accepted_result.is_ok(); - let (running, stream_is_open) = wait_for_foreground_turn( + let Some((running, stream_is_open)) = wait_for_foreground_turn( queued, &mut transport, stream_is_open, FOREGROUND_JOB_HEARTBEAT_INTERVAL, + Some(fallback_shutdown), ) - .await; + .await + else { + return accepted_result; + }; let scope = running.scope().clone(); - let result = stream_started_reconciliation_job( + let result = stream_started_reconciliation_job_with_fallback( paths, transport, stream_is_open, running.job_id(), scope, runtime_catalog, - running.timing(), + ForegroundReconciliationOptions { + timing: running.timing(), + fallback_shutdown: Some(fallback_shutdown), + }, ) .await; - running.finish(); + let result = match result { + ForegroundReconciliationCompletion::Finalized(result) => { + running.finish(); + result + } + ForegroundReconciliationCompletion::Cancelled => { + drop(running); + ReconciliationJobCompletion::Cancelled + .into_result() + .map(|_summary| ()) + } + }; foreground_reconciliation_result(accepted_result, result) } @@ -670,6 +743,7 @@ async fn run_update_job( queue: ReconciliationQueue, mut transport: DaemonTransport, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + fallback_shutdown: &watch::Receiver, ) -> Result<(), DaemonError> { let result = match enqueue_update_job(&paths, &queue) { Ok(result) => result, @@ -691,13 +765,17 @@ async fn run_update_job( .await .map_err(DaemonError::from); let stream_is_open = accepted_result.is_ok(); - let (running, stream_is_open) = wait_for_foreground_turn( + let Some((running, stream_is_open)) = wait_for_foreground_turn( queued, &mut transport, stream_is_open, FOREGROUND_JOB_HEARTBEAT_INTERVAL, + Some(fallback_shutdown), ) - .await; + .await + else { + return accepted_result; + }; let result = stream_started_update_job( paths, transport, @@ -705,10 +783,22 @@ async fn run_update_job( running.job_id(), runtime_catalog, running.timing(), + Some(fallback_shutdown), ) .await; - running.finish(); + let result = match result { + ForegroundReconciliationCompletion::Finalized(result) => { + running.finish(); + result + } + ForegroundReconciliationCompletion::Cancelled => { + drop(running); + ReconciliationJobCompletion::Cancelled + .into_result() + .map(|_summary| ()) + } + }; foreground_reconciliation_result(accepted_result, result) } @@ -757,7 +847,8 @@ async fn stream_started_update_job( job_id: &str, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, timing: ReconciliationJobTiming, -) -> Result<(), DaemonError> + fallback_shutdown: Option<&watch::Receiver>, +) -> ForegroundReconciliationCompletion where Stream: AsyncWrite + Unpin, { @@ -792,12 +883,19 @@ where let (event_sender, event_receiver) = channel(FOREGROUND_JOB_PROGRESS_BUFFER); let (phase_sender, phase_receiver) = watch::channel(Vec::new()); let progress = DaemonDownloadProgress::new(event_sender, phase_sender); - let completion = complete_streamed_job_with_heartbeat_and_events( + let completion = complete_streamed_output_with_heartbeat_and_events( &mut transport, job_id, "Managed Resource update still running", FOREGROUND_JOB_HEARTBEAT_INTERVAL, - complete_update_job_with_progress(&paths, job_id, runtime_catalog, progress, timing), + complete_update_job_with_progress( + &paths, + job_id, + runtime_catalog, + progress, + timing, + fallback_shutdown, + ), event_receiver, phase_receiver, ) @@ -812,42 +910,55 @@ where runtime_catalog, DaemonDownloadProgress::disabled(), timing, + fallback_shutdown, ) .await, false, ) }; - started_stream_result?; + if matches!(update_result, ReconciliationJobCompletion::Cancelled) { + return ForegroundReconciliationCompletion::Cancelled; + } + let update_result = update_result.into_result(); + if let Err(error) = started_stream_result { + return ForegroundReconciliationCompletion::Finalized(Err(error)); + } if !stream_is_open || !transport_is_open { - return update_result.map(|_summary| ()); + return ForegroundReconciliationCompletion::Finalized(update_result.map(|_summary| ())); } match update_result { Ok(summary) => { - write_foreground_terminal_event( + let result = write_foreground_terminal_event( &mut transport, &DaemonEvent::JobCompleted { job_id, summary: &summary, }, ) - .await?; + .await; + if let Err(error) = result { + return ForegroundReconciliationCompletion::Finalized(Err(error)); + } } Err(error) => { let error_message = error.to_string(); - write_foreground_terminal_event( + let result = write_foreground_terminal_event( &mut transport, &DaemonEvent::JobFailed { job_id, error: &error_message, }, ) - .await?; + .await; + if let Err(error) = result { + return ForegroundReconciliationCompletion::Finalized(Err(error)); + } } } - Ok(()) + ForegroundReconciliationCompletion::Finalized(Ok(())) } fn foreground_reconciliation_result( @@ -863,20 +974,29 @@ async fn wait_for_foreground_turn( transport: &mut DaemonTransport, mut stream_is_open: bool, heartbeat_interval: Duration, -) -> (RunningReconciliation, bool) + fallback_shutdown: Option<&watch::Receiver>, +) -> Option<(RunningReconciliation, bool)> where Stream: AsyncWrite + Unpin, { let job_id = queued.job_id().to_string(); let wait_for_turn = queued.wait_for_turn(); tokio::pin!(wait_for_turn); + let mut fallback_shutdown = fallback_shutdown.cloned(); let mut heartbeat = interval_at(Instant::now() + heartbeat_interval, heartbeat_interval); heartbeat.set_missed_tick_behavior(MissedTickBehavior::Delay); loop { tokio::select! { biased; - running = &mut wait_for_turn => return (running, stream_is_open), + _ = async { + if let Some(fallback_shutdown) = fallback_shutdown.as_mut() { + wait_for_fallback_shutdown(fallback_shutdown).await; + } else { + std::future::pending::<()>().await; + } + } => return None, + running = &mut wait_for_turn => return Some((running, stream_is_open)), _ = heartbeat.tick(), if stream_is_open => { let event = DaemonEvent::Log { job_id: &job_id, @@ -897,9 +1017,10 @@ where } } +#[cfg(test)] async fn stream_started_reconciliation_job( paths: PvPaths, - mut transport: DaemonTransport, + transport: DaemonTransport, stream_is_open: bool, job_id: &str, scope: ReconciliationScope, @@ -909,6 +1030,60 @@ async fn stream_started_reconciliation_job( where Stream: AsyncWrite + Unpin, { + stream_started_reconciliation_job_with_fallback( + paths, + transport, + stream_is_open, + job_id, + scope, + runtime_catalog, + ForegroundReconciliationOptions { + timing, + fallback_shutdown: None, + }, + ) + .await + .into_result() +} + +struct ForegroundReconciliationOptions<'a> { + timing: ReconciliationJobTiming, + fallback_shutdown: Option<&'a watch::Receiver>, +} + +enum ForegroundReconciliationCompletion { + Finalized(Result<(), DaemonError>), + Cancelled, +} + +impl ForegroundReconciliationCompletion { + #[cfg(test)] + fn into_result(self) -> Result<(), DaemonError> { + match self { + Self::Finalized(result) => result, + Self::Cancelled => ReconciliationJobCompletion::Cancelled + .into_result() + .map(|_summary| ()), + } + } +} + +async fn stream_started_reconciliation_job_with_fallback( + paths: PvPaths, + mut transport: DaemonTransport, + stream_is_open: bool, + job_id: &str, + scope: ReconciliationScope, + runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + options: ForegroundReconciliationOptions<'_>, +) -> ForegroundReconciliationCompletion +where + Stream: AsyncWrite + Unpin, +{ + let ForegroundReconciliationOptions { + timing, + fallback_shutdown, + } = options; let scope_text = scope.to_string(); let started_stream_result = if stream_is_open { async { @@ -936,18 +1111,22 @@ where let (event_sender, event_receiver) = channel(FOREGROUND_JOB_PROGRESS_BUFFER); let (phase_sender, phase_receiver) = watch::channel(Vec::new()); let progress = DaemonDownloadProgress::new(event_sender, phase_sender); - let completion = complete_streamed_job_with_heartbeat_and_events( + let completion = complete_streamed_output_with_heartbeat_and_events( &mut transport, job_id, "Reconciliation still running", FOREGROUND_JOB_HEARTBEAT_INTERVAL, - complete_reconciliation_job_with_progress( + complete_reconciliation_job_with_progress_outcome( &paths, job_id, &scope, runtime_catalog, progress, timing, + ReconciliationJobOptions { + fallback_shutdown, + ..ReconciliationJobOptions::default() + }, None, ), event_receiver, @@ -958,18 +1137,41 @@ where (completion.result, completion.transport_is_open) } else { ( - complete_reconciliation_job(&paths, job_id, &scope, runtime_catalog, timing, None) - .await, + complete_reconciliation_job_with_progress_outcome( + &paths, + job_id, + &scope, + runtime_catalog, + DaemonDownloadProgress::disabled(), + timing, + ReconciliationJobOptions { + fallback_shutdown, + ..ReconciliationJobOptions::default() + }, + None, + ) + .await, false, ) }; - started_stream_result?; + if matches!( + reconciliation_result, + ReconciliationJobCompletion::Cancelled + ) { + return ForegroundReconciliationCompletion::Cancelled; + } + let reconciliation_result = reconciliation_result.into_result(); + if let Err(error) = started_stream_result { + return ForegroundReconciliationCompletion::Finalized(Err(error)); + } if !stream_is_open || !transport_is_open { - return reconciliation_result.map(|_summary| ()); + return ForegroundReconciliationCompletion::Finalized( + reconciliation_result.map(|_summary| ()), + ); } - match reconciliation_result { + let result = match reconciliation_result { Ok(summary) => { write_foreground_terminal_event( &mut transport, @@ -978,7 +1180,7 @@ where summary: &summary, }, ) - .await?; + .await } Err(error) => { let error_message = error.to_string(); @@ -989,11 +1191,11 @@ where error: &error_message, }, ) - .await?; + .await } - } + }; - Ok(()) + ForegroundReconciliationCompletion::Finalized(result) } #[cfg(test)] @@ -1030,7 +1232,33 @@ where } } +#[cfg(test)] async fn complete_streamed_job_with_heartbeat_and_events( + transport: &mut DaemonTransport, + job_id: &str, + heartbeat_message: &'static str, + heartbeat_interval: Duration, + completion: Completion, + events: Receiver, + phases: watch::Receiver>, +) -> StreamedJobCompletion> +where + Stream: AsyncWrite + Unpin, + Completion: Future>, +{ + complete_streamed_output_with_heartbeat_and_events( + transport, + job_id, + heartbeat_message, + heartbeat_interval, + completion, + events, + phases, + ) + .await +} + +async fn complete_streamed_output_with_heartbeat_and_events( transport: &mut DaemonTransport, job_id: &str, heartbeat_message: &'static str, @@ -1038,10 +1266,10 @@ async fn complete_streamed_job_with_heartbeat_and_events( completion: Completion, mut events: Receiver, mut phases: watch::Receiver>, -) -> StreamedJobCompletion +) -> StreamedJobCompletion where Stream: AsyncWrite + Unpin, - Completion: Future>, + Completion: Future, { let mut heartbeat = interval_at(Instant::now() + heartbeat_interval, heartbeat_interval); heartbeat.set_missed_tick_behavior(MissedTickBehavior::Delay); @@ -1130,13 +1358,13 @@ where } } -async fn finish_streamed_job( +async fn finish_streamed_job( transport: &mut DaemonTransport, job_id: &str, phases: &mut watch::Receiver>, next_phase: &mut usize, - result: Result, -) -> StreamedJobCompletion + result: Output, +) -> StreamedJobCompletion where Stream: AsyncWrite + Unpin, { @@ -1306,8 +1534,10 @@ async fn complete_update_job( runtime_catalog, DaemonDownloadProgress::disabled(), ReconciliationJobTiming::immediate(), + None, ) .await + .into_result() } async fn complete_update_job_with_progress( @@ -1316,7 +1546,8 @@ async fn complete_update_job_with_progress( runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, progress: DaemonDownloadProgress, timing: ReconciliationJobTiming, -) -> Result { + fallback_shutdown: Option<&watch::Receiver>, +) -> ReconciliationJobCompletion { let phase_log = ReconciliationPhaseLog::new(paths, job_id, "update", "system") .with_progress(progress.phase_sender.clone()); phase_log.completed( @@ -1327,29 +1558,92 @@ async fn complete_update_job_with_progress( &[], ); let progress = progress.with_phase_log(phase_log.clone()); - let result = complete_update_job_inner(paths, runtime_catalog, progress, &phase_log).await; + let result = complete_update_job_inner( + paths, + runtime_catalog, + progress, + &phase_log, + fallback_shutdown, + ) + .await; + let result = match result { + Ok(ReconciliationOutcome::Completed(completed)) => Ok(completed), + Ok(ReconciliationOutcome::Cancelled) => return ReconciliationJobCompletion::Cancelled, + Err(error) => Err(error), + }; let finalization_timer = phase_log.start(ReconciliationPhase::Finalization, "job"); - match &result { + let completion = match result { Ok(completed) => { - let mut database = Database::open(paths)?; let mut coverage = vec![JobDiagnosticSubject::UpdateAssessment]; coverage.extend(completed.coverage.iter().cloned()); - database.complete_job_with_coverage(job_id, &completed.summary, &coverage)?; - structured_log::job_completed(paths, job_id, "update", "system", &completed.summary); + let recording_result = + Database::open(paths) + .map_err(DaemonError::from) + .and_then(|mut database| { + database + .complete_job_with_coverage(job_id, &completed.summary, &coverage) + .map_err(DaemonError::from) + }); + match recording_result { + Ok(()) => { + structured_log::job_completed( + paths, + job_id, + "update", + "system", + &completed.summary, + ); + ReconciliationJobCompletion::Succeeded(completed.summary) + } + Err(error) => { + structured_log::job_completion_recording_failed( + paths, + job_id, + "update", + "system", + &completed.summary, + &error.to_string(), + ); + ReconciliationJobCompletion::Failed { + error: Box::new(error), + recording_error: None, + } + } + } } - Err(error) => { - let error_message = error.error.to_string(); - let mut database = Database::open(paths)?; - database.fail_job_with_subject(job_id, &error_message, &error.subject)?; - structured_log::job_failed(paths, job_id, "update", "system", &error_message); + Err(failure) => { + let error_message = failure.error.to_string(); + let recording_error = Database::open(paths) + .map_err(DaemonError::from) + .and_then(|mut database| { + database + .fail_job_with_subject(job_id, &error_message, &failure.subject) + .map_err(DaemonError::from) + }) + .err() + .map(Box::new); + if let Some(recording_error) = &recording_error { + structured_log::job_failure_recording_failed( + paths, + job_id, + "update", + "system", + &error_message, + &recording_error.to_string(), + ); + } else { + structured_log::job_failed(paths, job_id, "update", "system", &error_message); + } + ReconciliationJobCompletion::Failed { + error: failure.error, + recording_error, + } } - } - finalization_timer.finish(PhaseOutcome::from_succeeded(result.is_ok()), &[]); + }; + finalization_timer.finish(PhaseOutcome::from_succeeded(completion.is_succeeded()), &[]); - result - .map(|completed| completed.summary) - .map_err(|failure| *failure.error) + completion } async fn complete_update_job_inner( @@ -1357,7 +1651,11 @@ async fn complete_update_job_inner( runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, progress: DaemonDownloadProgress, phase_log: &ReconciliationPhaseLog, -) -> Result { + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, FailedUpdateJob> { + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let report = if runtime_catalog.is_none() { let update_paths = paths.clone(); let update_progress = progress.clone(); @@ -1384,69 +1682,70 @@ async fn complete_update_job_inner( let report = match report.into_result() { Ok(report) => report, Err(update_error) => { + if fallback_shutdown_requested(fallback_shutdown) { + return Err(FailedUpdateJob::new( + update_error, + JobDiagnosticSubject::UpdateAssessment, + )); + } return Err(reconcile_partial_update_failure( paths, runtime_catalog, progress, phase_log, update_error, + fallback_shutdown, ) .await); } }; if report.updated_count == 0 { - return Ok(CompletedUpdateJob { + return Ok(ReconciliationOutcome::Completed(CompletedUpdateJob { summary: unchanged_update_summary(&report), coverage: Vec::new(), - }); + })); + } + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); } - let project_result = reconcile_system_projects_and_resources_with_progress( + let mut failure_subject = JobDiagnosticSubject::SystemReconciliation; + let reconciliation_result = complete_system_reconciliation_with_progress( paths, runtime_catalog, progress, phase_log, + None, + fallback_shutdown, + Some(&mut failure_subject), ) .await; - let gateway_result = reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await; - let (project_report, gateway_summary) = match (project_result, gateway_result) { - (Ok(project_report), Ok(gateway_summary)) => (project_report, gateway_summary), - (project_result, gateway_result) => { - let subject = if project_result.is_err() { - JobDiagnosticSubject::SystemReconciliation - } else { - JobDiagnosticSubject::GatewayRuntime - }; - let error = combined_system_reconciliation_error( - [project_result.err(), gateway_result.err()] - .into_iter() - .flatten() - .collect(), - ); - return Err(compensate_caddy_update_failure(paths, &report, error, subject).await); - } - }; - let reconciliation_summary = system_reconciliation_summary(&project_report, &gateway_summary); - let coverage = match completed_system_reconciliation_coverage(paths, &project_report) { - Ok(coverage) => coverage, + let completed = match reconciliation_result { + Ok(ReconciliationOutcome::Completed(completed)) => completed, + Ok(ReconciliationOutcome::Cancelled) => return Ok(ReconciliationOutcome::Cancelled), Err(error) => { return Err(compensate_caddy_update_failure( paths, &report, error, - JobDiagnosticSubject::SystemReconciliation, + failure_subject, + phase_log, + fallback_shutdown, ) .await); } }; let summary = format!( - "updated {} artifact(s); reconciled: {reconciliation_summary}", - report.updated_count + "updated {} artifact(s); reconciled: {}", + report.updated_count, completed.summary, ); - Ok(CompletedUpdateJob { summary, coverage }) + Ok(ReconciliationOutcome::Completed(CompletedUpdateJob { + summary, + coverage: completed.coverage, + })) } async fn compensate_caddy_update_failure( @@ -1454,21 +1753,52 @@ async fn compensate_caddy_update_failure( report: &ManagedResourceUpdateReport, original_error: DaemonError, subject: JobDiagnosticSubject, + phase_log: &ReconciliationPhaseLog, + fallback_shutdown: Option<&watch::Receiver>, ) -> FailedUpdateJob { match report.rollback_caddy(paths) { - Ok(true) => match reconcile_gateway_runtimes(paths).await { - Ok(_summary) => FailedUpdateJob::new( - original_error, - if subject == JobDiagnosticSubject::GatewayRuntime { - JobDiagnosticSubject::UpdateAssessment - } else { - subject - }, - ), - Err(recovery_error) => FailedUpdateJob::new( - caddy_compensation_error(original_error, recovery_error), - subject, - ), + Ok(true) if fallback_shutdown_requested(fallback_shutdown) => { + FailedUpdateJob::new(original_error, subject) + } + Ok(true) => match fallback_shutdown { + Some(fallback_shutdown) => { + match reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + phase_log, + fallback_shutdown, + ) + .await + { + Ok(ReconciliationOutcome::Completed(_) | ReconciliationOutcome::Cancelled) => { + FailedUpdateJob::new( + original_error, + if subject == JobDiagnosticSubject::GatewayRuntime { + JobDiagnosticSubject::UpdateAssessment + } else { + subject + }, + ) + } + Err(recovery_error) => FailedUpdateJob::new( + caddy_compensation_error(original_error, recovery_error), + subject, + ), + } + } + None => match reconcile_gateway_runtimes(paths).await { + Ok(_summary) => FailedUpdateJob::new( + original_error, + if subject == JobDiagnosticSubject::GatewayRuntime { + JobDiagnosticSubject::UpdateAssessment + } else { + subject + }, + ), + Err(recovery_error) => FailedUpdateJob::new( + caddy_compensation_error(original_error, recovery_error), + subject, + ), + }, }, Ok(false) => FailedUpdateJob::new(original_error, subject), Err(rollback_error) => FailedUpdateJob::new( @@ -1484,30 +1814,27 @@ async fn reconcile_partial_update_failure( progress: DaemonDownloadProgress, phase_log: &ReconciliationPhaseLog, update_error: DaemonError, + fallback_shutdown: Option<&watch::Receiver>, ) -> FailedUpdateJob { - let project_result = reconcile_system_projects_and_resources_with_progress( + let reconciliation_result = complete_system_reconciliation_with_progress( paths, runtime_catalog, progress, phase_log, + None, + fallback_shutdown, + None, ) .await; - let gateway_result = reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await; - let failures = [project_result.err(), gateway_result.err()] - .into_iter() - .flatten() - .collect::>(); - if !failures.is_empty() { - return FailedUpdateJob::new( - partial_update_reconciliation_error( - update_error, - combined_system_reconciliation_error(failures), - ), + match reconciliation_result { + Ok(ReconciliationOutcome::Completed(_) | ReconciliationOutcome::Cancelled) => { + FailedUpdateJob::new(update_error, JobDiagnosticSubject::UpdateAssessment) + } + Err(reconciliation_error) => FailedUpdateJob::new( + partial_update_reconciliation_error(update_error, reconciliation_error), JobDiagnosticSubject::UpdateAssessment, - ); + ), } - - FailedUpdateJob::new(update_error, JobDiagnosticSubject::UpdateAssessment) } fn partial_update_reconciliation_error( @@ -1671,6 +1998,7 @@ fn unready_established_resource_projects( Ok(failures) } +#[cfg(test)] async fn complete_managed_resource_reconciliation_with_progress( paths: &PvPaths, name: &crate::reconciliation::ReconciliationScopeComponent, @@ -1679,6 +2007,32 @@ async fn complete_managed_resource_reconciliation_with_progress( progress: DaemonDownloadProgress, phase_log: &ReconciliationPhaseLog, ) -> Result { + complete_managed_resource_reconciliation_with_progress_and_fallback_shutdown( + paths, + name, + track, + runtime_catalog, + progress, + phase_log, + None, + ) + .await? + .into_completed() +} + +async fn complete_managed_resource_reconciliation_with_progress_and_fallback_shutdown( + paths: &PvPaths, + name: &crate::reconciliation::ReconciliationScopeComponent, + track: &crate::reconciliation::ReconciliationScopeComponent, + runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + progress: DaemonDownloadProgress, + phase_log: &ReconciliationPhaseLog, + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } + let dependent_projects = Database::open(paths)? .projects_demanding_managed_resource_track(name.as_str(), track.as_str())?; let strict_failures = unready_established_resource_projects( @@ -1701,6 +2055,9 @@ async fn complete_managed_resource_reconciliation_with_progress( skip_projects.insert(project_id.clone()); } } + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_reconciliation_errors(strict_failures.into_values()); + } let record_timer = phase_log.start(ReconciliationPhase::ProjectApply, "linked_projects"); let record_result = reconcile_system_projects_with_progress( @@ -1712,10 +2069,16 @@ async fn complete_managed_resource_reconciliation_with_progress( ProjectApplyStage::RecordRequirements, &linked_projects(paths)?, &skip_projects, + fallback_shutdown, ) .await; finish_project_phase(record_timer, &record_result); - record_result?; + let record_failures = record_result?.failures; + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_reconciliation_errors( + strict_failures.into_values().chain(record_failures), + ); + } let resources_timer = phase_log.start( ReconciliationPhase::Resources, @@ -1728,6 +2091,7 @@ async fn complete_managed_resource_reconciliation_with_progress( runtime_catalog, &dependent_projects, progress.clone().suppressing_operation_phases(), + fallback_shutdown, ) .await; resources_timer.finish( @@ -1746,6 +2110,14 @@ async fn complete_managed_resource_reconciliation_with_progress( skip_projects.insert(project_id.clone()); } } + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_reconciliation_errors( + strict_failures + .into_values() + .chain(resource_failures.into_values()) + .chain(record_failures), + ); + } let project_timer = phase_log.start(ReconciliationPhase::ProjectApply, "linked_projects"); // The staged apply covers established dependents only: Projects that never @@ -1760,6 +2132,7 @@ async fn complete_managed_resource_reconciliation_with_progress( ProjectApplyStage::CompleteStagedApply, &dependent_projects, &skip_projects, + fallback_shutdown, ) .await; finish_project_phase(project_timer, &project_result); @@ -1783,6 +2156,14 @@ async fn complete_managed_resource_reconciliation_with_progress( source: Box::new(error), }); } + if fallback_shutdown_requested(fallback_shutdown) { + let failures = record_failures + .into_iter() + .chain(resource_failures.into_values()) + .chain(project_report.failures) + .collect::>(); + return cancel_or_preserve_reconciliation_errors(failures); + } let summary = managed_resource_reconciliation_summary(name.as_str(), track.as_str(), &project_report); let mut coverage = vec![JobDiagnosticSubject::Resource { @@ -1791,7 +2172,9 @@ async fn complete_managed_resource_reconciliation_with_progress( }]; coverage.extend(project_report.successful_project_coverage()); - Ok(CompletedReconciliationJob { summary, coverage }) + Ok(ReconciliationOutcome::Completed( + CompletedReconciliationJob { summary, coverage }, + )) } /// Applies persisted environments after a resource change. Only tests exercise this path @@ -1853,6 +2236,7 @@ fn reconcile_persisted_project_envs( Ok(report) } +#[cfg(test)] async fn complete_reconciliation_job( paths: &PvPaths, job_id: &str, @@ -1873,6 +2257,7 @@ async fn complete_reconciliation_job( .await } +#[cfg(test)] async fn complete_reconciliation_job_with_progress( paths: &PvPaths, job_id: &str, @@ -1901,6 +2286,7 @@ async fn complete_reconciliation_job_with_progress( enum ReconciliationJobCompletion { Succeeded(String), + Cancelled, Failed { error: Box, recording_error: Option>, @@ -1911,6 +2297,10 @@ impl ReconciliationJobCompletion { fn into_result(self) -> Result { match self { Self::Succeeded(summary) => Ok(summary), + Self::Cancelled => Err(DaemonError::UnexpectedProtocolResponse { + reason: "foreground reconciliation was cancelled without a shutdown signal" + .to_owned(), + }), Self::Failed { error, recording_error, @@ -1924,15 +2314,15 @@ impl ReconciliationJobCompletion { } #[derive(Default)] -struct ReconciliationJobOptions { +struct ReconciliationJobOptions<'a> { discard_obsolete_project: bool, pf_routing_state: Option, + fallback_shutdown: Option<&'a watch::Receiver>, } #[expect( clippy::too_many_arguments, - reason = "`shutdown` is only owned by the startup task; all other callers pass `None`, \ - so it cannot be derived from the job options." + reason = "the startup-only shutdown signal and cloneable test fallback have distinct scopes" )] async fn complete_reconciliation_job_with_progress_outcome( paths: &PvPaths, @@ -1941,9 +2331,12 @@ async fn complete_reconciliation_job_with_progress_outcome( runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, progress: DaemonDownloadProgress, timing: ReconciliationJobTiming, - options: ReconciliationJobOptions, + options: ReconciliationJobOptions<'_>, shutdown: Option<&oneshot::Receiver<()>>, ) -> ReconciliationJobCompletion { + let discard_obsolete_project = options.discard_obsolete_project; + let pf_routing_state = options.pf_routing_state; + let fallback_shutdown = options.fallback_shutdown; let scope_text = scope.to_string(); let phase_log = ReconciliationPhaseLog::new(paths, job_id, "reconcile", &scope_text) .with_progress(progress.phase_sender.clone()); @@ -1954,7 +2347,7 @@ async fn complete_reconciliation_job_with_progress_outcome( timing.queue_wait(), &[], ); - let obsolete_project = match (options.discard_obsolete_project, scope) { + let obsolete_project = match (discard_obsolete_project, scope) { (true, ReconciliationScope::Project { id }) => { project_exists(paths, id.as_str()).map(|exists| !exists) } @@ -1976,10 +2369,13 @@ async fn complete_reconciliation_job_with_progress_outcome( &[], ); } - Ok(CompletedReconciliationJob { - summary: "Project was removed before background reconciliation; skipped".to_owned(), - coverage: Vec::new(), - }) + Ok(ReconciliationOutcome::Completed( + CompletedReconciliationJob { + summary: "Project was removed before background reconciliation; skipped" + .to_owned(), + coverage: Vec::new(), + }, + )) } Ok(false) => match &effective_scope { ReconciliationScope::System => { @@ -1989,40 +2385,53 @@ async fn complete_reconciliation_job_with_progress_outcome( progress, &phase_log, shutdown, + fallback_shutdown, + None, ) .await } ReconciliationScope::Resource { name, .. } if gateway_runtime_resource(name.as_str()) => { - complete_gateway_reconciliation(paths, &phase_log).await + complete_gateway_reconciliation(paths, &phase_log, fallback_shutdown).await } ReconciliationScope::Resource { name, track } => { - complete_managed_resource_reconciliation_with_progress( + complete_managed_resource_reconciliation_with_progress_and_fallback_shutdown( paths, name, track, runtime_catalog, progress, &phase_log, + fallback_shutdown, ) .await } ReconciliationScope::Project { id } => { - complete_project_reconciliation_with_progress( + complete_project_reconciliation_with_progress_and_fallback( paths, id, runtime_catalog, progress, &phase_log, - options.pf_routing_state, &mut failure_subject, + ReconciliationJobOptions { + pf_routing_state, + fallback_shutdown, + ..ReconciliationJobOptions::default() + }, ) .await } }, }; + let result = match result { + Ok(ReconciliationOutcome::Completed(completed)) => Ok(completed), + Ok(ReconciliationOutcome::Cancelled) => return ReconciliationJobCompletion::Cancelled, + Err(error) => Err(error), + }; + let coverage_count = result .as_ref() .map_or(0, |completed| completed.coverage.len()); @@ -2121,13 +2530,30 @@ fn fail_reconciliation_job( async fn complete_gateway_reconciliation( paths: &PvPaths, phase_log: &ReconciliationPhaseLog, -) -> Result { - let summary = reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await?; + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { + let summary = match fallback_shutdown { + Some(fallback_shutdown) => { + match reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + phase_log, + fallback_shutdown, + ) + .await? + { + ReconciliationOutcome::Completed(summary) => summary, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + } + } + None => reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await?, + }; - Ok(CompletedReconciliationJob { - summary, - coverage: vec![JobDiagnosticSubject::GatewayRuntime], - }) + Ok(ReconciliationOutcome::Completed( + CompletedReconciliationJob { + summary, + coverage: vec![JobDiagnosticSubject::GatewayRuntime], + }, + )) } async fn complete_system_reconciliation_with_progress( @@ -2136,11 +2562,20 @@ async fn complete_system_reconciliation_with_progress( progress: DaemonDownloadProgress, phase_log: &ReconciliationPhaseLog, shutdown: Option<&oneshot::Receiver<()>>, -) -> Result { + fallback_shutdown: Option<&watch::Receiver>, + mut failure_subject: Option<&mut JobDiagnosticSubject>, +) -> Result, DaemonError> { + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } + let discovery_timer = phase_log.start(ReconciliationPhase::DemandDiscovery, "linked_projects"); let discovery_result = discover_system_project_demand(paths); finish_demand_discovery_phase(discovery_timer, &discovery_result); let demand = discovery_result?; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let resources_timer = phase_log.start(ReconciliationPhase::Resources, "desired_resources"); let mut resources_progress = progress.clone().suppressing_operation_phases(); @@ -2152,13 +2587,17 @@ async fn complete_system_reconciliation_with_progress( ) .await; resources_timer.finish(PhaseOutcome::from_succeeded(resources_result.is_ok()), &[]); + if fallback_shutdown_requested(fallback_shutdown) { + return resources_result.map(|()| ReconciliationOutcome::Cancelled); + } // Retry the install once before applying the Projects. No Project Apply downloads, so a // retry after one could never recover the apply that needed the artifact. The retry is its // own timed phase and suppresses nested operation records, so the log stays coherent. If it // still fails, the later read-only check decides whether current applied demand still needs it. // Skip the retry when shutdown was already requested: a second blocking download would // hold the shutdown drain with no one left to consume its result. - let shutdown_requested = shutdown.is_some_and(|shutdown| !shutdown.is_empty()); + let shutdown_requested = shutdown.is_some_and(|shutdown| !shutdown.is_empty()) + || fallback_shutdown_requested(fallback_shutdown); if resources_result.is_err() && !shutdown_requested { let retry_timer = phase_log.start(ReconciliationPhase::Resources, "desired_resources"); resources_progress = progress @@ -2172,11 +2611,17 @@ async fn complete_system_reconciliation_with_progress( ) .await; retry_timer.finish(PhaseOutcome::from_succeeded(resources_result.is_ok()), &[]); + if fallback_shutdown_requested(fallback_shutdown) { + return resources_result.map(|()| ReconciliationOutcome::Cancelled); + } } let mut resource_tracks = demand.resource_tracks; let mut project_demands = demand.project_demands; let has_late_resource_demand = discover_late_system_project_demand(paths, &mut resource_tracks, &mut project_demands)?; + if fallback_shutdown_requested(fallback_shutdown) { + return resources_result.map(|()| ReconciliationOutcome::Cancelled); + } if has_late_resource_demand { let late_timer = phase_log.start(ReconciliationPhase::Resources, "desired_resources"); resources_result = reconcile_system_resources_with_runtime_catalog_and_progress( @@ -2187,6 +2632,13 @@ async fn complete_system_reconciliation_with_progress( ) .await; late_timer.finish(PhaseOutcome::from_succeeded(resources_result.is_ok()), &[]); + if fallback_shutdown_requested(fallback_shutdown) { + return resources_result.map(|()| ReconciliationOutcome::Cancelled); + } + } + let projects = linked_projects(paths)?; + if fallback_shutdown_requested(fallback_shutdown) { + return resources_result.map(|()| ReconciliationOutcome::Cancelled); } let project_timer = phase_log.start(ReconciliationPhase::ProjectApply, "linked_projects"); let project_result = reconcile_system_projects_with_progress( @@ -2196,12 +2648,27 @@ async fn complete_system_reconciliation_with_progress( &project_demands, &progress, ProjectApplyStage::CompleteStagedApply, - &linked_projects(paths)?, + &projects, &BTreeSet::new(), + fallback_shutdown, ) .await; finish_project_phase(project_timer, &project_result); + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_system_reconciliation_errors( + resources_result, + project_result, + Ok(()), + ); + } let cleanup_result = stop_undemanded_system_resource_runtimes(paths, runtime_catalog).await; + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_system_reconciliation_errors( + resources_result, + project_result, + cleanup_result, + ); + } if resources_result.is_err() || project_result .as_ref() @@ -2214,17 +2681,62 @@ async fn complete_system_reconciliation_with_progress( &progress, ); } - let gateway_result = reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await; + if fallback_shutdown_requested(fallback_shutdown) { + return cancel_or_preserve_system_reconciliation_errors( + resources_result, + project_result, + cleanup_result, + ); + } + let has_system_failure = resources_result.is_err() + || project_result + .as_ref() + .map_or(true, |report| !report.failures.is_empty()) + || cleanup_result.is_err(); + if let Some(subject) = failure_subject.as_deref_mut() { + *subject = JobDiagnosticSubject::GatewayRuntime; + } + let gateway_result = match fallback_shutdown { + Some(fallback_shutdown) => { + reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + phase_log, + fallback_shutdown, + ) + .await + } + None => reconcile_gateway_runtimes_with_phase_log(paths, phase_log) + .await + .map(ReconciliationOutcome::Completed), + }; let (project_report, gateway_summary) = match ( resources_result, project_result, cleanup_result, gateway_result, ) { - (Ok(()), Ok(project_report), Ok(()), Ok(gateway_summary)) => { - (project_report, gateway_summary) + ( + Ok(()), + Ok(project_report), + Ok(()), + Ok(ReconciliationOutcome::Completed(gateway_summary)), + ) => (project_report, gateway_summary), + (Ok(()), Ok(project_report), Ok(()), Ok(ReconciliationOutcome::Cancelled)) => { + if !project_report.failures.is_empty() + && let Some(subject) = failure_subject.as_deref_mut() + { + *subject = JobDiagnosticSubject::SystemReconciliation; + } + return cancel_or_preserve_system_reconciliation_errors( + Ok(()), + Ok(project_report), + Ok(()), + ); } (resources_result, project_result, cleanup_result, gateway_result) => { + if has_system_failure && let Some(subject) = failure_subject.as_deref_mut() { + *subject = JobDiagnosticSubject::SystemReconciliation; + } let project_failures = match project_result { Ok(report) => report.failures, Err(error) => vec![error], @@ -2241,11 +2753,45 @@ async fn complete_system_reconciliation_with_progress( } }; let summary = system_reconciliation_summary(&project_report, &gateway_summary); + if let Some(subject) = failure_subject { + *subject = JobDiagnosticSubject::SystemReconciliation; + } let coverage = completed_system_reconciliation_coverage(paths, &project_report)?; - Ok(CompletedReconciliationJob { summary, coverage }) + Ok(ReconciliationOutcome::Completed( + CompletedReconciliationJob { summary, coverage }, + )) +} + +fn cancel_or_preserve_system_reconciliation_errors( + resources_result: Result<(), DaemonError>, + project_result: Result, + cleanup_result: Result<(), DaemonError>, +) -> Result, DaemonError> { + let project_failures = match project_result { + Ok(report) => report.failures, + Err(error) => vec![error], + }; + let failures = resources_result + .err() + .into_iter() + .chain(project_failures) + .chain(cleanup_result.err()); + cancel_or_preserve_reconciliation_errors(failures) +} + +fn cancel_or_preserve_reconciliation_errors( + failures: impl IntoIterator, +) -> Result, DaemonError> { + let failures = failures.into_iter().collect::>(); + if failures.is_empty() { + Ok(ReconciliationOutcome::Cancelled) + } else { + Err(combined_system_reconciliation_error(failures)) + } } +#[cfg(test)] async fn complete_project_reconciliation_with_progress( paths: &PvPaths, id: &crate::reconciliation::ReconciliationScopeComponent, @@ -2255,32 +2801,99 @@ async fn complete_project_reconciliation_with_progress( pf_routing_state: Option, failure_subject: &mut Option, ) -> Result { - let project_result = reconcile_project_env_and_missing_resources_with_progress( + complete_project_reconciliation_with_progress_and_fallback( paths, - id.as_str(), + id, runtime_catalog, - progress.clone(), + progress, phase_log, + failure_subject, + ReconciliationJobOptions { + pf_routing_state, + ..ReconciliationJobOptions::default() + }, ) - .await; + .await? + .into_completed() +} + +async fn complete_project_reconciliation_with_progress_and_fallback( + paths: &PvPaths, + id: &crate::reconciliation::ReconciliationScopeComponent, + runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, + progress: DaemonDownloadProgress, + phase_log: &ReconciliationPhaseLog, + failure_subject: &mut Option, + options: ReconciliationJobOptions<'_>, +) -> Result, DaemonError> { + let pf_routing_state = options.pf_routing_state; + let fallback_shutdown = options.fallback_shutdown; + let project_result = + reconcile_project_env_and_missing_resources_with_progress_and_fallback_shutdown( + paths, + id.as_str(), + runtime_catalog, + progress.clone(), + phase_log, + fallback_shutdown, + ) + .await; let project_env_summary = match project_result { - Ok(summary) => summary, + Ok(ReconciliationOutcome::Completed(summary)) => summary, + Ok(ReconciliationOutcome::Cancelled) => return Ok(ReconciliationOutcome::Cancelled), Err(project_error) => { - let gateway_result = reconcile_gateway_runtimes_with_phase_log(paths, phase_log).await; - return Err(combined_system_reconciliation_error( - std::iter::once(project_error) - .chain(gateway_result.err()) - .collect(), - )); + if fallback_shutdown_requested(fallback_shutdown) { + return Err(project_error); + } + let gateway_result = match fallback_shutdown { + Some(fallback_shutdown) => { + reconcile_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + phase_log, + fallback_shutdown, + ) + .await + } + None => reconcile_gateway_runtimes_with_phase_log(paths, phase_log) + .await + .map(ReconciliationOutcome::Completed), + }; + return match gateway_result { + Ok(ReconciliationOutcome::Completed(_) | ReconciliationOutcome::Cancelled) => { + Err(project_error) + } + Err(gateway_error) => Err(combined_system_reconciliation_error(vec![ + project_error, + gateway_error, + ])), + }; + } + }; + let gateway_outcome = match fallback_shutdown { + Some(fallback_shutdown) => { + match reconcile_project_gateway_runtimes_with_phase_log_and_fallback_shutdown( + paths, + id.as_str(), + pf_routing_state, + phase_log, + Some(fallback_shutdown), + ) + .await? + { + ReconciliationOutcome::Completed(outcome) => outcome, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + } + } + None => { + reconcile_project_gateway_runtimes_with_phase_log( + paths, + id.as_str(), + pf_routing_state, + phase_log, + ) + .await? } }; - let gateway_outcome = reconcile_project_gateway_runtimes_with_phase_log( - paths, - id.as_str(), - pf_routing_state, - phase_log, - ) - .await?; let (gateway_summary, gateway_evaluated) = match gateway_outcome { ProjectGatewayReconciliationOutcome::Reconciled { summary, @@ -2294,6 +2907,8 @@ async fn complete_project_reconciliation_with_progress( progress, phase_log, None, + fallback_shutdown, + None, ) .await; } @@ -2310,7 +2925,9 @@ async fn complete_project_reconciliation_with_progress( coverage.push(JobDiagnosticSubject::GatewayRuntime); } - Ok(CompletedReconciliationJob { summary, coverage }) + Ok(ReconciliationOutcome::Completed( + CompletedReconciliationJob { summary, coverage }, + )) } fn finish_project_phase( @@ -2370,7 +2987,7 @@ async fn reconcile_project_env_and_missing_resources( project_id: &str, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, ) -> Result { - reconcile_project_env_and_missing_resources_with_progress( + reconcile_project_env_and_missing_resources_with_progress_and_fallback_shutdown( paths, project_id, runtime_catalog, @@ -2381,33 +2998,51 @@ async fn reconcile_project_env_and_missing_resources( "reconcile", &format!("project:{project_id}"), ), + None, ) - .await + .await? + .into_completed() } -async fn reconcile_project_env_and_missing_resources_with_progress( +async fn reconcile_project_env_and_missing_resources_with_progress_and_fallback_shutdown( paths: &PvPaths, project_id: &str, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, progress: DaemonDownloadProgress, phase_log: &ReconciliationPhaseLog, -) -> Result { + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> +{ + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } + let record_timer = phase_log.start(ReconciliationPhase::ProjectApply, project_id); - let record_result = reconcile_project_env_with_runtime_catalog_and_progress( + let record_result = reconcile_project_env_with_runtime_catalog_and_progress_outcome( paths, project_id, runtime_catalog, None, &BTreeSet::new(), progress.clone(), - ProjectApplyStage::RecordRequirements, + ProjectApplyOptions::new(ProjectApplyStage::RecordRequirements, fallback_shutdown), ) .await; record_timer.finish( - PhaseOutcome::from_succeeded(record_result.is_ok()), + match &record_result { + Ok(ReconciliationOutcome::Completed(_)) => PhaseOutcome::Succeeded, + Ok(ReconciliationOutcome::Cancelled) => PhaseOutcome::Skipped, + Err(_) => PhaseOutcome::Failed, + }, &[("project_count", 1)], ); - let recorded = record_result?; + let recorded = match record_result? { + ReconciliationOutcome::Completed(recorded) => recorded, + ReconciliationOutcome::Cancelled => return Ok(ReconciliationOutcome::Cancelled), + }; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let requested_php_extensions = recorded.requested_php_extensions(); let recorded_tracks = recorded.recorded_tracks().clone(); @@ -2431,25 +3066,38 @@ async fn reconcile_project_env_and_missing_resources_with_progress( &[], ); let deferred_resources_error = resources_result?.deferred_error; + if fallback_shutdown_requested(fallback_shutdown) { + return match deferred_resources_error { + Some(error) => Err(error), + None => Ok(ReconciliationOutcome::Cancelled), + }; + } let apply_timer = phase_log.start(ReconciliationPhase::ProjectApply, project_id); - let apply_result = reconcile_project_env_with_runtime_catalog_and_progress( + let apply_result = reconcile_project_env_with_runtime_catalog_and_progress_outcome( paths, project_id, runtime_catalog, None, &BTreeSet::new(), progress, - ProjectApplyStage::CompleteStagedApply, + ProjectApplyOptions::new(ProjectApplyStage::CompleteStagedApply, fallback_shutdown), ) .await; apply_timer.finish( - PhaseOutcome::from_succeeded(apply_result.is_ok()), + match &apply_result { + Ok(ReconciliationOutcome::Completed(_)) => PhaseOutcome::Succeeded, + Ok(ReconciliationOutcome::Cancelled) => PhaseOutcome::Skipped, + Err(_) => PhaseOutcome::Failed, + }, &[("project_count", 1)], ); - match (apply_result, deferred_resources_error) { - (Ok(summary), None) => Ok(summary), + let result = match (apply_result, deferred_resources_error) { + (Ok(ReconciliationOutcome::Completed(summary)), None) => Ok(summary), + (Ok(ReconciliationOutcome::Cancelled), None) => { + return Ok(ReconciliationOutcome::Cancelled); + } (Ok(_), Some(repair_error)) => Err(repair_error), (Err(apply_error), None) => Err(apply_error), // Both stages failed. The apply is reported as primary because it is the later, @@ -2461,7 +3109,13 @@ async fn reconcile_project_env_and_missing_resources_with_progress( repair: Box::new(repair_error), }) } + }; + let summary = result?; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); } + + Ok(ReconciliationOutcome::Completed(summary)) } /// Installs the Managed Resource tracks the Project declares, then, under the original @@ -2584,6 +3238,7 @@ async fn reconcile_system_projects_with_progress( stage: ProjectApplyStage, projects: &[ProjectRecord], skip_project_ids: &BTreeSet, + fallback_shutdown: Option<&watch::Receiver>, ) -> Result { let mut report = SystemProjectReconciliationReport { total: projects.len(), @@ -2593,25 +3248,29 @@ async fn reconcile_system_projects_with_progress( let empty_demand = ProjectDemand::default(); for project in projects { + if fallback_shutdown_requested(fallback_shutdown) { + break; + } if skip_project_ids.contains(&project.id) { continue; } - match reconcile_project_env_with_runtime_catalog_and_progress( + match reconcile_project_env_with_runtime_catalog_and_progress_outcome( paths, &project.id, runtime_catalog, Some(project_demands.get(&project.id).unwrap_or(&empty_demand)), demanded_tracks, progress.clone(), - stage, + ProjectApplyOptions::new(stage, fallback_shutdown), ) .await { - Ok(summary) => { + Ok(ReconciliationOutcome::Completed(summary)) => { report.succeeded += 1; report.successful_project_ids.push(project.id.clone()); report.summaries.push(summary.as_str().to_owned()); } + Ok(ReconciliationOutcome::Cancelled) => break, Err(error @ DaemonError::ProjectEnvFailureRecordingFailed { .. }) => { return Err(error); } @@ -2651,6 +3310,7 @@ fn discover_late_system_project_demand( Ok(has_late_resource_demand) } +#[cfg(test)] async fn reconcile_system_projects_and_resources_with_progress( paths: &PvPaths, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, @@ -2711,6 +3371,7 @@ async fn reconcile_system_projects_and_resources_with_progress( ProjectApplyStage::CompleteStagedApply, &linked_projects(paths)?, &BTreeSet::new(), + None, ) .await; finish_project_phase(project_timer, &project_result); @@ -3157,18 +3818,20 @@ mod tests { #[cfg(target_os = "macos")] use super::run_reconciliation_job; use super::{ - DaemonDownloadProgress, FOREGROUND_JOB_PROGRESS_BUFFER, - FOREGROUND_JOB_STREAM_WRITE_TIMEOUT, ForegroundJobEvent, SystemProjectReconciliationReport, - abandon_reconciliation_job, complete_managed_resource_reconciliation_with_progress, + BackgroundReconciliationError, DaemonDownloadProgress, FOREGROUND_JOB_PROGRESS_BUFFER, + FOREGROUND_JOB_STREAM_WRITE_TIMEOUT, ForegroundJobEvent, ReconciliationJobCompletion, + SystemProjectReconciliationReport, abandon_reconciliation_job, + cancel_or_preserve_system_reconciliation_errors, + complete_managed_resource_reconciliation_with_progress, complete_or_fail_background_reconciliation, complete_project_reconciliation_with_progress, - complete_reconciliation_job_with_progress, complete_streamed_job_with_heartbeat, - complete_streamed_job_with_heartbeat_and_events, + complete_queued_background_reconciliation_job, complete_reconciliation_job_with_progress, + complete_streamed_job_with_heartbeat, complete_streamed_job_with_heartbeat_and_events, complete_system_reconciliation_with_progress, complete_update_job, - completed_system_reconciliation_coverage, discover_system_project_demand, - enqueue_foreground_reconciliation_job, enqueue_reconciliation_job, - enqueue_startup_reconciliation_job, enqueue_update_job, foreground_reconciliation_result, - linked_projects, managed_resource_reconciliation_summary, reconcile_persisted_project_envs, - reconcile_project_env_and_missing_resources, + complete_update_job_with_progress, completed_system_reconciliation_coverage, + discover_system_project_demand, enqueue_foreground_reconciliation_job, + enqueue_reconciliation_job, enqueue_startup_reconciliation_job, enqueue_update_job, + foreground_reconciliation_result, linked_projects, managed_resource_reconciliation_summary, + reconcile_persisted_project_envs, reconcile_project_env_and_missing_resources, reconcile_project_env_with_runtime_catalog_and_progress, reconcile_system_projects_and_resources_with_progress, reconcile_system_projects_with_progress, @@ -3180,7 +3843,7 @@ mod tests { wait_for_foreground_turn, wait_for_startup_reconciliation_turn, write_coalesced_update_response, write_foreground_terminal_event, }; - use crate::project_env::ProjectApplyStage; + use crate::project_env::{ProjectApplyOptions, ProjectApplyStage}; use crate::reconciliation::{ EnqueueResult, ReconciliationJobTiming, ReconciliationQueue, ReconciliationScope, }; @@ -5337,6 +6000,8 @@ mod tests { progress, &phase_log, None, + None, + None, ) .await .map(|_| ()) @@ -5701,6 +6366,8 @@ mod tests { super::DaemonDownloadProgress::disabled(), &phase_log, None, + None, + None, ) .await?; @@ -5807,6 +6474,8 @@ mod tests { super::DaemonDownloadProgress::disabled(), &phase_log, None, + None, + None, ) .await; @@ -5948,7 +6617,7 @@ mod tests { None, &BTreeSet::new(), super::DaemonDownloadProgress::disabled(), - ProjectApplyStage::CompleteApply, + ProjectApplyOptions::new(ProjectApplyStage::CompleteApply, None), ) .await?; let verification = async { @@ -5985,6 +6654,8 @@ mod tests { super::DaemonDownloadProgress::disabled(), &phase_log, None, + None, + None, ) .await; let database = Database::open(&paths)?; @@ -6151,6 +6822,8 @@ mod tests { progress, &phase_log, None, + None, + None, ) .await .map(|_| ()); @@ -6366,6 +7039,8 @@ mod tests { progress, &phase_log, None, + None, + None, ) .await .map(|_| ()) @@ -6537,6 +7212,8 @@ mod tests { super::DaemonDownloadProgress::disabled(), &phase_log, None, + None, + None, ) .await; @@ -6750,6 +7427,7 @@ mod tests { ProjectApplyStage::CompleteStagedApply, &linked_projects(&paths)?, &BTreeSet::new(), + None, ) .await?; stop_undemanded_system_resource_runtimes(&paths, Some(&catalog)).await?; @@ -6943,6 +7621,7 @@ mod tests { ProjectApplyStage::CompleteStagedApply, &linked_projects(&paths)?, &BTreeSet::new(), + None, ) .await?; @@ -7533,7 +8212,7 @@ mod tests { None, &BTreeSet::new(), DaemonDownloadProgress::disabled(), - ProjectApplyStage::CompleteStagedApply, + ProjectApplyOptions::new(ProjectApplyStage::CompleteStagedApply, None), ) .await; @@ -7635,8 +8314,10 @@ mod tests { &job_id, Some(&catalog), running.timing(), + None, ) - .await?; + .await + .into_result()?; running.finish(); let phases = reconciliation_phase_events(&paths, &job_id)?; @@ -8209,6 +8890,7 @@ mod tests { let held_client = HeldManifestArtifactClient { inner: resource_client, release_receiver: Mutex::new(release_receiver), + started: None, }; let mut database = Database::open(&paths)?; database.record_managed_resource_track_desired( @@ -9589,6 +10271,7 @@ mod tests { let paths = PvPaths::for_home(tempdir.path().join("home")); state::fs::write_sensitive_file(paths.db(), "not a database")?; let (shutdown_sender, shutdown_receiver) = oneshot::channel(); + let (_fallback_shutdown_sender, fallback_shutdown_receiver) = watch::channel(false); shutdown_sender .send(()) .map_err(|()| anyhow::anyhow!("startup shutdown receiver was dropped"))?; @@ -9598,6 +10281,7 @@ mod tests { ReconciliationQueue::new(), None, shutdown_receiver, + fallback_shutdown_receiver, ) .await; @@ -9799,7 +10483,14 @@ mod tests { let (client, server) = duplex(1024); let task = tokio::spawn(async move { let mut transport = protocol::transport(server); - wait_for_foreground_turn(waiting, &mut transport, true, Duration::from_millis(5)).await + wait_for_foreground_turn( + waiting, + &mut transport, + true, + Duration::from_millis(5), + None, + ) + .await }); let mut reader = protocol::transport(client); @@ -9818,7 +10509,9 @@ mod tests { } running.finish(); - let (waiting_running, stream_is_open) = task.await?; + let (waiting_running, stream_is_open) = task + .await? + .ok_or_else(|| anyhow::anyhow!("queued foreground job was cancelled"))?; assert!(stream_is_open); assert_eq!(waiting_running.job_id(), waiting_job_id); waiting_running.finish(); @@ -9843,14 +10536,23 @@ mod tests { let task = tokio::spawn(async move { let mut transport = protocol::transport(FailingWriteStream::with_signal(failed_write_sender)); - wait_for_foreground_turn(waiting, &mut transport, true, Duration::from_millis(5)).await + wait_for_foreground_turn( + waiting, + &mut transport, + true, + Duration::from_millis(5), + None, + ) + .await }); timeout(Duration::from_millis(100), failed_write_receiver) .await? .map_err(|_error| anyhow::anyhow!("queued heartbeat writer dropped"))?; running.finish(); - let (waiting_running, stream_is_open) = task.await?; + let (waiting_running, stream_is_open) = task + .await? + .ok_or_else(|| anyhow::anyhow!("queued foreground job was cancelled"))?; assert!(!stream_is_open); assert_eq!(waiting_running.job_id(), waiting_job_id); @@ -9871,6 +10573,7 @@ mod tests { let task_paths = paths.clone(); let task_queue = queue.clone(); let scope = ReconciliationScope::resource("caddy", "2")?; + let (_fallback_sender, fallback_receiver) = watch::channel(false); let task = tokio::spawn(async move { run_reconciliation_job( task_paths, @@ -9878,6 +10581,7 @@ mod tests { protocol::transport(server), scope, None, + &fallback_receiver, ) .await }); @@ -9926,6 +10630,391 @@ mod tests { Ok(()) } + #[cfg(target_os = "macos")] + #[tokio::test] + async fn cancelled_queued_foreground_reconciliation_is_abandoned_before_its_turn() + -> anyhow::Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let (client, server) = UnixStream::pair()?; + let mut client = protocol::transport(client); + let queue = ReconciliationQueue::new(); + let active = queued(enqueue_update_job(&paths, &queue)?)? + .wait_for_turn() + .await; + let (fallback_sender, fallback_receiver) = watch::channel(false); + let task_paths = paths.clone(); + let mut task = tokio::spawn(async move { + run_reconciliation_job( + task_paths, + queue, + protocol::transport(server), + ReconciliationScope::resource("caddy", "2")?, + None, + &fallback_receiver, + ) + .await + }); + + let accepted = timeout(Duration::from_millis(300), client.next()) + .await? + .ok_or_else(|| anyhow::anyhow!("foreground stream ended before queue admission"))??; + let accepted = serde_json::from_str::(&accepted)?; + assert_eq!(accepted["status"], "accepted"); + assert!(!task.is_finished()); + fallback_sender + .send(true) + .map_err(|_| anyhow::anyhow!("queued foreground reconciliation stopped early"))?; + let result = timeout(Duration::from_millis(300), &mut task).await??; + + assert!(result.is_ok()); + let job = Database::open(&paths)? + .recent_jobs()? + .into_iter() + .find(|job| job.kind == "reconcile") + .ok_or_else(|| anyhow::anyhow!("missing queued reconciliation job"))?; + assert_eq!(job.status, JobStatus::Failed); + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + let coverage_count = Connection::open(paths.db().as_std_path())?.query_row( + "SELECT COUNT(*) FROM job_diagnostic_outcomes WHERE job_id = ?1 AND outcome = 'success'", + [&job.id], + |row| row.get::<_, i64>(0), + )?; + assert_eq!(coverage_count, 0); + active.finish(); + + Ok(()) + } + + #[tokio::test] + async fn cancelled_queued_background_reconciliation_is_abandoned() -> anyhow::Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let queue = ReconciliationQueue::new(); + let active = queued(enqueue_update_job(&paths, &queue)?)? + .wait_for_turn() + .await; + let waiting = queued(enqueue_reconciliation_job( + &paths, + &queue, + ReconciliationScope::resource("caddy", "2")?, + )?)?; + let waiting_job_id = waiting.job_id().to_owned(); + let (fallback_sender, fallback_receiver) = watch::channel(false); + let completion_paths = paths.clone(); + let completion_task = tokio::spawn(async move { + complete_queued_background_reconciliation_job( + &completion_paths, + waiting, + None, + Some(&fallback_receiver), + ) + .await + }); + + tokio::task::yield_now().await; + assert!(!completion_task.is_finished()); + fallback_sender + .send(true) + .map_err(|_| anyhow::anyhow!("queued reconciliation stopped before cancellation"))?; + timeout(Duration::from_millis(300), completion_task) + .await?? + .map_err(BackgroundReconciliationError::into_error)?; + + let job = Database::open(&paths)? + .recent_jobs()? + .into_iter() + .find(|job| job.id == waiting_job_id) + .ok_or_else(|| anyhow::anyhow!("missing queued reconciliation job"))?; + assert_eq!(job.status, JobStatus::Failed); + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + active.finish(); + + Ok(()) + } + + #[tokio::test] + async fn fallback_shutdown_finishes_resource_phase_before_abandoning_system_job() + -> anyhow::Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + seed_installed_caddy(&paths)?; + let _caddy_guard = SeededCaddyGuard::new(paths.clone()); + let (client, _archive_size) = scripted_artifact_client( + tempdir.path(), + "composer", + COMPOSER_TEST_TRACK, + COMPOSER_TEST_ARTIFACT_VERSION, + COMPOSER_TEST_ARCHIVE_FILE_NAME, + "composer.phar", + )?; + let started = Arc::new(AtomicBool::new(false)); + let (release_sender, release_receiver) = mpsc::channel(); + let client = HeldManifestArtifactClient { + inner: client, + release_receiver: Mutex::new(release_receiver), + started: Some(Arc::clone(&started)), + }; + Database::open(&paths)?.record_managed_resource_track_desired( + "composer", + COMPOSER_TEST_TRACK, + ManagedResourceDesiredState::Installed, + )?; + let catalog = Arc::new( + crate::managed_resources::ManagedResourceRuntimeCatalog::without_adapters_with_manifest_client( + OFFLINE_TEST_MANIFEST_URL, + client, + )?, + ); + let queue = ReconciliationQueue::new(); + let queued = queued(enqueue_reconciliation_job( + &paths, + &queue, + ReconciliationScope::System, + )?)?; + let job_id = queued.job_id().to_owned(); + let (fallback_sender, fallback_receiver) = watch::channel(false); + let task_paths = paths.clone(); + let task_catalog = Arc::clone(&catalog); + let mut completion_task = tokio::spawn(async move { + complete_queued_background_reconciliation_job( + &task_paths, + queued, + Some(task_catalog.as_ref()), + Some(&fallback_receiver), + ) + .await + }); + timeout(Duration::from_secs(5), async { + while !started.load(Ordering::SeqCst) { + tokio::task::yield_now().await; + } + }) + .await?; + + fallback_sender + .send(true) + .map_err(|_| anyhow::anyhow!("system reconciliation stopped before cancellation"))?; + assert!( + timeout(Duration::from_millis(100), &mut completion_task) + .await + .is_err() + ); + assert!(matches!( + JobsLock::acquire(&paths), + Err(StateError::CoordinationLockHeld { .. }) + )); + release_sender.send(())?; + timeout(Duration::from_secs(5), completion_task) + .await?? + .map_err(BackgroundReconciliationError::into_error)?; + + let job = Database::open(&paths)? + .recent_jobs()? + .into_iter() + .find(|job| job.id == job_id) + .ok_or_else(|| anyhow::anyhow!("missing system reconciliation job"))?; + assert_eq!(job.status, JobStatus::Failed); + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + let completed_phases = reconciliation_phase_events(&paths, &job.id)? + .into_iter() + .filter_map(|event| event["phase"].as_str().map(str::to_owned)) + .collect::>(); + assert_eq!(completed_phases, ["queue", "demand_discovery", "resources"]); + let coverage_count = Connection::open(paths.db().as_std_path())?.query_row( + "SELECT COUNT(*) FROM job_diagnostic_outcomes WHERE job_id = ?1 AND outcome = 'success'", + [&job.id], + |row| row.get::<_, i64>(0), + )?; + assert_eq!(coverage_count, 0); + assert!(!paths.gateway_pid().exists()); + assert!(!paths.gateway_runtime_metadata().exists()); + assert!(!paths.gateway_root_config().exists()); + + Ok(()) + } + + #[test] + fn fallback_after_update_mutation_preserves_changed_and_no_op_boundaries() -> anyhow::Result<()> + { + for (artifact_version, archive_file_name, expected_summary) in [ + ("2.8.1-pv1", "composer-2.8.1-pv1-any.tar.gz", None), + ( + COMPOSER_TEST_ARTIFACT_VERSION, + COMPOSER_TEST_ARCHIVE_FILE_NAME, + Some("current"), + ), + ] { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + seed_installed_artifact( + &paths, + "composer", + COMPOSER_TEST_TRACK, + COMPOSER_TEST_ARTIFACT_VERSION, + "composer.phar", + )?; + let (client, _archive_size) = scripted_artifact_client( + tempdir.path(), + "composer", + COMPOSER_TEST_TRACK, + artifact_version, + archive_file_name, + "composer.phar", + )?; + let started = Arc::new(AtomicBool::new(false)); + let (release_sender, release_receiver) = mpsc::channel(); + let client = HeldManifestArtifactClient { + inner: client, + release_receiver: Mutex::new(release_receiver), + started: Some(Arc::clone(&started)), + }; + let catalog = Arc::new( + crate::managed_resources::ManagedResourceRuntimeCatalog::without_adapters_with_manifest_client( + OFFLINE_TEST_MANIFEST_URL, + client, + )?, + ); + let queue = ReconciliationQueue::new(); + let runtime = crate::build_runtime()?; + let waiting = queued(enqueue_update_job(&paths, &queue)?)?; + let running = runtime.block_on(waiting.wait_for_turn()); + let job_id = running.job_id().to_owned(); + let (fallback_sender, fallback_receiver) = watch::channel(false); + let completion_paths = paths.clone(); + let completion_catalog = Arc::clone(&catalog); + let completion_job_id = job_id.clone(); + let completion_thread = std::thread::Builder::new() + .name("held-update-fallback".to_owned()) + .spawn(move || -> anyhow::Result<_> { + let runtime = crate::build_runtime()?; + let completion = runtime.block_on(complete_update_job_with_progress( + &completion_paths, + &completion_job_id, + Some(completion_catalog.as_ref()), + DaemonDownloadProgress::disabled(), + running.timing(), + Some(&fallback_receiver), + )); + if matches!(completion, ReconciliationJobCompletion::Cancelled) { + drop(running); + } else { + running.finish(); + } + + Ok(completion) + })?; + let deadline = Instant::now() + Duration::from_secs(5); + while !started.load(Ordering::SeqCst) { + if Instant::now() >= deadline { + return Err(anyhow::anyhow!( + "update manifest request did not reach its test gate" + )); + } + std::thread::sleep(Duration::from_millis(10)); + } + assert!(matches!( + JobsLock::acquire(&paths), + Err(StateError::CoordinationLockHeld { path }) if path == paths.jobs_lock() + )); + + fallback_sender + .send(true) + .map_err(|_| anyhow::anyhow!("update stopped before fallback was signalled"))?; + assert!( + !completion_thread.is_finished(), + "update returned before its non-interruptible mutation completed" + ); + release_sender + .send(()) + .map_err(|_| anyhow::anyhow!("update dropped its manifest gate"))?; + let completion = completion_thread + .join() + .map_err(|_| anyhow::anyhow!("update completion thread panicked"))??; + let job = Database::open(&paths)? + .recent_jobs()? + .into_iter() + .find(|job| job.id == job_id) + .ok_or_else(|| anyhow::anyhow!("missing update job {job_id}"))?; + + match (expected_summary, completion) { + (Some(expected_summary), ReconciliationJobCompletion::Succeeded(summary)) => { + assert_eq!(summary, expected_summary); + assert_eq!(job.status, JobStatus::Succeeded); + } + (None, ReconciliationJobCompletion::Cancelled) => { + assert_eq!(job.status, JobStatus::Failed); + assert_eq!( + job.error.as_deref(), + Some("Managed Resource update was abandoned before completion") + ); + let current_path = state::fs::read_link( + &paths + .resources() + .join("composer") + .join(COMPOSER_TEST_TRACK) + .join("current"), + )?; + assert_eq!( + current_path, + Utf8PathBuf::from(format!("releases/{artifact_version}")) + ); + let phases = reconciliation_phase_events(&paths, &job_id)?; + for phase in [ + "demand_discovery", + "resources", + "project_apply", + "workers", + "gateway", + ] { + assert!( + !phases.iter().any(|event| event["phase"] == phase), + "update entered {phase} after fallback: {phases:#?}" + ); + } + } + (expected_summary, _completion) => { + return Err(anyhow::anyhow!( + "unexpected update completion for expected summary {expected_summary:?}" + )); + } + } + let _jobs_lock = JobsLock::acquire(&paths)?; + } + + Ok(()) + } + + #[test] + fn system_project_failures_outrank_gateway_cancellation() -> anyhow::Result<()> { + let sentinel = DaemonError::UnexpectedProtocolResponse { + reason: "project sentinel".to_owned(), + }; + let result = cancel_or_preserve_system_reconciliation_errors( + Ok(()), + Ok(SystemProjectReconciliationReport { + failures: vec![sentinel], + ..SystemProjectReconciliationReport::default() + }), + Ok(()), + ); + let error = result + .err() + .ok_or_else(|| anyhow::anyhow!("cancellation replaced the Project failure"))?; + assert_eq!(error.to_string(), "daemon protocol error: project sentinel"); + + Ok(()) + } + #[tokio::test] async fn background_reconciliation_coalesces_under_daemon_jobs_lock() -> anyhow::Result<()> { let tempdir = tempdir()?; @@ -10399,8 +11488,10 @@ mod tests { job_id, Some(catalog), ReconciliationJobTiming::immediate(), + None, ) - .await?; + .await + .into_result()?; let mut reader = protocol::transport(client); let mut events = Vec::new(); @@ -10799,10 +11890,14 @@ mod tests { struct HeldManifestArtifactClient { inner: ScriptedArtifactClient, release_receiver: Mutex>, + started: Option>, } impl resources::ResourceHttpClient for HeldManifestArtifactClient { fn get_text(&self, url: &str) -> resources::Result { + if let Some(started) = &self.started { + started.store(true, Ordering::SeqCst); + } let release_receiver = match self.release_receiver.lock() { Ok(release_receiver) => release_receiver, Err(poisoned) => poisoned.into_inner(), diff --git a/crates/daemon/src/lib.rs b/crates/daemon/src/lib.rs index 712af385..367f56d0 100644 --- a/crates/daemon/src/lib.rs +++ b/crates/daemon/src/lib.rs @@ -17,14 +17,14 @@ mod watcher; use std::future::Future; use std::io; -use std::sync::Arc; +use std::sync::{Arc, mpsc}; use managed_resources::ManagedResourceRuntimeCatalog; use platform::PlatformTarget; use serde::Serialize; use state::{Database, PvPaths, StateError}; use tokio::runtime::Runtime; -use tokio::sync::oneshot; +use tokio::sync::{oneshot, watch}; use tokio::task::JoinHandle; pub use caddy_admin::{ @@ -53,8 +53,10 @@ pub use supervisor::{ pub struct RunningDaemon { paths: PvPaths, shutdown: oneshot::Sender<()>, + fallback_shutdown: watch::Sender, task: JoinHandle>, dns: dns::RunningDnsResolver, + blocked_request_release_signal: Option>, } impl RunningDaemon { @@ -98,11 +100,47 @@ impl RunningDaemon { .await } + #[doc(hidden)] + pub async fn start_without_managed_resource_adapters_with_manifest_client_and_blocked_request_release( + paths: PvPaths, + manifest_url: impl Into, + client: impl resources::ResourceHttpClient + Send + Sync + 'static, + blocked_request_release_signal: mpsc::Sender<()>, + ) -> Result { + ipc::require_ipc_for(PlatformTarget::current()?)?; + Self::start_with_runtime_catalog_and_blocked_request_release( + paths, + Some( + ManagedResourceRuntimeCatalog::without_adapters_with_manifest_client( + manifest_url, + client, + )?, + ), + Some(blocked_request_release_signal), + ) + .await + } + async fn start_with_runtime_catalog( paths: PvPaths, runtime_catalog: Option, ) -> Result { - match Self::start_with_runtime_catalog_inner(paths.clone(), runtime_catalog).await { + Self::start_with_runtime_catalog_and_blocked_request_release(paths, runtime_catalog, None) + .await + } + + async fn start_with_runtime_catalog_and_blocked_request_release( + paths: PvPaths, + runtime_catalog: Option, + blocked_request_release_signal: Option>, + ) -> Result { + match Self::start_with_runtime_catalog_inner( + paths.clone(), + runtime_catalog, + blocked_request_release_signal, + ) + .await + { Ok(daemon) => Ok(daemon), Err(error) => { write_startup_failure_marker(&paths, &error); @@ -115,6 +153,7 @@ impl RunningDaemon { async fn start_with_runtime_catalog_inner( paths: PvPaths, runtime_catalog: Option, + blocked_request_release_signal: Option>, ) -> Result { let mut database = Database::open(&paths)?; ipc::prepare_endpoint(&paths).await?; @@ -135,20 +174,24 @@ impl RunningDaemon { }; structured_log::daemon_started(&paths); let (shutdown, shutdown_receiver) = oneshot::channel(); + let (fallback_shutdown, fallback_shutdown_receiver) = watch::channel(false); let server_paths = paths.clone(); let runtime_catalog = runtime_catalog.map(Arc::new); let task = tokio::spawn(server::serve( server_paths, listener, shutdown_receiver, + fallback_shutdown_receiver, runtime_catalog, )); Ok(Self { paths, shutdown, + fallback_shutdown, task, dns, + blocked_request_release_signal, }) } @@ -166,6 +209,35 @@ impl RunningDaemon { Ok(()) } + + /// Signals shutdown and removes the IPC endpoint without waiting for daemon tasks. + /// + /// This is a test-only escape hatch for a fixture whose owning Tokio runtime cannot + /// continue driving those tasks during fallback cleanup. It requests cooperative cancellation + /// but does not wait for it to finish. A matching reload that may already have been sent keeps + /// its promoted files and pending marker instead of issuing a competing load. The fixture must + /// still clean up the exact established and partially started runtimes it owns. + #[doc(hidden)] + pub fn shutdown_without_waiting_for_test(self) -> Result<(), DaemonError> { + let Self { + paths, + shutdown, + fallback_shutdown, + task, + mut dns, + blocked_request_release_signal, + } = self; + let _ = fallback_shutdown.send(true); + if let Some(signal) = blocked_request_release_signal { + let _sent = signal.send(()); + } + let _ = shutdown.send(()); + dns.signal_shutdown(); + drop(task); + drop(dns); + + ipc::remove_endpoint(&paths) + } } #[derive(Serialize)] @@ -247,8 +319,10 @@ async fn wait_for_shutdown( let RunningDaemon { paths, shutdown, + fallback_shutdown: _fallback_shutdown, mut task, mut dns, + blocked_request_release_signal: _blocked_request_release_signal, } = daemon; tokio::pin!(shutdown_signal); @@ -323,7 +397,7 @@ mod tests { use insta::assert_debug_snapshot; use platform::{PlatformCapability, PlatformError, PlatformTarget}; use state::PvPaths; - use tokio::sync::oneshot; + use tokio::sync::{oneshot, watch}; use tokio::time::timeout; use super::{ @@ -416,13 +490,16 @@ mod tests { async fn shutdown_wait_returns_when_server_task_fails_before_signal() { let paths = PvPaths::for_home("/tmp/pv-daemon-test-home"); let (shutdown, _shutdown_receiver) = oneshot::channel(); + let (fallback_shutdown, _fallback_shutdown_receiver) = watch::channel(false); let task = tokio::spawn(async { Err(DaemonError::Io(io::Error::other("server stopped early"))) }); let daemon = RunningDaemon { paths, shutdown, + fallback_shutdown, task, dns: super::dns::RunningDnsResolver::pending_for_test(), + blocked_request_release_signal: None, }; let result = wait_for_shutdown(daemon, future::pending::>()).await; @@ -441,6 +518,7 @@ mod tests { let stale_listener = tokio::net::UnixListener::bind(paths.daemon_socket())?; drop(stale_listener); let (shutdown, shutdown_receiver) = oneshot::channel(); + let (fallback_shutdown, _fallback_shutdown_receiver) = watch::channel(false); let task = tokio::spawn(async { let _ = shutdown_receiver.await; Ok(()) @@ -448,10 +526,12 @@ mod tests { let daemon = RunningDaemon { paths: paths.clone(), shutdown, + fallback_shutdown, task, dns: super::dns::RunningDnsResolver::failed_for_test(io::Error::other( "dns stopped early", )), + blocked_request_release_signal: None, }; let result = timeout( @@ -477,13 +557,16 @@ mod tests { let stale_listener = tokio::net::UnixListener::bind(paths.daemon_socket())?; drop(stale_listener); let (shutdown, _shutdown_receiver) = oneshot::channel(); + let (fallback_shutdown, _fallback_shutdown_receiver) = watch::channel(false); let task = tokio::spawn(future::pending::>()); task.abort(); let daemon = RunningDaemon { paths: paths.clone(), shutdown, + fallback_shutdown, task, dns: super::dns::RunningDnsResolver::aborted_for_test(), + blocked_request_release_signal: None, }; let result = daemon.shutdown().await; diff --git a/crates/daemon/src/managed_resources/mod.rs b/crates/daemon/src/managed_resources/mod.rs index 7c83744f..a134ba72 100644 --- a/crates/daemon/src/managed_resources/mod.rs +++ b/crates/daemon/src/managed_resources/mod.rs @@ -32,8 +32,10 @@ use state::{ RUNTIME_PORT_FALLBACK_START, ResourceAllocationRecord, RuntimeObservedStatus, RuntimeSubject, StateError, }; +use tokio::sync::watch; use tokio::time::{sleep, timeout}; +use crate::gateway::ReconciliationOutcome; use crate::jobs::DaemonDownloadProgress; use crate::project_env::DemandedResourceTrack; use crate::supervisor::{ @@ -351,6 +353,24 @@ pub(crate) enum ArtifactInstall { Forbidden, } +#[derive(Clone, Copy)] +pub(crate) struct ResourceReconciliationOptions<'a> { + artifact_install: ArtifactInstall, + fallback_shutdown: Option<&'a watch::Receiver>, +} + +impl<'a> ResourceReconciliationOptions<'a> { + pub(crate) fn new( + artifact_install: ArtifactInstall, + fallback_shutdown: Option<&'a watch::Receiver>, + ) -> Self { + Self { + artifact_install, + fallback_shutdown, + } + } +} + pub(crate) async fn reconcile_project_resources_with_progress( paths: &PvPaths, database: &mut Database, @@ -358,8 +378,8 @@ pub(crate) async fn reconcile_project_resources_with_progress( plan: &crate::project_env::ProjectResourcePlan, demanded_tracks: &BTreeSet, progress: DaemonDownloadProgress, - artifact_install: ArtifactInstall, -) -> Result<(), DaemonError> { + options: ResourceReconciliationOptions<'_>, +) -> Result, DaemonError> { let catalog = ManagedResourceRuntimeCatalog::production()?; reconcile_project_resources_with_catalog_and_progress( @@ -370,7 +390,7 @@ pub(crate) async fn reconcile_project_resources_with_progress( &catalog, demanded_tracks, progress, - artifact_install, + options, ) .await } @@ -389,16 +409,26 @@ pub(crate) async fn reconcile_project_resources_with_catalog_and_progress( catalog: &ManagedResourceRuntimeCatalog, demanded_tracks: &BTreeSet, progress: DaemonDownloadProgress, - artifact_install: ArtifactInstall, -) -> Result<(), DaemonError> { + options: ResourceReconciliationOptions<'_>, +) -> Result, DaemonError> { + let ResourceReconciliationOptions { + artifact_install, + fallback_shutdown, + } = options; let supervisor = ProcessSupervisor::new(paths.clone()); let mut demanded_tracks = demanded_tracks.clone(); demanded_tracks.extend(plan.resources.iter().map(|resource| { DemandedResourceTrack::new(resource.resource_name.clone(), resource.track.clone()) })); + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } stop_undemanded_catalog_runtimes(paths, database, catalog, &supervisor, &demanded_tracks) .await?; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let install_requests = missing_project_install_requests(database, plan, catalog); let mut prefetched_installs = prefetch_missing_project_installs(paths, catalog, install_requests, progress.clone()) @@ -409,6 +439,7 @@ pub(crate) async fn reconcile_project_resources_with_catalog_and_progress( progress: &progress, prefetched_installs: &mut prefetched_installs, artifact_install, + fallback_shutdown, }; reconcile_resource_tracks(paths, database, &mut context, project, plan).await @@ -434,6 +465,7 @@ pub(crate) async fn reconcile_persisted_resource_track_with_progress( runtime_catalog, &projects, progress, + None, ) .await?; @@ -450,6 +482,7 @@ pub(crate) async fn reconcile_persisted_resource_track_for_projects_with_progres runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, projects: &[ProjectRecord], progress: DaemonDownloadProgress, + fallback_shutdown: Option<&watch::Receiver>, ) -> Result<(bool, BTreeMap), DaemonError> { use crate::project_env::record_project_env_failure; use state::ResourceAllocationStatus; @@ -514,14 +547,22 @@ pub(crate) async fn reconcile_persisted_resource_track_for_projects_with_progres progress: &progress, prefetched_installs: &mut prefetched_installs, artifact_install: ArtifactInstall::Forbidden, + fallback_shutdown, }; reconcile_resource_track(paths, &mut database, &mut context, &resource, &[]).await?; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(installed); + } + if let Some(adapter) = catalog.adapter(resource_name) { let runtime_context = persisted_resource_runtime_context(paths, &mut database, adapter, &resource)?; for project in projects { + if fallback_shutdown_requested(fallback_shutdown) { + break; + } let allocations = database .resource_allocations(&project.id, resource_name)? .into_iter() @@ -1559,6 +1600,7 @@ struct ResourceTrackReconciliationContext<'context> { progress: &'context DaemonDownloadProgress, prefetched_installs: &'context mut BTreeMap, artifact_install: ArtifactInstall, + fallback_shutdown: Option<&'context watch::Receiver>, } type ProjectTrackKey = (String, String); @@ -1604,6 +1646,11 @@ async fn reconcile_resource_track( return Ok(()); }; let completed = prepared.wait().await; + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return completed + .readiness + .map_err(|error| record_resource_runtime_failure(database, resource, error)); + } match finish_resource_track(paths, database, reconciliation.catalog, completed).await { Ok(()) => Ok(()), Err(error) => Err(record_resource_runtime_failure(database, resource, error)), @@ -1616,16 +1663,25 @@ async fn reconcile_resource_tracks( reconciliation: &mut ResourceTrackReconciliationContext<'_>, project: &ProjectRecord, plan: &crate::project_env::ProjectResourcePlan, -) -> Result<(), DaemonError> { +) -> Result, DaemonError> { let mut pending = FuturesUnordered::new(); let mut ready = VecDeque::new(); let mut failures = Vec::new(); + let mut cancelled = false; for resource in &plan.resources { + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + cancelled = true; + break; + } if pending.len() == RUNTIME_READINESS_CONCURRENCY_LIMIT && let Some(runtime) = pending.next().await { ready.push_back(runtime); } + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + cancelled = true; + break; + } let allocations = match desired_allocations(database, project, plan, resource) { Ok(allocations) => allocations, Err(error) => { @@ -1655,6 +1711,17 @@ async fn reconcile_resource_tracks( } } } + if cancelled { + for resource in &plan.resources { + let key = resource_key(resource); + if let Some(PrefetchedProjectInstall::Failed(error)) = + reconciliation.prefetched_installs.remove(&key) + { + let error = record_resource_runtime_failure(database, resource, error); + failures.push((key, error)); + } + } + } loop { let runtime = if let Some(runtime) = ready.pop_front() { @@ -1665,6 +1732,18 @@ async fn reconcile_resource_tracks( break; }; let key = runtime.key(); + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + cancelled = true; + if let Err(error) = runtime.readiness { + let resource = state::ProjectManagedResourceInput { + resource_name: key.0.clone(), + track: key.1.clone(), + }; + let error = record_resource_runtime_failure(database, &resource, error); + failures.push((key, error)); + } + continue; + } let result = { let finalization = finish_resource_track(paths, database, reconciliation.catalog, runtime); @@ -1687,18 +1766,21 @@ async fn reconcile_resource_tracks( } failures.sort_by(|left, right| left.0.cmp(&right.0)); - if failures.is_empty() { - return Ok(()); + if !failures.is_empty() { + return Err(combined_project_resource_error( + failures + .into_iter() + .map(|((resource_name, track), error)| { + ManagedResourceProjectFailure::new(resource_name, track, error) + }) + .collect(), + )); + } + if cancelled || fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); } - Err(combined_project_resource_error( - failures - .into_iter() - .map(|((resource_name, track), error)| { - ManagedResourceProjectFailure::new(resource_name, track, error) - }) - .collect(), - )) + Ok(ReconciliationOutcome::Completed(())) } struct PreparedResourceRuntime { @@ -1770,9 +1852,15 @@ async fn prepare_resource_track( track: resource.track.clone(), }); }; + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } let mut attempt = 0; loop { + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } attempt += 1; let ports = assign_named_ports(database, adapter, &resource.resource_name, &resource.track)?; @@ -1803,30 +1891,44 @@ async fn prepare_resource_track( }; let env = adapter.resource_env(&context)?; let context = ManagedResourceRuntimeContext { env, ..context }; + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } database.record_managed_resource_track_env_context( &resource.resource_name, &resource.track, &context.env, )?; let spec = adapter.build_process_spec(paths, &context)?; + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } adapter.prepare_runtime(paths, &context).await?; + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } let readiness = adapter.readiness(&context)?; let readiness_timeout = adapter_readiness_timeout(adapter); + if fallback_shutdown_requested(reconciliation.fallback_shutdown) { + return Ok(None); + } match start_or_adopt_runtime( reconciliation.supervisor, spec, readiness, readiness_timeout, + reconciliation.fallback_shutdown, ) .await { - Ok(readiness) => { + Ok(Some(readiness)) => { return Ok(Some(PreparedResourceRuntime { context, allocations: allocations.to_vec(), readiness, })); } + Ok(None) => return Ok(None), Err(DaemonError::NonPvManagedResourceRuntimeListener { .. }) if attempt < RESOURCE_START_ATTEMPTS => { @@ -2325,14 +2427,15 @@ async fn start_or_adopt_runtime( spec: ProcessSpec, readiness: ManagedResourceReadiness, readiness_timeout: Duration, -) -> Result { + fallback_shutdown: Option<&watch::Receiver>, +) -> Result, DaemonError> { if supervisor.adopt(&spec)?.is_some() { - return Ok(PendingManagedResourceReadiness { + return Ok(Some(PendingManagedResourceReadiness { spec, readiness, readiness_timeout, process: None, - }); + })); } if let Some(adopted) = supervisor.adopt_recorded(&spec.pid_path, &spec.metadata_path)? { adopted.stop(RESOURCE_STOP_GRACE_PERIOD).await?; @@ -2344,14 +2447,22 @@ async fn start_or_adopt_runtime( return Err(DaemonError::NonPvManagedResourceRuntimeListener { name: spec.name }); } + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(None); + } + let process = supervisor.start(spec.clone()).await?; - Ok(PendingManagedResourceReadiness { + Ok(Some(PendingManagedResourceReadiness { spec, readiness, readiness_timeout, process: Some(process), - }) + })) +} + +fn fallback_shutdown_requested(shutdown: Option<&watch::Receiver>) -> bool { + shutdown.is_some_and(|shutdown| *shutdown.borrow()) } struct PendingManagedResourceReadiness { diff --git a/crates/daemon/src/managed_resources/tests.rs b/crates/daemon/src/managed_resources/tests.rs index a4859dcd..1f3499ca 100644 --- a/crates/daemon/src/managed_resources/tests.rs +++ b/crates/daemon/src/managed_resources/tests.rs @@ -8,11 +8,13 @@ use std::time::Duration; use crate::{ DaemonError, ProcessSpec, ProcessSupervisor, ReadinessCheck, + gateway::ReconciliationOutcome, jobs::DaemonDownloadProgress, managed_resources::{ManagedResourceRuntimeAdapter, ManagedResourceRuntimeContext}, project_env::{ - DemandedResourceTrack, discover_project_demand, + DemandedResourceTrack, ProjectApplyOptions, ProjectApplyStage, discover_project_demand, reconcile_project_env_with_runtime_catalog_and_progress, + reconcile_project_env_with_runtime_catalog_and_progress_outcome, }, reconciliation::{ReconciliationQueue, ReconciliationScope}, }; @@ -38,7 +40,7 @@ use state::{ ResourceAllocationStatus, RuntimeObservedStatus, RuntimeSubject, StateError, }; use time::{Duration as CertificateDuration, OffsetDateTime}; -use tokio::sync::Semaphore; +use tokio::sync::{Semaphore, watch}; use tokio::time::timeout; const FAKE_MAILPIT_TRACK: &str = "1.0"; @@ -330,6 +332,7 @@ struct GatedArtifactClient { maximum_active_downloads: Arc, download_started_sender: mpsc::Sender<()>, release_downloads: Arc<(Mutex, Condvar)>, + download_failure: Option, } impl resources::ResourceHttpClient for GatedArtifactClient { @@ -374,6 +377,12 @@ impl resources::ResourceHttpClient for GatedArtifactClient { }) }); let result = released.and_then(|_released| { + if let Some(reason) = &self.download_failure { + return Err(resources::ResourcesError::HttpRequestFailed { + url: url.to_owned(), + reason: reason.clone(), + }); + } std::io::Write::write_all(writer, archive).map_err(|error| { resources::ResourcesError::DownloadWriteFailed { url: url.to_owned(), @@ -1686,7 +1695,10 @@ async fn system_resource_reconciliation_stops_unlinked_project_runtime() -> Resu None, &demanded_tracks, DaemonDownloadProgress::disabled(), - crate::project_env::ProjectApplyStage::CompleteApply, + crate::project_env::ProjectApplyOptions::new( + crate::project_env::ProjectApplyStage::CompleteApply, + None, + ), ) .await?; @@ -1865,6 +1877,7 @@ fn system_setup_default_downloads_are_parallel_and_bounded_at_four() -> Result<( maximum_active_downloads: Arc::clone(&maximum_active_downloads), download_started_sender, release_downloads: Arc::clone(&release_downloads), + download_failure: None, }); let http_client: Arc = client; let install_paths = paths.clone(); @@ -2108,7 +2121,7 @@ async fn project_allocation_failure_preserves_shared_runtime_health() -> Result< &catalog, &BTreeSet::new(), crate::jobs::DaemonDownloadProgress::disabled(), - super::ArtifactInstall::Allowed, + super::ResourceReconciliationOptions::new(super::ArtifactInstall::Allowed, None), ) .await; let Err(DaemonError::State(StateError::InvalidEnvJson { .. })) = result else { @@ -2130,6 +2143,373 @@ async fn project_allocation_failure_preserves_shared_runtime_health() -> Result< Ok(()) } +#[tokio::test] +async fn fallback_after_resource_artifact_install_skips_runtime_preparation() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project = link_project( + &paths, + &tempdir.path().join("project"), + "acme.test", + "serve: false\nmysql:\n version: \"8.0\"\n", + )?; + let fixture = SetupDefaultFixture { + resource_name: "mysql", + track: FAKE_SQL_TRACK, + artifact_version: FAKE_SQL_ARTIFACT_VERSION, + archive_file_name: "mysql-8.0.0-pv1-any.tar.gz", + executable_relative_path: "bin/pv-fake-sql", + support_files: &[], + }; + let (manifest, archives) = remote_setup_default_fixtures(tempdir.path(), &[fixture])?; + let manifest_requests = Arc::new(AtomicUsize::new(0)); + let active_downloads = Arc::new(AtomicUsize::new(0)); + let maximum_active_downloads = Arc::new(AtomicUsize::new(0)); + let release_downloads = Arc::new((Mutex::new(false), Condvar::new())); + let (download_started_sender, download_started_receiver) = mpsc::channel(); + let client = Arc::new(GatedArtifactClient { + manifest, + archives, + manifest_requests, + active_downloads, + maximum_active_downloads, + download_started_sender, + release_downloads: Arc::clone(&release_downloads), + download_failure: None, + }); + let gate = Arc::new(ReadinessWaveGate::with_gated_preparation(FAKE_SQL_TRACK)); + let allocation_events = Arc::new(Mutex::new(Vec::new())); + let mut catalog = super::ManagedResourceRuntimeCatalog::with_adapter( + super::ManagedResourceInstallOptions { + manifest_url: TEST_ARTIFACT_MANIFEST_URL.to_owned(), + target_platform: resources::TargetPlatform::current()?, + }, + GatedSqlRuntimeAdapter::new(Arc::clone(&gate), Arc::clone(&allocation_events))?, + ); + catalog.http_client = Some(client); + let (fallback_sender, fallback_receiver) = watch::channel(false); + + let reconciliation = super::reconcile_persisted_resource_track_for_projects_with_progress( + &paths, + "mysql", + FAKE_SQL_TRACK, + Some(&catalog), + std::slice::from_ref(&project), + DaemonDownloadProgress::disabled(), + Some(&fallback_receiver), + ); + tokio::pin!(reconciliation); + let download_started = tokio::task::spawn_blocking(move || { + download_started_receiver.recv_timeout(Duration::from_secs(5)) + }); + tokio::select! { + result = &mut reconciliation => { + bail!("resource reconciliation finished before artifact cancellation: {result:#?}"); + } + started = download_started => { + started??; + } + } + fallback_sender.send(true)?; + { + let mut released = release_downloads + .0 + .lock() + .map_err(|_poison| anyhow!("download release gate lock poisoned"))?; + *released = true; + release_downloads.1.notify_all(); + } + gate.preparation.add_permits(1); + let (installed, failures) = timeout(Duration::from_secs(5), &mut reconciliation).await??; + + assert!(installed); + assert!(failures.is_empty()); + assert_eq!(gate.preparation_started.load(Ordering::SeqCst), 0); + assert_eq!(gate.started.load(Ordering::SeqCst), 0); + assert!(cloned_hook_events(&allocation_events)?.is_empty()); + assert!( + Database::open(&paths)? + .managed_resource_track("mysql", FAKE_SQL_TRACK)? + .installed_version + .is_some() + ); + assert!(!paths.resource_pid("mysql", FAKE_SQL_TRACK).exists()); + assert!( + !paths + .resource_runtime_metadata("mysql", FAKE_SQL_TRACK) + .exists() + ); + assert!( + !paths + .resource_runtime_config("mysql", FAKE_SQL_TRACK) + .exists() + ); + + Ok(()) +} + +#[tokio::test] +async fn artifact_failure_completed_during_fallback_outranks_cancellation() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project = link_project( + &paths, + &tempdir.path().join("project"), + "acme.test", + "serve: false\nmysql:\n version: \"8.0\"\n", + )?; + let fixture = SetupDefaultFixture { + resource_name: "mysql", + track: FAKE_SQL_TRACK, + artifact_version: FAKE_SQL_ARTIFACT_VERSION, + archive_file_name: "mysql-8.0.0-pv1-any.tar.gz", + executable_relative_path: "bin/pv-fake-sql", + support_files: &[], + }; + let (manifest, archives) = remote_setup_default_fixtures(tempdir.path(), &[fixture])?; + let release_downloads = Arc::new((Mutex::new(false), Condvar::new())); + let (download_started_sender, download_started_receiver) = mpsc::channel(); + let client = Arc::new(GatedArtifactClient { + manifest, + archives, + manifest_requests: Arc::new(AtomicUsize::new(0)), + active_downloads: Arc::new(AtomicUsize::new(0)), + maximum_active_downloads: Arc::new(AtomicUsize::new(0)), + download_started_sender, + release_downloads: Arc::clone(&release_downloads), + download_failure: Some("fixture download failed".to_owned()), + }); + let gate = Arc::new(ReadinessWaveGate::with_gated_preparation(FAKE_SQL_TRACK)); + let allocation_events = Arc::new(Mutex::new(Vec::new())); + let mut catalog = super::ManagedResourceRuntimeCatalog::with_adapter( + super::ManagedResourceInstallOptions { + manifest_url: TEST_ARTIFACT_MANIFEST_URL.to_owned(), + target_platform: resources::TargetPlatform::current()?, + }, + GatedSqlRuntimeAdapter::new(Arc::clone(&gate), Arc::clone(&allocation_events))?, + ); + catalog.http_client = Some(client); + let plan = crate::project_env::ProjectResourcePlan { + resources: vec![ProjectManagedResourceInput { + resource_name: "mysql".to_owned(), + track: FAKE_SQL_TRACK.to_owned(), + }], + allocations: BTreeMap::new(), + }; + let (fallback_sender, fallback_receiver) = watch::channel(false); + let mut database = Database::open(&paths)?; + let demanded_tracks = BTreeSet::new(); + let reconciliation = super::reconcile_project_resources_with_catalog_and_progress( + &paths, + &mut database, + &project, + &plan, + &catalog, + &demanded_tracks, + DaemonDownloadProgress::disabled(), + super::ResourceReconciliationOptions::new( + super::ArtifactInstall::Allowed, + Some(&fallback_receiver), + ), + ); + tokio::pin!(reconciliation); + let download_started = tokio::task::spawn_blocking(move || { + download_started_receiver.recv_timeout(Duration::from_secs(5))?; + Ok::<_, mpsc::RecvTimeoutError>(download_started_receiver) + }); + let _download_started_receiver = tokio::select! { + result = &mut reconciliation => { + bail!("resource reconciliation finished before artifact failure cancellation: {result:#?}"); + } + started = download_started => { + started?? + } + }; + fallback_sender.send(true)?; + { + let mut released = release_downloads + .0 + .lock() + .map_err(|_poison| anyhow!("download release gate lock poisoned"))?; + *released = true; + release_downloads.1.notify_all(); + } + gate.preparation.add_permits(1); + let result = timeout(Duration::from_secs(5), &mut reconciliation).await?; + + let error = match result { + Err(error) => error, + Ok(outcome) => { + bail!("artifact failure did not outrank fallback cancellation: {outcome:#?}"); + } + }; + assert!( + format!("{error:#?}").contains("fixture download failed"), + "fallback masked the artifact failure: {error:#?}" + ); + assert_eq!(gate.preparation_started.load(Ordering::SeqCst), 0); + assert_eq!(gate.started.load(Ordering::SeqCst), 0); + assert!(cloned_hook_events(&allocation_events)?.is_empty()); + assert!(!paths.resource_pid("mysql", FAKE_SQL_TRACK).exists()); + assert!( + !paths + .resource_runtime_metadata("mysql", FAKE_SQL_TRACK) + .exists() + ); + assert!( + !paths + .resource_runtime_config("mysql", FAKE_SQL_TRACK) + .exists() + ); + + Ok(()) +} + +#[tokio::test] +async fn fallback_during_project_resource_readiness_preserves_pending_env_state() -> Result<()> { + assert_fallback_during_project_resource_readiness(None).await +} + +#[tokio::test] +async fn readiness_error_during_fallback_outranks_project_cancellation() -> Result<()> { + assert_fallback_during_project_resource_readiness(Some("fixture readiness failed")).await +} + +async fn assert_fallback_during_project_resource_readiness( + readiness_error: Option<&str>, +) -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project = link_project( + &paths, + &tempdir.path().join("project"), + "acme.test", + "serve: false\nenv:\n REVISION: changed\nmysql:\n version: \"8.0\"\n", + )?; + let baseline_env = "REVISION=baseline\n"; + state::fs::write_sensitive_file(&project.path.join(".env"), baseline_env)?; + seed_fake_sql_artifact(&paths, "mysql", FAKE_SQL_TRACK)?; + let port_guard = TcpListener::bind(("127.0.0.1", 0))?; + let port = port_guard.local_addr()?.port(); + let mut database = Database::open(&paths)?; + database.assign_port( + PortRequest::resource_port("mysql", FAKE_SQL_TRACK, "mysql", port, port, port), + |candidate| candidate == port, + )?; + let project_before = database + .project_by_id(&project.id)? + .ok_or_else(|| anyhow!("missing linked Project"))?; + let observed_before = database.project_env_observed_state(&project.id)?; + drop(database); + drop(port_guard); + + let gate = Arc::new(match readiness_error { + Some(reason) => ReadinessWaveGate::with_failing_readiness(reason), + None => ReadinessWaveGate::new(), + }); + let allocation_events = Arc::new(Mutex::new(Vec::new())); + let catalog = super::ManagedResourceRuntimeCatalog::with_adapter( + super::ManagedResourceInstallOptions { + manifest_url: OFFLINE_TEST_MANIFEST_URL.to_owned(), + target_platform: resources::TargetPlatform::current()?, + }, + GatedSqlRuntimeAdapter::new(Arc::clone(&gate), Arc::clone(&allocation_events))?, + ); + let (fallback_sender, fallback_receiver) = watch::channel(false); + let demanded_tracks = BTreeSet::new(); + let reconciliation = reconcile_project_env_with_runtime_catalog_and_progress_outcome( + &paths, + &project.id, + Some(&catalog), + None, + &demanded_tracks, + DaemonDownloadProgress::disabled(), + ProjectApplyOptions::new(ProjectApplyStage::CompleteApply, Some(&fallback_receiver)), + ); + tokio::pin!(reconciliation); + + timeout(Duration::from_secs(5), async { + while gate.started.load(Ordering::SeqCst) == 0 { + tokio::select! { + result = &mut reconciliation => { + return Err(anyhow!("Project reconciliation finished before readiness cancellation: {result:#?}")); + } + () = tokio::task::yield_now() => {} + } + } + Ok(()) + }) + .await??; + fallback_sender.send(true)?; + gate.proceed.add_permits(1); + gate.finish.add_permits(1); + let result = timeout(Duration::from_secs(5), &mut reconciliation).await?; + + let database = Database::open(&paths)?; + let env_after = read_dotenv(&project)?; + let observed_after = database.project_env_observed_state(&project.id)?; + let project_after = database + .project_by_id(&project.id)? + .ok_or_else(|| anyhow!("missing linked Project after cancellation"))?; + drop(database); + let runtime_files_after_reconciliation = + runtime_files_exist_for_resource(&paths, "mysql", FAKE_SQL_TRACK)?; + let supervisor = ProcessSupervisor::new(paths.clone()); + if let Some(adopted) = supervisor.adopt_recorded( + &paths.resource_pid("mysql", FAKE_SQL_TRACK), + &paths.resource_runtime_metadata("mysql", FAKE_SQL_TRACK), + )? { + adopted.stop(Duration::from_secs(1)).await?; + } + super::cleanup_resource_runtime_files( + &paths, + &ProjectManagedResourceInput { + resource_name: "mysql".to_owned(), + track: FAKE_SQL_TRACK.to_owned(), + }, + )?; + + match (readiness_error, result) { + (None, Ok(ReconciliationOutcome::Cancelled)) => { + assert_eq!(observed_after, observed_before); + } + ( + Some(expected), + Err(DaemonError::ReadinessTimedOut { + last_error: Some(actual), + .. + }), + ) => { + assert!(actual.contains(expected)); + let observed = observed_after + .ok_or_else(|| anyhow!("readiness failure was not recorded for the Project"))?; + assert_eq!(observed.status, ProjectEnvObservedStatus::Failed); + assert!( + observed + .message + .as_deref() + .is_some_and(|message| message.contains(expected)) + ); + assert_eq!( + runtime_files_after_reconciliation, + RuntimeFilePresence { + pid: false, + metadata: false, + config: false, + } + ); + } + (expected, actual) => { + bail!("unexpected readiness outcome for {expected:?}: {actual:#?}"); + } + } + assert_eq!(env_after, baseline_env); + assert_eq!(project_after, project_before); + assert!(cloned_hook_events(&allocation_events)?.is_empty()); + + Ok(()) +} + #[tokio::test] async fn project_download_failures_follow_original_plan_order() -> Result<()> { let tempdir = tempdir()?; @@ -2172,7 +2552,7 @@ async fn project_download_failures_follow_original_plan_order() -> Result<()> { &catalog, &BTreeSet::new(), crate::jobs::DaemonDownloadProgress::disabled(), - super::ArtifactInstall::Allowed, + super::ResourceReconciliationOptions::new(super::ArtifactInstall::Allowed, None), ) .await; let Err(DaemonError::ManagedResourceProjectFailures { failures }) = result else { @@ -2274,7 +2654,10 @@ async fn project_php_demand_does_not_overwrite_concurrent_pair_removal() -> Resu None, &BTreeSet::new(), crate::jobs::DaemonDownloadProgress::disabled(), - crate::project_env::ProjectApplyStage::RecordRequirements, + crate::project_env::ProjectApplyOptions::new( + crate::project_env::ProjectApplyStage::RecordRequirements, + None, + ), ) .await; if crate::project_env::clear_project_php_demand_test_barrier(track) { @@ -2352,7 +2735,7 @@ async fn project_manifest_failure_preserves_earlier_installed_resource_work() -> &catalog, &BTreeSet::new(), crate::jobs::DaemonDownloadProgress::disabled(), - super::ArtifactInstall::Allowed, + super::ResourceReconciliationOptions::new(super::ArtifactInstall::Allowed, None), ) .await; let states = database.runtime_observed_states()?; @@ -2368,7 +2751,7 @@ async fn project_manifest_failure_preserves_earlier_installed_resource_work() -> &catalog, &BTreeSet::new(), crate::jobs::DaemonDownloadProgress::disabled(), - super::ArtifactInstall::Allowed, + super::ResourceReconciliationOptions::new(super::ArtifactInstall::Allowed, None), ) .await?; let Err(DaemonError::ManagedResourceProjectFailures { failures }) = result else { @@ -2487,7 +2870,10 @@ async fn project_application_pins_resource_track_until_selector_changes() -> Res Some(&demand), &demand.resource_tracks, progress.clone(), - crate::project_env::ProjectApplyStage::CompleteApply, + crate::project_env::ProjectApplyOptions::new( + crate::project_env::ProjectApplyStage::CompleteApply, + None, + ), ) .await; let database = Database::open(&paths)?; @@ -2514,7 +2900,10 @@ async fn project_application_pins_resource_track_until_selector_changes() -> Res Some(&demand), &demand.resource_tracks, progress, - crate::project_env::ProjectApplyStage::CompleteApply, + crate::project_env::ProjectApplyOptions::new( + crate::project_env::ProjectApplyStage::CompleteApply, + None, + ), ) .await; let mut database = Database::open(&paths)?; @@ -4521,6 +4910,7 @@ async fn resource_readiness_slots_include_start_and_poll_during_preparation() -> progress: &progress, prefetched_installs: &mut prefetched_installs, artifact_install: super::ArtifactInstall::Allowed, + fallback_shutdown: None, }; let reconciliation = super::reconcile_resource_tracks(&paths, &mut database, &mut context, &project, &plan); @@ -4645,6 +5035,7 @@ async fn resource_readiness_wave_recovers_after_cancellation_and_stays_db_free() progress: &progress, prefetched_installs: &mut prefetched_installs, artifact_install: super::ArtifactInstall::Allowed, + fallback_shutdown: None, }; let reconciliation = super::reconcile_resource_tracks(&paths, &mut database, &mut context, &project, &plan); @@ -4726,6 +5117,7 @@ async fn resource_readiness_wave_recovers_after_cancellation_and_stays_db_free() progress: &progress, prefetched_installs: &mut prefetched_installs, artifact_install: super::ArtifactInstall::Allowed, + fallback_shutdown: None, }; let result = { let reconciliation = @@ -7265,6 +7657,7 @@ struct ReadinessWaveGate { preparation_track: Option, preparation_started: AtomicUsize, preparation: Arc, + readiness_error: Option, } impl ReadinessWaveGate { @@ -7279,6 +7672,7 @@ impl ReadinessWaveGate { preparation_track: None, preparation_started: AtomicUsize::new(0), preparation: Arc::new(Semaphore::new(0)), + readiness_error: None, } } @@ -7288,6 +7682,13 @@ impl ReadinessWaveGate { ..Self::new() } } + + fn with_failing_readiness(reason: &str) -> Self { + Self { + readiness_error: Some(reason.to_owned()), + ..Self::new() + } + } } #[derive(Clone)] @@ -7331,6 +7732,14 @@ impl super::ManagedResourceRuntimeAdapter for GatedSqlRuntimeAdapter { }] } + fn readiness_timeout(&self) -> Duration { + if self.gate.readiness_error.is_some() { + Duration::from_millis(100) + } else { + super::RESOURCE_READINESS_TIMEOUT + } + } + fn prepare_runtime<'a>( &'a self, _paths: &'a PvPaths, @@ -7384,13 +7793,20 @@ impl super::ManagedResourceRuntimeAdapter for GatedSqlRuntimeAdapter { let gate = Arc::clone(&gate); Box::pin(async move { - gate.started.fetch_add(1, Ordering::SeqCst); - let active = gate.active.fetch_add(1, Ordering::SeqCst) + 1; - gate.maximum_active.fetch_max(active, Ordering::SeqCst); - acquire_test_gate(Arc::clone(&gate.proceed)).await?; - gate.ready_to_return.fetch_add(1, Ordering::SeqCst); - acquire_test_gate(Arc::clone(&gate.finish)).await?; - gate.active.fetch_sub(1, Ordering::SeqCst); + let attempt = gate.started.fetch_add(1, Ordering::SeqCst); + if gate.readiness_error.is_none() || attempt == 0 { + let active = gate.active.fetch_add(1, Ordering::SeqCst) + 1; + gate.maximum_active.fetch_max(active, Ordering::SeqCst); + acquire_test_gate(Arc::clone(&gate.proceed)).await?; + gate.ready_to_return.fetch_add(1, Ordering::SeqCst); + acquire_test_gate(Arc::clone(&gate.finish)).await?; + gate.active.fetch_sub(1, Ordering::SeqCst); + } + if let Some(reason) = &gate.readiness_error { + return Err(crate::DaemonError::UnexpectedProtocolResponse { + reason: reason.clone(), + }); + } Ok(()) }) diff --git a/crates/daemon/src/project_env.rs b/crates/daemon/src/project_env.rs index 6bbb55e5..51c447a4 100644 --- a/crates/daemon/src/project_env.rs +++ b/crates/daemon/src/project_env.rs @@ -18,8 +18,10 @@ use state::{ ProjectMode, ProjectPhpRuntimeInput, ProjectReconciliationStateInput, ProjectRecord, PvPaths, ResourceAllocationInput, ResourceAllocationRecord, ResourceAllocationStatus, StateError, }; +use tokio::sync::watch; use crate::DaemonError; +use crate::gateway::ReconciliationOutcome; use crate::jobs::DaemonDownloadProgress; use crate::managed_resources::ManagedResourceRuntimeCatalog; use crate::structured_log; @@ -79,6 +81,24 @@ pub(crate) enum ProjectApplyStage { CompleteApply, } +#[derive(Clone, Copy)] +pub(crate) struct ProjectApplyOptions<'a> { + stage: ProjectApplyStage, + fallback_shutdown: Option<&'a watch::Receiver>, +} + +impl<'a> ProjectApplyOptions<'a> { + pub(crate) fn new( + stage: ProjectApplyStage, + fallback_shutdown: Option<&'a watch::Receiver>, + ) -> Self { + Self { + stage, + fallback_shutdown, + } + } +} + impl ProjectApplyStage { fn artifact_install(self) -> crate::managed_resources::ArtifactInstall { match self { @@ -214,11 +234,12 @@ pub(crate) async fn reconcile_project_env( None, &BTreeSet::new(), DaemonDownloadProgress::disabled(), - ProjectApplyStage::CompleteApply, + ProjectApplyOptions::new(ProjectApplyStage::CompleteApply, None), ) .await } +#[cfg(test)] pub(crate) async fn reconcile_project_env_with_runtime_catalog_and_progress( paths: &PvPaths, project_id: &str, @@ -226,31 +247,54 @@ pub(crate) async fn reconcile_project_env_with_runtime_catalog_and_progress( discovered_demand: Option<&ProjectDemand>, demanded_tracks: &BTreeSet, progress: DaemonDownloadProgress, - stage: ProjectApplyStage, + options: ProjectApplyOptions<'_>, ) -> Result { + reconcile_project_env_with_runtime_catalog_and_progress_outcome( + paths, + project_id, + catalog, + discovered_demand, + demanded_tracks, + progress, + options, + ) + .await? + .into_completed() +} + +pub(crate) async fn reconcile_project_env_with_runtime_catalog_and_progress_outcome( + paths: &PvPaths, + project_id: &str, + catalog: Option<&ManagedResourceRuntimeCatalog>, + discovered_demand: Option<&ProjectDemand>, + demanded_tracks: &BTreeSet, + progress: DaemonDownloadProgress, + options: ProjectApplyOptions<'_>, +) -> Result, DaemonError> { let mut database = Database::open(paths)?; - let result: Result = async { - let project = - database - .project_by_id(project_id)? - .ok_or_else(|| StateError::ProjectNotFound { - target: project_id.to_string(), - })?; - reconcile_loaded_project( - paths, - &mut database, - &project, - catalog, - discovered_demand, - demanded_tracks, - progress, - stage, - ) - .await - } - .await; + let result: Result, DaemonError> = + async { + let project = + database + .project_by_id(project_id)? + .ok_or_else(|| StateError::ProjectNotFound { + target: project_id.to_string(), + })?; + reconcile_loaded_project( + paths, + &mut database, + &project, + catalog, + discovered_demand, + demanded_tracks, + progress, + options, + ) + .await + } + .await; match result { - Ok(summary) => Ok(summary), + Ok(outcome) => Ok(outcome), // No project row exists, so there is no env to attribute a failure to; // propagate bare as reviewed instead of recording against nothing. Err(reconciliation @ DaemonError::State(StateError::ProjectNotFound { .. })) => { @@ -367,11 +411,11 @@ pub(crate) async fn reconcile_project_env_with_catalog( None, &BTreeSet::new(), DaemonDownloadProgress::disabled(), - ProjectApplyStage::CompleteApply, + ProjectApplyOptions::new(ProjectApplyStage::CompleteApply, None), ) .await { - Ok(summary) => Ok(summary), + Ok(outcome) => outcome.into_completed(), Err(error) => { let message = error.to_string(); record_project_env_failure(database, &project.id, &message)?; @@ -533,8 +577,15 @@ async fn reconcile_loaded_project( discovered_demand: Option<&ProjectDemand>, demanded_tracks: &BTreeSet, progress: DaemonDownloadProgress, - stage: ProjectApplyStage, -) -> Result { + options: ProjectApplyOptions<'_>, +) -> Result, DaemonError> { + let ProjectApplyOptions { + stage, + fallback_shutdown, + } = options; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let config_file = match ProjectConfigFile::read_from_root(&project.path) { Ok(config_file) => config_file, Err(error) => { @@ -612,6 +663,9 @@ async fn reconcile_loaded_project( source: Box::new(error), }); } + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } let resolved_php_runtime = match php_track .map(|track| { resolve_project_php_runtime_for_track(database, config_file.config.php.as_ref(), track) @@ -646,6 +700,9 @@ async fn reconcile_loaded_project( // Record Requirements resolves what the Project needs and returns it; it does not replace // usage, because the Resources phase has not installed the artifacts yet. if stage != ProjectApplyStage::RecordRequirements { + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } // Fail before replacing usage when an artifact the Resources phase should have // installed is missing, so the runtime this apply would replace stays demanded. if stage.artifact_install() == crate::managed_resources::ArtifactInstall::Forbidden { @@ -660,8 +717,10 @@ async fn reconcile_loaded_project( // The Resources phase owns artifact work, so a staged apply refuses to install and an // unstaged apply installs what its caller never provisioned. - let resource_result = if stage == ProjectApplyStage::RecordRequirements { - Ok(()) + let resource_outcome = if stage == ProjectApplyStage::RecordRequirements { + ReconciliationOutcome::Completed(()) + } else if fallback_shutdown_requested(fallback_shutdown) { + ReconciliationOutcome::Cancelled } else if let Some(catalog) = catalog { crate::managed_resources::reconcile_project_resources_with_catalog_and_progress( paths, @@ -671,9 +730,12 @@ async fn reconcile_loaded_project( catalog, demanded_tracks, progress, - stage.artifact_install(), + crate::managed_resources::ResourceReconciliationOptions::new( + stage.artifact_install(), + fallback_shutdown, + ), ) - .await + .await? } else { crate::managed_resources::reconcile_project_resources_with_progress( paths, @@ -682,22 +744,30 @@ async fn reconcile_loaded_project( &plan, demanded_tracks, progress, - stage.artifact_install(), + crate::managed_resources::ResourceReconciliationOptions::new( + stage.artifact_install(), + fallback_shutdown, + ), ) - .await + .await? }; - resource_result?; + if matches!(resource_outcome, ReconciliationOutcome::Cancelled) + || fallback_shutdown_requested(fallback_shutdown) + { + return Ok(ReconciliationOutcome::Cancelled); + } let runtime_warnings = resolved_php_runtime .as_ref() .map(|runtime| ignored_php_extension_warnings(&runtime.ignored_extensions)) .unwrap_or_default(); - Ok::<_, DaemonError>((plan, runtime_warnings)) + Ok::<_, DaemonError>(ReconciliationOutcome::Completed((plan, runtime_warnings))) } .await; - let (plan, runtime_warnings) = match pre_render_result { - Ok(values) => values, + let completed = match pre_render_result { + Ok(ReconciliationOutcome::Completed(values)) => Some(values), + Ok(ReconciliationOutcome::Cancelled) => None, Err(error) => { if let Some(Err(tls_error)) = tls_maintenance_result.as_ref() { structured_log::project_tls_maintenance_failed( @@ -713,13 +783,21 @@ async fn reconcile_loaded_project( if let Some(Err(error)) = tls_maintenance_result { return Err(error); } + let Some((plan, runtime_warnings)) = completed else { + return Ok(ReconciliationOutcome::Cancelled); + }; + if fallback_shutdown_requested(fallback_shutdown) { + return Ok(ReconciliationOutcome::Cancelled); + } if stage == ProjectApplyStage::RecordRequirements { - return Ok(ProjectEnvReconciliationSummary { - message: "project requirements recorded", - requested_php_extensions, - recorded_tracks, - }); + return Ok(ReconciliationOutcome::Completed( + ProjectEnvReconciliationSummary { + message: "project requirements recorded", + requested_php_extensions, + recorded_tracks, + }, + )); } let context = has_env_mappings @@ -749,11 +827,17 @@ async fn reconcile_loaded_project( &rendered.warnings, )?; - Ok(ProjectEnvReconciliationSummary { - message: rendered.summary, - requested_php_extensions, - recorded_tracks: BTreeSet::new(), - }) + Ok(ReconciliationOutcome::Completed( + ProjectEnvReconciliationSummary { + message: rendered.summary, + requested_php_extensions, + recorded_tracks: BTreeSet::new(), + }, + )) +} + +fn fallback_shutdown_requested(shutdown: Option<&watch::Receiver>) -> bool { + shutdown.is_some_and(|shutdown| *shutdown.borrow()) } /// The resource tracks a Record Requirements pass resolves from the current config: diff --git a/crates/daemon/src/server.rs b/crates/daemon/src/server.rs index bbcc8556..b0a405df 100644 --- a/crates/daemon/src/server.rs +++ b/crates/daemon/src/server.rs @@ -33,6 +33,7 @@ pub(crate) async fn serve( paths: PvPaths, listener: LocalListener, mut shutdown: oneshot::Receiver<()>, + fallback_shutdown: watch::Receiver, runtime_catalog: Option>, ) -> Result<(), DaemonError> { let mut connections = JoinSet::new(); @@ -41,6 +42,7 @@ pub(crate) async fn serve( let startup_paths = paths.clone(); let startup_queue = queue.clone(); let startup_runtime_catalog = runtime_catalog.clone(); + let startup_fallback_shutdown = fallback_shutdown.clone(); let (startup_shutdown, startup_shutdown_receiver) = oneshot::channel(); let mut startup_shutdown = Some(startup_shutdown); let mut startup_task = Some(tokio::spawn(async move { @@ -49,6 +51,7 @@ pub(crate) async fn serve( startup_queue, startup_runtime_catalog.as_deref(), startup_shutdown_receiver, + startup_fallback_shutdown, ) .await })); @@ -150,11 +153,13 @@ pub(crate) async fn serve( Ok(Some(EnqueueResult::Queued(queued))) => { let completion_paths = paths.clone(); let completion_runtime_catalog = runtime_catalog.clone(); - let _task = tokio::spawn(async move { + let completion_fallback_shutdown = fallback_shutdown.clone(); + background_tasks.spawn(async move { let result = complete_queued_background_reconciliation_job( &completion_paths, queued, completion_runtime_catalog.as_deref(), + Some(&completion_fallback_shutdown), ) .await; let _result = handle_background_reconciliation_result( @@ -210,6 +215,7 @@ pub(crate) async fn serve( let task_queue = queue.clone(); let task_runtime_catalog = runtime_catalog.clone(); let task_shutdown = background_shutdown_receiver.clone(); + let task_fallback_shutdown = fallback_shutdown.clone(); background_tasks.spawn(async move { let scope_text = scope.to_string(); @@ -219,6 +225,7 @@ pub(crate) async fn serve( scope, task_runtime_catalog.as_deref(), task_shutdown, + task_fallback_shutdown, ) .await; let _result = handle_background_reconciliation_result( @@ -239,6 +246,7 @@ pub(crate) async fn serve( let connection_paths = paths.clone(); let connection_queue = queue.clone(); let connection_runtime_catalog = runtime_catalog.clone(); + let connection_fallback_shutdown = fallback_shutdown.clone(); connections.spawn(async move { handle_connection( @@ -246,6 +254,7 @@ pub(crate) async fn serve( connection_queue, stream, connection_runtime_catalog, + connection_fallback_shutdown, ) .await }); @@ -284,7 +293,9 @@ pub(crate) async fn serve( let _send_result = background_shutdown.send(true); let startup_result = stop_startup_task(&paths, startup_shutdown.take(), startup_task.take()).await; - connections.abort_all(); + if !*fallback_shutdown.borrow() { + connections.abort_all(); + } while background_tasks.join_next().await.is_some() {} while connections.join_next().await.is_some() {} @@ -298,6 +309,7 @@ async fn run_debounced_reconciliation_job( scope: ReconciliationScope, runtime_catalog: Option<&ManagedResourceRuntimeCatalog>, mut shutdown: watch::Receiver, + fallback_shutdown: watch::Receiver, ) -> Result<(), BackgroundReconciliationError> { let result = loop { match enqueue_background_reconciliation_job(&paths, &queue, scope.clone()) { @@ -328,7 +340,14 @@ async fn run_debounced_reconciliation_job( running = queued.wait_for_turn() => running, }; - complete_running_background_reconciliation_job(&paths, running, runtime_catalog, None).await + complete_running_background_reconciliation_job( + &paths, + running, + runtime_catalog, + None, + Some(&fallback_shutdown), + ) + .await } /// Resolves once daemon shutdown is requested, or once the shutdown signal can no @@ -432,9 +451,15 @@ async fn handle_connection( queue: ReconciliationQueue, stream: LocalStream, runtime_catalog: Option>, + mut fallback_shutdown: watch::Receiver, ) -> Result<(), DaemonError> { let mut transport = protocol::transport(stream); - let Some(line) = read_request_line(&mut transport, REQUEST_LINE_TIMEOUT).await? else { + let line = tokio::select! { + biased; + _ = wait_for_fallback_shutdown(&mut fallback_shutdown) => return Ok(()), + line = read_request_line(&mut transport, REQUEST_LINE_TIMEOUT) => line?, + }; + let Some(line) = line else { return Ok(()); }; let request = serde_json::from_str::(&line)?; @@ -463,6 +488,7 @@ async fn handle_connection( &kind, &scope, runtime_catalog.as_deref(), + &fallback_shutdown, ) .await } @@ -494,6 +520,12 @@ async fn handle_connection( } } +async fn wait_for_fallback_shutdown(shutdown: &mut watch::Receiver) { + if shutdown.wait_for(|requested| *requested).await.is_err() { + std::future::pending::<()>().await; + } +} + async fn read_request_line( transport: &mut DaemonTransport, read_timeout: Duration, @@ -700,6 +732,7 @@ mod tests { let task_paths = paths.clone(); let scope = ReconciliationScope::project(project.id)?; let (_shutdown_sender, shutdown_receiver) = watch::channel(false); + let (_fallback_sender, fallback_receiver) = watch::channel(false); let task = tokio::spawn(async move { run_debounced_reconciliation_job( task_paths, @@ -707,6 +740,7 @@ mod tests { scope, None, shutdown_receiver, + fallback_receiver, ) .await }); @@ -739,6 +773,7 @@ mod tests { let task_paths = paths.clone(); let scope = ReconciliationScope::project(project.id)?; let (shutdown_sender, shutdown_receiver) = watch::channel(false); + let (_fallback_sender, fallback_receiver) = watch::channel(false); let task = tokio::spawn(async move { run_debounced_reconciliation_job( task_paths, @@ -746,6 +781,7 @@ mod tests { scope, None, shutdown_receiver, + fallback_receiver, ) .await }); diff --git a/crates/daemon/src/structured_log.rs b/crates/daemon/src/structured_log.rs index c155d2a4..1114373a 100644 --- a/crates/daemon/src/structured_log.rs +++ b/crates/daemon/src/structured_log.rs @@ -310,6 +310,37 @@ pub(crate) fn job_failure_recording_failed( } } +pub(crate) fn job_completion_recording_failed( + paths: &PvPaths, + job_id: &str, + kind: &str, + scope: &str, + summary: &str, + recording_error: &str, +) { + let result = append( + paths, + "error", + "reconciliation", + "job_completion_recording_failed", + "failed to persist job completion", + &[ + ("job_id", job_id), + ("kind", kind), + ("scope", scope), + ("summary", summary), + ("recording_error", recording_error), + ], + ); + if let Err(log_error) = result { + let mut standard_error = io::stderr().lock(); + let _fallback_result = writeln!( + standard_error, + "PV {kind} job {job_id} completed with {summary}; failed to persist the completion with {recording_error} and to write the daemon log with {log_error}" + ); + } +} + pub(crate) fn job_abandonment_failed(paths: &PvPaths, job_id: &str, kind: &str, error: &str) { let result = append( paths, diff --git a/crates/daemon/tests/daemon_foundation.rs b/crates/daemon/tests/daemon_foundation.rs index c4ec6f23..55ecd476 100644 --- a/crates/daemon/tests/daemon_foundation.rs +++ b/crates/daemon/tests/daemon_foundation.rs @@ -1,4 +1,4 @@ -use anyhow::{Result, anyhow}; +use anyhow::{Context, Result, anyhow}; use camino::{Utf8Path, Utf8PathBuf}; use camino_tempfile::tempdir; use hickory_proto::op::{Message, MessageType, OpCode, Query, ResponseCode}; @@ -8,25 +8,37 @@ use hickory_proto::serialize::binary::BinEncodable; use insta::{Settings, assert_debug_snapshot}; use rcgen::generate_simple_self_signed; use rusqlite::{Connection, params}; +#[cfg(unix)] +use rustix::fs::FlockOperation; +use rustix::io::Errno; +#[cfg(target_os = "macos")] +use rustix::process::getpgid; +use rustix::process::{Pid, test_kill_process, test_kill_process_group}; use serde_json::{Value, json}; use state::{ AppReleaseLayout, DNS_PREFERRED_PORT, Database, GatewayPort, JobRecord, JobStatus, JobsLock, LinkProjectInput, PortOwner, PortRequest, PvPaths, RUNTIME_PORT_FALLBACK_END, RUNTIME_PORT_FALLBACK_START, RuntimeObservedStatus, RuntimeSubject, UpdateLock, }; +use std::collections::BTreeMap; +use std::future::Future; use std::io::{self, ErrorKind, Write as _}; use std::net::{Ipv4Addr, SocketAddr, TcpListener as StdTcpListener, UdpSocket as StdUdpSocket}; +#[cfg(unix)] +use std::os::fd::OwnedFd; use std::str::FromStr; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex, mpsc}; use std::time::{Duration, Instant}; use tokio::io::{AsyncBufReadExt, AsyncReadExt, AsyncWriteExt, BufReader}; use tokio::net::{TcpStream, UdpSocket, UnixListener, UnixStream}; -use tokio::time::{sleep, timeout}; +use tokio::time::{Instant as TokioInstant, sleep, timeout, timeout_at}; const EXPECTED_DNS_TTL_SECONDS: u32 = 5; const JOB_STATUS_WAIT_TIMEOUT: Duration = Duration::from_secs(10); const JOB_STATUS_POLL_INTERVAL: Duration = Duration::from_millis(50); +const REQUEST_LINES_TIMEOUT: Duration = Duration::from_secs(30); +const TARGETED_SCENARIO_TIMEOUT: Duration = Duration::from_secs(60); const TEST_ARTIFACT_MANIFEST_URL: &str = "https://artifacts.example.test/manifest.json"; const FAKE_CADDY_SCRIPT: &str = r#"#!/bin/sh set -eu @@ -104,6 +116,10 @@ const FOUNDATION_FAKE_CADDY_SERVER_SCRIPT: &str = include_str!(concat!( )); const SEEDED_GATEWAY_CLEANUP_TIMEOUT: Duration = Duration::from_millis(500); const SEEDED_GATEWAY_CLEANUP_POLL_INTERVAL: Duration = Duration::from_millis(25); +const FALLBACK_SUBPROCESS_HOME: &str = "PV_DAEMON_FALLBACK_SUBPROCESS_HOME"; +const FALLBACK_SUBPROCESS_RELEASE: &str = "PV_DAEMON_FALLBACK_SUBPROCESS_RELEASE"; +#[cfg(unix)] +const FOUNDATION_WORKER_PORT_HANDOFF_LOCK: &str = "daemon-foundation-worker-port-handoff.lock"; #[tokio::test] async fn socket_protocol_streams_job_progress_and_persists_final_status() -> Result<()> { @@ -422,7 +438,8 @@ async fn daemon_shutdown_cancels_startup_reconciliation_waiting_for_jobs_lock() async fn daemon_shutdown_drains_active_startup_reconciliation() -> Result<()> { let tempdir = tempdir()?; let paths = PvPaths::for_home(tempdir.path().join("home")); - let [validation_started, release_validation] = seed_barrier_foundation_caddy(&paths)?; + let [validation_started, release_validation, runtime_started] = + seed_barrier_foundation_caddy(&paths)?; let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); let result = async { @@ -460,6 +477,9 @@ async fn daemon_shutdown_drains_active_startup_reconciliation() -> Result<()> { assert!(shutdown_was_pending); assert_eq!(job.status, JobStatus::Succeeded); assert_eq!(job.error, None); + assert!(runtime_started.exists()); + assert!(paths.gateway_pid().exists()); + assert!(paths.gateway_runtime_metadata().exists()); Ok::<(), anyhow::Error>(()) } @@ -468,6 +488,690 @@ async fn daemon_shutdown_drains_active_startup_reconciliation() -> Result<()> { propagate_after_cleanup(result, cleanup_result) } +#[tokio::test] +async fn seeded_gateway_drop_does_not_block_current_thread_runtime() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + state::fs::ensure_user_dir(paths.home())?; + let output_path = tempdir.path().join("nested.output"); + let nested_pid_path = paths.run().join("nested-test.pid"); + let nested_metadata_path = paths.run().join("nested-test.json"); + let nested_release_path = paths.run().join("nested-test.release"); + let fallback_ready_path = paths.run().join("nested-fallback-ready"); + let fallback_release_path = paths.run().join("nested-fallback-release"); + let mut private_environment = BTreeMap::new(); + private_environment.insert( + FALLBACK_SUBPROCESS_HOME.to_owned(), + paths.home().to_string(), + ); + private_environment.insert( + FALLBACK_SUBPROCESS_RELEASE.to_owned(), + nested_release_path.to_string(), + ); + let mut child = daemon::ProcessSupervisor::new(paths.clone()) + .start(daemon::ProcessSpec { + name: "nested seeded gateway regression".to_owned(), + command: current_test_binary()?, + arguments: vec![ + "--exact".to_owned(), + "seeded_gateway_drop_current_thread_inner".to_owned(), + "--ignored".to_owned(), + "--nocapture".to_owned(), + ], + private_environment, + config_path: paths.home().to_owned(), + config_fingerprint: None, + log_path: output_path.clone(), + pid_path: nested_pid_path.clone(), + metadata_path: nested_metadata_path.clone(), + resource_name: "test".to_owned(), + track: "fallback".to_owned(), + }) + .await?; + if let Err(error) = state::fs::write_sensitive_file(&nested_release_path, "release\n") { + let cleanup_result = async { + child.stop(Duration::from_millis(100)).await?; + state::fs::remove_file_if_exists(&nested_pid_path)?; + state::fs::remove_file_if_exists(&nested_metadata_path)?; + Ok(()) + } + .await; + return propagate_after_cleanup(Err(error.into()), cleanup_result); + } + + let setup_deadline = TokioInstant::now() + TARGETED_SCENARIO_TIMEOUT; + while !fallback_ready_path.exists() && !child.has_exited()? { + if TokioInstant::now() >= setup_deadline { + let cleanup_result = async { + let child_cleanup = async { + child.stop(Duration::from_millis(100)).await?; + state::fs::remove_file_if_exists(&nested_pid_path)?; + state::fs::remove_file_if_exists(&nested_metadata_path)?; + Ok(()) + } + .await; + let runtime_cleanup = emergency_cleanup_seeded_runtimes(&paths).await; + + combine_cleanup_results(child_cleanup, runtime_cleanup) + } + .await; + return propagate_after_cleanup( + Err(anyhow!("nested fallback fixture setup timed out")), + cleanup_result, + ); + } + sleep(Duration::from_millis(20)).await; + } + state::fs::write_sensitive_file(&fallback_release_path, "release\n")?; + let deadline = TokioInstant::now() + Duration::from_secs(5); + + loop { + if child.has_exited()? { + break; + } + if TokioInstant::now() >= deadline { + let mut cleanup_failures = Vec::new(); + if let Err(error) = async { + child.stop(Duration::from_millis(100)).await?; + state::fs::remove_file_if_exists(&nested_pid_path)?; + state::fs::remove_file_if_exists(&nested_metadata_path)?; + Ok::<(), anyhow::Error>(()) + } + .await + { + cleanup_failures.push(format!("child stop failed: {error}")); + } + if let Err(error) = emergency_cleanup_seeded_runtimes(&paths).await { + cleanup_failures.push(format!("fixture cleanup failed: {error}")); + } + let output = state::fs::read_to_string(&output_path).unwrap_or_else(|error| { + cleanup_failures.push(format!("output capture failed: {error}")); + "".to_owned() + }); + let cleanup_diagnostic = if cleanup_failures.is_empty() { + String::new() + } else { + format!("; cleanup failures: {}", cleanup_failures.join("; ")) + }; + return Err(anyhow!( + "nested fallback regression timed out; output={output}{cleanup_diagnostic}" + )); + } + sleep(Duration::from_millis(20)).await; + } + + child.stop(Duration::from_millis(100)).await?; + state::fs::remove_file_if_exists(&nested_pid_path)?; + state::fs::remove_file_if_exists(&nested_metadata_path)?; + let output = state::fs::read_to_string(&output_path)?; + assert!( + output.contains("seeded gateway fallback sentinel"), + "nested test did not report its sentinel; output={output}" + ); + assert!(!paths.daemon_socket().exists()); + assert!(!paths.gateway_pid().exists()); + assert!(!paths.gateway_runtime_metadata().exists()); + let gateway_group = recorded_test_pid(&paths.run().join("captured-gateway-leader.pid"))?; + let gateway_descendant = recorded_test_pid(&paths.run().join("gateway-descendant.pid"))?; + let validation_group = recorded_test_pid(&paths.run().join("worker-validation-group.pid"))?; + let validation_descendant = + recorded_test_pid(&paths.run().join("worker-validation-descendant.pid"))?; + let root_candidate = Utf8PathBuf::from( + state::fs::read_to_string(&paths.run().join("worker-validation-root-candidate.path"))? + .trim(), + ); + let fragment_candidate = Utf8PathBuf::from( + state::fs::read_to_string( + &paths + .run() + .join("worker-validation-fragment-candidate.path"), + )? + .trim(), + ); + assert_eq!(test_kill_process_group(gateway_group), Err(Errno::SRCH)); + assert_eq!(test_kill_process(gateway_descendant), Err(Errno::SRCH)); + assert_eq!(test_kill_process_group(validation_group), Err(Errno::SRCH)); + assert_eq!(test_kill_process(validation_descendant), Err(Errno::SRCH)); + assert!(!root_candidate.exists()); + assert!(!fragment_candidate.exists()); + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +#[ignore = "run by seeded_gateway_drop_does_not_block_current_thread_runtime"] +async fn seeded_gateway_drop_current_thread_inner() -> Result<()> { + let home = std::env::vars_os() + .find_map(|(key, value)| (key == FALLBACK_SUBPROCESS_HOME).then_some(value)) + .ok_or_else(|| anyhow!("nested fallback subprocess home was missing"))?; + let home = Utf8PathBuf::from_path_buf(home.into()) + .map_err(|path| anyhow!("nested fallback subprocess home is not UTF-8: {path:?}"))?; + let release = std::env::vars_os() + .find_map(|(key, value)| (key == FALLBACK_SUBPROCESS_RELEASE).then_some(value)) + .ok_or_else(|| anyhow!("nested fallback subprocess release path was missing"))?; + let release = Utf8PathBuf::from_path_buf(release.into()).map_err(|path| { + anyhow!("nested fallback subprocess release path is not UTF-8: {path:?}") + })?; + wait_for_path(&release).await?; + let paths = PvPaths::for_home(home); + seed_foundation_caddy(&paths)?; + make_gateway_descendant_observable(&paths)?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_succeeded_job_scope(&paths, "system").await?; + state::fs::write_sensitive_file( + &paths.run().join("captured-gateway-leader.pid"), + &state::fs::read_to_string(&paths.gateway_pid())?, + )?; + wait_for_path(&paths.run().join("gateway-descendant.pid")).await?; + + let project_path = paths.home().join("validator-project"); + let (project_id, _worker_port_handoff) = FoundationWorkerPortHandoff::new(|| { + seed_foundation_php_project_after_caddy( + &paths, + &project_path, + "php: \"8.4\"\n", + 45_000, + 49_999, + ) + })?; + let [validation_started, _release_validation, _runtime_started] = + install_worker_validation_barrier(&paths, false)?; + gateway_guard.attach_worker("8.4"); + let request_paths = paths.clone(); + let _request_task = tokio::spawn(async move { + request_lines( + &request_paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "run_job", + "kind": "reconcile", + "scope": format!("project:{project_id}"), + }), + ) + .await + }); + wait_for_path(&validation_started).await?; + wait_for_path(&paths.run().join("worker-validation-leader.pid")).await?; + wait_for_path(&paths.run().join("worker-validation-descendant.pid")).await?; + let validation_leader = recorded_test_pid(&paths.run().join("worker-validation-leader.pid"))?; + let validation_group = captured_validation_process_group(validation_leader)?; + state::fs::write_sensitive_file( + &paths.run().join("worker-validation-group.pid"), + &validation_group.as_raw_pid().to_string(), + )?; + wait_for_path(&paths.run().join("worker-validation-root-candidate.path")).await?; + wait_for_path( + &paths + .run() + .join("worker-validation-fragment-candidate.path"), + ) + .await?; + state::fs::write_sensitive_file(&paths.run().join("nested-fallback-ready"), "ready\n")?; + wait_for_path(&paths.run().join("nested-fallback-release")).await?; + + Err(anyhow!("seeded gateway fallback sentinel")) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_prevents_late_gateway_startup() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let [validation_started, release_validation, runtime_started] = + seed_barrier_foundation_caddy(&paths)?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + timeout(Duration::from_secs(5), async { + loop { + if validation_started.exists() { + return; + } + sleep(JOB_STATUS_POLL_INTERVAL).await; + } + }) + .await?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + + gateway_guard.shutdown_daemon_without_waiting()?; + state::fs::write_sensitive_file(&release_validation, "release\n")?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert!(!runtime_started.exists()); + assert!(!paths.gateway_root_config().exists()); + gateway_guard.shutdown_and_cleanup().await?; + + assert!(!paths.daemon_socket().exists()); + assert!(!paths.gateway_pid().exists()); + assert!(!paths.gateway_runtime_metadata().exists()); + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_prevents_late_worker_startup() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + let ([validation_started, release_validation, runtime_started], _port_reservation) = + seed_barrier_foundation_worker(&paths, &project_path, false)?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_path(&validation_started).await?; + let validation_leader_path = paths.run().join("worker-validation-leader.pid"); + let validation_descendant_path = paths.run().join("worker-validation-descendant.pid"); + wait_for_path(&validation_leader_path).await?; + wait_for_path(&validation_descendant_path).await?; + let validation_leader = recorded_test_pid(&validation_leader_path)?; + let validation_group = captured_validation_process_group(validation_leader)?; + let validation_descendant = recorded_test_pid(&validation_descendant_path)?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert_eq!(test_kill_process_group(validation_group), Err(Errno::SRCH)); + assert_eq!(test_kill_process(validation_descendant), Err(Errno::SRCH)); + assert!(!runtime_started.exists()); + assert!(!paths.worker_root_config("8.4").exists()); + state::fs::write_sensitive_file(&release_validation, "release\n")?; + gateway_guard.shutdown_and_cleanup().await?; + + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + assert!(!paths.daemon_socket().exists()); + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_dominates_worker_validation_failure() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + let ([validation_started, release_validation, runtime_started], _port_reservation) = + seed_barrier_foundation_worker(&paths, &project_path, true)?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_path(&validation_started).await?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + + gateway_guard.shutdown_daemon_without_waiting()?; + state::fs::write_sensitive_file(&release_validation, "release\n")?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert!(!runtime_started.exists()); + assert!(!paths.worker_root_config("8.4").exists()); + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_cancels_fresh_worker_readiness() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + let (_project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\n", + 50_000, + 54_999, + )?; + let worker_port = port_handoff.port(); + let [readiness_started, readiness_gate] = install_worker_readiness_barrier(&paths)?; + port_handoff.release_for_runtime_start(); + let worker_root_config = paths.worker_root_config("8.4"); + state::fs::write_sensitive_file(&readiness_gate, "blocked\n")?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + wait_for_path(&readiness_started).await?; + wait_for_path(&paths.worker_pid("8.4")).await?; + wait_for_runtime_replacement_required(&paths.worker_runtime_metadata("8.4")).await?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let worker_pid = recorded_test_pid(&paths.worker_pid("8.4"))?; + + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert_eq!(test_kill_process_group(worker_pid), Err(Errno::SRCH)); + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + assert!(!worker_root_config.exists()); + if platform::loopback_tcp_port_has_listener(worker_port)? { + return Err(anyhow!( + "worker port {worker_port} still has a TCP listener" + )); + } + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_preserves_pending_matching_worker_reload() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + state::fs::ensure_user_dir(&project_path.join("public"))?; + state::fs::ensure_user_dir(&project_path.join("web"))?; + let (project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\ndocument_root: public\n", + 55_000, + 59_999, + )?; + let [load_started, release_load, load_requests] = install_worker_load_barrier(&paths)?; + port_handoff.release_for_runtime_start(); + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + daemon::gateway::reconcile_gateway_runtimes(&paths).await?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let worker_pid = recorded_test_pid(&paths.worker_pid("8.4"))?; + let previous_config = state::fs::read_to_string(&paths.worker_root_config("8.4"))?; + let worker_fragment = paths + .worker_projects_config_dir("8.4") + .join(format!("{project_id}.Caddyfile")); + let previous_fragment = state::fs::read_to_string(&worker_fragment)?; + state::fs::write_sensitive_file( + &project_path.join("pv.yml"), + "php: \"8.4\"\ndocument_root: web\n", + )?; + + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_path(&load_started).await?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert_eq!(recorded_test_pid(&paths.worker_pid("8.4"))?, worker_pid); + assert_eq!( + state::fs::read_to_string(&paths.worker_root_config("8.4"))?, + previous_config + ); + let promoted_fragment = state::fs::read_to_string(&worker_fragment)?; + assert_ne!(promoted_fragment, previous_fragment); + assert!(promoted_fragment.contains(project_path.join("web").as_str())); + assert!(!promoted_fragment.contains(project_path.join("public").as_str())); + assert_eq!(state::fs::read_to_string(&load_requests)?, "load\n"); + let metadata: Value = serde_json::from_str(&state::fs::read_to_string( + &paths.worker_runtime_metadata("8.4"), + )?)?; + assert_eq!(metadata["replacement_required"], true); + assert!(metadata["applied_config_fingerprint"].is_null()); + assert!(metadata["staged_config_fingerprint"].is_string()); + assert_eq!( + metadata["staged_config_fingerprint"], + metadata["desired_config_fingerprint"] + ); + let supervisor = daemon::ProcessSupervisor::new(paths.clone()); + assert!( + supervisor + .adopt_recorded( + &paths.worker_pid("8.4"), + &paths.worker_runtime_metadata("8.4") + )? + .is_some() + ); + state::fs::write_sensitive_file(&release_load, "release\n")?; + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_cancels_watcher_reload_and_preserves_pending_worker() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + state::fs::ensure_user_dir(&project_path.join("public"))?; + state::fs::ensure_user_dir(&project_path.join("web"))?; + let (project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\ndocument_root: public\n", + 55_000, + 59_999, + )?; + let [load_started, release_load, load_requests] = install_worker_load_barrier(&paths)?; + port_handoff.release_for_runtime_start(); + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + daemon::gateway::reconcile_gateway_runtimes(&paths).await?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let worker_pid = recorded_test_pid(&paths.worker_pid("8.4"))?; + let previous_config = state::fs::read_to_string(&paths.worker_root_config("8.4"))?; + let worker_fragment = paths + .worker_projects_config_dir("8.4") + .join(format!("{project_id}.Caddyfile")); + let previous_fragment = state::fs::read_to_string(&worker_fragment)?; + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_succeeded_job_scope(&paths, "system").await?; + + state::fs::write_sensitive_file( + &project_path.join("pv.yml"), + "php: \"8.4\"\ndocument_root: web\n", + )?; + wait_for_path(&load_started).await?; + let job = + wait_for_job_scope_status(&paths, &format!("project:{project_id}"), JobStatus::Running) + .await?; + + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert_eq!(state::fs::read_to_string(&load_requests)?, "load\n"); + assert_eq!(recorded_test_pid(&paths.worker_pid("8.4"))?, worker_pid); + assert_eq!( + state::fs::read_to_string(&paths.worker_root_config("8.4"))?, + previous_config + ); + let promoted_fragment = state::fs::read_to_string(&worker_fragment)?; + assert_ne!(promoted_fragment, previous_fragment); + assert!(promoted_fragment.contains(project_path.join("web").as_str())); + assert!(!promoted_fragment.contains(project_path.join("public").as_str())); + let metadata: Value = serde_json::from_str(&state::fs::read_to_string( + &paths.worker_runtime_metadata("8.4"), + )?)?; + assert_eq!(metadata["replacement_required"], true); + assert!(metadata["applied_config_fingerprint"].is_null()); + assert!(metadata["staged_config_fingerprint"].is_string()); + assert_eq!( + metadata["staged_config_fingerprint"], + metadata["desired_config_fingerprint"] + ); + let supervisor = daemon::ProcessSupervisor::new(paths.clone()); + assert!( + supervisor + .adopt_recorded( + &paths.worker_pid("8.4"), + &paths.worker_runtime_metadata("8.4") + )? + .is_some() + ); + + state::fs::write_sensitive_file(&release_load, "release\n")?; + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn matching_cancellation_preserves_pending_runtime_without_restore_proof() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + state::fs::ensure_user_dir(&project_path.join("public"))?; + state::fs::ensure_user_dir(&project_path.join("web"))?; + let (project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\ndocument_root: public\n", + 60_000, + 64_999, + )?; + let [load_started, release_load, load_requests] = install_worker_load_barrier(&paths)?; + port_handoff.release_for_runtime_start(); + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + daemon::gateway::reconcile_gateway_runtimes(&paths) + .await + .context("start missing-proof fixture runtimes")?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let worker_pid = recorded_test_pid(&paths.worker_pid("8.4"))?; + let previous_config = state::fs::read_to_string(&paths.worker_root_config("8.4"))?; + let worker_fragment = paths + .worker_projects_config_dir("8.4") + .join(format!("{project_id}.Caddyfile")); + let previous_fragment = state::fs::read_to_string(&worker_fragment)?; + let metadata_path = paths.worker_runtime_metadata("8.4"); + let mut metadata: Value = serde_json::from_str(&state::fs::read_to_string(&metadata_path)?)?; + metadata + .as_object_mut() + .ok_or_else(|| anyhow!("worker runtime metadata was not an object"))? + .remove("applied_config_fingerprint"); + state::fs::write_sensitive_file(&metadata_path, &serde_json::to_string_pretty(&metadata)?) + .context("remove matching runtime fingerprint")?; + state::fs::write_sensitive_file( + &project_path.join("pv.yml"), + "php: \"8.4\"\ndocument_root: web\n", + )?; + + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_path(&load_started).await?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + assert_eq!(recorded_test_pid(&paths.worker_pid("8.4"))?, worker_pid); + assert_eq!( + state::fs::read_to_string(&paths.worker_root_config("8.4"))?, + previous_config + ); + let promoted_fragment = state::fs::read_to_string(&worker_fragment)?; + assert_ne!(promoted_fragment, previous_fragment); + assert!(promoted_fragment.contains(project_path.join("web").as_str())); + assert!(!promoted_fragment.contains(project_path.join("public").as_str())); + assert_eq!(state::fs::read_to_string(&load_requests)?, "load\n"); + let metadata: Value = serde_json::from_str(&state::fs::read_to_string(&metadata_path)?)?; + assert_eq!(metadata["replacement_required"], true); + assert!(metadata["applied_config_fingerprint"].is_null()); + assert!(metadata["staged_config_fingerprint"].is_string()); + assert_eq!( + metadata["staged_config_fingerprint"], + metadata["desired_config_fingerprint"] + ); + let supervisor = daemon::ProcessSupervisor::new(paths.clone()); + assert!( + supervisor + .adopt_recorded(&paths.worker_pid("8.4"), &metadata_path)? + .is_some() + ); + state::fs::write_sensitive_file(&release_load, "release\n")?; + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +fn assert_job_has_no_coverage(paths: &PvPaths, job_id: &str) -> Result<()> { + let coverage_count = Connection::open(paths.db().as_std_path())?.query_row( + "SELECT COUNT(*) FROM job_diagnostic_outcomes WHERE job_id = ?1 AND outcome = 'success'", + [job_id], + |row| row.get::<_, i64>(0), + )?; + assert_eq!(coverage_count, 0); + + Ok(()) +} + +fn current_test_binary() -> Result { + let binary = std::env::args_os() + .next() + .ok_or_else(|| anyhow!("test binary path was missing"))?; + Utf8PathBuf::from_path_buf(binary.into()) + .map_err(|path| anyhow!("test binary path is not UTF-8: {path:?}")) +} + +async fn emergency_cleanup_seeded_runtimes(paths: &PvPaths) -> Result<()> { + cleanup_seeded_runtimes(paths, None).await?; + state::fs::remove_file_if_exists(&paths.daemon_socket())?; + + Ok(()) +} + +fn recorded_test_pid(path: &Utf8Path) -> Result { + let raw_pid = state::fs::read_to_string(path)?.trim().parse::()?; + Pid::from_raw(raw_pid).ok_or_else(|| anyhow!("invalid recorded test pid {raw_pid}")) +} + +#[cfg(target_os = "macos")] +fn captured_validation_process_group(leader: Pid) -> Result { + Ok(getpgid(Some(leader))?) +} + +#[cfg(any(target_os = "linux", target_os = "windows"))] +fn captured_validation_process_group(leader: Pid) -> Result { + Ok(leader) +} + #[tokio::test] async fn runtime_health_scanning_waits_for_startup_completion() -> Result<()> { let tempdir = tempdir()?; @@ -560,6 +1264,146 @@ async fn runtime_health_scanning_waits_for_startup_completion() -> Result<()> { Ok(()) } +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_cancels_foreground_socket_reconciliation() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + seed_foundation_caddy(&paths)?; + let mut port_handoff; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_succeeded_job_scope(&paths, "system").await?; + let gateway_pid = recorded_test_pid(&paths.gateway_pid())?; + + let project_path = tempdir.path().join("project"); + let (project_id, worker_port_handoff) = FoundationWorkerPortHandoff::new(|| { + seed_foundation_php_project_after_caddy( + &paths, + &project_path, + "php: \"8.4\"\n", + 40_000, + 44_999, + ) + })?; + let [validation_started, release_validation, runtime_started] = + install_worker_validation_barrier(&paths, false)?; + gateway_guard.attach_worker("8.4"); + port_handoff = worker_port_handoff; + port_handoff.release_for_runtime_start(); + let request_paths = paths.clone(); + let request_scope = format!("project:{project_id}"); + let request_task = tokio::spawn(async move { + request_lines( + &request_paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "run_job", + "kind": "reconcile", + "scope": request_scope, + }), + ) + .await + }); + wait_for_path(&validation_started).await?; + let job = + wait_for_job_scope_status(&paths, &format!("project:{project_id}"), JobStatus::Running) + .await?; + + gateway_guard.shutdown_daemon_without_waiting()?; + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert_eq!( + job.error.as_deref(), + Some("reconciliation was abandoned before completion") + ); + assert_job_has_no_coverage(&paths, &job.id)?; + state::fs::write_sensitive_file(&release_validation, "release\n")?; + let lines = timeout(Duration::from_secs(5), request_task).await???; + assert_eq!(required_response_job_id(&lines)?, job.id); + sleep(Duration::from_millis(100)).await; + assert!(!runtime_started.exists()); + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + assert_eq!(recorded_test_pid(&paths.gateway_pid())?, gateway_pid); + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn daemon_shutdown_joins_health_triggered_worker_recovery() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + let (project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\n", + 30_000, + 34_999, + )?; + port_handoff.release_for_runtime_start(); + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_succeeded_job_scope(&paths, "system").await?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let [validation_started, release_validation, runtime_started] = + install_worker_validation_barrier(&paths, true)?; + let supervisor = daemon::ProcessSupervisor::new(paths.clone()); + let worker = supervisor + .adopt_recorded( + &paths.worker_pid("8.4"), + &paths.worker_runtime_metadata("8.4"), + )? + .ok_or_else(|| anyhow!("worker was not running before health recovery"))?; + worker.stop(Duration::from_secs(1)).await?; + state::fs::remove_file_if_exists(&paths.worker_pid("8.4"))?; + state::fs::remove_file_if_exists(&paths.worker_runtime_metadata("8.4"))?; + + tokio::time::pause(); + tokio::time::advance(Duration::from_secs(30)).await; + tokio::task::yield_now().await; + tokio::time::resume(); + wait_for_path(&validation_started).await?; + let job = + wait_for_job_scope_status(&paths, &format!("project:{project_id}"), JobStatus::Running) + .await?; + + let daemon = gateway_guard + .daemon + .take() + .ok_or_else(|| anyhow!("daemon was not attached before health recovery shutdown"))?; + let mut shutdown_task = tokio::spawn(daemon.shutdown()); + let early_shutdown = timeout(Duration::from_millis(100), &mut shutdown_task).await; + let shutdown_completed_before_recovery = early_shutdown.is_ok(); + state::fs::write_sensitive_file(&release_validation, "release\n")?; + match early_shutdown { + Ok(result) => result??, + Err(_elapsed) => shutdown_task.await??, + } + let job = wait_for_job_id_status(&paths, &job.id, JobStatus::Failed).await?; + assert!( + job.error + .as_deref() + .is_some_and(|error| error.contains("FrankenPHP config validation failed")) + ); + assert_job_has_no_coverage(&paths, &job.id)?; + sleep(Duration::from_millis(100)).await; + assert!(!runtime_started.exists()); + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + gateway_guard.shutdown_and_cleanup().await?; + assert!(!shutdown_completed_before_recovery); + + Ok(()) +} + struct BlockedStartupDownloadClient { started: Arc, release: Mutex>, @@ -591,6 +1435,43 @@ impl resources::ResourceHttpClient for BlockedStartupDownloadClient { } } +struct BlockedForegroundUpdateClient { + block_update: Arc, + update_started: Arc, + release_receiver: Mutex>, +} + +impl resources::ResourceHttpClient for BlockedForegroundUpdateClient { + fn get_text(&self, url: &str) -> resources::Result { + if !self.block_update.load(Ordering::SeqCst) { + return Ok(CADDY_ARTIFACT_MANIFEST.to_owned()); + } + self.update_started.store(true, Ordering::SeqCst); + let release_receiver = match self.release_receiver.lock() { + Ok(release_receiver) => release_receiver, + Err(poisoned) => poisoned.into_inner(), + }; + release_receiver + .recv_timeout(Duration::from_secs(5)) + .map_err(|error| resources::ResourcesError::HttpRequestFailed { + url: url.to_owned(), + reason: format!("blocked update release failed: {error}"), + })?; + + Err(resources::ResourcesError::HttpRequestFailed { + url: url.to_owned(), + reason: "blocked update sentinel".to_owned(), + }) + } + + fn download(&self, url: &str, _writer: &mut dyn io::Write) -> resources::Result<()> { + Err(resources::ResourcesError::HttpStatusFailed { + url: url.to_owned(), + status_code: 404, + }) + } +} + #[tokio::test] async fn daemon_shutdown_keeps_jobs_lock_until_blocking_install_finishes() -> Result<()> { let tempdir = tempdir()?; @@ -646,6 +1527,160 @@ async fn daemon_shutdown_keeps_jobs_lock_until_blocking_install_finishes() -> Re Ok(()) } +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_wakes_blocked_resource_request() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let started = Arc::new(AtomicBool::new(false)); + let (cancel, blocked) = mpsc::channel(); + let client = BlockedStartupDownloadClient { + started: Arc::clone(&started), + release: Mutex::new(blocked), + }; + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters_with_manifest_client_and_blocked_request_release( + paths.clone(), + TEST_ARTIFACT_MANIFEST_URL, + client, + cancel, + ) + .await?; + timeout(Duration::from_secs(5), async { + while !started.load(Ordering::SeqCst) { + sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let job = wait_for_job_scope_status(&paths, "system", JobStatus::Running).await?; + assert!(matches!( + JobsLock::acquire(&paths), + Err(state::StateError::CoordinationLockHeld { .. }) + )); + + daemon.shutdown_without_waiting_for_test()?; + + let job = timeout( + Duration::from_secs(5), + wait_for_job_id_status(&paths, &job.id, JobStatus::Failed), + ) + .await??; + assert!( + job.error + .as_deref() + .is_some_and(|error| error.contains("HTTP status 404")), + "resource error was not preserved: {:?}", + job.error + ); + let lock_deadline = Instant::now() + Duration::from_secs(5); + let _jobs_lock = loop { + match JobsLock::acquire(&paths) { + Ok(lock) => break lock, + Err(state::StateError::CoordinationLockHeld { .. }) + if Instant::now() < lock_deadline => + { + sleep(JOB_STATUS_POLL_INTERVAL).await; + } + Err(error) => return Err(error.into()), + } + }; + assert!(!paths.daemon_socket().exists()); + assert!(!paths.gateway_pid().exists()); + assert!(!paths.gateway_runtime_metadata().exists()); + + Ok(()) +} + +#[tokio::test(flavor = "current_thread")] +async fn fallback_shutdown_drains_blocked_foreground_update_and_preserves_its_error() -> Result<()> +{ + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + seed_foundation_caddy(&paths)?; + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + let block_update = Arc::new(AtomicBool::new(false)); + let update_started = Arc::new(AtomicBool::new(false)); + let (release_sender, release_receiver) = mpsc::channel(); + let client = BlockedForegroundUpdateClient { + block_update: Arc::clone(&block_update), + update_started: Arc::clone(&update_started), + release_receiver: Mutex::new(release_receiver), + }; + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters_with_manifest_client_and_blocked_request_release( + paths.clone(), + TEST_ARTIFACT_MANIFEST_URL, + client, + release_sender, + ) + .await?; + wait_for_succeeded_job_scope(&paths, "system").await?; + + block_update.store(true, Ordering::SeqCst); + let shutdown_paths = paths.clone(); + let shutdown_update_started = Arc::clone(&update_started); + let shutdown_task = tokio::task::spawn_blocking(move || { + let deadline = Instant::now() + Duration::from_secs(5); + while !shutdown_update_started.load(Ordering::SeqCst) { + if Instant::now() >= deadline { + return Err(anyhow!( + "foreground update did not reach the blocked request" + )); + } + std::thread::sleep(Duration::from_millis(10)); + } + let jobs_lock_held = matches!( + JobsLock::acquire(&shutdown_paths), + Err(state::StateError::CoordinationLockHeld { .. }) + ); + daemon.shutdown_without_waiting_for_test()?; + + Ok::(jobs_lock_held) + }); + let request_paths = paths.clone(); + let request_task = tokio::spawn(async move { + request_lines( + &request_paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "run_job", + "kind": "update", + "scope": "system", + }), + ) + .await + }); + let lines = timeout(Duration::from_secs(5), request_task).await???; + let job_id = required_response_job_id(&lines)?; + let jobs_lock_held = shutdown_task.await??; + assert!(jobs_lock_held); + let job = wait_for_job_id_status(&paths, job_id, JobStatus::Failed).await?; + assert!( + job.error + .as_deref() + .is_some_and(|error| error.contains("blocked update sentinel")), + "update error was not preserved: {:?}", + job.error + ); + assert_eq!(job.id, job_id); + let lock_deadline = Instant::now() + Duration::from_secs(5); + let _jobs_lock = loop { + match JobsLock::acquire(&paths) { + Ok(lock) => break lock, + Err(state::StateError::CoordinationLockHeld { .. }) + if Instant::now() < lock_deadline => + { + sleep(JOB_STATUS_POLL_INTERVAL).await; + } + Err(error) => return Err(error.into()), + } + }; + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + gateway_guard.shutdown_and_cleanup().await?; + + Ok(()) +} + #[tokio::test] async fn startup_reconciliation_records_non_contention_enqueue_failure() -> Result<()> { let tempdir = tempdir()?; @@ -719,7 +1754,8 @@ async fn startup_reconciliation_starts_then_adopts_gateway_across_daemon_restart async fn repeated_system_requests_during_startup_create_one_trailing_job() -> Result<()> { let tempdir = tempdir()?; let paths = PvPaths::for_home(tempdir.path().join("home")); - let [validation_started, release_validation] = seed_barrier_foundation_caddy(&paths)?; + let [validation_started, release_validation, _runtime_started] = + seed_barrier_foundation_caddy(&paths)?; let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); let result = async { @@ -744,13 +1780,50 @@ async fn repeated_system_requests_during_startup_create_one_trailing_job() -> Re }))?; let mut readers = Vec::new(); let mut responses = Vec::new(); - for _ in 0..3 { - let mut stream = UnixStream::connect(paths.daemon_socket()).await?; - stream.write_all(request.as_bytes()).await?; - stream.write_all(b"\n").await?; + let exchange_deadline = TokioInstant::now() + REQUEST_LINES_TIMEOUT; + for reader_index in 0..3 { + let mut stream = request_step( + exchange_deadline, + UnixStream::connect(paths.daemon_socket()), + &paths.daemon_socket(), + &request, + &format!("connect for reader {reader_index}"), + REQUEST_LINES_TIMEOUT, + &responses, + ) + .await?; + request_step( + exchange_deadline, + stream.write_all(request.as_bytes()), + &paths.daemon_socket(), + &request, + &format!("request write for reader {reader_index}"), + REQUEST_LINES_TIMEOUT, + &responses, + ) + .await?; + request_step( + exchange_deadline, + stream.write_all(b"\n"), + &paths.daemon_socket(), + &request, + &format!("newline write for reader {reader_index}"), + REQUEST_LINES_TIMEOUT, + &responses, + ) + .await?; let mut reader = BufReader::new(stream); let mut line = String::new(); - reader.read_line(&mut line).await?; + request_step( + exchange_deadline, + reader.read_line(&mut line), + &paths.daemon_socket(), + &request, + &format!("initial response read for reader {reader_index}"), + REQUEST_LINES_TIMEOUT, + &responses, + ) + .await?; responses.push(serde_json::from_str::(line.trim_end())?); readers.push(reader); } @@ -763,10 +1836,20 @@ async fn repeated_system_requests_during_startup_create_one_trailing_job() -> Re })); state::fs::write_sensitive_file(&release_validation, "release\n")?; - for mut reader in readers { + for (reader_index, mut reader) in readers.into_iter().enumerate() { loop { let mut line = String::new(); - if reader.read_line(&mut line).await? == 0 { + let bytes = request_step( + exchange_deadline, + reader.read_line(&mut line), + &paths.daemon_socket(), + &request, + &format!("response drain for reader {reader_index}"), + REQUEST_LINES_TIMEOUT, + &responses, + ) + .await?; + if bytes == 0 { break; } } @@ -1101,12 +2184,58 @@ fn seed_foundation_caddy(paths: &PvPaths) -> Result<()> { Ok(()) } +fn make_gateway_descendant_observable(paths: &PvPaths) -> Result<()> { + let executable = paths.home().join("fake-caddy-release/bin/caddy"); + let descendant_pid_path = paths.run().join("gateway-descendant.pid"); + let source = paths.home().join("observable-fake-caddy"); + let script = FOUNDATION_FAKE_CADDY_SCRIPT.replace( + " child=\"$!\"\n", + &format!(" child=\"$!\"\n printf '%s\\n' \"$child\" > \"{descendant_pid_path}\"\n"), + ); + if script == FOUNDATION_FAKE_CADDY_SCRIPT { + return Err(anyhow!("fake Caddy child launch was not instrumented")); + } + state::fs::write_sensitive_file(&source, &script)?; + let install = AppReleaseLayout::new(paths.clone()).install_release_binary("0.0.2", &source)?; + state::fs::rename(install.binary_path(), &executable)?; + + Ok(()) +} + fn seed_foundation_php_project( paths: &PvPaths, project_path: &Utf8Path, config: &str, +) -> Result<(String, FoundationWorkerPortHandoff)> { + seed_foundation_php_project_in_range(paths, project_path, config, 40_000, 44_999) +} + +fn seed_foundation_php_project_in_range( + paths: &PvPaths, + project_path: &Utf8Path, + config: &str, + port_range_start: u16, + port_range_end: u16, +) -> Result<(String, FoundationWorkerPortHandoff)> { + FoundationWorkerPortHandoff::new(|| { + seed_foundation_caddy(paths)?; + seed_foundation_php_project_after_caddy( + paths, + project_path, + config, + port_range_start, + port_range_end, + ) + }) +} + +fn seed_foundation_php_project_after_caddy( + paths: &PvPaths, + project_path: &Utf8Path, + config: &str, + port_range_start: u16, + port_range_end: u16, ) -> Result<(String, StdTcpListener)> { - seed_foundation_caddy(paths)?; let certified_key = generate_simple_self_signed(vec![ "project.test".to_owned(), "pv-gateway.localhost".to_owned(), @@ -1163,7 +2292,8 @@ fn seed_foundation_php_project( "8.4.8-pv1", &frankenphp_release, )?; - let mut worker_port_reservations = reserve_foundation_ports(1, 40_000, 44_999)?; + let mut worker_port_reservations = + reserve_foundation_ports(1, port_range_start, port_range_end)?; let worker_port_reservation = worker_port_reservations .pop() .ok_or_else(|| anyhow!("expected one reserved worker port"))?; @@ -1190,29 +2320,125 @@ fn seed_foundation_php_project( })? .project; - Ok((project.id, worker_port_reservation)) + Ok((project.id, worker_port_reservation)) +} + +fn seed_barrier_foundation_caddy(paths: &PvPaths) -> Result<[Utf8PathBuf; 3]> { + seed_foundation_caddy(paths)?; + let executable = paths.home().join("fake-caddy-release/bin/caddy"); + let validation_started = paths.run().join("startup-validation-started"); + let release_validation = paths.run().join("release-startup-validation"); + let runtime_started = paths.run().join("gateway-runtime-started"); + let wrapper_source = paths.home().join("caddy-startup-barrier"); + let caddy_script = FOUNDATION_FAKE_CADDY_SCRIPT + .strip_prefix("#!/bin/sh\n") + .ok_or_else(|| anyhow!("fake Caddy script is missing its shebang"))?; + state::fs::write_sensitive_file( + &wrapper_source, + &format!( + "#!/bin/sh\nset -eu\nif [ \"${{1:-}}\" = \"validate\" ]; then\n : > \"{validation_started}\"\n while [ ! -f \"{release_validation}\" ]; do sleep 0.01; done\nelif [ \"${{1:-}}\" = \"run\" ]; then\n : > \"{runtime_started}\"\nfi\n{caddy_script}" + ), + )?; + let wrapper_install = + AppReleaseLayout::new(paths.clone()).install_release_binary("0.0.1", &wrapper_source)?; + state::fs::rename(wrapper_install.binary_path(), &executable)?; + + Ok([validation_started, release_validation, runtime_started]) +} + +fn seed_barrier_foundation_worker( + paths: &PvPaths, + project_path: &Utf8Path, + fail_validation: bool, +) -> Result<([Utf8PathBuf; 3], FoundationWorkerPortHandoff)> { + let (_project_id, port_handoff) = + seed_foundation_php_project(paths, project_path, "php: \"8.4\"\n")?; + let barrier = install_worker_validation_barrier(paths, fail_validation)?; + + Ok((barrier, port_handoff)) } -fn seed_barrier_foundation_caddy(paths: &PvPaths) -> Result<[Utf8PathBuf; 2]> { - seed_foundation_caddy(paths)?; - let executable = paths.home().join("fake-caddy-release/bin/caddy"); - let validation_started = paths.run().join("startup-validation-started"); - let release_validation = paths.run().join("release-startup-validation"); - let wrapper_source = paths.home().join("caddy-startup-barrier"); - let caddy_script = FOUNDATION_FAKE_CADDY_SCRIPT +fn install_worker_validation_barrier( + paths: &PvPaths, + fail_validation: bool, +) -> Result<[Utf8PathBuf; 3]> { + let executable = paths.home().join("8.4-frankenphp-release/bin/frankenphp"); + let validation_started = paths.run().join("worker-validation-started"); + let release_validation = paths.run().join("release-worker-validation"); + let runtime_started = paths.run().join("worker-runtime-started"); + let validation_leader = paths.run().join("worker-validation-leader.pid"); + let validation_descendant = paths.run().join("worker-validation-descendant.pid"); + let validation_root_candidate = paths.run().join("worker-validation-root-candidate.path"); + let validation_fragment_candidate = paths + .run() + .join("worker-validation-fragment-candidate.path"); + let wrapper_source = paths.home().join("worker-startup-barrier"); + let worker_script = state::fs::read_to_string(&executable)?; + let worker_script = worker_script .strip_prefix("#!/bin/sh\n") - .ok_or_else(|| anyhow!("fake Caddy script is missing its shebang"))?; + .ok_or_else(|| anyhow!("fake FrankenPHP script is missing its shebang"))?; + let validation_outcome = if fail_validation { "exit 7" } else { ":" }; state::fs::write_sensitive_file( &wrapper_source, &format!( - "#!/bin/sh\nset -eu\nif [ \"${{1:-}}\" = \"validate\" ]; then\n : > \"{validation_started}\"\n while [ ! -f \"{release_validation}\" ]; do sleep 0.01; done\nfi\n{caddy_script}" + "#!/bin/sh\nset -eu\nif [ \"${{1:-}}\" = \"validate\" ]; then\n printf '%s\\n' \"$$\" > \"{validation_leader}\"\n printf '%s\\n' \"$3\" > \"{validation_root_candidate}\"\n sed -n 's|^import \"\\(.*\\)/\\*\\.Caddyfile\"$|\\1|p' \"$3\" > \"{validation_fragment_candidate}\"\n (while [ ! -f \"{release_validation}\" ]; do sleep 0.01; done) &\n validation_child=\"$!\"\n printf '%s\\n' \"$validation_child\" > \"{validation_descendant}\"\n : > \"{validation_started}\"\n wait \"$validation_child\"\n {validation_outcome}\nelif [ \"${{1:-}}\" = \"run\" ]; then\n : > \"{runtime_started}\"\nfi\n{worker_script}" ), )?; let wrapper_install = - AppReleaseLayout::new(paths.clone()).install_release_binary("0.0.1", &wrapper_source)?; + AppReleaseLayout::new(paths.clone()).install_release_binary("0.0.3", &wrapper_source)?; state::fs::rename(wrapper_install.binary_path(), &executable)?; - Ok([validation_started, release_validation]) + Ok([validation_started, release_validation, runtime_started]) +} + +fn install_worker_load_barrier(paths: &PvPaths) -> Result<[Utf8PathBuf; 3]> { + let server_path = paths + .home() + .join("8.4-frankenphp-release/bin/frankenphp.server.py"); + let load_started = paths.run().join("worker-load-started"); + let release_load = paths.run().join("release-worker-load"); + let load_consumed = paths.run().join("worker-load-consumed"); + let load_requests = paths.run().join("worker-load-requests"); + let server = state::fs::read_to_string(&server_path)?; + let insertion = format!( + " with open({load_requests:?}, \"a\", encoding=\"utf-8\") as request_file:\n request_file.write(\"load\\n\")\n if not os.path.exists({load_consumed:?}):\n with open({load_started:?}, \"w\", encoding=\"utf-8\") as marker_file:\n marker_file.write(\"started\\n\")\n while not os.path.exists({release_load:?}):\n time.sleep(0.01)\n with open({load_consumed:?}, \"w\", encoding=\"utf-8\") as marker_file:\n marker_file.write(\"consumed\\n\")\n\n content_length = int(self.headers.get(\"Content-Length\", \"0\"))", + load_requests = load_requests.as_str(), + load_consumed = load_consumed.as_str(), + load_started = load_started.as_str(), + release_load = release_load.as_str(), + ); + let instrumented = server.replace( + " content_length = int(self.headers.get(\"Content-Length\", \"0\"))", + &insertion, + ); + if instrumented == server { + return Err(anyhow!("fake FrankenPHP load handler was not instrumented")); + } + state::fs::write_sensitive_file(&server_path, &instrumented)?; + + Ok([load_started, release_load, load_requests]) +} + +fn install_worker_readiness_barrier(paths: &PvPaths) -> Result<[Utf8PathBuf; 2]> { + let server_path = paths + .home() + .join("8.4-frankenphp-release/bin/frankenphp.server.py"); + let readiness_started = paths.run().join("worker-readiness-started"); + let readiness_gate = paths.run().join("worker-readiness-gate"); + let server = state::fs::read_to_string(&server_path)?; + let binding = "servers = [Server((\"127.0.0.1\", http_port), Handler)]"; + let insertion = format!( + "{binding}\nwith open({readiness_started:?}, \"w\", encoding=\"utf-8\") as marker_file:\n marker_file.write(\"started\\n\")\nwhile os.path.exists({readiness_gate:?}):\n time.sleep(0.01)", + readiness_started = readiness_started.as_str(), + readiness_gate = readiness_gate.as_str(), + ); + let instrumented = server.replacen(binding, &insertion, 1); + if instrumented == server { + return Err(anyhow!("fake FrankenPHP readiness was not instrumented")); + } + state::fs::write_sensitive_file(&server_path, &instrumented)?; + + Ok([readiness_started, readiness_gate]) } fn available_foundation_gateway_ports() -> Result<[u16; 2]> { @@ -1237,6 +2463,94 @@ fn available_foundation_gateway_ports() -> Result<[u16; 2]> { .map_err(|_| anyhow!("expected two available gateway ports")) } +struct FoundationWorkerPortHandoff { + port: u16, + reservation: Option, + #[cfg(unix)] + lock: Option, +} + +impl FoundationWorkerPortHandoff { + fn new(reserve: impl FnOnce() -> Result<(String, StdTcpListener)>) -> Result<(String, Self)> { + #[cfg(unix)] + let lock = { + let lock_path = Utf8Path::new(env!("CARGO_TARGET_TMPDIR")) + .join(FOUNDATION_WORKER_PORT_HANDOFF_LOCK); + let file = state::fs::open_append_file(&lock_path)?; + rustix::fs::flock(&file, FlockOperation::LockExclusive).map_err(io::Error::from)?; + file.into() + }; + let (output, reservation) = reserve()?; + let port = reservation.local_addr()?.port(); + + Ok(( + output, + Self { + port, + reservation: Some(reservation), + #[cfg(unix)] + lock: Some(lock), + }, + )) + } + + fn release_for_runtime_start(&mut self) { + self.reservation = None; + } + + fn port(&self) -> u16 { + self.port + } + + async fn verify_publication_and_release_lock( + &mut self, + paths: &PvPaths, + php_track: &str, + ) -> Result { + if self.reservation.is_some() { + return Err(anyhow!( + "foundation worker port reservation was not released before publication" + )); + } + + let supervisor = daemon::ProcessSupervisor::new(paths.clone()); + let recorded_pid = supervisor + .adopt_recorded( + &paths.worker_pid(php_track), + &paths.worker_runtime_metadata(php_track), + )? + .ok_or_else(|| anyhow!("seeded FrankenPHP runtime was not adoptable"))? + .pid(); + daemon::wait_for_readiness( + daemon::ReadinessCheck::Tcp { + host: Ipv4Addr::LOCALHOST.to_string(), + port: self.port, + }, + Duration::from_secs(1), + ) + .await?; + let published_pid = supervisor + .adopt_recorded( + &paths.worker_pid(php_track), + &paths.worker_runtime_metadata(php_track), + )? + .ok_or_else(|| anyhow!("seeded FrankenPHP runtime was not adoptable after readiness"))? + .pid(); + if published_pid != recorded_pid { + return Err(anyhow!( + "seeded FrankenPHP runtime changed during publication from PID {recorded_pid} to {published_pid}" + )); + } + + #[cfg(unix)] + { + self.lock = None; + } + + Ok(published_pid) + } +} + fn reserve_foundation_ports(count: usize, start: u16, end: u16) -> Result> { let mut listeners = Vec::with_capacity(count); @@ -1290,6 +2604,16 @@ impl SeededGatewayGuard { daemon.shutdown().await.map_err(|error| anyhow!(error)) } + fn shutdown_daemon_without_waiting(&mut self) -> Result<()> { + let Some(daemon) = self.daemon.take() else { + return Ok(()); + }; + + daemon + .shutdown_without_waiting_for_test() + .map_err(anyhow::Error::from) + } + async fn shutdown_and_cleanup(&mut self) -> Result<()> { let result = shutdown_seeded_gateway( self.daemon.take(), @@ -1303,6 +2627,18 @@ impl SeededGatewayGuard { result } + + async fn shutdown_without_waiting_and_cleanup(&mut self) -> Result<()> { + let shutdown_result = self.shutdown_daemon_without_waiting(); + let cleanup_result = + cleanup_seeded_runtimes(&self.paths, self.worker_track.as_deref()).await; + let result = combine_cleanup_results(shutdown_result, cleanup_result); + if result.is_ok() { + self.cleanup_complete = true; + } + + result + } } impl Drop for SeededGatewayGuard { @@ -1313,7 +2649,11 @@ impl Drop for SeededGatewayGuard { let paths = self.paths.clone(); let diagnostic_paths = paths.clone(); - let daemon = self.daemon.take(); + let daemon_shutdown_result = self.daemon.take().map_or(Ok(()), |daemon| { + daemon + .shutdown_without_waiting_for_test() + .map_err(anyhow::Error::from) + }); let worker_track = self.worker_track.clone(); let cleanup_panicked = std::thread::scope(|scope| { let cleanup_thread = scope.spawn(move || { @@ -1331,11 +2671,11 @@ impl Drop for SeededGatewayGuard { } }; - if let Err(error) = runtime.block_on(shutdown_seeded_gateway( - daemon, - &paths, - worker_track.as_deref(), - )) { + let runtime_cleanup_result = + runtime.block_on(cleanup_seeded_runtimes(&paths, worker_track.as_deref())); + if let Err(error) = + combine_cleanup_results(daemon_shutdown_result, runtime_cleanup_result) + { report_seeded_gateway_cleanup_failure( &paths, &format!("cleanup failed: {error}"), @@ -1360,6 +2700,7 @@ fn report_seeded_gateway_cleanup_failure(paths: &PvPaths, message: &str) { if let Ok(mut log) = state::fs::open_append_file(&paths.daemon_log()) { let _write_result = log.write_all(format!("{record}\n").as_bytes()); } + let _write_result = writeln!(io::stderr().lock(), "{record}"); } async fn shutdown_seeded_gateway( @@ -1371,9 +2712,15 @@ async fn shutdown_seeded_gateway( Some(daemon) => daemon.shutdown().await.map_err(|error| anyhow!(error)), None => Ok(()), }; + let cleanup_result = cleanup_seeded_runtimes(paths, worker_track).await; + + combine_cleanup_results(shutdown_result, cleanup_result) +} + +async fn cleanup_seeded_runtimes(paths: &PvPaths, worker_track: Option<&str>) -> Result<()> { let worker_cleanup_result = stop_seeded_worker(paths, worker_track).await; let gateway_cleanup_result = stop_seeded_gateway(paths).await; - let cleanup_result = match (worker_cleanup_result, gateway_cleanup_result) { + match (worker_cleanup_result, gateway_cleanup_result) { (Ok(()), Ok(())) => Ok(()), (Err(worker_error), Ok(())) => { Err(anyhow!("seeded FrankenPHP cleanup failed: {worker_error}")) @@ -1382,14 +2729,19 @@ async fn shutdown_seeded_gateway( (Err(worker_error), Err(gateway_error)) => Err(anyhow!( "seeded FrankenPHP cleanup failed: {worker_error}; seeded Caddy cleanup failed: {gateway_error}" )), - }; + } +} - match (shutdown_result, cleanup_result) { +fn combine_cleanup_results( + shutdown_result: Result<()>, + runtime_cleanup_result: Result<()>, +) -> Result<()> { + match (shutdown_result, runtime_cleanup_result) { (Ok(()), Ok(())) => Ok(()), (Err(shutdown_error), Ok(())) => Err(anyhow!("daemon shutdown failed: {shutdown_error}")), (Ok(()), Err(cleanup_error)) => Err(cleanup_error), (Err(shutdown_error), Err(cleanup_error)) => Err(anyhow!( - "daemon shutdown failed: {shutdown_error}; seeded Caddy cleanup failed: {cleanup_error}" + "daemon shutdown failed: {shutdown_error}; runtime cleanup failed: {cleanup_error}" )), } } @@ -1448,11 +2800,29 @@ fn propagate_after_cleanup( (Err(operation_error), Ok(())) => Err(operation_error), (Ok(_), Err(cleanup_error)) => Err(cleanup_error), (Err(operation_error), Err(cleanup_error)) => Err(anyhow!( - "operation failed: {operation_error}; seeded Caddy cleanup failed: {cleanup_error}" + "operation failed: {operation_error}; fixture cleanup failed: {cleanup_error}" )), } } +#[test] +fn fixture_cleanup_preserves_operation_error_precedence() { + let outcomes = [ + propagate_after_cleanup(Ok("value"), Ok(())).map(str::to_owned), + propagate_after_cleanup::<&str>(Err(anyhow!("operation sentinel")), Ok(())) + .map(str::to_owned), + propagate_after_cleanup(Ok("value"), Err(anyhow!("cleanup sentinel"))).map(str::to_owned), + propagate_after_cleanup::<&str>( + Err(anyhow!("operation sentinel")), + Err(anyhow!("cleanup sentinel")), + ) + .map(str::to_owned), + ] + .map(|result| result.map_err(|error| error.to_string())); + + assert_debug_snapshot!(outcomes); +} + async fn stop_seeded_gateway(paths: &PvPaths) -> Result<()> { let supervisor = daemon::ProcessSupervisor::new(paths.clone()); let deadline = Instant::now() + SEEDED_GATEWAY_CLEANUP_TIMEOUT; @@ -1579,26 +2949,33 @@ async fn system_reconciliation_reconciles_linked_project_env() -> Result<()> { let tempdir = tempdir()?; let paths = PvPaths::for_home(tempdir.path().join("home")); let project_path = tempdir.path().join("project"); - let (_project_id, worker_port_reservation) = seed_foundation_php_project( + let (_project_id, mut worker_port_handoff) = seed_foundation_php_project( &paths, &project_path, "php: \"8.4\"\nenv:\n APP_URL: \"${project_url}\"\n APP_NAME: setup\n", )?; let php_track = "8.4"; let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker(php_track); + worker_port_handoff.release_for_runtime_start(); let daemon = daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; gateway_guard.attach_daemon(daemon); - gateway_guard.attach_worker(php_track); - drop(worker_port_reservation); let client_paths = paths.clone(); - let completed_result = tokio::task::spawn_blocking(move || { - daemon::run_job_blocking(client_paths, "reconcile", "system") - }) - .await - .map_err(anyhow::Error::from) - .and_then(|result| result.map_err(anyhow::Error::from)); + let completed_result = async { + let completed = tokio::task::spawn_blocking(move || { + daemon::run_job_blocking(client_paths, "reconcile", "system") + }) + .await + .map_err(anyhow::Error::from) + .and_then(|result| result.map_err(anyhow::Error::from))?; + worker_port_handoff + .verify_publication_and_release_lock(&paths, php_track) + .await?; + Ok(completed) + } + .await; let cleanup_result = gateway_guard.shutdown_and_cleanup().await; let completed = propagate_after_cleanup(completed_result, cleanup_result)?; @@ -1719,6 +3096,64 @@ async fn targeted_gateway_phases_are_disjoint() -> Result<()> { Ok(()) } +#[tokio::test(flavor = "current_thread")] +async fn targeted_scenario_timeout_still_cleans_owned_state() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let project_path = tempdir.path().join("project"); + let (_project_id, mut port_handoff) = seed_foundation_php_project_in_range( + &paths, + &project_path, + "php: \"8.4\"\n", + 35_000, + 39_999, + )?; + let [readiness_started, readiness_gate] = install_worker_readiness_barrier(&paths)?; + state::fs::write_sensitive_file(&readiness_gate, "blocked\n")?; + port_handoff.release_for_runtime_start(); + let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); + gateway_guard.attach_worker("8.4"); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + gateway_guard.attach_daemon(daemon); + wait_for_path(&readiness_started).await?; + wait_for_path(&paths.worker_pid("8.4")).await?; + port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; + let worker_group = recorded_test_pid(&paths.worker_pid("8.4"))?; + + let operation_result = timeout( + Duration::from_millis(25), + std::future::pending::>(), + ) + .await + .map_err(|_elapsed| anyhow!("targeted gateway scenario timed out after 25ms")) + .and_then(|result| result); + let cleanup_result = timeout( + Duration::from_secs(5), + gateway_guard.shutdown_without_waiting_and_cleanup(), + ) + .await + .map_err(|_elapsed| anyhow!("nonwaiting targeted scenario cleanup timed out"))?; + let error = propagate_after_cleanup(operation_result, cleanup_result) + .err() + .ok_or_else(|| anyhow!("pending scenario unexpectedly completed"))?; + + assert_eq!( + error.to_string(), + "targeted gateway scenario timed out after 25ms" + ); + assert!(!paths.daemon_socket().exists()); + assert!(!paths.gateway_pid().exists()); + assert!(!paths.gateway_runtime_metadata().exists()); + assert!(!paths.worker_pid("8.4").exists()); + assert!(!paths.worker_runtime_metadata("8.4").exists()); + assert_eq!(test_kill_process_group(worker_group), Err(Errno::SRCH)); + + Ok(()) +} + async fn run_targeted_gateway_phase_scenario( scenario: TargetedGatewayPhaseScenario, ) -> Result<(JobStatus, Vec)> { @@ -1772,7 +3207,7 @@ async fn run_targeted_gateway_phase_scenario( "8.4.8-pv1", &frankenphp_release, )?; - let worker_port_reservations = reserve_foundation_ports(1, 40_000, 44_999)?; + let worker_port_reservations = reserve_foundation_ports(1, 25_000, 29_999)?; let worker_service_port = worker_port_reservations[0].local_addr()?.port(); database.assign_port( PortRequest::php_worker( @@ -1815,46 +3250,58 @@ async fn run_targeted_gateway_phase_scenario( daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; gateway_guard.attach_daemon(daemon); gateway_guard.attach_worker(php_track); - drop(worker_port_reservations); - let initial_lines = request_lines( - &paths, - json!({ - "protocol_version": daemon::PROTOCOL_VERSION, - "command": "run_job", - "kind": "reconcile", - "scope": "system", - }), - ) - .await?; - let initial_job_id = required_response_job_id(&initial_lines)?; - wait_for_succeeded_job_id(&paths, initial_job_id).await?; - - state::fs::write_sensitive_file(&target_config_path, "serve: false\n")?; - match scenario { - TargetedGatewayPhaseScenario::Success => {} - TargetedGatewayPhaseScenario::GatewayFailure => { - state::fs::write_sensitive_file( - &paths.home().join("fake-caddy-release/bin/caddy"), - "#!/bin/sh\nexit 2\n", - )?; - } - TargetedGatewayPhaseScenario::StaleWorkerFailure => { - state::fs::write_sensitive_file(&frankenphp_executable, "#!/bin/sh\nexit 2\n")?; + let operation_result = timeout(TARGETED_SCENARIO_TIMEOUT, async { + drop(worker_port_reservations); + let initial_lines = request_lines( + &paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "run_job", + "kind": "reconcile", + "scope": "system", + }), + ) + .await?; + let initial_job_id = required_response_job_id(&initial_lines)?; + wait_for_succeeded_job_id(&paths, initial_job_id).await?; + + state::fs::write_sensitive_file(&target_config_path, "serve: false\n")?; + match scenario { + TargetedGatewayPhaseScenario::Success => {} + TargetedGatewayPhaseScenario::GatewayFailure => { + state::fs::write_sensitive_file( + &paths.home().join("fake-caddy-release/bin/caddy"), + "#!/bin/sh\nexit 2\n", + )?; + } + TargetedGatewayPhaseScenario::StaleWorkerFailure => { + state::fs::write_sensitive_file(&frankenphp_executable, "#!/bin/sh\nexit 2\n")?; + } } - } - let lines = request_lines( - &paths, - json!({ - "protocol_version": daemon::PROTOCOL_VERSION, - "command": "run_job", - "kind": "reconcile", - "scope": format!("project:{}", target.id), - }), - ) - .await?; - let job_id = required_response_job_id(&lines)?.to_owned(); - let cleanup_result = gateway_guard.shutdown_and_cleanup().await; + let lines = request_lines( + &paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "run_job", + "kind": "reconcile", + "scope": format!("project:{}", target.id), + }), + ) + .await?; + required_response_job_id(&lines).map(str::to_owned) + }) + .await + .map_err(|_elapsed| { + anyhow!("targeted gateway scenario timed out after {TARGETED_SCENARIO_TIMEOUT:?}") + }) + .and_then(|result| result); + let cleanup_result = if operation_result.is_err() { + gateway_guard.shutdown_without_waiting_and_cleanup().await + } else { + gateway_guard.shutdown_and_cleanup().await + }; + let job_id = propagate_after_cleanup(operation_result, cleanup_result)?; let database = Database::open(&paths)?; let job = database @@ -1863,7 +3310,7 @@ async fn run_targeted_gateway_phase_scenario( .find(|job| job.id == job_id) .ok_or_else(|| anyhow!("missing targeted reconciliation job {job_id}")); let log = state::fs::read_to_string(&paths.daemon_log()); - let job = propagate_after_cleanup(job, cleanup_result)?; + let job = job?; let phases = log? .lines() .map(serde_json::from_str::) @@ -1891,12 +3338,11 @@ async fn daemon_health_automatically_recovers_killed_worker_with_invalid_project let paths = PvPaths::for_home(tempdir.path().join("home")); let project_path = tempdir.path().join("project"); let config_path = project_path.join("pv.yml"); - let (project_id, worker_port_reservation) = + let (project_id, mut worker_port_handoff) = seed_foundation_php_project(&paths, &project_path, "php: \"8.4\"\n")?; - let worker_service_port = worker_port_reservation.local_addr()?.port(); - drop(worker_port_reservation); let mut gateway_guard = SeededGatewayGuard::new(paths.clone()); gateway_guard.attach_worker("8.4"); + worker_port_handoff.release_for_runtime_start(); let initial_daemon = daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; @@ -1927,20 +3373,9 @@ async fn daemon_health_automatically_recovers_killed_worker_with_invalid_project wait_for_job_scope_status(&paths, &format!("project:{project_id}"), JobStatus::Failed) .await?; - let recovered_worker = supervisor - .adopt_recorded( - &paths.worker_pid("8.4"), - &paths.worker_runtime_metadata("8.4"), - )? - .ok_or_else(|| anyhow!("health recovery did not restart the worker"))?; - daemon::wait_for_readiness( - daemon::ReadinessCheck::Tcp { - host: "127.0.0.1".to_owned(), - port: worker_service_port, - }, - Duration::from_secs(1), - ) - .await?; + let recovered_worker_pid = worker_port_handoff + .verify_publication_and_release_lock(&paths, "8.4") + .await?; let database = Database::open(&paths)?; let assignments = database.assigned_ports()?; @@ -1962,7 +3397,7 @@ async fn daemon_health_automatically_recovers_killed_worker_with_invalid_project .await?; let runtime_states = database.runtime_observed_states()?; - assert_ne!(recovered_worker.pid(), initial_worker_pid); + assert_ne!(recovered_worker_pid, initial_worker_pid); assert_eq!(job.status, JobStatus::Failed); assert_eq!(state::fs::read_to_string(&config_path)?, "php: [\n"); assert!(runtime_states.iter().any(|state| { @@ -2207,6 +3642,37 @@ async fn idle_client_without_newline_does_not_block_health_requests() -> Result< Ok(()) } +#[tokio::test] +async fn fallback_shutdown_drains_idle_client_without_waiting_for_request_timeout() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + let daemon = + daemon::RunningDaemon::start_without_managed_resource_adapters(paths.clone()).await?; + let mut idle_stream = UnixStream::connect(paths.daemon_socket()).await?; + idle_stream.write_all(b"{").await?; + request_lines( + &paths, + json!({ + "protocol_version": daemon::PROTOCOL_VERSION, + "command": "health", + }), + ) + .await?; + + daemon.shutdown_without_waiting_for_test()?; + let mut response = Vec::new(); + timeout( + Duration::from_secs(5), + idle_stream.read_to_end(&mut response), + ) + .await??; + + assert!(response.is_empty()); + assert!(!paths.daemon_socket().exists()); + + Ok(()) +} + #[tokio::test] async fn start_removes_stale_socket_before_binding() -> Result<()> { let tempdir = tempdir()?; @@ -2590,17 +4056,62 @@ async fn send_raw_request(paths: &PvPaths, request: &str) -> Result<()> { } async fn request_lines(paths: &PvPaths, request: Value) -> Result> { - let mut stream = UnixStream::connect(paths.daemon_socket()).await?; - let request = serde_json::to_string(&request)?; - stream.write_all(request.as_bytes()).await?; - stream.write_all(b"\n").await?; + request_lines_with_timeout(paths, request, REQUEST_LINES_TIMEOUT).await +} - let mut reader = BufReader::new(stream); +async fn request_lines_with_timeout( + paths: &PvPaths, + request: Value, + limit: Duration, +) -> Result> { + let deadline = TokioInstant::now() + limit; + let request_payload = serde_json::to_string(&request)?; let mut lines = Vec::new(); + let mut stream = request_step( + deadline, + UnixStream::connect(paths.daemon_socket()), + &paths.daemon_socket(), + &request_payload, + "connect", + limit, + &lines, + ) + .await?; + request_step( + deadline, + stream.write_all(request_payload.as_bytes()), + &paths.daemon_socket(), + &request_payload, + "request write", + limit, + &lines, + ) + .await?; + request_step( + deadline, + stream.write_all(b"\n"), + &paths.daemon_socket(), + &request_payload, + "newline write", + limit, + &lines, + ) + .await?; + + let mut reader = BufReader::new(stream); loop { let mut line = String::new(); - let bytes = reader.read_line(&mut line).await?; + let bytes = request_step( + deadline, + reader.read_line(&mut line), + &paths.daemon_socket(), + &request_payload, + "response read", + limit, + &lines, + ) + .await?; if bytes == 0 { break; @@ -2612,7 +4123,73 @@ async fn request_lines(paths: &PvPaths, request: Value) -> Result> { Ok(lines) } +#[tokio::test] +async fn request_lines_timeout_reports_stage_request_and_partial_responses() -> Result<()> { + let tempdir = tempdir()?; + let paths = PvPaths::for_home(tempdir.path().join("home")); + state::fs::ensure_layout(&paths)?; + let listener = UnixListener::bind(paths.daemon_socket())?; + let peer = tokio::spawn(async move { + let (stream, _address) = listener.accept().await?; + let mut reader = BufReader::new(stream); + let mut request = String::new(); + reader.read_line(&mut request).await?; + reader + .get_mut() + .write_all(b"{\"status\":\"accepted\"}\n") + .await?; + std::future::pending::<()>().await; + Ok::<(), io::Error>(()) + }); + + let error = request_lines_with_timeout( + &paths, + json!({"command": "pending"}), + Duration::from_millis(50), + ) + .await + .err() + .ok_or_else(|| anyhow!("request unexpectedly completed"))?; + peer.abort(); + let _peer_result = peer.await; + + assert_debug_snapshot!(error.to_string()); + + Ok(()) +} + +async fn request_step( + deadline: TokioInstant, + operation: impl Future>, + endpoint: &Utf8Path, + request_payload: &str, + stage: &str, + limit: Duration, + partial_responses: &[Value], +) -> Result +where + E: Into, +{ + match timeout_at(deadline, operation).await { + Ok(Ok(value)) => Ok(value), + Ok(Err(error)) => Err(error.into()).with_context(|| { + format!( + "daemon request {request_payload} failed during {stage} at {endpoint}; partial responses: {}", + Value::Array(partial_responses.to_vec()) + ) + }), + Err(_elapsed) => Err(anyhow!( + "daemon request {request_payload} exceeded its {limit:?} deadline during {stage}; partial responses: {}", + Value::Array(partial_responses.to_vec()) + )), + } +} + async fn wait_for_succeeded_job_id(paths: &PvPaths, id: &str) -> Result { + wait_for_job_id_status(paths, id, JobStatus::Succeeded).await +} + +async fn wait_for_job_id_status(paths: &PvPaths, id: &str, status: JobStatus) -> Result { let deadline = Instant::now() + JOB_STATUS_WAIT_TIMEOUT; loop { @@ -2620,7 +4197,7 @@ async fn wait_for_succeeded_job_id(paths: &PvPaths, id: &str) -> Result Result Result<()> { + timeout(TARGETED_SCENARIO_TIMEOUT, async { + loop { + if path.exists() { + return; + } + sleep(JOB_STATUS_POLL_INTERVAL).await; + } + }) + .await + .map_err(|_elapsed| anyhow!("file {path} was not created")) +} + +async fn wait_for_runtime_replacement_required(metadata_path: &Utf8Path) -> Result<()> { + timeout(TARGETED_SCENARIO_TIMEOUT, async { + loop { + if let Ok(content) = state::fs::read_to_string(metadata_path) + && let Ok(metadata) = serde_json::from_str::(&content) + && metadata["replacement_required"] == true + { + return; + } + sleep(JOB_STATUS_POLL_INTERVAL).await; + } + }) + .await + .map_err(|_elapsed| { + anyhow!("runtime metadata {metadata_path} did not record a pending replacement") + }) } async fn wait_for_succeeded_job_scope(paths: &PvPaths, scope: &str) -> Result { diff --git a/crates/daemon/tests/gateway_reconciliation.rs b/crates/daemon/tests/gateway_reconciliation.rs index 4612bb80..b7d608e2 100644 --- a/crates/daemon/tests/gateway_reconciliation.rs +++ b/crates/daemon/tests/gateway_reconciliation.rs @@ -3624,6 +3624,52 @@ async fn frankenphp_config_validation_timeout_stops_validator_process_group() -> Ok(()) } +#[cfg(target_os = "macos")] +#[tokio::test] +async fn config_validation_stops_descendant_that_retains_output_after_leader_exit() -> Result<()> { + let tempdir = tempdir()?; + let validator = tempdir.path().join("detached-output-validator"); + let validator_child_pid = tempdir.path().join("validator-child.pid"); + let config_path = tempdir.path().join("Caddyfile"); + + fs::write_sensitive_file( + &validator, + &format!( + r#"#!/bin/sh +set -eu + +if [ "$1" = "validate" ]; then + sleep 30 & + echo "$!" > {} + exit 0 +fi + +exit 2 +"#, + shell_single_quoted(validator_child_pid.as_str()) + ), + )?; + set_executable(&validator)?; + fs::write_sensitive_file(&config_path, "{}\n")?; + + timeout( + Duration::from_secs(5), + validate_config( + &CaddyCliCommand::frankenphp(&validator), + &config_path, + &BTreeMap::new(), + ), + ) + .await??; + + let child_pid = state::testing::read_to_string(&validator_child_pid)? + .trim() + .parse::()?; + wait_for_process_exit(child_pid).await?; + + Ok(()) +} + #[tokio::test] async fn retired_worker_cleanup_removes_runtime_identity() -> Result<()> { let tempdir = tempdir()?; diff --git a/crates/daemon/tests/project_env_reconciliation.rs b/crates/daemon/tests/project_env_reconciliation.rs index 1bb82b59..43e712b7 100644 --- a/crates/daemon/tests/project_env_reconciliation.rs +++ b/crates/daemon/tests/project_env_reconciliation.rs @@ -1,9 +1,11 @@ use std::collections::BTreeMap; +use std::future::Future; use std::io; use std::net::{Ipv4Addr, TcpListener, UdpSocket}; use std::os::unix::fs::PermissionsExt; +use std::time::Duration as StdDuration; -use anyhow::{Result, anyhow}; +use anyhow::{Context, Result, anyhow}; use camino::Utf8Path; use camino_tempfile::tempdir; use insta::{Settings, assert_debug_snapshot}; @@ -22,6 +24,9 @@ use state::{ use time::{Duration, OffsetDateTime}; use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader}; use tokio::net::UnixStream; +use tokio::time::{Instant as TokioInstant, timeout_at}; + +const REQUEST_LINES_TIMEOUT: StdDuration = StdDuration::from_secs(30); #[tokio::test] async fn resource_only_project_uses_custom_env_file_and_no_php_worker() -> Result<()> { @@ -2063,17 +2068,55 @@ fn bind_loopback_tcp_udp_pair() -> Result<(u16, TcpListener, UdpSocket)> { } async fn request_lines(paths: &PvPaths, request: Value) -> Result> { - let mut stream = UnixStream::connect(paths.daemon_socket()).await?; - let request = serde_json::to_string(&request)?; - stream.write_all(request.as_bytes()).await?; - stream.write_all(b"\n").await?; + let limit = REQUEST_LINES_TIMEOUT; + let deadline = TokioInstant::now() + limit; + let request_payload = serde_json::to_string(&request)?; + let mut lines = Vec::new(); + let mut stream = request_step( + deadline, + UnixStream::connect(paths.daemon_socket()), + &paths.daemon_socket(), + &request_payload, + "connect", + limit, + &lines, + ) + .await?; + request_step( + deadline, + stream.write_all(request_payload.as_bytes()), + &paths.daemon_socket(), + &request_payload, + "request write", + limit, + &lines, + ) + .await?; + request_step( + deadline, + stream.write_all(b"\n"), + &paths.daemon_socket(), + &request_payload, + "newline write", + limit, + &lines, + ) + .await?; let mut reader = BufReader::new(stream); - let mut lines = Vec::new(); loop { let mut line = String::new(); - let bytes = reader.read_line(&mut line).await?; + let bytes = request_step( + deadline, + reader.read_line(&mut line), + &paths.daemon_socket(), + &request_payload, + "response read", + limit, + &lines, + ) + .await?; if bytes == 0 { break; @@ -2085,6 +2128,33 @@ async fn request_lines(paths: &PvPaths, request: Value) -> Result> { Ok(lines) } +async fn request_step( + deadline: TokioInstant, + operation: impl Future>, + endpoint: &Utf8Path, + request_payload: &str, + stage: &str, + limit: StdDuration, + partial_responses: &[Value], +) -> Result +where + E: Into, +{ + match timeout_at(deadline, operation).await { + Ok(Ok(value)) => Ok(value), + Ok(Err(error)) => Err(error.into()).with_context(|| { + format!( + "daemon request {request_payload} failed during {stage} at {endpoint}; partial responses: {}", + Value::Array(partial_responses.to_vec()) + ) + }), + Err(_elapsed) => Err(anyhow!( + "daemon request {request_payload} exceeded its {limit:?} deadline during {stage}; partial responses: {}", + Value::Array(partial_responses.to_vec()) + )), + } +} + fn link_project( paths: &PvPaths, project_path: &Utf8Path, diff --git a/crates/daemon/tests/snapshots/daemon_foundation__fixture_cleanup_preserves_operation_error_precedence.snap b/crates/daemon/tests/snapshots/daemon_foundation__fixture_cleanup_preserves_operation_error_precedence.snap new file mode 100644 index 00000000..4d4f1892 --- /dev/null +++ b/crates/daemon/tests/snapshots/daemon_foundation__fixture_cleanup_preserves_operation_error_precedence.snap @@ -0,0 +1,18 @@ +--- +source: crates/daemon/tests/daemon_foundation.rs +expression: outcomes +--- +[ + Ok( + "value", + ), + Err( + "operation sentinel", + ), + Err( + "cleanup sentinel", + ), + Err( + "operation failed: operation sentinel; fixture cleanup failed: cleanup sentinel", + ), +] diff --git a/crates/daemon/tests/snapshots/daemon_foundation__request_lines_timeout_reports_stage_request_and_partial_responses.snap b/crates/daemon/tests/snapshots/daemon_foundation__request_lines_timeout_reports_stage_request_and_partial_responses.snap new file mode 100644 index 00000000..9fd96f8d --- /dev/null +++ b/crates/daemon/tests/snapshots/daemon_foundation__request_lines_timeout_reports_stage_request_and_partial_responses.snap @@ -0,0 +1,6 @@ +--- +source: crates/daemon/tests/daemon_foundation.rs +assertion_line: 2900 +expression: error.to_string() +--- +"daemon request {\"command\":\"pending\"} exceeded its 50ms deadline during response read; partial responses: [{\"status\":\"accepted\"}]"