Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
64 changes: 55 additions & 9 deletions crates/engram-coordinator/src/api/snapshot.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1106,8 +1106,8 @@ pub(crate) async fn resume_from_idle(
// instead of declaring the session dead. This is also the documented
// recovery path for `ColdBootUnavailable` Idle fallbacks: re-enable
// the image, then /resume lands here.
if session.live_disk_manifest.is_some() {
return resume_disk_only_cold_boot(ctx, session).await;
if let Some(rootfs) = session.live_disk_manifest {
return resume_disk_only_cold_boot(ctx, session, rootfs).await;
}
tracing::warn!(
session_id = %id,
Expand Down Expand Up @@ -1146,13 +1146,16 @@ pub(crate) async fn destroy_retained_sandbox(
}

/// ADR 0028 Fix B — manual-resume flavor of the disk-only cold boot:
/// fresh kernel boot mounting the session's `live_disk_manifest` on
/// whichever host can take it, fresh harness. On-disk work survives;
/// in-RAM context does not (this path only exists because no coherent
/// memory snapshot was ever recorded).
/// fresh kernel boot mounting `rootfs` on whichever host can take it,
/// fresh harness. On-disk work survives; in-RAM context does not. This
/// path runs when no coherent memory snapshot was recorded (`rootfs` is
/// the session's `live_disk_manifest`), or when the host refused the
/// memory image as unusable (`rootfs` is the newest disk of the live
/// and snapshot lineages).
async fn resume_disk_only_cold_boot(
ctx: &crate::session_ops::OpCtx<'_>,
session: Session,
rootfs: engram_core::types::manifest::ManifestRef,
) -> Result<SnapshotResponse, ApiError> {
let state = ctx.state;
let id = session.id;
Expand All @@ -1162,15 +1165,19 @@ async fn resume_disk_only_cold_boot(
let mut spec = crate::boot_materializer::materialize_cold_boot(state, &session)
.await?
.ok_or_else(|| ApiError::Conflict("disk-only recovery requires an enabled image".into()))?;
spec.rootfs_manifest = session.live_disk_manifest;
spec.rootfs_manifest = Some(rootfs);
let (repo, tag) = engram_core::types::session::split_image_ref(&session.image);
// The same budgets the memory restore was admitted with: a cold boot
// needs the RAM and CPU too, so placement must check that they fit.
let budget =
crate::boot_materializer::resolve_resume_budget(&state.services.meta, &session).await;
let context = crate::placement::ScheduleContext {
nbd_slot_need: 1 + u32::from(spec.swap_mib.unwrap_or(0) > 0),
repo,
image_version: tag,
snapshot_host: None,
memory_mib: None,
cpu_budget_vcpus: None,
memory_mib: budget.map(|(mib, _)| mib),
cpu_budget_vcpus: budget.map(|(_, vcpus)| vcpus),
required_image_digest: None,
exclude_host: None,
prefer_host: session.host_id,
Expand Down Expand Up @@ -1949,6 +1956,45 @@ async fn resume_from_fc_snapshot(
.await
{
Ok(v) => v,
// The host refused the memory image (for example, one that can
// reference swap pages no snapshot holds). Every host refuses
// it the same way, so a retry cannot help, but the disk is
// intact: boot a fresh kernel on the newest disk. The memory
// pairing rule above does not apply, because no memory is
// restored, so the live lineage wins when it is newer.
Err(SandboxError::MemoryImageUnusable(reason)) => {
let Some(rootfs) =
effective_resume_disk_manifest(session.live_disk_manifest, record.disk_manifest)
else {
return Err(ApiError::Internal(format!(
"memory image unusable and the session has no disk manifest: {reason}"
)));
};
tracing::warn!(
session_id = %id,
snapshot_id = %record.id,
rootfs = ?rootfs,
%reason,
"resume: host refused the memory image; recovering with a disk-only cold boot",
);
let response = resume_disk_only_cold_boot(op_ctx, session, rootfs).await?;
::metrics::counter!(crate::metrics::SESSION_RESUME_MEMORY_IMAGE_FALLBACK_TOTAL)
.increment(1);
// Tell the user why their processes are gone. Best-effort: the
// session is already Active, and a failed emit must not undo that.
let _ = state
.emit_fenced(
id,
op_ctx.fence(),
SessionEvent::ResumedFromDisk {
disk_manifest: rootfs,
reason,
at: state.services.clock.now_utc(),
},
)
.await;
return Ok(response);
}
Err(SandboxError::Snapshot(msg)) => {
// Lost local artifacts on every viable host — chunked
// restore couldn't rehydrate from the manifest either.
Expand Down
7 changes: 7 additions & 0 deletions crates/engram-coordinator/src/metrics.rs
Original file line number Diff line number Diff line change
Expand Up @@ -388,6 +388,13 @@ pub const SESSION_OP_RESUME_BUDGET_EXHAUSTED_TOTAL: &str =
/// Should be ~0; each increment is one session honestly declared
/// unresumable instead of churning the outbox shim forever.
pub const SESSION_UNRESUMABLE_DEMOTED_TOTAL: &str = "engram_session_unresumable_demoted_total";
/// Resumes whose memory image the host refused as unusable (for
/// example, a memory image captured before swap was a chunked disk)
/// that then completed with a disk-only cold boot. Counted only on
/// success: each increment is one session that lost its in-RAM context
/// but kept its disk.
pub const SESSION_RESUME_MEMORY_IMAGE_FALLBACK_TOTAL: &str =
"engram_session_resume_memory_image_fallback_total";
/// ADR 0079 (review finding #5): orphaned Pending sessions (placed but
/// no active create_boot op) re-enqueued by the reclaim sweep backstop.
pub const SESSION_OP_PENDING_ORPHANS_RECOVERED_TOTAL: &str =
Expand Down
12 changes: 12 additions & 0 deletions crates/engram-coordinator/src/state.rs
Original file line number Diff line number Diff line change
Expand Up @@ -420,6 +420,17 @@ pub enum SessionEvent {
reason: String,
at: DateTime<Utc>,
},
/// A resume recovered with a disk-only cold boot because the host
/// refused the session's memory image. The disk and every file on it
/// are kept; running processes, shells and in-memory state are gone.
/// Coordinator-authoritative, so a rewind keeps it.
ResumedFromDisk {
/// The disk the fresh kernel booted on.
disk_manifest: engram_core::types::manifest::ManifestRef,
/// Why the host refused the memory image.
reason: String,
at: DateTime<Utc>,
},
/// ADR 0107: a validated session-mode directive rode a prompt (e.g.
/// `plan`). Coordinator-authoritative — the user genuinely selected the
/// mode — so `rewind_session_to_cursor` excludes this kind from its
Expand Down Expand Up @@ -549,6 +560,7 @@ impl SessionEvent {
Self::FileShared { .. } => "file_shared",
Self::RecoveredFromCheckpoint { .. } => "recovered_from_checkpoint",
Self::DurabilityRollback { .. } => "durability_rollback",
Self::ResumedFromDisk { .. } => "resumed_from_disk",
}
}

Expand Down
9 changes: 8 additions & 1 deletion crates/engram-coordinator/src/teleport.rs
Original file line number Diff line number Diff line change
Expand Up @@ -525,7 +525,14 @@ mod steps {
Err(SandboxError::NotFound | SandboxError::AlreadyExists) => {
rollback_begin(ctx, row, "live_export_lost".into()).await
}
Err(e) if e.to_string().contains("postcopy-never-loaded") => {
// The destination refused the guest's memory image (for
// example, a guest still on a raw swap file). Every host
// refuses it the same way, so a retry cannot help; abort the
// export and keep the source running.
Err(e)
if matches!(e, SandboxError::MemoryImageUnusable(_))
|| e.to_string().contains("postcopy-never-loaded") =>
{
ctx.state
.host_registry
.backend_for(row.source_host_id)
Expand Down
8 changes: 8 additions & 0 deletions crates/engram-coordinator/tests/support/teleport.rs
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,8 @@ pub struct ScriptedHost {
pub clock: Arc<dyn Clock>,
pub capture_fails: AtomicBool,
pub restore_fails: AtomicBool,
/// `restore` refuses the memory image (`MemoryImageUnusable`).
pub memory_refused: AtomicBool,
pub resume_fails: AtomicBool,
/// `resume` answers NotFound: the source sandbox no longer exists.
pub resume_not_found: AtomicBool,
Expand Down Expand Up @@ -50,6 +52,7 @@ impl ScriptedHost {
clock,
capture_fails: AtomicBool::new(false),
restore_fails: AtomicBool::new(false),
memory_refused: AtomicBool::new(false),
resume_fails: AtomicBool::new(false),
resume_not_found: AtomicBool::new(false),
abort_not_found: AtomicBool::new(false),
Expand Down Expand Up @@ -161,6 +164,11 @@ impl HostClient for ScriptedHost {
_fence: engram_core::traits::SessionFence,
) -> Result<SandboxId, SandboxError> {
self.restores.fetch_add(1, Ordering::SeqCst);
if self.memory_refused.load(Ordering::SeqCst) {
return Err(SandboxError::MemoryImageUnusable(
"swap restore has no manifest".into(),
));
}
if self.restore_fails.load(Ordering::SeqCst) {
return Err(SandboxError::Snapshot("restore failed".into()));
}
Expand Down
27 changes: 27 additions & 0 deletions crates/engram-coordinator/tests/support/teleport_scenarios.rs
Original file line number Diff line number Diff line change
Expand Up @@ -442,6 +442,33 @@ async fn lost_live_export_rolls_back_with_durable_error(rig: Rig) {
}
scenario!(lost_live_export_rolls_back_with_durable_error);

/// A destination that refuses the guest's memory image refuses it on
/// every attempt. The move rolls back at once and the source keeps
/// running; it never retries into `fail_move`, which would destroy it.
async fn refused_memory_image_rolls_back_live_move(rig: Rig) {
rig.make_live().await;
rig.dest.memory_refused.store(true, Ordering::SeqCst);
// One drive settles the move: before the typed arm, the refusal
// returned Retry here and went to `fail_move` after three attempts.
assert!(matches!(rig.drive().await, OpOutcome::Done));
assert_eq!(rig.dest.restores.load(Ordering::SeqCst), 1, "no retry");
assert!(
rig.source.aborts.load(Ordering::SeqCst) >= 1,
"the live export is released through migration_abort",
);
assert_eq!(
rig.source.destroys.load(Ordering::SeqCst),
0,
"the source guest is never destroyed",
);
assert_eq!(rig.phase().await, None);
assert_eq!(
meta_session(&rig).await.status,
engram_core::types::SessionState::Active
);
}
scenario!(refused_memory_image_rolls_back_live_move);

/// A consumed export (an earlier abort passed its point of no return) is
/// not a resumed guest: the rollback still needs the resume ack before it
/// declares Active.
Expand Down
10 changes: 10 additions & 0 deletions crates/engram-core/src/error.rs
Original file line number Diff line number Diff line change
Expand Up @@ -172,6 +172,15 @@ pub enum SandboxError {
kind: String,
message: String,
},
/// The host refused a snapshot restore because the memory image
/// cannot be restored exactly on this fleet, for example a memory
/// image that can reference swap pages no snapshot holds. The
/// refusal is deterministic: every host gives the same answer, so
/// a retry cannot succeed. The disk is unaffected, and the resume
/// recovers with a disk-only cold boot. Crosses the host→coord
/// gRPC boundary as a `failed_precondition` with a marker message
/// (the `HarnessSpawn` precedent).
MemoryImageUnusable(String),
}

/// ADR 0116 B-D4: is a harness-spawn failure of this `kind`
Expand Down Expand Up @@ -214,6 +223,7 @@ impl fmt::Display for SandboxError {
Self::HarnessSpawn { kind, message } => {
write!(f, "harness_spawn: kind={kind} {message}")
}
Self::MemoryImageUnusable(msg) => write!(f, "memory_image_unusable: {msg}"),
}
}
}
Expand Down
27 changes: 26 additions & 1 deletion crates/engram-dst/src/world.rs
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,16 @@ pub struct SimHostState {
pub rpc_hang: Option<std::time::Duration>,
/// sandbox -> owning session (as told to us via create's spec).
pub sandboxes: BTreeMap<SandboxId, Option<SessionId>>,
/// Refuse every snapshot `restore` as `MemoryImageUnusable`, the
/// answer a real host gives for a memory image it cannot restore
/// exactly (for example, one captured before swap was a chunked
/// disk). Off by default, so seeds are unchanged.
pub refuse_memory_images: bool,
/// How many restores `refuse_memory_images` refused.
pub memory_refusals: u32,
/// The root disk manifest each `create` was asked to mount, in call
/// order. A disk-only cold boot carries `Some`.
pub created_rootfs: Vec<Option<engram_core::types::manifest::ManifestRef>>,
}

/// A mutating host-verb's WORLD-side effect (ADR 0098 R2). Every
Expand Down Expand Up @@ -863,13 +873,16 @@ impl SimHostClient {

#[async_trait]
impl HostClient for SimHostClient {
async fn create(&self, _spec: SandboxSpec) -> Result<SandboxId, SandboxError> {
async fn create(&self, spec: SandboxSpec) -> Result<SandboxId, SandboxError> {
self.maybe_hang().await;
// Draw the id BEFORE the liveness gate so the entropy stream is
// identical to the pre-effect-queue world even on a down-host
// failure (Calm determinism).
let id = SandboxId::from(self.entropy.uuid());
self.world.require_up(self.host_id)?;
if let Some(h) = self.world.hosts.lock().get_mut(&self.host_id) {
h.created_rootfs.push(spec.rootfs_manifest);
}
// Ownership is learned at bind_session time (the spec is a
// template, not a binding — see sandbox.rs's type docs).
self.world.record_effect(
Expand Down Expand Up @@ -1003,6 +1016,18 @@ impl HostClient for SimHostClient {
self.maybe_hang().await;
let id = SandboxId::from(self.entropy.uuid());
self.world.require_up(self.host_id)?;
if let Some(h) = self
.world
.hosts
.lock()
.get_mut(&self.host_id)
.filter(|h| h.refuse_memory_images)
{
h.memory_refusals += 1;
return Err(SandboxError::MemoryImageUnusable(
"sim: memory image refused".into(),
));
}
// ADR 0108 E: a snapshot restore resumes a captured-warm harness
// (ADR 0037) which re-dials on the vsock epoch bump. The dial is
// scheduled by the effect's APPLICATION (the harness lives inside
Expand Down
Loading
Loading