runtime: canonicalize Worker aggregates

This commit is contained in:
2026-08-11 22:38:29 +09:00
parent 33d3d1a43f
commit 34f8949e85
16 changed files with 2013 additions and 201 deletions
+47 -2
View File
@@ -42,6 +42,8 @@ pub struct SpawnedWorkerRecord {
pub struct RuntimeDir {
path: PathBuf,
write_legacy_snapshots: bool,
preserve_on_drop: bool,
socket_file_name: &'static str,
}
impl RuntimeDir {
@@ -56,6 +58,8 @@ impl RuntimeDir {
Ok(Self {
path,
write_legacy_snapshots: true,
preserve_on_drop: false,
socket_file_name: "sock",
})
}
@@ -69,6 +73,36 @@ impl RuntimeDir {
Ok(Self {
path,
write_legacy_snapshots: false,
preserve_on_drop: false,
socket_file_name: "sock",
})
}
/// Create an exact, persistent generation-scoped Worker run directory.
/// Existing directories are rejected so stale artifacts cannot be reused.
pub async fn create_worker_run(path: &Path) -> Result<Self, io::Error> {
let parent = path
.parent()
.ok_or_else(|| io::Error::other("run path has no parent"))?;
fs::create_dir_all(parent).await?;
fs::create_dir(path).await?;
fs::create_dir(path.join("artifacts")).await?;
fs::create_dir(path.join("spawned")).await?;
for log in ["worker.out.log", "worker.err.log"] {
let file = fs::OpenOptions::new()
.write(true)
.create_new(true)
.open(path.join(log))
.await?;
file.sync_all().await?;
}
std::fs::File::open(path)?.sync_all()?;
std::fs::File::open(parent)?.sync_all()?;
Ok(Self {
path: path.to_path_buf(),
write_legacy_snapshots: false,
preserve_on_drop: true,
socket_file_name: "worker.sock",
})
}
@@ -116,13 +150,24 @@ impl RuntimeDir {
/// that only know the worker name (e.g. the TUI's attach flow)
/// predict the same path via [`manifest::paths::worker_socket_path`].
pub fn socket_path(&self) -> PathBuf {
self.path.join("sock")
self.path.join(self.socket_file_name)
}
pub async fn close_socket(&self) -> Result<(), io::Error> {
match fs::remove_file(self.socket_path()).await {
Ok(()) => Ok(()),
Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()),
Err(error) => Err(error),
}
}
}
impl Drop for RuntimeDir {
fn drop(&mut self) {
let _ = std::fs::remove_dir_all(&self.path);
let _ = std::fs::remove_file(self.socket_path());
if !self.preserve_on_drop {
let _ = std::fs::remove_dir_all(&self.path);
}
}
}
@@ -120,15 +120,15 @@ pub fn adopt_allocation(
/// The Worker's in-memory `segment_id` can change underneath the
/// allocation in two normal places:
///
/// - `Worker::compact` mints a fresh session and swaps it in.
/// - `session_store::ensure_head_or_fork` auto-forks when another
/// - `Worker::compact` mints a fresh Segment in the same Session.
/// - `session_store::ensure_head_or_fork` auto-forks within that Session when another
/// writer has advanced the store head behind our back.
///
/// Both paths must call this so subsequent [`lookup_segment`] queries
/// find the live session id, not the old one. Without this update a
/// find the live Segment id, not the old one. Without this update a
/// concurrent `restore_from_manifest(new_id)` would see "no live
/// writer" and proceed to register a competing allocation on the
/// session this Worker just moved into.
/// Segment lineage this Worker just moved into.
///
/// The lock is opened once and the allocation is rewritten inside the
/// guard, so the segment_id collision check is atomic with the