FDS/OS 1.0

This commit is contained in:
2026-09-21 22:29:23 +08:00
commit 99bc3d15c5
430 changed files with 34876 additions and 0 deletions
View File
+23
View File
@@ -0,0 +1,23 @@
[package]
name = "fds-cartridged"
version = "0.1.0"
edition = "2024"
license = "MIT"
description = "FDS event-driven cartridge and physical bay manager"
[dependencies]
clap.workspace = true
fds-common = { path = "../fds-common" }
fds-software = { path = "../fds-software" }
fds-burn = { path = "../fds-burn" }
serde = { version = "1", features = ["derive"] }
libc = "0.2"
serde_json = "1"
[[bin]]
name = "fds-cartridged"
path = "src/main.rs"
[[bin]]
name = "fds-profile"
path = "src/profile-main.rs"
+358
View File
@@ -0,0 +1,358 @@
//! One background writer, with root-owned operation records retained until the
//! physical insertion changes. Restart never turns an interrupted write SAFE.
use crate::media::checked;
use fds_burn::{device::Disk, image, worker::Preparation};
use fds_common::{
Bay, Error, Result,
control::{LIMIT, MediaJob},
manifest::Class,
};
use serde::{Deserialize, Serialize};
use std::{
collections::BTreeMap,
fs,
io::{self, Read, Write},
os::{
fd::AsRawFd,
unix::{fs::PermissionsExt, process::CommandExt},
},
path::Path,
process::{Child, ChildStdout, Command, Stdio},
};
const DIRECTORY: &str = "/run/fds/burn";
#[derive(Serialize, Deserialize)]
struct Record {
disk: Disk,
job: MediaJob,
owner: u32,
}
struct Worker {
child: Child,
output: ChildStdout,
input: Vec<u8>,
bay: Bay,
confirmed: bool,
cancelled: bool,
}
#[derive(Default)]
pub struct Manager {
records: BTreeMap<Bay, Record>,
active: Option<Worker>,
}
impl Manager {
pub fn active(&self) -> bool {
self.active.is_some()
}
pub fn load() -> Result<Self> {
fs::create_dir_all(DIRECTORY)?;
fs::set_permissions(DIRECTORY, fs::Permissions::from_mode(0o700))?;
let mut this = Self::default();
for n in 1..=12 {
let bay = Bay::try_from(n)?;
let path = format!("{DIRECTORY}/{bay}.json");
if Path::new(&path).exists() {
let mut record: Record =
serde_json::from_str(&fds_common::read_text(Path::new(&path), LIMIT as u64)?)
.map_err(|e| Error(format!("Invalid burn recovery record: {e}")))?;
if record.job.bay != bay {
return Err(Error("Burn recovery bay mismatch".into()));
}
if !record.job.finished() {
record.job.phase = "failed".into();
record.job.sequence += 1;
record.job.confirmation = None;
record.job.error = Some(
"Media service restarted during an operation; cartridge is not SAFE".into(),
);
}
this.records.insert(bay, record);
this.save(bay)?;
}
}
Ok(this)
}
fn save(&self, bay: Bay) -> Result<()> {
let path = format!("{DIRECTORY}/{bay}.json");
fs::write(
format!("{path}.next"),
serde_json::to_vec(&self.records[&bay]).map_err(|e| Error(e.to_string()))?,
)?;
fs::rename(format!("{path}.next"), path)?;
Ok(())
}
pub fn reserved(&self, bay: Bay) -> Option<&MediaJob> {
self.records
.get(&bay)
.filter(|r| r.disk.present())
.map(|r| &r.job)
}
pub fn status(&self, id: &str) -> Result<MediaJob> {
self.records
.values()
.find(|r| r.job.id == id)
.map(|r| r.job.clone())
.ok_or_else(|| Error("Unknown media operation".into()))
}
pub fn begin(
&mut self,
bay: Bay,
usb: &Path,
path: &str,
class: Class,
uid: u32,
) -> Result<MediaJob> {
if self.active.is_some() {
return Err(Error(
"Another media operation is active; finish or cancel it first".into(),
));
}
image::label(class)?;
if !path.starts_with('/') || path.len() > 4096 || path.contains('\0') {
return Err(Error(
"Image path must be an absolute path of at most 4096 bytes".into(),
));
}
let disk = Disk::select_current(usb)?;
disk.protect(Path::new("/sys"), Path::new("/proc"))?;
let job = MediaJob {
id: image::hex(&image::random_id()?),
bay,
sequence: 0,
phase: "inspecting".into(),
diskseq: disk.diskseq,
target_bytes: disk.bytes,
model: disk.model.clone(),
serial: disk.serial.clone(),
image_class: class,
image_bytes: None,
image_sha256: None,
progress_bytes: 0,
confirmation: None,
error: None,
};
let preparation = Preparation {
disk: disk.clone(),
image_path: path.into(),
uid,
job: job.clone(),
};
self.records.insert(
bay,
Record {
disk,
job: job.clone(),
owner: uid,
},
);
self.save(bay)?;
let parent = unsafe { libc::getpid() };
let mut command = Command::new("/usr/bin/fds-burn");
command
.arg("--worker")
.env_clear()
.env("PATH", "/usr/bin:/bin")
.env("LC_ALL", "C")
.stdin(Stdio::piped())
.stdout(Stdio::piped())
.stderr(Stdio::inherit());
unsafe {
command.pre_exec(move || {
let mut mask: libc::sigset_t = std::mem::zeroed();
libc::sigemptyset(&mut mask);
if libc::sigprocmask(libc::SIG_SETMASK, &mask, std::ptr::null_mut()) < 0
|| libc::prctl(libc::PR_SET_PDEATHSIG, libc::SIGKILL) < 0
{
return Err(io::Error::last_os_error());
}
if libc::getppid() != parent {
return Err(io::Error::other("Cartridge service exited"));
}
Ok(())
});
}
let started = (|| -> Result<Worker> {
let mut child = command.spawn()?;
let output = child.stdout.take().unwrap();
let setup = (|| -> Result<()> {
checked(
unsafe { libc::fcntl(output.as_raw_fd(), libc::F_SETFL, libc::O_NONBLOCK) },
"make burn status asynchronous",
)?;
let mut data =
serde_json::to_vec(&preparation).map_err(|e| Error(e.to_string()))?;
data.push(b'\n');
child.stdin.as_mut().unwrap().write_all(&data)?;
Ok(())
})();
if let Err(e) = setup {
let _ = child.kill();
let _ = child.wait();
return Err(e);
}
Ok(Worker {
child,
output,
input: Vec::new(),
bay,
confirmed: false,
cancelled: false,
})
})();
match started {
Ok(worker) => self.active = Some(worker),
Err(e) => {
let record = self.records.get_mut(&bay).unwrap();
record.job.phase = "failed".into();
record.job.error = Some(e.to_string());
record.job.sequence += 1;
self.save(bay)?;
return Err(e);
}
}
Ok(job)
}
pub fn descriptor(&self) -> i32 {
self.active.as_ref().map_or(-1, |w| w.output.as_raw_fd())
}
pub fn poll(&mut self) -> Result<()> {
let Some(worker) = self.active.as_mut() else {
return Ok(());
};
let bay = worker.bay;
let mut changed = false;
loop {
let mut chunk = [0; 8192];
match worker.output.read(&mut chunk) {
Ok(0) => break,
Ok(n) => {
worker.input.extend_from_slice(&chunk[..n]);
if worker.input.len() > LIMIT {
return Err(Error("Media worker status exceeded protocol limit".into()));
}
while let Some(end) = worker.input.iter().position(|b| *b == b'\n') {
let report: MediaJob = serde_json::from_slice(&worker.input[..end])
.map_err(|e| Error(format!("Invalid media worker status: {e}")))?;
worker.input.drain(..=end);
let record = self.records.get_mut(&bay).unwrap();
if report.id != record.job.id
|| report.bay != bay
|| report.diskseq != record.disk.diskseq
|| report.sequence <= record.job.sequence
{
return Err(Error("Media worker identity or sequence changed".into()));
}
record.job = report;
changed = true;
}
}
Err(e) if e.kind() == io::ErrorKind::WouldBlock => break,
Err(e) => return Err(e.into()),
}
}
let exited = worker.child.try_wait()?;
if let Some(status) = exited {
// The child can write its last report between EAGAIN and waitpid.
// Once it has exited, drain that final pipe data before deciding
// whether completion was reported.
worker.output.read_to_end(&mut worker.input)?;
while let Some(end) = worker.input.iter().position(|b| *b == b'\n') {
let report: MediaJob = serde_json::from_slice(&worker.input[..end])
.map_err(|e| Error(e.to_string()))?;
worker.input.drain(..=end);
let record = self.records.get_mut(&bay).unwrap();
if report.id != record.job.id
|| report.bay != bay
|| report.diskseq != record.disk.diskseq
|| report.sequence <= record.job.sequence
{
return Err(Error(
"Final media worker identity or sequence changed".into(),
));
}
record.job = report;
changed = true;
}
let record = self.records.get_mut(&bay).unwrap();
if !status.success() || !record.job.finished() {
record.job.phase = "failed".into();
record.job.sequence += 1;
record.job.confirmation = None;
record.job.error = Some(format!(
"Media worker exited without verified completion ({status}); no SAFE status issued"
));
changed = true;
}
self.active = None;
}
if changed {
self.save(bay)?;
}
Ok(())
}
fn control(&mut self, id: &str, uid: u32, line: &str, confirm: bool) -> Result<MediaJob> {
let job = self.status(id)?;
let record = &self.records[&job.bay];
if uid != 0 && uid != record.owner {
return Err(Error(
"Only the operation's owner may confirm or cancel it".into(),
));
}
if confirm
&& (job.phase != "awaiting_confirmation" || job.confirmation.as_deref() != Some(line))
{
return Err(Error(
"Confirmation must exactly match this cartridge and image".into(),
));
}
let worker = self
.active
.as_mut()
.filter(|w| w.bay == job.bay)
.ok_or_else(|| Error("Media operation is no longer active".into()))?;
if job.finished() {
return Err(Error("Media operation already finished".into()));
}
if (confirm && worker.confirmed) || worker.cancelled {
return Err(Error("Media control command was already sent".into()));
}
if confirm {
worker.confirmed = true;
} else {
worker.cancelled = true;
}
let input = worker
.child
.stdin
.as_mut()
.ok_or_else(|| Error("Media worker input closed".into()))?;
input.write_all(line.as_bytes())?;
input.write_all(b"\n")?;
Ok(job)
}
pub fn confirm(&mut self, id: &str, uid: u32, phrase: &str) -> Result<MediaJob> {
self.control(id, uid, phrase, true)
}
pub fn cancel(&mut self, id: &str, uid: u32) -> Result<MediaJob> {
self.control(id, uid, "CANCEL", false)
}
pub fn stop(&mut self) -> Result<()> {
if let Some(worker) = self.active.as_mut() {
worker.child.stdin.take();
}
while self.active.is_some() {
self.poll()?;
if self.active.is_some() {
let mut fd = libc::pollfd {
fd: self.descriptor(),
events: libc::POLLIN,
revents: 0,
};
let result = unsafe { libc::poll(&mut fd, 1, 1000) };
if result < 0 && io::Error::last_os_error().kind() != io::ErrorKind::Interrupted {
return Err(io::Error::last_os_error().into());
}
}
}
Ok(())
}
}
+257
View File
@@ -0,0 +1,257 @@
//! Managed jobs enter a root-owned cgroup before losing privileges or executing.
//! Descendants inherit membership; they cannot escape by double-forking.
use crate::media::{c, checked};
use fds_common::{Bay, Error, Result, read_text};
use std::{
fs::{self, File, OpenOptions},
io::{self, Read, Seek, SeekFrom},
os::{
fd::{AsRawFd, FromRawFd, OwnedFd},
unix::{
fs::{OpenOptionsExt, PermissionsExt},
process::CommandExt,
},
},
path::{Path, PathBuf},
process::{Child, Command, Stdio},
time::{Duration, Instant},
};
const ROOT: &str = "/sys/fs/cgroup/fds";
pub fn prepare() -> Result<()> {
let mut stat: libc::statfs = unsafe { std::mem::zeroed() };
checked(
unsafe { libc::statfs(c("/sys/fs/cgroup")?.as_ptr(), &mut stat) },
"inspect cgroup filesystem",
)?;
if stat.f_type as u64 == 0x6265_6572 {
checked(
unsafe {
libc::mount(
c("none")?.as_ptr(),
c("/sys/fs/cgroup")?.as_ptr(),
c("cgroup2")?.as_ptr(),
libc::MS_NOSUID | libc::MS_NODEV | libc::MS_NOEXEC,
std::ptr::null(),
)
},
"mount managed process hierarchy",
)?;
} else if stat.f_type as u64 != 0x6367_7270 {
return Err(Error(
"Managed programs require the cgroup v2 filesystem".into(),
));
}
fs::create_dir_all(ROOT)?;
fs::set_permissions(ROOT, fs::Permissions::from_mode(0o755))?;
fs::write("/sys/fs/cgroup/cgroup.subtree_control", b"+pids")?;
fs::write(Path::new(ROOT).join("cgroup.subtree_control"), b"+pids")?;
Ok(())
}
fn directory(name: &str) -> PathBuf {
Path::new(ROOT).join(name)
}
pub fn enter_service(name: &str) -> Result<()> {
if unsafe { libc::geteuid() } != 0 || !fds_common::manifest::identifier(name) {
return Err(Error("Invalid privileged service group".into()));
}
let path = directory(name);
fs::create_dir_all(&path)?;
fs::write(path.join("pids.max"), b"256")?;
fs::write(path.join("cgroup.procs"), b"0")?;
Ok(())
}
pub fn start(
bay: Bay,
arguments: &[String],
working: &str,
environment: &[(String, String)],
) -> Result<Child> {
start_group(&format!("bay{bay}"), arguments, working, environment)
}
pub fn start_group(
name: &str,
arguments: &[String],
working: &str,
environment: &[(String, String)],
) -> Result<Child> {
if !fds_common::manifest::identifier(name) {
return Err(Error("Invalid process group".into()));
}
if arguments.is_empty()
|| !arguments[0].starts_with('/')
|| arguments.len() > 128
|| arguments.iter().any(|a| a.contains('\0'))
{
return Err(Error(
"Managed run requires an absolute executable path and at most 128 arguments".into(),
));
}
let path = directory(name);
fs::create_dir_all(&path)?;
fs::write(path.join("pids.max"), b"256")?;
let group = OpenOptions::new()
.write(true)
.custom_flags(libc::O_CLOEXEC)
.open(path.join("cgroup.procs"))?;
let mut command = Command::new(&arguments[0]);
command
.args(&arguments[1..])
.current_dir(working)
.env_clear()
.env("PATH", "/usr/bin:/bin")
.env("HOME", "/home/fds")
.env("USER", "fds")
.env("LOGNAME", "fds")
.env("LANG", "en_US.UTF-8")
.env("SHELL", "/bin/bash")
.stdin(Stdio::null())
.stdout(Stdio::inherit())
.stderr(Stdio::inherit());
command.envs(environment.iter().cloned());
// Only async-signal-safe syscalls are used in the forked child. Writing 0
// moves the child itself, avoiding PID reuse and parent/child migration races.
unsafe {
command.pre_exec(move || {
if libc::write(group.as_raw_fd(), b"0".as_ptr().cast(), 1) != 1 {
return Err(io::Error::last_os_error());
}
if libc::setsid() < 0 {
return Err(io::Error::last_os_error());
}
if libc::setgroups(0, std::ptr::null()) < 0
|| libc::setgid(1000) < 0
|| libc::setuid(1000) < 0
{
return Err(io::Error::last_os_error());
}
if libc::prctl(libc::PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) < 0 {
return Err(io::Error::last_os_error());
}
let mut mask: libc::sigset_t = std::mem::zeroed();
libc::sigemptyset(&mut mask);
if libc::sigprocmask(libc::SIG_SETMASK, &mask, std::ptr::null_mut()) < 0 {
return Err(io::Error::last_os_error());
}
Ok(())
});
}
command.spawn().map_err(Into::into)
}
pub fn count(bay: Bay) -> Result<usize> {
let path = directory(&format!("bay{bay}")).join("cgroup.procs");
if !path.exists() {
return Ok(0);
}
Ok(read_text(&path, 1024 * 1024)?.lines().count())
}
fn populated(events: &mut File) -> Result<bool> {
events.seek(SeekFrom::Start(0))?;
let mut text = String::new();
events.take(4096).read_to_string(&mut text)?;
if text.lines().any(|s| s == "populated 0") {
Ok(false)
} else if text.lines().any(|s| s == "populated 1") {
Ok(true)
} else {
Err(Error("Invalid cgroup event state".into()))
}
}
fn wait_empty(events: &mut File, limit: Duration) -> Result<bool> {
let deadline = Instant::now() + limit;
loop {
if !populated(events)? {
return Ok(true);
}
let remaining = deadline.saturating_duration_since(Instant::now());
if remaining.is_zero() {
return Ok(false);
}
let mut descriptor = libc::pollfd {
fd: events.as_raw_fd(),
events: libc::POLLPRI | libc::POLLERR,
revents: 0,
};
let result = unsafe {
libc::poll(
&mut descriptor,
1,
remaining.as_millis().max(1).min(i32::MAX as u128) as i32,
)
};
if result < 0 && io::Error::last_os_error().kind() == io::ErrorKind::Interrupted {
continue;
}
checked(result, "wait for cartridge consumers")?;
}
}
pub fn stop(bay: Bay) -> Result<()> {
stop_group(&format!("bay{bay}"))
}
pub fn stop_group(name: &str) -> Result<()> {
if !fds_common::manifest::identifier(name) {
return Err(Error("Invalid process group".into()));
}
let path = directory(name);
if !path.exists() {
return Ok(());
}
let mut events = File::open(path.join("cgroup.events"))?;
if !populated(&mut events)? {
return Ok(());
}
let membership = format!("0::/fds/{name}");
for value in read_text(&path.join("cgroup.procs"), 1024 * 1024)?.lines() {
let pid: i32 = value
.parse()
.map_err(|_| Error("Invalid consumer process ID".into()))?;
let fd = unsafe { libc::syscall(libc::SYS_pidfd_open, pid, 0) } as libc::c_int;
if fd < 0 {
if io::Error::last_os_error().raw_os_error() == Some(libc::ESRCH) {
continue;
}
checked(fd, "open consumer process handle")?;
}
let fd = unsafe { OwnedFd::from_raw_fd(fd) };
// Verify membership after opening the stable handle. A recycled PID in
// another cgroup must never receive a signal intended for a consumer.
let current = match read_text(
&Path::new("/proc").join(pid.to_string()).join("cgroup"),
16384,
) {
Ok(s) => s,
Err(_) => continue,
};
if current.lines().any(|s| s == membership) {
let sent = unsafe {
libc::syscall(
libc::SYS_pidfd_send_signal,
fd.as_raw_fd(),
libc::SIGTERM,
std::ptr::null::<libc::siginfo_t>(),
0,
)
} as libc::c_int;
if sent < 0 && io::Error::last_os_error().raw_os_error() != Some(libc::ESRCH) {
checked(sent, "stop cartridge consumer")?;
}
}
}
if !wait_empty(&mut events, Duration::from_secs(1))? {
// The kernel kills the entire cgroup atomically, including forks made
// after the TERM snapshot. This is an exit deadline, not a fixed wait.
fs::write(path.join("cgroup.kill"), b"1")?;
if !wait_empty(&mut events, Duration::from_secs(2))? {
return Err(Error(
"Cartridge consumers did not exit; media remains mounted".into(),
));
}
}
Ok(())
}
pub fn stop_all() -> Result<()> {
for n in 1..=12 {
stop(Bay::try_from(n)?)?;
}
Ok(())
}
+87
View File
@@ -0,0 +1,87 @@
//! A DATA session outlives the daemon process. Losing its writeback descriptor
//! must never turn an earlier I/O error into a fresh, apparently healthy session.
use fds_common::{Bay, Error, Result, read_text};
use serde::{Deserialize, Serialize};
use std::{fs, io, os::unix::fs::PermissionsExt, path::Path};
const DIRECTORY: &str = "/run/fds/data-sessions";
#[derive(Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
struct Record {
key: String,
fault: Option<String>,
}
impl Record {
fn fault_for(&self, key: &str) -> Option<String> {
(self.key == key).then(|| {
self.fault.clone().unwrap_or_else(|| {
"DATA service was interrupted before verified unmount; recover this cartridge before writable use".into()
})
})
}
}
pub fn prepare() -> Result<()> {
fs::create_dir_all(DIRECTORY)?;
fs::set_permissions(DIRECTORY, fs::Permissions::from_mode(0o700))?;
Ok(())
}
fn path(bay: Bay) -> String {
format!("{DIRECTORY}/{bay}.json")
}
pub fn begin(bay: Bay, key: &str) -> Result<()> {
save(bay, key, None)
}
pub fn save(bay: Bay, key: &str, fault: Option<&str>) -> Result<()> {
let path = path(bay);
let record = Record {
key: key.into(),
fault: fault.map(Into::into),
};
fs::write(
format!("{path}.next"),
serde_json::to_vec(&record).map_err(|e| Error(e.to_string()))?,
)?;
fs::rename(format!("{path}.next"), path)?;
Ok(())
}
pub fn recovered_fault(bay: Bay, key: &str) -> Result<Option<String>> {
let path = path(bay);
if !Path::new(&path).try_exists()? {
return Ok(None);
}
let record: Record = serde_json::from_str(&read_text(Path::new(&path), 16 * 1024)?)
.map_err(|e| Error(format!("Invalid DATA session record: {e}")))?;
Ok(record.fault_for(key))
}
pub fn clear(bay: Bay) -> Result<()> {
match fs::remove_file(path(bay)) {
Ok(()) => Ok(()),
Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(()),
Err(e) => Err(e.into()),
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn interrupted_sessions_and_faults_follow_only_the_same_insertion() {
let mut record = Record {
key: "usb/2:8:1:91".into(),
fault: None,
};
assert!(
record
.fault_for(&record.key)
.unwrap()
.contains("interrupted")
);
assert!(record.fault_for("usb/2:8:1:92").is_none());
record.fault = Some("flush failed: I/O error".into());
let decoded: Record =
serde_json::from_slice(&serde_json::to_vec(&record).unwrap()).unwrap();
assert_eq!(
decoded.fault_for(&record.key).as_deref(),
Some("flush failed: I/O error")
);
}
}
+66
View File
@@ -0,0 +1,66 @@
mod burning;
mod consumers;
mod data_sessions;
mod media;
mod power;
mod profiles;
mod recovery;
mod server;
mod software;
use clap::Parser;
use fds_common::{Error, Result, topology};
use std::{path::PathBuf, process::ExitCode};
#[derive(Parser)]
#[command(
version,
about = "Run the root cartridge service or inspect USB topology"
)]
struct Cli {
/// Notify s6 when the service is ready.
#[arg(long, conflicts_with = "topology")]
notify: bool,
/// Inspect topology without starting the service or mounting devices.
#[arg(long, value_name = "SYSFS_ROOT")]
topology: Option<PathBuf>,
}
fn run() -> Result<()> {
let cli = Cli::parse();
if let Some(sys) = cli.topology {
println!(
"{}",
serde_json::to_string_pretty(&topology::devices(&sys)?)
.map_err(|e| Error(e.to_string()))?
);
Ok(())
} else {
server::run(cli.notify)
}
}
fn main() -> ExitCode {
match run() {
Ok(()) => ExitCode::SUCCESS,
Err(e) => {
eprintln!("fds-cartridged: {e}");
ExitCode::FAILURE
}
}
}
#[cfg(test)]
mod cli_tests {
use super::*;
use clap::{CommandFactory, Parser};
#[test]
fn typed_command_contract() {
Cli::command().debug_assert();
assert!(!Cli::try_parse_from(["fds-cartridged"]).unwrap().notify);
assert!(
Cli::try_parse_from(["fds-cartridged", "--notify"])
.unwrap()
.notify
);
assert!(Cli::try_parse_from(["fds-cartridged", "--topology", "/sys"]).is_ok());
assert!(Cli::try_parse_from(["fds-cartridged", "--notify", "--topology", "/sys"]).is_err());
}
}
+548
View File
@@ -0,0 +1,548 @@
use crate::data_sessions;
use fds_common::{
Bay, Error, Result,
manifest::{Class, Manifest},
read_text,
sysfs::{self, BlockPartition},
};
use std::{
ffi::CString,
fs::{self, File, OpenOptions},
io::{self, Read},
os::{
fd::{AsRawFd, FromRawFd},
unix::fs::{FileTypeExt, MetadataExt, OpenOptionsExt},
},
path::Path,
};
pub fn checked(value: libc::c_int, action: &str) -> Result<()> {
if value < 0 {
Err(Error(format!("{action}: {}", io::Error::last_os_error())))
} else {
Ok(())
}
}
pub fn c(value: &str) -> Result<CString> {
CString::new(value).map_err(|_| Error("NUL in syscall argument".into()))
}
pub fn unmount(path: &str, removed: bool) -> Result<()> {
// Detach is restricted to confirmed surprise-removal cleanup, never safe eject.
checked(
unsafe {
libc::umount2(
c(path)?.as_ptr(),
if removed { libc::MNT_DETACH } else { 0 },
)
},
"unmount cartridge",
)
}
pub fn manifest(root: &str) -> Result<Manifest> {
Manifest::parse(&metadata(root, "CARTRIDGE.TOML")?)
}
pub fn metadata(root: &str, name: &str) -> Result<String> {
// Each component is opened relative to an existing directory descriptor;
// a cartridge cannot redirect privileged metadata reads through symlinks.
let root = OpenOptions::new()
.read(true)
.custom_flags(libc::O_DIRECTORY | libc::O_NOFOLLOW | libc::O_CLOEXEC)
.open(root)?;
let dir = unsafe {
libc::openat(
root.as_raw_fd(),
c("FDS")?.as_ptr(),
libc::O_RDONLY | libc::O_DIRECTORY | libc::O_NOFOLLOW | libc::O_CLOEXEC,
)
};
checked(dir, "open FDS metadata directory")?;
let dir = unsafe { File::from_raw_fd(dir) };
let fd = unsafe {
libc::openat(
dir.as_raw_fd(),
c(name)?.as_ptr(),
libc::O_RDONLY | libc::O_NOFOLLOW | libc::O_CLOEXEC | libc::O_NONBLOCK,
)
};
checked(fd, "open cartridge manifest")?;
let file = unsafe { File::from_raw_fd(fd) };
if !file.metadata()?.is_file() {
return Err(Error("Cartridge manifest is not a regular file".into()));
}
let mut text = String::new();
file.take(fds_common::MAX_CONFIG_BYTES + 1)
.read_to_string(&mut text)?;
if text.len() as u64 > fds_common::MAX_CONFIG_BYTES {
return Err(Error("Cartridge metadata exceeds 64 KiB".into()));
}
Ok(text)
}
pub struct Mounted {
pub bay: Bay,
pub key: String,
pub path: String,
pub manifest: Manifest,
pub protected: bool,
pub source: File,
pub writable: bool,
pub sync_handle: Option<File>,
pub fault: Option<String>,
pub partition: BlockPartition,
pub software: Option<crate::software::Mounted>,
}
impl Mounted {
pub fn present(&self) -> bool {
key(&self.partition).is_ok_and(|key| key == self.key)
}
pub fn activate_data(&mut self) -> Result<()> {
if self.manifest.cartridge.class != Class::Data || self.protected {
return Err(Error("This is not a DATA cartridge".into()));
}
if let Some(fault) = &self.fault {
return Err(Error(format!("DATA is quarantined: {fault}")));
}
if self.writable {
return Ok(());
}
self.ensure_exclusive_mount()?;
data_sessions::begin(self.bay, &self.key)?;
if let Err(error) = unmount(&self.path, false) {
data_sessions::clear(self.bay)?;
return Err(error);
}
let result = checked(
unsafe {
libc::mount(
c(&format!("/proc/self/fd/{}", self.source.as_raw_fd()))?.as_ptr(),
c("/data")?.as_ptr(),
c("ext4")?.as_ptr(),
libc::MS_NOSUID | libc::MS_NODEV | libc::MS_NOEXEC,
std::ptr::null(),
)
},
"mount writable DATA",
);
if let Err(error) = result {
// Restore read-only visibility when activation fails. If restoration
// also fails, retain a fault and never report this insertion SAFE.
let restored = checked(
unsafe {
libc::mount(
c(&format!("/proc/self/fd/{}", self.source.as_raw_fd()))?.as_ptr(),
c(&self.path)?.as_ptr(),
c("ext4")?.as_ptr(),
libc::MS_RDONLY | libc::MS_NOSUID | libc::MS_NODEV | libc::MS_NOEXEC,
c("noload")?.as_ptr().cast(),
)
},
"restore read-only DATA",
);
if let Err(restore_error) = restored {
self.record_fault(&restore_error)?;
} else {
data_sessions::clear(self.bay)?;
}
return Err(error);
}
self.path = "/data".into();
self.writable = true;
// Retain a filesystem descriptor from activation to observe writeback
// errors across the session. A block-device descriptor cannot do this.
match File::open(&self.path) {
Ok(handle) => self.sync_handle = Some(handle),
Err(error) => {
let error = Error(format!("Open DATA writeback handle: {error}"));
self.record_fault(&error)?;
return Err(error);
}
}
Ok(())
}
fn record_fault(&mut self, error: &Error) -> Result<()> {
self.fault = Some(error.to_string());
data_sessions::save(self.bay, &self.key, self.fault.as_deref())
}
fn data_readonly(&self, readonly: bool) -> io::Result<()> {
let result = unsafe {
libc::mount(
std::ptr::null(),
c(&self.path).map_err(io::Error::other)?.as_ptr(),
std::ptr::null(),
libc::MS_REMOUNT
| libc::MS_NOSUID
| libc::MS_NODEV
| libc::MS_NOEXEC
| if readonly { libc::MS_RDONLY } else { 0 },
std::ptr::null(),
)
};
if result < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
pub fn program(
&mut self,
arguments: &[String],
) -> Result<(Vec<String>, Vec<(String, String)>)> {
if self.manifest.cartridge.class != Class::Program
|| self.fault.is_some()
|| !self.present()
{
return Err(Error("No healthy PROGRAM cartridge in this bay".into()));
}
self.ensure_exclusive_mount()?;
if let Some(software) = &mut self.software {
return software.program(arguments);
}
let name = arguments
.first()
.ok_or_else(|| Error("Specify an executable name from app/bin".into()))?;
if !fds_common::manifest::identifier(name) {
return Err(Error(
"PROGRAM executable must be a simple name from app/bin".into(),
));
}
let app = fs::canonicalize(Path::new(&self.path).join("app"))?;
let executable = fs::canonicalize(app.join("bin").join(name))?;
if !app.starts_with(&self.path) || !executable.starts_with(&app) || !executable.is_file() {
return Err(Error("PROGRAM executable escapes its app directory".into()));
}
self.ensure_exclusive_mount()?;
checked(
unsafe {
libc::mount(
std::ptr::null(),
c(&self.path)?.as_ptr(),
std::ptr::null(),
libc::MS_REMOUNT | libc::MS_RDONLY | libc::MS_NODEV | libc::MS_NOSUID,
std::ptr::null(),
)
},
"enable read-only PROGRAM execution",
)?;
let mut args = arguments.to_vec();
args[0] = executable.to_string_lossy().into_owned();
let env = vec![
("FDS_APP".into(), app.display().to_string()),
(
"PATH".into(),
format!("{}/bin:/usr/bin:/bin", app.display()),
),
(
"LD_LIBRARY_PATH".into(),
app.join("lib").display().to_string(),
),
(
"XDG_DATA_DIRS".into(),
format!("{}/share:/usr/share", app.display()),
),
("DISPLAY".into(), ":0".into()),
("XAUTHORITY".into(), "/run/fds/x11/authority".into()),
];
Ok((args, env))
}
pub fn eject(&mut self) -> Result<()> {
if let Some(fault) = &self.fault {
return Err(Error(format!(
"DATA I/O fault: {fault}; no SAFE status issued"
)));
}
self.ensure_exclusive_mount()?;
if self.writable {
let handle = self
.sync_handle
.as_ref()
.ok_or_else(|| Error("Missing DATA writeback handle".into()))?;
if let Err(error) = checked(
unsafe { libc::syncfs(handle.as_raw_fd()) },
"flush DATA filesystem",
) {
self.record_fault(&error)?;
return Err(error);
}
// Stop new writers before releasing the descriptor that tracks
// writeback errors. A busy writer leaves this descriptor open.
// MS_REMOUNT without MS_BIND makes the filesystem itself read-only.
if let Err(error) = self.data_readonly(true) {
let busy = error.raw_os_error() == Some(libc::EBUSY);
let error = Error(format!("Make DATA read-only before unmount: {error}"));
if !busy {
self.record_fault(&error)?;
}
return Err(error);
}
if let Err(error) = checked(
unsafe { libc::syncfs(handle.as_raw_fd()) },
"verify DATA writeback after read-only transition",
) {
self.record_fault(&error)?;
return Err(error);
}
// Our descriptor itself makes the mount busy. Close it only after
// syncfs succeeds, immediately before the ordinary unmount.
self.sync_handle.take();
}
if let Some(software) = &mut self.software {
software.release(false)?;
}
match unmount(&self.path, false) {
Ok(()) => {
if self.writable {
data_sessions::clear(self.bay)?;
}
Ok(())
}
Err(error) => {
if self.writable {
// Still read-only: establish the new error cursor before
// restoring writes, so no writeback failure is skipped.
let restored = File::open(&self.path).and_then(|handle| {
self.sync_handle = Some(handle);
self.data_readonly(false)
});
if let Err(restore_error) = restored {
let restore_error =
Error(format!("Restore DATA after busy unmount: {restore_error}"));
self.record_fault(&restore_error)?;
return Err(restore_error);
}
}
Err(error)
}
}
}
pub fn removed(&mut self) -> Result<()> {
if let Some(software) = &mut self.software {
software.release(true)?;
}
self.sync_handle.take();
unmount(&self.path, true)
}
pub fn ensure_exclusive_mount(&self) -> Result<()> {
if let Some(software) = &self.software {
software.exclusive()?;
}
let dev = self.source.metadata()?.rdev();
let identity = format!("{}:{}", libc::major(dev), libc::minor(dev));
let table = read_text(Path::new("/proc/self/mountinfo"), 4 * 1024 * 1024)?;
let matching: Vec<_> = table
.lines()
.filter(|line| line.split_whitespace().nth(2) == Some(identity.as_str()))
.collect();
if matching.len() != 1 || matching[0].split_whitespace().nth(4) != Some(self.path.as_str())
{
return Err(Error(
"Cartridge has additional or unexpected mounts; close them before eject".into(),
));
}
Ok(())
}
}
pub fn partitions_for(usb: &Path) -> Result<Vec<BlockPartition>> {
Ok(sysfs::partitions(Path::new("/sys"))?
.into_iter()
.filter(|part| {
let name = Path::new(&part.device).file_name().unwrap();
fs::canonicalize(Path::new("/sys/class/block").join(name))
.is_ok_and(|path| path.starts_with(usb))
})
.collect())
}
pub fn key(part: &BlockPartition) -> Result<String> {
let path = fs::canonicalize(
Path::new("/sys/dev/block").join(format!("{}:{}", part.major, part.minor)),
)?;
let disk = path
.parent()
.ok_or_else(|| Error("Missing parent disk".into()))?;
let sequence = read_text(&disk.join("diskseq"), 64)?;
Ok(format!(
"{}:{}:{}:{}",
path.display(),
part.major,
part.minor,
sequence.trim()
))
}
/// Partition-table rereads can briefly remove every partition without removing
/// its physical disk. Preserve an eject record until that insertion is gone.
pub fn key_disk_present(key: &str) -> Result<bool> {
let fields: Vec<_> = key.rsplitn(4, ':').collect();
if fields.len() != 4 || fields[..3].iter().any(|s| s.parse::<u64>().is_err()) {
return Err(Error("Invalid saved cartridge insertion identity".into()));
}
let partition = Path::new(fields[3]);
let parent = partition
.parent()
.ok_or_else(|| Error("Missing saved disk path".into()))?;
if !parent.starts_with("/sys/devices") {
return Err(Error(
"Saved disk identity is outside kernel devices".into(),
));
}
let sequence = match read_text(&parent.join("diskseq"), 64) {
Ok(sequence) => sequence,
Err(_) if !parent.exists() => return Ok(false),
Err(error) => return Err(error),
};
Ok(sequence.trim() == fields[0])
}
pub fn mount(bay: Bay, part: &BlockPartition, key: String) -> Result<Mounted> {
let fault = if part.partition_name == "FDS_DATA" {
data_sessions::recovered_fault(bay, &key)?
} else {
None
};
let source = OpenOptions::new()
.read(true)
.custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK | libc::O_CLOEXEC)
.open(&part.device)?;
let st = source.metadata()?;
if !st.file_type().is_block_device()
|| libc::major(st.rdev()) != part.major
|| libc::minor(st.rdev()) != part.minor
|| self::key(part)? != key
{
return Err(Error("Device changed during inspection".into()));
}
let root = fs::metadata("/")?;
let protected = root.dev() == st.rdev();
let expected = match part.partition_name.as_str() {
"FDS_SYSTEM" => Class::System,
"FDS_DATA" => Class::Data,
"FDS_PROGRAM" | "FDS_METADATA" => Class::Program,
"FDS_ENVIRONMENT" => Class::Environment,
"FDS_UTILITY" => Class::Utility,
_ => return Err(Error("Unrecognized cartridge partition name".into())),
};
let published = format!("/run/fds/media/{bay}");
let path = if protected {
"/".into()
} else {
format!("/run/fds/probe/{bay}")
};
if !protected {
fs::create_dir_all(&path)?;
let kind = if expected == Class::Data {
"ext4"
} else {
"erofs"
};
let data = c(if expected == Class::Data {
"noload"
} else {
""
})?;
checked(
unsafe {
libc::mount(
c(&format!("/proc/self/fd/{}", source.as_raw_fd()))?.as_ptr(),
c(&path)?.as_ptr(),
c(kind)?.as_ptr(),
libc::MS_RDONLY | libc::MS_NOSUID | libc::MS_NODEV | libc::MS_NOEXEC,
data.as_ptr().cast(),
)
},
"mount cartridge read-only",
)?;
}
let result = manifest(&path).and_then(|value| {
if value.cartridge.class != expected {
return Err(Error(
"Manifest class disagrees with GPT partition name".into(),
));
}
Ok(value)
});
match result {
Ok(manifest) => {
let published = if expected == Class::Program {
let destination = format!("/run/fds/apps/{}", manifest.cartridge.id);
let mounts = read_text(Path::new("/proc/self/mountinfo"), 4 * 1024 * 1024)?;
if mounts
.lines()
.any(|l| l.split_whitespace().nth(4) == Some(destination.as_str()))
{
unmount(&path, false)?;
return Err(Error("A PROGRAM with this id is already mounted".into()));
}
destination
} else {
published
};
let path = if protected {
path
} else {
fs::create_dir_all(&published)?;
let moved = checked(
unsafe {
libc::mount(
c(&path)?.as_ptr(),
c(&published)?.as_ptr(),
std::ptr::null(),
libc::MS_MOVE,
std::ptr::null(),
)
},
"publish validated cartridge",
);
if let Err(error) = moved {
unmount(&path, false)?;
return Err(error);
}
published
};
Ok(Mounted {
bay,
key,
path,
manifest,
protected,
source,
writable: false,
sync_handle: None,
fault,
partition: part.clone(),
software: None,
})
}
Err(error) => {
if !protected {
unmount(&path, false)?;
}
Err(error)
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::os::unix::fs::symlink;
#[test]
fn metadata_symlinks_and_special_files_are_rejected() {
let root = std::env::temp_dir().join(format!("fds-manifest-{}", std::process::id()));
fs::create_dir_all(root.join("FDS")).unwrap();
let metadata = root.join("FDS/CARTRIDGE.TOML");
fs::write(
&metadata,
include_str!("../../../tests/fixtures/manifests/windowmaker.toml"),
)
.unwrap();
assert!(manifest(root.to_str().unwrap()).is_ok());
fs::remove_file(&metadata).unwrap();
symlink("/etc/passwd", &metadata).unwrap();
assert!(manifest(root.to_str().unwrap()).is_err());
fs::remove_file(&metadata).unwrap();
let name = c(metadata.to_str().unwrap()).unwrap();
assert_eq!(unsafe { libc::mkfifo(name.as_ptr(), 0o600) }, 0);
assert!(manifest(root.to_str().unwrap()).is_err());
fs::remove_file(&metadata).unwrap();
fs::remove_dir(root.join("FDS")).unwrap();
symlink("/etc", root.join("FDS")).unwrap();
assert!(manifest(root.to_str().unwrap()).is_err());
fs::remove_dir_all(root).unwrap();
}
}
+139
View File
@@ -0,0 +1,139 @@
//! Shutdown preparation is persistent for this boot. A failed step leaves new
//! operations frozen; only an explicit retry or resume changes that decision.
use fds_common::{
Error, Result,
control::{PowerEvent, PowerState},
trace,
};
use std::{
fs, io,
os::unix::{fs::PermissionsExt, process::CommandExt},
path::Path,
process::Command,
};
pub const DIRECTORY: &str = "/run/fds/power";
pub const RECORD: &str = "/run/fds/power/state.json";
pub const NATIVE: &str = "/run/fds/power/native-pending";
pub struct Manager {
pub state: PowerState,
}
impl Manager {
pub fn load() -> Result<Self> {
fs::create_dir_all(DIRECTORY)?;
fs::set_permissions(DIRECTORY, fs::Permissions::from_mode(0o700))?;
let state = if Path::new(RECORD).try_exists()? {
serde_json::from_str(&fds_common::read_text(Path::new(RECORD), 32 * 1024)?)
.map_err(|e| Error(format!("Invalid shutdown recovery record: {e}")))?
} else {
PowerState::default()
};
let mut this = Self { state };
this.state.native_pending = Path::new(NATIVE).try_exists()?;
if this.state.phase != "idle" && this.state.phase != "prepared" {
this.failed(&Error(
"Cartridge service restarted during shutdown preparation; retry required".into(),
))?;
} else if this.state.native_pending && this.state.phase == "idle" {
this.failed(&Error(
"Native shutdown is waiting for DATA preparation".into(),
))?;
}
Ok(this)
}
pub fn frozen(&self) -> bool {
self.state.phase != "idle"
}
fn save(&self) -> Result<()> {
fs::write(
format!("{RECORD}.next"),
serde_json::to_vec(&self.state).map_err(|e| Error(e.to_string()))?,
)?;
fs::rename(format!("{RECORD}.next"), RECORD)?;
Ok(())
}
pub fn phase(&mut self, phase: &str) -> Result<()> {
self.state.phase = phase.into();
self.state.events.push(PowerEvent {
phase: phase.into(),
at_ns: trace::now()?,
});
if self.state.events.len() > 64 {
self.state.events.remove(0);
}
self.save()?;
eprintln!("FDS shutdown: {phase}");
Ok(())
}
pub fn begin(&mut self, action: Option<&str>, native: bool) -> Result<()> {
if native {
self.native_started()?;
}
if !self.state.native_pending {
self.state.action = action.map(Into::into);
}
self.state.error = None;
self.phase("frozen")
}
pub fn native_started(&mut self) -> Result<()> {
fs::write(NATIVE, b"native shutdown requested\n")?;
self.state.native_pending = true;
self.save()
}
pub fn failed(&mut self, error: &Error) -> Result<()> {
self.state.error = Some(
error
.to_string()
.chars()
.map(|c| if c.is_control() { ' ' } else { c })
.take(2048)
.collect(),
);
self.phase("blocked")
}
pub fn resume(&mut self) -> Result<()> {
if self.state.native_pending || Path::new(NATIVE).try_exists()? {
return Err(Error("Native shutdown has already started and cannot be cancelled; resolve the reported problem and retry fds poweroff".into()));
}
self.state = PowerState::default();
self.save()
}
pub fn commit(&mut self, reboot: bool) -> Result<()> {
if self.state.phase != "prepared" {
return Err(Error("Shutdown preparation is incomplete".into()));
}
if self.state.native_pending {
return Ok(());
}
self.state.native_pending = true;
self.save()?;
let mut command = Command::new("/usr/bin/s6-linux-init-shutdown");
command
.args([if reboot { "-r" } else { "-p" }, "-t", "0", "now"])
.env_clear()
.env("PATH", "/usr/bin:/bin");
unsafe {
command.pre_exec(|| {
let mut mask = std::mem::zeroed();
libc::sigemptyset(&mut mask);
if libc::sigprocmask(libc::SIG_SETMASK, &mask, std::ptr::null_mut()) < 0 {
return Err(io::Error::last_os_error());
}
Ok(())
});
}
let result = command.status().map_err(Error::from).and_then(|status| {
if status.success() {
Ok(())
} else {
Err(Error(format!("Native shutdown request failed: {status}")))
}
});
if let Err(error) = result {
self.state.native_pending = false;
self.failed(&error)?;
return Err(error);
}
Ok(())
}
}
+328
View File
@@ -0,0 +1,328 @@
//! Profile client and privileged s6 helpers. No cartridge supplies root commands.
#[allow(dead_code)]
mod consumers;
#[allow(dead_code)]
mod data_sessions;
#[allow(dead_code)]
mod media;
#[allow(dead_code)]
mod software;
mod x11;
use fds_common::{
Error, Result,
control::{self, Request},
trace,
};
use media::{c, checked};
use std::{
fs::{self, OpenOptions},
io::{Read, Write},
os::{
fd::{AsRawFd, FromRawFd, OwnedFd},
unix::{
fs::{OpenOptionsExt, PermissionsExt},
process::CommandExt,
},
},
path::Path,
process::{Command, ExitCode},
time::{Duration, Instant},
};
fn root() -> Result<()> {
if unsafe { libc::geteuid() } != 0 {
return Err(Error("This s6 service helper requires root".into()));
}
Ok(())
}
fn xserver() -> Result<()> {
root()?;
fs::create_dir_all("/run/fds/x11")?;
checked(
unsafe { libc::chown(c("/run/fds/x11")?.as_ptr(), 0, 1000) },
"set X authority directory group",
)?;
fs::set_permissions("/run/fds/x11", fs::Permissions::from_mode(0o750))?;
let mut cookie = [0u8; 16];
fs::File::open("/dev/urandom")?.read_exact(&mut cookie)?;
let mut bytes = vec![0xff, 0xff]; // FamilyWild: Unix socket access still requires the secret cookie.
for field in [&b""[..], &b"0"[..], &b"MIT-MAGIC-COOKIE-1"[..], &cookie[..]] {
bytes.extend_from_slice(&(field.len() as u16).to_be_bytes());
bytes.extend_from_slice(field);
}
let temporary = "/run/fds/x11/authority.next";
match fs::remove_file(temporary) {
Ok(()) => (),
Err(e) if e.kind() == std::io::ErrorKind::NotFound => (),
Err(e) => return Err(e.into()),
}
let mut authority = OpenOptions::new()
.write(true)
.create_new(true)
.mode(0o640)
.custom_flags(libc::O_NOFOLLOW)
.open(temporary)?;
authority.write_all(&bytes)?;
checked(
unsafe { libc::fchown(authority.as_raw_fd(), 0, 1000) },
"set X authority group",
)?;
drop(authority);
fs::rename(temporary, "/run/fds/x11/authority")?;
let backend = fs::read_to_string("/etc/fds/xserver").unwrap_or_else(|_| "xorg".into());
let mut command = match backend.trim() {
"xorg" => {
let mut c = Command::new("/usr/libexec/Xorg");
c.args([":0", "vt2", "-logfile", "/run/log/xserver/Xorg.0.log"]);
c
}
"xvfb" => {
let mut c = Command::new("/usr/bin/Xvfb");
c.args([":0", "-screen", "0", "1600x1200x24"]);
c
}
_ => return Err(Error("Unknown X server backend".into())),
};
command.args([
"-displayfd",
"3",
"-nolisten",
"tcp",
"-noreset",
"-auth",
"/run/fds/x11/authority",
"-dpi",
"120",
"-extension",
"Composite",
]);
Err(command.exec().into())
}
fn session() -> Result<()> {
root()?;
checked(
unsafe { libc::fcntl(3, libc::F_GETFD) },
"require s6 readiness descriptor",
)?;
checked(
unsafe { libc::fcntl(3, libc::F_SETFD, libc::FD_CLOEXEC) },
"protect readiness descriptor",
)?;
consumers::prepare()?;
consumers::stop_group("desktop")?;
let _ = fs::remove_file("/run/fds/desktop-ready-ns");
let env = vec![
("DISPLAY".into(), ":0".into()),
("XAUTHORITY".into(), "/run/fds/x11/authority".into()),
];
let mut mask: libc::sigset_t = unsafe { std::mem::zeroed() };
unsafe {
libc::sigemptyset(&mut mask);
for sig in [libc::SIGTERM, libc::SIGINT, libc::SIGCHLD] {
libc::sigaddset(&mut mask, sig);
}
}
checked(
unsafe { libc::sigprocmask(libc::SIG_BLOCK, &mask, std::ptr::null_mut()) },
"block session signals",
)?;
let fd = unsafe { libc::signalfd(-1, &mask, libc::SFD_CLOEXEC | libc::SFD_NONBLOCK) };
checked(fd, "observe session processes")?;
let signals = unsafe { OwnedFd::from_raw_fd(fd) };
let mut observer = Some(x11::Observer::connect()?);
let mut wm = consumers::start_group(
"desktop",
&["/usr/libexec/fds/windowmaker-session".into()],
"/home/fds",
&env,
)?;
let deadline = Instant::now() + Duration::from_secs(15);
let mut ready = false;
let mut terminal = None;
let result = (|| -> Result<()> {
loop {
if wm.try_wait()?.is_some() {
return Err(Error("WindowMaker exited".into()));
}
if !ready && Instant::now() >= deadline {
return Err(Error("WindowMaker readiness deadline expired".into()));
}
let mut fds = [
libc::pollfd {
fd: signals.as_raw_fd(),
events: libc::POLLIN,
revents: 0,
},
libc::pollfd {
fd: observer.as_ref().map_or(-1, AsRawFd::as_raw_fd),
events: libc::POLLIN,
revents: 0,
},
];
let timeout = if ready {
-1
} else {
deadline
.saturating_duration_since(Instant::now())
.as_millis()
.min(i32::MAX as u128) as i32
};
let count = unsafe { libc::poll(fds.as_mut_ptr(), 2, timeout) };
if count < 0
&& std::io::Error::last_os_error().kind() == std::io::ErrorKind::Interrupted
{
continue;
}
checked(count, "wait for desktop events")?;
if fds[0].revents != 0 {
let mut info: libc::signalfd_siginfo = unsafe { std::mem::zeroed() };
while unsafe {
libc::read(
signals.as_raw_fd(),
(&mut info as *mut libc::signalfd_siginfo).cast(),
std::mem::size_of_val(&info),
)
} > 0
{
if info.ssi_signo == libc::SIGTERM as u32
|| info.ssi_signo == libc::SIGINT as u32
{
return Ok(());
}
}
if let Some(child) = &mut terminal {
let _: Option<std::process::ExitStatus> = std::process::Child::try_wait(child)?;
}
}
if !ready && fds[1].revents != 0 && observer.as_mut().unwrap().event()? {
let now = trace::now()?;
trace::save(Path::new(trace::RUNTIME), trace::Point::DesktopReady, now)?;
fs::write("/run/fds/desktop-ready-ns", format!("{now}\n"))?;
terminal = Some(consumers::start_group(
"desktop",
&["/usr/libexec/fds/terminal".into()],
"/home/fds",
&env,
)?);
checked(
unsafe { libc::write(3, b"\n".as_ptr().cast(), 1) } as i32,
"notify desktop readiness",
)?;
unsafe {
libc::close(3);
}
observer.take();
ready = true;
}
}
})();
consumers::stop_group("desktop")?;
let _ = wm.wait();
if let Some(mut child) = terminal {
let _ = child.wait();
}
let _ = fs::remove_file("/run/fds/desktop-ready-ns");
result
}
#[derive(clap::Parser)]
#[command(
version,
about = "Control the desktop and network profile",
after_help = "Examples: fds-profile activate windowmaker; fds-profile deactivate"
)]
struct Cli {
#[command(subcommand)]
command: Option<Action>,
}
#[derive(clap::Subcommand)]
enum Action {
/// Show desktop and network state (the default).
Status,
/// Activate a profile.
Activate {
#[arg(value_parser = ["windowmaker", "cli"])]
name: String,
},
/// Return to the console.
Deactivate,
#[command(long_flag = "xserver", hide = true)]
Xserver,
#[command(long_flag = "network", hide = true)]
Network,
#[command(long_flag = "cleanup-network", hide = true)]
CleanupNetwork,
#[command(long_flag = "session", hide = true)]
Session,
#[command(long_flag = "cleanup-session", hide = true)]
CleanupSession,
}
fn run() -> Result<()> {
use clap::Parser;
let request = match Cli::parse().command.unwrap_or(Action::Status) {
Action::Xserver => return xserver(),
Action::Network => {
root()?;
consumers::enter_service("network")?;
return Err(Command::new("/usr/libexec/fds/network-run").exec().into());
}
Action::CleanupNetwork => {
root()?;
return consumers::stop_group("network");
}
Action::Session => return session(),
Action::CleanupSession => {
root()?;
consumers::stop_group("desktop")?;
let _ = fs::remove_file("/run/fds/desktop-ready-ns");
return Ok(());
}
Action::Activate { name } => Request::Profile { profile: name },
Action::Deactivate => Request::Profile {
profile: "cli".into(),
},
Action::Status => Request::Profiles,
};
let response = control::request(&request)?;
println!(
"{}",
serde_json::to_string_pretty(&response.profiles).map_err(|e| Error(e.to_string()))?
);
Ok(())
}
fn main() -> ExitCode {
match run() {
Ok(()) => ExitCode::SUCCESS,
Err(e) => {
eprintln!("fds-profile: {e}");
ExitCode::FAILURE
}
}
}
#[cfg(test)]
mod cli_tests {
use super::*;
use clap::{CommandFactory, Parser};
#[test]
fn typed_command_contract() {
Cli::command().debug_assert();
assert!(
Cli::try_parse_from(["fds-profile"])
.unwrap()
.command
.is_none()
);
for flag in [
"--xserver",
"--network",
"--cleanup-network",
"--session",
"--cleanup-session",
] {
assert!(Cli::try_parse_from(["fds-profile", flag]).is_ok());
assert!(Cli::try_parse_from(["fds-profile", flag, "activate", "cli"]).is_err());
}
assert!(Cli::try_parse_from(["fds-profile", "activate", "windowmaker"]).is_ok());
assert!(Cli::try_parse_from(["fds-profile", "activate"]).is_err());
}
}
+288
View File
@@ -0,0 +1,288 @@
//! Only SYSTEM-owned, named service bundles may be activated by media.
use crate::{consumers, media::Mounted};
use fds_common::{
Bay, Error, Result, control::ProfileState, manifest::Class, topology::UsbDevice, trace,
};
use std::{
collections::{BTreeMap, BTreeSet},
fs,
os::unix::process::CommandExt,
path::Path,
process::{Child, Command},
};
fn begin_change(service: &str, up: bool) -> Result<Child> {
let mut command = Command::new("/usr/bin/s6-rc");
command.args([
"-b",
"-l",
"/run/s6-rc",
"-t",
"20000",
if up { "-u" } else { "-d" },
"change",
service,
]);
unsafe {
command.pre_exec(|| {
let mut mask = std::mem::zeroed();
libc::sigemptyset(&mut mask);
if libc::sigprocmask(libc::SIG_SETMASK, &mask, std::ptr::null_mut()) < 0 {
return Err(std::io::Error::last_os_error());
}
Ok(())
});
}
command.spawn().map_err(Into::into)
}
pub fn change(service: &str, up: bool) -> Result<()> {
if begin_change(service, up)?.wait()?.success() {
Ok(())
} else {
Err(Error(format!(
"Service transition failed: {service}; inspect /run/log/{service}"
)))
}
}
#[derive(Default)]
pub struct Manager {
automatic: bool,
desktop: bool,
pending: Option<Child>,
owner: Option<(Bay, String)>,
attempted: BTreeSet<String>,
pub manual_network: bool,
network_disabled: bool,
network: Vec<String>,
network_attempted: Vec<String>,
activation: Option<u64>,
error: Option<String>,
}
impl Manager {
pub fn new(automatic: bool) -> Self {
Self {
automatic,
..Self::default()
}
}
pub fn desktop(&mut self, name: &str, mounts: &BTreeMap<Bay, Mounted>) -> Result<()> {
match name {
"windowmaker" => {
self.owner = None;
self.start_desktop()
}
"cli" => {
self.attempted
.extend(mounts.values().map(|m| m.key.clone()));
self.stop_desktop()
}
_ => Err(Error(
"Unknown profile; available profiles: cli, windowmaker".into(),
)),
}
}
fn start_desktop(&mut self) -> Result<()> {
if self.desktop {
return Ok(());
}
if !console_ready() {
return Err(Error("The console is not ready yet".into()));
}
let instant = trace::now()?;
self.activation = Some(instant);
self.error = None;
match fs::remove_file("/run/fds/desktop-ready-ns") {
Ok(()) => (),
Err(e) if e.kind() == std::io::ErrorKind::NotFound => (),
Err(e) => return Err(e.into()),
}
self.pending = Some(begin_change("desktop", true)?);
self.desktop = true;
Ok(())
}
pub fn poll(&mut self) -> Result<()> {
let status = if let Some(child) = &mut self.pending {
child.try_wait()?
} else {
None
};
if let Some(status) = status {
self.pending.take();
if !status.success() {
self.desktop = false;
self.owner = None;
self.error = Some(
"Desktop startup failed; inspect /run/log/xserver and /run/log/desktop".into(),
);
change("desktop", false)?;
}
}
Ok(())
}
pub fn stop_desktop(&mut self) -> Result<()> {
if let Some(mut child) = self.pending.take() {
let _ = child.wait()?;
}
// s6-rc also tears down a partly started bundle after an activation error.
change("desktop", false)?;
self.desktop = false;
self.owner = None;
Ok(())
}
pub fn before_eject(
&mut self,
bay: Bay,
data: bool,
mounts: &BTreeMap<Bay, Mounted>,
) -> Result<()> {
if data && self.desktop || self.owner.as_ref().is_some_and(|(b, _)| *b == bay) {
self.desktop("cli", mounts)?;
}
Ok(())
}
pub fn network(&mut self, enabled: bool) -> Result<()> {
let interfaces = if enabled {
interfaces(None)?
} else {
Vec::new()
};
if enabled && interfaces.is_empty() {
return Err(Error("No Ethernet interface is available".into()));
}
self.set_network(interfaces)?;
self.manual_network = enabled;
self.network_disabled = !enabled;
Ok(())
}
fn set_network(&mut self, interfaces: Vec<String>) -> Result<()> {
if self.network == interfaces && !interfaces.is_empty() {
return Ok(());
}
change("network", false)?;
consumers::stop_group("network")?;
self.network.clear();
if interfaces.is_empty() {
return Ok(());
}
fs::create_dir_all("/run/fds/network")?;
fs::write("/run/fds/network/interfaces", interfaces.join("\n") + "\n")?;
change("network", true)?;
self.network = interfaces;
Ok(())
}
pub fn reconcile(
&mut self,
mounts: &BTreeMap<Bay, Mounted>,
devices: &[(Bay, UsbDevice)],
) -> Result<()> {
self.poll()?;
let keys: BTreeSet<_> = mounts.values().map(|m| m.key.clone()).collect();
self.attempted.retain(|key| keys.contains(key));
if let Some((bay, key)) = &self.owner {
if !mounts.get(bay).is_some_and(|m| &m.key == key) {
self.stop_desktop()?;
}
}
if self.automatic && !self.desktop && console_ready() {
let candidates: Vec<_> = mounts
.iter()
.filter(|(_, m)| {
m.manifest.cartridge.class == Class::Environment
&& m.fault.is_none()
&& m.manifest
.activation
.as_ref()
.is_some_and(|a| a.profile == "windowmaker")
})
.collect();
if candidates.len() == 1 {
let (bay, mount) = candidates[0];
if self.attempted.insert(mount.key.clone()) {
match self.start_desktop() {
Ok(()) => self.owner = Some((*bay, mount.key.clone())),
Err(error) => {
eprintln!("ENVIRONMENT activation: {error}");
self.error = Some(error.to_string());
}
}
}
}
}
let usb_paths: Vec<_> = devices.iter().map(|(_, d)| d.path.as_path()).collect();
let automatic = interfaces(Some(&usb_paths))?;
if automatic.is_empty() {
self.network_disabled = false;
}
let wanted = if self.manual_network {
interfaces(None)?
} else if self.network_disabled || !self.automatic {
Vec::new()
} else {
automatic
};
if wanted != self.network_attempted {
self.network_attempted = wanted.clone();
if let Err(error) = self.set_network(wanted) {
eprintln!("Network activation: {error}");
self.error = Some(error.to_string());
}
}
Ok(())
}
pub fn status(&self) -> ProfileState {
let ready = fs::read_to_string("/run/fds/desktop-ready-ns")
.ok()
.and_then(|s| s.trim().parse().ok());
ProfileState {
desktop: if self.desktop { "windowmaker" } else { "cli" }.into(),
environment_bay: self.owner.as_ref().map(|(b, _)| *b),
network: self.network.clone(),
manual_network: self.manual_network,
activation_ns: self.activation,
ready_ns: if self.desktop { ready } else { None },
error: self.error.clone(),
}
}
pub fn shutdown(&mut self) -> Result<()> {
self.stop_desktop()?;
change("network", false)?;
consumers::stop_group("network")?;
self.network.clear();
Ok(())
}
}
pub fn console_ready() -> bool {
Path::new(trace::RUNTIME)
.join("console/console-ready.json")
.is_file()
}
fn interfaces(usb_paths: Option<&[&Path]>) -> Result<Vec<String>> {
let mut names = Vec::new();
for entry in fs::read_dir("/sys/class/net")? {
let entry = entry?;
let path = entry.path();
let name = entry.file_name().to_string_lossy().into_owned();
if name == "lo"
|| name.len() > 15
|| !name
.bytes()
.all(|c| c.is_ascii_alphanumeric() || b"_.-".contains(&c))
{
continue;
}
if fs::read_to_string(path.join("type")).is_ok_and(|s| s.trim() == "1")
&& !path.join("wireless").exists()
{
if let Some(parents) = usb_paths {
if !fs::canonicalize(&path)
.is_ok_and(|p| parents.iter().any(|parent| p.starts_with(parent)))
{
continue;
}
}
names.push(name);
}
}
names.sort();
Ok(names)
}
+291
View File
@@ -0,0 +1,291 @@
//! Explicit, local recovery maintenance. Normal boot never runs a checker.
use crate::media::{self, checked};
use fds_burn::device::Disk;
use fds_common::{
Bay, Error, Result,
control::{BayDisk, RecoveryReport},
manifest::Class,
read_text,
sysfs::BlockPartition,
};
use std::{
fs::{self, File, OpenOptions},
os::{
fd::AsRawFd,
unix::{
fs::{FileTypeExt, MetadataExt, OpenOptionsExt, PermissionsExt},
process::CommandExt,
},
},
path::Path,
process::{Command, Stdio},
};
pub struct Selection {
bay: Bay,
disk: Disk,
partition: BlockPartition,
pub key: String,
confirmation: String,
log: String,
}
// Linux UAPI include/uapi/linux/loop.h. LOOP_CONFIGURE atomically binds the
// already verified partition descriptor and enables automatic cleanup.
#[repr(C)]
struct LoopInfo {
device: u64,
inode: u64,
rdevice: u64,
offset: u64,
size_limit: u64,
number: u32,
encrypt_type: u32,
encrypt_key_size: u32,
flags: u32,
file_name: [u8; 64],
crypt_name: [u8; 64],
encrypt_key: [u8; 32],
init: [u64; 2],
}
#[repr(C)]
struct LoopConfig {
fd: u32,
block_size: u32,
info: LoopInfo,
reserved: [u64; 8],
}
const _: () = assert!(std::mem::size_of::<LoopInfo>() == 232);
const _: () = assert!(std::mem::size_of::<LoopConfig>() == 304);
fn checker_device(partition: &File, repair: bool) -> Result<File> {
let control = OpenOptions::new()
.read(true)
.write(true)
.custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
.open("/dev/loop-control")?;
for _ in 0..8 {
let number = unsafe { libc::ioctl(control.as_raw_fd(), 0x4c82u32 as _) };
checked(number, "allocate recovery loop device")?;
let device = OpenOptions::new()
.read(true)
.write(true)
.custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
.open(format!("/dev/loop{number}"))?;
let metadata = device.metadata()?;
if !metadata.file_type().is_block_device()
|| libc::major(metadata.rdev()) != 7
|| libc::minor(metadata.rdev()) != number as u32
{
return Err(Error("Unexpected recovery loop device identity".into()));
}
let mut config: LoopConfig = unsafe { std::mem::zeroed() };
config.fd = partition.as_raw_fd() as u32;
config.info.flags = 4 | if repair { 0 } else { 1 }; // AUTOCLEAR, READ_ONLY
let result = unsafe { libc::ioctl(device.as_raw_fd(), 0x4c0au32 as _, &config) };
if result == 0 {
return Ok(device);
}
if std::io::Error::last_os_error().raw_os_error() != Some(libc::EBUSY) {
checked(result, "bind verified DATA partition for recovery")?;
}
// Retry an actual allocation race, without a time-based delay.
}
Err(Error(
"Recovery loop allocation repeatedly raced with another user".into(),
))
}
pub fn select(bay: Bay, usb: &Path) -> Result<Selection> {
let disk = Disk::select_current(usb)?;
let mut parts = media::partitions_for(usb)?;
if parts.len() != 1 || parts[0].partition_name != "FDS_DATA" {
return Err(Error(
"Recovery requires one FDS_DATA partition; SYSTEM and internal storage are excluded"
.into(),
));
}
let partition = parts.remove(0);
let key = media::key(&partition)?;
let confirmation = format!(
"REPAIR BAY{bay} DISK{} BOOT{}",
disk.diskseq,
fds_common::trace::boot_id()?
);
Ok(Selection {
bay,
disk,
partition,
key,
confirmation,
log: format!(
"/run/fds/recovery/data-{bay}-{}.log",
fds_common::trace::now()?
),
})
}
impl Selection {
pub fn confirm(&self, repair: bool, confirmation: Option<&str>) -> Result<()> {
if repair && confirmation != Some(self.confirmation.as_str())
|| !repair && confirmation.is_some()
{
return Err(Error(
"Recovery confirmation does not match this bay, insertion and boot".into(),
));
}
Ok(())
}
pub fn report(&self, checked: bool, repaired: bool) -> RecoveryReport {
RecoveryReport {
disk: BayDisk {
bay: self.bay,
diskseq: self.disk.diskseq,
bytes: self.disk.bytes,
sector_bytes: self.disk.sector_bytes,
model: self.disk.model.clone(),
serial: self.disk.serial.clone(),
protected: None,
},
confirmation: if checked {
None
} else {
Some(self.confirmation.clone())
},
checked,
repaired,
log: checked.then(|| self.log.clone()),
}
}
pub fn run(&self, repair: bool) -> Result<()> {
let result = self.check_filesystem(repair);
result.map_err(|error| {
Error(format!(
"DATA recovery failed: {error}; no SAFE status issued. Log: {}",
self.log
))
})
}
fn check_filesystem(&self, repair: bool) -> Result<()> {
// The whole-disk reservation excludes mounts and other FDS writers.
// A child inherits it so daemon death cannot release it before fsck exits.
let disk = self.disk.open_exclusive()?;
let image = fds_burn::image::inspect(&disk, self.disk.bytes)?;
if image.class != Class::Data || image.filesystem != "ext4" {
return Err(Error(
"Recovery accepts only a valid single-partition DATA GPT with ext4".into(),
));
}
let part = OpenOptions::new()
.read(true)
.write(repair)
.custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW | libc::O_NONBLOCK)
.open(&self.partition.device)?;
let metadata = part.metadata()?;
let mut diskseq = 0u64;
checked(
unsafe { libc::ioctl(part.as_raw_fd(), 0x80081280u32 as _, &mut diskseq) },
"verify DATA insertion",
)?;
let sysfs = fs::canonicalize(
Path::new("/sys/dev/block")
.join(format!("{}:{}", self.partition.major, self.partition.minor)),
)?;
let number = |name: &str| -> Result<u64> {
read_text(&sysfs.join(name), 128)?
.trim()
.parse()
.map_err(|_| Error("Invalid partition geometry".into()))
};
if !metadata.file_type().is_block_device()
|| libc::major(metadata.rdev()) != self.partition.major
|| libc::minor(metadata.rdev()) != self.partition.minor
|| diskseq != self.disk.diskseq
|| media::key(&self.partition)? != self.key
|| number("start")?.checked_mul(512) != Some(image.partition_start)
|| number("size")?.checked_mul(512) != Some(image.partition_bytes)
{
return Err(Error("DATA identity or partition geometry changed".into()));
}
// e2fsck claims its device exclusively too. Give it an automatically
// removed loop view while retaining the physical whole-disk claim;
// releasing that claim would allow an unrelated mount during repairs.
let checker = checker_device(&part, repair)?;
fs::create_dir_all("/run/fds/recovery")?;
fs::set_permissions("/run/fds/recovery", fs::Permissions::from_mode(0o700))?;
let log = OpenOptions::new()
.write(true)
.create_new(true)
.mode(0o600)
.custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW)
.open(&self.log)?;
let run = |fix: bool| -> Result<i32> {
let parent = unsafe { libc::getpid() };
let disk_fd = disk.as_raw_fd();
let part_fd = checker.as_raw_fd();
let mut command = Command::new("/usr/bin/e2fsck");
command
.args(["-f", if fix { "-p" } else { "-n" }])
.arg(format!("/proc/self/fd/{part_fd}"))
.env_clear()
.env("PATH", "/usr/bin:/bin")
.env("LC_ALL", "C")
.stdin(Stdio::null())
.stdout(log.try_clone()?)
.stderr(log.try_clone()?);
unsafe {
command.pre_exec(move || {
if libc::prctl(libc::PR_SET_PDEATHSIG, libc::SIGKILL) < 0
|| libc::getppid() != parent
{
return Err(std::io::Error::other("Recovery parent disappeared"));
}
let mut mask = std::mem::zeroed();
libc::sigemptyset(&mut mask);
let limit = libc::rlimit {
rlim_cur: 16 * 1024 * 1024,
rlim_max: 16 * 1024 * 1024,
};
if libc::sigprocmask(libc::SIG_SETMASK, &mask, std::ptr::null_mut()) < 0
|| libc::setrlimit(libc::RLIMIT_FSIZE, &limit) < 0
|| libc::fcntl(disk_fd, libc::F_SETFD, 0) < 0
|| libc::fcntl(part_fd, libc::F_SETFD, 0) < 0
{
return Err(std::io::Error::last_os_error());
}
Ok(())
});
}
command
.status()?
.code()
.ok_or_else(|| Error("Filesystem checker was terminated".into()))
};
let status = run(repair)?;
if status != 0 && !(repair && status == 1) {
return Err(Error(format!(
"e2fsck returned {status}; unresolved errors require review"
)));
}
if repair {
checker.sync_all()?;
part.sync_all()?;
checked(
unsafe { libc::ioctl(checker.as_raw_fd(), 0x1261u32 as _) },
"invalidate recovery loop cache",
)?;
checked(
unsafe { libc::ioctl(part.as_raw_fd(), 0x1261u32 as _) },
"flush and invalidate DATA block cache",
)?;
let verified = run(false)?;
if verified != 0 {
return Err(Error(format!(
"Post-repair read-only check returned {verified}"
)));
}
}
if !self.disk.present() || media::key(&self.partition)? != self.key {
return Err(Error("DATA disappeared during recovery".into()));
}
Ok(())
}
}
File diff suppressed because it is too large Load Diff
+318
View File
@@ -0,0 +1,318 @@
//! Metadata-first software media. Building is exclusively a workstation operation.
use crate::media::{self, c, checked};
use fds_burn::{device::Disk, image};
use fds_common::{Bay, Error, Result, read_text, sysfs::BlockPartition};
use fds_software::{Catalogue, archive};
use std::{
collections::{BTreeMap, BTreeSet},
fs::{self, File, OpenOptions},
os::{
fd::AsRawFd,
unix::fs::{FileTypeExt, MetadataExt, OpenOptionsExt, PermissionsExt},
},
path::{Path, PathBuf},
};
struct Payload {
path: String,
source: File,
partition: BlockPartition,
key: String,
}
pub struct Mounted {
pub catalogue: Catalogue,
root: String,
payloads: BTreeMap<u8, Payload>,
caches: BTreeMap<String, String>,
}
fn mount(source: &str, path: &str, kind: &str, flags: libc::c_ulong, options: &str) -> Result<()> {
checked(
unsafe {
libc::mount(
c(source)?.as_ptr(),
c(path)?.as_ptr(),
c(kind)?.as_ptr(),
flags,
c(options)?.as_ptr().cast(),
)
},
"mount software storage",
)
}
fn exclusive(source: &File, paths: &[&str]) -> Result<()> {
let meta = source.metadata()?;
let dev = if meta.file_type().is_block_device() {
meta.rdev()
} else {
meta.dev()
};
let identity = format!("{}:{}", libc::major(dev), libc::minor(dev));
let mounts = read_text(Path::new("/proc/self/mountinfo"), 4 * 1024 * 1024)?;
let found: BTreeSet<_> = mounts
.lines()
.filter(|line| line.split_whitespace().nth(2) == Some(identity.as_str()))
.filter_map(|line| line.split_whitespace().nth(4))
.collect();
if found != paths.iter().copied().collect() {
return Err(Error(
"Software storage has additional or unexpected mounts; close them before eject".into(),
));
}
Ok(())
}
impl Mounted {
pub fn open(bay: Bay, metadata: &str, usb: &Path, parts: &[BlockPartition]) -> Result<Self> {
let catalogue = Catalogue::parse(&media::metadata(metadata, "SOFTWARE.TOML")?)?;
let disk = Disk::select_current(usb)?;
let source = OpenOptions::new()
.read(true)
.custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK | libc::O_CLOEXEC)
.open(&disk.path)?;
let stat = source.metadata()?;
if !stat.file_type().is_block_device()
|| libc::major(stat.rdev()) != disk.major
|| libc::minor(stat.rdev()) != disk.minor
{
return Err(Error("Software disk identity changed".into()));
}
let layout = image::inspect(&source, disk.bytes)?;
if layout.partitions.first().map(|p| p.name.as_str()) != Some("FDS_METADATA")
|| layout.partitions.len() != catalogue.partition_count()
|| parts.len() != layout.partitions.len()
{
return Err(Error(
"Software catalogue does not match the complete GPT layout".into(),
));
}
let disk_path = fs::canonicalize(format!("/sys/dev/block/{}:{}", disk.major, disk.minor))?;
let root = format!("/run/fds/software/{bay}");
fs::create_dir_all(&root)?;
fs::set_permissions(&root, fs::Permissions::from_mode(0o755))?;
let mut result = Self {
catalogue,
root,
payloads: BTreeMap::new(),
caches: BTreeMap::new(),
};
for spec in &layout.partitions {
let part = parts
.iter()
.find(|p| p.partition_name == spec.name)
.ok_or_else(|| Error("Missing software partition".into()))?;
let sys = fs::canonicalize(format!("/sys/dev/block/{}:{}", part.major, part.minor))?;
let number = |name: &str| -> Result<u64> {
read_text(&sys.join(name), 64)?
.trim()
.parse()
.map_err(|_| Error("Invalid kernel partition geometry".into()))
};
if sys.parent() != Some(disk_path.as_path())
|| number("partition")? != u64::from(spec.number)
|| number("start")? != spec.start / 512
|| number("size")? != spec.bytes / 512
{
return Err(Error(
"Kernel partition geometry disagrees with verified GPT".into(),
));
}
if spec.number == 1 {
continue;
}
let key = media::key(part)?;
let source = OpenOptions::new()
.read(true)
.custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK | libc::O_CLOEXEC)
.open(&part.device)?;
let stat = source.metadata()?;
if !stat.file_type().is_block_device()
|| libc::major(stat.rdev()) != part.major
|| libc::minor(stat.rdev()) != part.minor
|| media::key(part)? != key
{
return Err(Error("Software payload changed during inspection".into()));
}
let path = format!("{}/payload{:02}", result.root, spec.number);
fs::create_dir_all(&path)?;
mount(
&format!("/proc/self/fd/{}", source.as_raw_fd()),
&path,
"erofs",
libc::MS_RDONLY | libc::MS_NOEXEC | libc::MS_NOSUID | libc::MS_NODEV,
"",
)?;
result.payloads.insert(
spec.number,
Payload {
path: path.clone(),
source,
partition: part.clone(),
key,
},
);
let bundles = Path::new(&path).join("bundles");
if !fs::symlink_metadata(&bundles)?.is_dir() {
return Err(Error("Payload bundles must be a real directory".into()));
}
let expected: BTreeSet<_> = result
.catalogue
.software
.iter()
.filter(|s| s.partition == spec.number)
.map(|s| format!("{}.tar.xz", s.id))
.collect();
let actual: BTreeSet<_> = fs::read_dir(&bundles)?
.map(|e| Ok(e?.file_name().to_string_lossy().into_owned()))
.collect::<Result<_>>()?;
if actual != expected {
return Err(Error(
"Payload archive inventory disagrees with catalogue".into(),
));
}
for software in result
.catalogue
.software
.iter()
.filter(|s| s.partition == spec.number)
{
if archive::open(&Path::new(&path).join(software.archive_path()))?
.metadata()?
.len()
!= software.archive_bytes
{
return Err(Error(
"Software archive size disagrees with catalogue".into(),
));
}
}
}
if Disk::select_current(usb)? != disk {
return Err(Error("Software disk changed during validation".into()));
}
result.exclusive()?;
Ok(result)
}
pub fn exclusive(&self) -> Result<()> {
for payload in self.payloads.values() {
exclusive(&payload.source, &[&payload.path])?;
}
for path in self.caches.values() {
exclusive(&File::open(path)?, &[path])?;
}
Ok(())
}
pub fn program(
&mut self,
arguments: &[String],
) -> Result<(Vec<String>, Vec<(String, String)>)> {
let (id, command) = arguments
.first()
.and_then(|s| s.split_once(':'))
.ok_or_else(|| {
Error("Use SOFTWARE-ID:COMMAND; fds bay BAY lists available commands".into())
})?;
let software = self
.catalogue
.software
.iter()
.find(|s| s.id == id)
.ok_or_else(|| Error("Unknown software id".into()))?;
let executable = software
.commands
.get(command)
.ok_or_else(|| Error("Unknown software command".into()))?;
let payload = self
.payloads
.get(&software.partition)
.ok_or_else(|| Error("Software is being ejected; remove and reinsert it".into()))?;
if media::key(&payload.partition)? != payload.key {
return Err(Error("Software payload was removed".into()));
}
if !self.caches.contains_key(id) {
// Each executable tree receives its own bounded, read-only tmpfs.
// Root owns every path; the consumer only receives ordinary UID 1000.
let capacity = software
.unpacked_bytes
.checked_add(u64::from(software.entries) * 4096 + 1024 * 1024)
.ok_or_else(|| Error("Runtime cache size overflow".into()))?;
if capacity > 256 * 1024 * 1024 {
return Err(Error("Software exceeds the 256 MiB runtime cache limit; use a smaller workstation bundle".into()));
}
let path = format!("{}/cache-{}", self.root, id);
fs::create_dir_all(&path)?;
mount(
"tmpfs",
&path,
"tmpfs",
libc::MS_NOSUID | libc::MS_NODEV,
&format!(
"mode=0700,size={capacity},nr_inodes={}",
software.entries + 1024
),
)?;
self.caches.insert(id.to_owned(), path.clone());
let result = (|| {
archive::verify(
&archive::open(&PathBuf::from(&payload.path).join(software.archive_path()))?,
software,
Some(Path::new(&path)),
)?;
fs::set_permissions(&path, fs::Permissions::from_mode(0o755))?;
mount(
"tmpfs",
&path,
"tmpfs",
libc::MS_REMOUNT | libc::MS_RDONLY | libc::MS_NOSUID | libc::MS_NODEV,
"",
)?;
Ok(())
})();
if let Err(error) = result {
media::unmount(&path, false)?;
self.caches.remove(id);
fs::remove_dir(&path)?;
return Err(error);
}
}
let root = &self.caches[id];
let mut args = arguments.to_vec();
args[0] = format!("{root}/{executable}");
Ok((
args,
vec![
("FDS_APP".into(), root.clone()),
("PATH".into(), format!("{root}/bin:/usr/bin:/bin")),
("LD_LIBRARY_PATH".into(), format!("{root}/lib")),
("XDG_DATA_DIRS".into(), format!("{root}/share:/usr/share")),
("DISPLAY".into(), ":0".into()),
("XAUTHORITY".into(), "/run/fds/x11/authority".into()),
],
))
}
pub fn release(&mut self, removed: bool) -> Result<()> {
if !removed {
self.exclusive()?;
}
while let Some((id, path)) = self.caches.last_key_value() {
media::unmount(path, removed)?;
let id = id.clone();
self.caches.remove(&id);
}
while let Some((number, payload)) = self.payloads.last_key_value() {
media::unmount(&payload.path, removed)?;
let number = *number;
self.payloads.remove(&number);
}
if Path::new(&self.root).exists() {
fs::remove_dir_all(&self.root)?;
}
Ok(())
}
}
impl Drop for Mounted {
fn drop(&mut self) {
if !self.payloads.is_empty() || !self.caches.is_empty() {
if let Err(error) = self.release(false) {
eprintln!("Software cleanup: {error}");
}
}
}
}
+194
View File
@@ -0,0 +1,194 @@
//! Minimal authenticated X11 readiness observer. Subscribe before the snapshot,
//! including when the EWMH atom has not been created yet. No Xlib dependency.
//! Wire formats: X Window System Protocol, Appendix B (connection and requests).
use fds_common::{Error, Result};
use std::{
fs,
io::{Read, Write},
os::{
fd::{AsRawFd, RawFd},
unix::net::UnixStream,
},
time::Duration,
};
fn u16le(data: &[u8], at: usize) -> Result<u16> {
Ok(u16::from_le_bytes(
data.get(at..at + 2)
.ok_or_else(|| Error("Short X11 packet".into()))?
.try_into()
.unwrap(),
))
}
fn u32le(data: &[u8], at: usize) -> Result<u32> {
Ok(u32::from_le_bytes(
data.get(at..at + 4)
.ok_or_else(|| Error("Short X11 packet".into()))?
.try_into()
.unwrap(),
))
}
fn root_window(data: &[u8]) -> Result<u32> {
if data.len() < 32 || data[20] == 0 {
return Err(Error("X11 server has no screen".into()));
}
let start = 32 + (usize::from(u16le(data, 16)?) + 3) / 4 * 4 + usize::from(data[21]) * 8;
u32le(data, start)
}
struct Packet {
header: [u8; 32],
body: Vec<u8>,
}
pub struct Observer {
stream: UnixStream,
root: u32,
atom: u32,
sequence: u16,
}
impl AsRawFd for Observer {
fn as_raw_fd(&self) -> RawFd {
self.stream.as_raw_fd()
}
}
impl Observer {
pub fn connect() -> Result<Self> {
let authority = fs::read("/run/fds/x11/authority")?;
if authority.len() != 45 || &authority[9..27] != b"MIT-MAGIC-COOKIE-1" {
return Err(Error("Invalid FDS X authority record".into()));
}
let mut stream = UnixStream::connect("/tmp/.X11-unix/X0")?;
stream.set_read_timeout(Some(Duration::from_secs(2)))?;
stream.set_write_timeout(Some(Duration::from_secs(2)))?;
let mut setup = vec![b'l', 0, 11, 0, 0, 0, 18, 0, 16, 0, 0, 0];
setup.extend_from_slice(b"MIT-MAGIC-COOKIE-1");
setup.extend_from_slice(&[0, 0]);
setup.extend_from_slice(&authority[29..]);
stream.write_all(&setup)?;
let mut header = [0u8; 8];
stream.read_exact(&mut header)?;
if header[0] != 1 || u16le(&header, 2)? != 11 {
return Err(Error("X11 authentication/setup failed".into()));
}
let mut body = vec![0; usize::from(u16le(&header, 6)?) * 4];
stream.read_exact(&mut body)?;
let root = root_window(&body)?;
let mut observer = Self {
stream,
root,
atom: 0,
sequence: 0,
};
// Event selection and InternAtom are ordered on this same connection.
// Its reply proves the subscription is installed before WindowMaker starts.
observer.send(
2,
0,
&[
root.to_le_bytes(),
(1u32 << 11).to_le_bytes(),
(1u32 << 22).to_le_bytes(),
]
.concat(),
)?;
let name = b"_NET_SUPPORTING_WM_CHECK";
let mut data = Vec::from((name.len() as u16).to_le_bytes());
data.extend_from_slice(&[0, 0]);
data.extend_from_slice(name);
data.resize((data.len() + 3) / 4 * 4, 0);
observer.send(16, 0, &data)?;
let packet = observer.reply()?;
observer.atom = u32le(&packet.header, 8)?;
if observer.atom == 0 {
return Err(Error("X11 could not allocate the readiness atom".into()));
}
Ok(observer)
}
fn send(&mut self, opcode: u8, detail: u8, body: &[u8]) -> Result<()> {
let mut data = vec![opcode, detail];
data.extend_from_slice(&((body.len() / 4 + 1) as u16).to_le_bytes());
data.extend_from_slice(body);
self.stream.write_all(&data)?;
self.sequence = self.sequence.wrapping_add(1);
Ok(())
}
fn packet(&mut self) -> Result<Packet> {
let mut header = [0u8; 32];
self.stream.read_exact(&mut header)?;
if header[0] == 0 {
return Err(Error(format!(
"X11 request failed with error {}",
header[1]
)));
}
let length = if header[0] == 1 || header[0] & 0x7f == 35 {
u32le(&header, 4)? as usize * 4
} else {
0
};
if length > 65536 {
return Err(Error("X11 reply exceeds readiness protocol limit".into()));
}
let mut body = vec![0; length];
self.stream.read_exact(&mut body)?;
Ok(Packet { header, body })
}
fn reply(&mut self) -> Result<Packet> {
for _ in 0..1024 {
let packet = self.packet()?;
if packet.header[0] == 1 {
if u16le(&packet.header, 2)? != self.sequence {
return Err(Error("Unexpected X11 reply sequence".into()));
}
return Ok(packet);
}
}
Err(Error(
"Excessive X11 events while waiting for a reply".into(),
))
}
pub fn ready(&mut self) -> Result<bool> {
self.send(
20,
0,
&[
self.root.to_le_bytes(),
self.atom.to_le_bytes(),
0u32.to_le_bytes(),
0u32.to_le_bytes(),
1u32.to_le_bytes(),
]
.concat(),
)?;
let packet = self.reply()?;
Ok(packet.header[1] == 32
&& u32le(&packet.header, 8)? == 33
&& u32le(&packet.header, 16)? == 1
&& packet.body.len() == 4
&& u32le(&packet.body, 0)? != 0)
}
pub fn event(&mut self) -> Result<bool> {
let packet = self.packet()?;
if packet.header[0] & 0x7f == 28
&& u32le(&packet.header, 4)? == self.root
&& u32le(&packet.header, 8)? == self.atom
{
self.ready()
} else {
Ok(false)
}
}
}
#[cfg(test)]
mod tests {
#[test]
fn setup_offsets_account_for_vendor_padding_and_pixmap_formats() {
let mut data = vec![0; 56];
data[16] = 3;
data[20] = 1;
data[21] = 2;
data[52..56].copy_from_slice(&0x1234u32.to_le_bytes());
assert_eq!(super::root_window(&data).unwrap(), 0x1234);
assert!(super::root_window(&data[..55]).is_err());
assert!(super::root_window(&[]).is_err());
}
}