Files
OpenPXE/crates/iso-store/src/nfs.rs
T
2026-05-21 02:13:08 -04:00

559 lines
20 KiB
Rust

//! NFS share manager.
//!
//! Lets an operator mount a remote NFS export as an ISO source instead of
//! uploading every ISO into the container's PVC. Supports NFSv3 and
//! NFSv4.1 — the two versions the user explicitly asked for.
//!
//! ## How it works
//!
//! 1. Operator submits a mount spec via the Storage tab:
//! `{ server: "10.0.0.20", export: "/srv/isos", version: "v41" }`.
//! 2. We slugify a stable id, mkdir `<work_dir>/nfs/<id>/`, then shell out
//! to `/bin/mount -t nfs -o vers=...,ro,nolock server:export local`.
//! 3. On success we walk the mount point looking for `*.iso` files and
//! register each one with the `IsoStore` as an external source — same
//! introspection pipeline as a web upload, but no sha256 (the bytes
//! live on a remote machine; hashing them would suck them through the
//! network on every restart).
//! 4. On failure we record `last_error` on the spec and persist anyway
//! so the UI can show a row in red rather than silently dropping it.
//!
//! ## Operational notes
//!
//! - Mounting NFS inside a container needs `CAP_SYS_ADMIN` and the
//! `nfs-common` package. The default image ships these (see Dockerfile).
//! - On OpenShift, the SCC must allow `CAP_SYS_ADMIN`. The bundled SCC
//! doesn't — operators have to opt in by switching to a more privileged
//! SCC or running NFS mounts as a CSI driver outside the pod.
//! - Mount commands are issued sequentially under a single mutex to avoid
//! `mount` racing on the same target dir.
//!
//! ## Persistence
//!
//! Mount specs (without runtime state) live at `<work_dir>/nfs.json`,
//! re-mounted on startup. Mounts that fail to come back online keep their
//! spec and their `last_error` so the operator sees what happened.
use crate::introspect::{introspect, IntrospectionReport};
use crate::store::{generate_boot_entries_for, slugify_str, IsoSource, IsoStore};
use openpxe_core::{Error, Result};
use parking_lot::Mutex;
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use time::OffsetDateTime;
use tokio::process::Command;
/// Wire-protocol versions we support. Keep this enum closed — silently
/// accepting "auto" or letting the kernel negotiate would mean operators
/// could never confirm which version is in use.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum NfsVersion {
/// NFSv3 — UDP/TCP, separate `mountd` protocol. Required for many
/// older NAS appliances.
V3,
/// NFSv4.1 — single TCP port (2049), session-based. Modern default.
V41,
}
impl NfsVersion {
fn vers_arg(self) -> &'static str {
match self {
Self::V3 => "vers=3",
Self::V41 => "vers=4.1",
}
}
}
/// One configured mount. The id is generated from server+export so the
/// operator can re-add the same export idempotently.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct NfsMount {
pub id: String,
pub server: String,
pub export: String,
pub version: NfsVersion,
/// Read-only by default — most ISO libraries are. Operators that need
/// write can flip this off but OpenPXE itself never writes.
pub read_only: bool,
/// Local mount point under `<work_dir>/nfs/`.
pub local_path: PathBuf,
/// Whether the mount is currently active.
pub mounted: bool,
/// Last error encountered on a `mount` or `umount` attempt; cleared on
/// success.
pub last_error: Option<String>,
#[serde(with = "time::serde::rfc3339::option")]
pub last_attempt: Option<OffsetDateTime>,
/// Number of `.iso` files found on the share (re-counted on each scan).
pub iso_count: u32,
}
/// Spec submitted by the UI. Server and export are normalized before use.
#[derive(Debug, Clone, Deserialize)]
pub struct NfsAddRequest {
pub server: String,
pub export: String,
#[serde(default = "default_version")]
pub version: NfsVersion,
#[serde(default = "default_ro")]
pub read_only: bool,
}
fn default_version() -> NfsVersion {
NfsVersion::V41
}
fn default_ro() -> bool {
true
}
#[derive(Debug, Default)]
struct Inner {
mounts: HashMap<String, NfsMount>,
}
/// Manages NFS mounts and surfaces them as ISO sources.
///
/// Cheap to clone — internal state is `Arc<Mutex<...>>`.
#[derive(Debug, Clone)]
pub struct NfsManager {
work_root: Arc<PathBuf>,
state_path: Arc<PathBuf>,
inner: Arc<Mutex<Inner>>,
iso_store: IsoStore,
/// Single-writer lock around the actual `mount`/`umount` shell-outs;
/// avoids racing on the same target directory.
mount_lock: Arc<tokio::sync::Mutex<()>>,
}
impl NfsManager {
/// Construct a manager rooted at `work_dir`. Mount points live under
/// `<work_dir>/nfs/<id>/`. State persists to `<work_dir>/nfs.json`.
#[must_use]
pub fn new(work_dir: &Path, iso_store: IsoStore) -> Self {
let work_root = work_dir.join("nfs");
let state_path = work_dir.join("nfs.json");
Self {
work_root: Arc::new(work_root),
state_path: Arc::new(state_path),
inner: Arc::new(Mutex::new(Inner::default())),
iso_store,
mount_lock: Arc::new(tokio::sync::Mutex::new(())),
}
}
/// Where this manager mounts shares. Used by `IsoStore` to resolve
/// NFS-backed `IsoMeta`s to their on-disk path.
#[must_use]
pub fn mount_root(&self) -> PathBuf {
self.work_root.as_ref().clone()
}
/// Load persisted state and re-attempt every mount. Errors are logged
/// per-mount but never fail the call — startup must not block on a
/// remote NFS server being slow.
pub async fn load_and_remount(&self) -> Result<()> {
tokio::fs::create_dir_all(self.work_root.as_path()).await?;
let mounts = match tokio::fs::read_to_string(self.state_path.as_path()).await {
Ok(text) => serde_json::from_str::<Vec<NfsMount>>(&text).unwrap_or_default(),
Err(_) => Vec::new(),
};
for mut m in mounts {
// Always start from "not mounted" — the kernel state was lost
// when the process died. We'll try to remount each one.
m.mounted = false;
m.last_error = None;
self.inner.lock().mounts.insert(m.id.clone(), m.clone());
if let Err(e) = self.try_mount(&m.id).await {
tracing::warn!(
target: "openpxe::nfs",
id = %m.id, error = %e,
"could not remount NFS share on startup"
);
}
}
Ok(())
}
/// Add a new mount. Returns the resulting `NfsMount` (with `mounted`
/// reflecting reality) or an error if the spec was invalid.
pub async fn add(&self, req: NfsAddRequest) -> Result<NfsMount> {
let server = req.server.trim().to_string();
let export = req.export.trim().to_string();
if server.is_empty() {
return Err(Error::Invalid("server is required".into()));
}
if !export.starts_with('/') {
return Err(Error::Invalid("export path must start with '/'".into()));
}
let id = mount_id(&server, &export);
let local_path = self.work_root.join(&id);
tokio::fs::create_dir_all(&local_path).await?;
let mount = NfsMount {
id: id.clone(),
server,
export,
version: req.version,
read_only: req.read_only,
local_path,
mounted: false,
last_error: None,
last_attempt: None,
iso_count: 0,
};
self.inner.lock().mounts.insert(id.clone(), mount);
self.persist_locked();
self.try_mount(&id).await?;
Ok(self.get(&id).expect("mount just inserted"))
}
/// Unmount and forget a share. Removes any ISOs it contributed from
/// the IsoStore and deletes the local mount point. Idempotent.
pub async fn remove(&self, id: &str) -> Result<()> {
// Best-effort umount; even if it fails (e.g. server unreachable)
// we still want to drop the in-memory record.
let _ = self.umount_one(id).await;
let local_path = {
let mut g = self.inner.lock();
g.mounts.remove(id).map(|m| m.local_path)
};
self.persist_locked();
self.iso_store.drop_external_source(id);
if let Some(p) = local_path {
// rmdir only — never recurse, the mount could still be live
// on some kernel error path and we don't want to nuke a
// remote filesystem.
let _ = tokio::fs::remove_dir(&p).await;
}
Ok(())
}
/// Re-scan a mounted share for ISOs, refreshing the IsoStore.
pub async fn rescan(&self, id: &str) -> Result<u32> {
let mount = self
.get(id)
.ok_or_else(|| Error::Invalid(format!("no such mount '{id}'")))?;
if !mount.mounted {
return Err(Error::Invalid(format!("mount '{id}' is not active")));
}
let count = self.scan_and_register(&mount).await?;
if let Some(m) = self.inner.lock().mounts.get_mut(id) {
m.iso_count = count;
}
self.persist_locked();
Ok(count)
}
/// Snapshot of every configured mount.
#[must_use]
pub fn list(&self) -> Vec<NfsMount> {
let g = self.inner.lock();
let mut v: Vec<_> = g.mounts.values().cloned().collect();
v.sort_by(|a, b| a.id.cmp(&b.id));
v
}
/// Look up a single mount by id.
#[must_use]
pub fn get(&self, id: &str) -> Option<NfsMount> {
self.inner.lock().mounts.get(id).cloned()
}
// ── internals ─────────────────────────────────────────────────────
async fn try_mount(&self, id: &str) -> Result<()> {
let _g = self.mount_lock.lock().await;
let m = self
.get(id)
.ok_or_else(|| Error::Invalid(format!("no such mount '{id}'")))?;
let now = OffsetDateTime::now_utc();
// Already mounted? Skip — `mount` would error on a busy target
// and confuse the operator's UI status.
if is_mountpoint(&m.local_path).await {
self.update_status(id, true, None, now);
// Even though already mounted, we still want a fresh ISO count.
let count = self.scan_and_register(&m).await.unwrap_or(0);
self.update_iso_count(id, count);
return Ok(());
}
let opts = mount_options(&m);
let target = format!("{}:{}", m.server, m.export);
let output = Command::new("mount")
.arg("-t")
.arg("nfs")
.arg("-o")
.arg(&opts)
.arg(&target)
.arg(&m.local_path)
.output()
.await;
match output {
Ok(out) if out.status.success() => {
tracing::info!(
target: "openpxe::nfs",
id = %id, server = %m.server, export = %m.export,
version = ?m.version,
"NFS mount succeeded"
);
self.update_status(id, true, None, now);
let count = self.scan_and_register(&m).await.unwrap_or(0);
self.update_iso_count(id, count);
Ok(())
}
Ok(out) => {
let err = format!(
"mount exit {}: {}",
out.status.code().unwrap_or(-1),
String::from_utf8_lossy(&out.stderr).trim()
);
tracing::warn!(target: "openpxe::nfs", id = %id, "{err}");
self.update_status(id, false, Some(err.clone()), now);
Err(Error::Invalid(err))
}
Err(e) => {
let err = format!("could not exec /bin/mount: {e}");
tracing::error!(target: "openpxe::nfs", id = %id, "{err}");
self.update_status(id, false, Some(err.clone()), now);
Err(Error::Invalid(err))
}
}
}
async fn umount_one(&self, id: &str) -> Result<()> {
let _g = self.mount_lock.lock().await;
let Some(m) = self.get(id) else { return Ok(()) };
if !is_mountpoint(&m.local_path).await {
self.update_status(id, false, None, OffsetDateTime::now_utc());
return Ok(());
}
// -l = lazy: detach immediately, finish when no process has a
// handle. Important if a stale ISO read is still in flight.
let out = Command::new("umount")
.arg("-l")
.arg(&m.local_path)
.output()
.await;
match out {
Ok(o) if o.status.success() => {
self.update_status(id, false, None, OffsetDateTime::now_utc());
Ok(())
}
Ok(o) => {
let e = format!(
"umount exit {}: {}",
o.status.code().unwrap_or(-1),
String::from_utf8_lossy(&o.stderr).trim()
);
self.update_status(id, false, Some(e.clone()), OffsetDateTime::now_utc());
Err(Error::Invalid(e))
}
Err(e) => {
let e = format!("could not exec /bin/umount: {e}");
self.update_status(id, false, Some(e.clone()), OffsetDateTime::now_utc());
Err(Error::Invalid(e))
}
}
}
/// Walk the mount point for `*.iso` files, introspect each one, and
/// register it with the IsoStore as an NFS-sourced entry. Returns the
/// count of ISOs registered.
async fn scan_and_register(&self, m: &NfsMount) -> Result<u32> {
// Drop any prior entries from this mount before re-registering, so
// a removed file disappears from the store.
self.iso_store.drop_external_source(&m.id);
let mut walker = tokio::fs::read_dir(&m.local_path).await?;
let mut count = 0u32;
while let Some(entry) = walker.next_entry().await? {
let p = entry.path();
if p.extension()
.and_then(|e| e.to_str())
.map(str::to_ascii_lowercase)
.as_deref()
!= Some("iso")
{
continue;
}
let filename = match p.file_name().and_then(|s| s.to_str()) {
Some(f) => f.to_string(),
None => continue,
};
let size = tokio::fs::metadata(&p).await?.len();
// Introspection is sync + IO-bound (reads ISO9660 PVD). Push
// it to a blocking thread so the runtime stays responsive on
// a slow share.
let p_owned = p.clone();
let report: IntrospectionReport =
tokio::task::spawn_blocking(move || introspect(&p_owned))
.await
.map_err(|e| Error::Other(e.into()))?;
let id = format!("nfs-{}-{}", m.id, slugify_str(&filename));
let boot_entries = generate_boot_entries_for(&id, &filename, &report);
let source = IsoSource::Nfs {
mount_id: m.id.clone(),
relative_path: filename.clone(),
};
self.iso_store
.register_external(id, filename, size, report, boot_entries, source);
count += 1;
}
Ok(count)
}
fn update_status(&self, id: &str, mounted: bool, err: Option<String>, ts: OffsetDateTime) {
if let Some(m) = self.inner.lock().mounts.get_mut(id) {
m.mounted = mounted;
m.last_error = err;
m.last_attempt = Some(ts);
}
self.persist_locked();
}
fn update_iso_count(&self, id: &str, count: u32) {
if let Some(m) = self.inner.lock().mounts.get_mut(id) {
m.iso_count = count;
}
self.persist_locked();
}
/// Atomically replace the on-disk JSON with the current state.
/// Persistence errors are logged, never propagated — settings live in
/// memory authoritatively, matching the SettingsStore policy.
fn persist_locked(&self) {
let mounts: Vec<NfsMount> = self.inner.lock().mounts.values().cloned().collect();
let path = self.state_path.as_path();
let tmp = path.with_extension("json.tmp");
let body = match serde_json::to_vec_pretty(&mounts) {
Ok(b) => b,
Err(e) => {
tracing::warn!(target: "openpxe::nfs", "serialize NFS state: {e}");
return;
}
};
if let Some(parent) = path.parent() {
let _ = std::fs::create_dir_all(parent);
}
if let Err(e) = std::fs::write(&tmp, body) {
tracing::warn!(target: "openpxe::nfs", "write NFS state tmp: {e}");
return;
}
if let Err(e) = std::fs::rename(&tmp, path) {
tracing::warn!(target: "openpxe::nfs", "rename NFS state: {e}");
}
}
}
fn mount_options(m: &NfsMount) -> String {
let mut opts = vec![m.version.vers_arg().to_string()];
if m.read_only {
opts.push("ro".into());
} else {
opts.push("rw".into());
}
// `nolock` for v3 — many storage appliances disable lockd; we don't
// need locking for read-only ISO access anyway.
if matches!(m.version, NfsVersion::V3) {
opts.push("nolock".into());
}
// Soft mount with a generous timeout — better to surface a hung share
// as a user-visible error than to wedge the iPXE client forever on a
// dead NFS server.
opts.push("soft".into());
opts.push("timeo=100".into());
opts.push("retrans=3".into());
opts.join(",")
}
fn mount_id(server: &str, export: &str) -> String {
let raw = format!("{server}{export}");
slugify_str(&raw)
}
/// Detect whether `path` is currently a mount point. We don't have
/// `is_mountpoint(2)`, so compare the parent's device id to the dir's;
/// if they differ the dir is a mount.
async fn is_mountpoint(path: &Path) -> bool {
let Some(parent) = path.parent() else {
return false;
};
let Ok(m1) = tokio::fs::metadata(path).await else {
return false;
};
let Ok(m2) = tokio::fs::metadata(parent).await else {
return false;
};
use std::os::unix::fs::MetadataExt;
m1.dev() != m2.dev()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn version_arg() {
assert_eq!(NfsVersion::V3.vers_arg(), "vers=3");
assert_eq!(NfsVersion::V41.vers_arg(), "vers=4.1");
}
#[test]
fn mount_options_v3_includes_nolock() {
let m = NfsMount {
id: "x".into(),
server: "s".into(),
export: "/e".into(),
version: NfsVersion::V3,
read_only: true,
local_path: PathBuf::from("/tmp/x"),
mounted: false,
last_error: None,
last_attempt: None,
iso_count: 0,
};
let opts = mount_options(&m);
assert!(opts.contains("vers=3"));
assert!(opts.contains("ro"));
assert!(opts.contains("nolock"));
assert!(opts.contains("soft"));
}
#[test]
fn mount_options_v41_no_nolock() {
let m = NfsMount {
id: "x".into(),
server: "s".into(),
export: "/e".into(),
version: NfsVersion::V41,
read_only: false,
local_path: PathBuf::from("/tmp/x"),
mounted: false,
last_error: None,
last_attempt: None,
iso_count: 0,
};
let opts = mount_options(&m);
assert!(opts.contains("vers=4.1"));
assert!(opts.contains("rw"));
assert!(!opts.contains("nolock"));
}
#[test]
fn mount_id_is_stable_and_safe() {
let a = mount_id("10.0.0.5", "/srv/isos");
let b = mount_id("10.0.0.5", "/srv/isos");
assert_eq!(a, b);
assert!(!a.contains('/'));
assert!(!a.contains('.'));
}
}