Discover applications on the cluster and the host for backups

Kubernetes workloads are grouped by their Helm instance label into
applications, each offering what is worth backing up: data volumes, a
PostgreSQL dump instead of the database's own volume, and optionally the
namespace manifests. Caches are listed but not preselected. Host
applications come from running systemd services that declare a state or
working directory. Selecting components creates one strategy each.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Dennis Nemec
2026-09-03 20:50:11 +02:00
parent 4efff51aa7
commit 51364fdd76
16 changed files with 1050 additions and 17 deletions

View File

@ -0,0 +1,463 @@
//! Applications that run on the server, and what of them is worth backing up.
//!
//! Kubernetes applications are grouped by the Helm/recommended label
//! `app.kubernetes.io/instance`; host applications come from running systemd services that
//! declare a state or working directory. Only applications that have something to back up
//! are reported.
use serde::{Deserialize, Serialize};
use crate::backup::BackupSource;
use crate::cluster::ClusterOverview;
use crate::host::HostService;
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum ApplicationKind {
Kubernetes,
Host,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum ComponentKind {
/// Persistent data: a volume or a directory.
Data,
/// A database that is dumped instead of copied.
Database,
/// The Kubernetes objects of the namespace.
Manifests,
}
/// One backup-worthy part of an application, ready to become a strategy.
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct BackupComponent {
pub id: String,
pub kind: ComponentKind,
pub label: String,
pub source: BackupSource,
/// Preselected in the UI. Caches and manifests are not.
pub recommended: bool,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct Application {
pub id: String,
pub name: String,
pub kind: ApplicationKind,
pub namespace: Option<String>,
pub detail: String,
pub components: Vec<BackupComponent>,
}
const INSTANCE_LABEL: &str = "app.kubernetes.io/instance";
const NAME_LABEL: &str = "app.kubernetes.io/name";
/// Workload roles whose data is a cache and does not need a backup.
const CACHES: [&str; 4] = ["valkey", "redis", "memcached", "keydb"];
/// Workload roles that are dumped with `pg_dumpall`.
const POSTGRES: [&str; 2] = ["postgresql", "postgres"];
fn title_case(s: &str) -> String {
let mut c = s.chars();
match c.next() {
Some(f) => f.to_uppercase().collect::<String>() + c.as_str(),
None => String::new(),
}
}
/// Applications running on the cluster, with the volumes and databases they own.
pub fn from_cluster(overview: &ClusterOverview) -> Vec<Application> {
let mut apps: Vec<Application> = Vec::new();
let mut groups: Vec<(String, String)> = Vec::new(); // (namespace, instance)
for w in &overview.workloads {
let instance = w
.labels
.get(INSTANCE_LABEL)
.cloned()
.unwrap_or_else(|| w.name.clone());
let key = (w.namespace.clone(), instance);
if !groups.contains(&key) {
groups.push(key);
}
}
for (namespace, instance) in groups {
let members: Vec<_> = overview
.workloads
.iter()
.filter(|w| {
w.namespace == namespace
&& w.labels
.get(INSTANCE_LABEL)
.map(|i| i == &instance)
.unwrap_or(w.name == instance)
})
.collect();
let mut components = Vec::new();
for w in &members {
let role = w
.labels
.get(NAME_LABEL)
.cloned()
.unwrap_or_else(|| w.name.clone());
let is_cache = CACHES.iter().any(|c| role.contains(c));
for claim in &w.claims {
let size = overview
.volume_claims
.iter()
.find(|p| p.namespace == namespace && &p.name == claim)
.map(|p| format!(" ({})", p.capacity))
.unwrap_or_default();
components.push(BackupComponent {
id: format!("data:{claim}"),
kind: ComponentKind::Data,
label: format!("{} volume {claim}{size}", title_case(&role)),
source: BackupSource::VolumeClaim {
namespace: namespace.clone(),
pvc: claim.clone(),
},
recommended: !is_cache,
});
}
if POSTGRES.iter().any(|p| role.contains(p)) {
let pod = format!("{}-0", w.name);
components.push(BackupComponent {
id: format!("db:{pod}"),
kind: ComponentKind::Database,
label: format!("PostgreSQL dump of {pod}"),
source: BackupSource::PostgresDump {
namespace: namespace.clone(),
pod,
},
recommended: true,
});
}
}
if components.is_empty() {
continue; // nothing to back up, e.g. a stateless controller
}
// a database is dumped, so its own volume would be a redundant second copy
if components.iter().any(|c| c.kind == ComponentKind::Database) {
components
.retain(|c| !(c.kind == ComponentKind::Data && is_database_volume(c, &members)));
}
// volumes first, then databases, then the manifests
components.sort_by_key(|c| match c.kind {
ComponentKind::Data => 0,
ComponentKind::Database => 1,
ComponentKind::Manifests => 2,
});
components.push(BackupComponent {
id: "manifests".into(),
kind: ComponentKind::Manifests,
label: format!("Kubernetes objects of namespace {namespace}"),
source: BackupSource::KubernetesManifests {
namespace: namespace.clone(),
},
recommended: false,
});
let roles: Vec<String> = members
.iter()
.map(|w| {
w.labels
.get(NAME_LABEL)
.cloned()
.unwrap_or_else(|| w.name.clone())
})
.collect();
apps.push(Application {
id: format!("k8s:{namespace}/{instance}"),
name: title_case(&instance),
kind: ApplicationKind::Kubernetes,
namespace: Some(namespace.clone()),
detail: format!("{} workload(s): {}", members.len(), roles.join(", ")),
components,
});
}
apps.sort_by(|a, b| a.name.cmp(&b.name));
apps
}
/// True when the volume belongs to a workload that is dumped as a database anyway.
fn is_database_volume(component: &BackupComponent, members: &[&crate::cluster::Workload]) -> bool {
let BackupSource::VolumeClaim { pvc, .. } = &component.source else {
return false;
};
members.iter().any(|w| {
let role = w
.labels
.get(NAME_LABEL)
.cloned()
.unwrap_or_else(|| w.name.clone());
POSTGRES.iter().any(|p| role.contains(p)) && w.claims.contains(pvc)
})
}
/// Units that are part of the operating system or the container runtime rather than an
/// application whose data a user would back up.
fn is_system_unit(unit: &str) -> bool {
[
"snap.",
"systemd-",
"user@",
"getty@",
"dbus",
"cron",
"ssh",
"qemu-",
"polkit",
"rsyslog",
"unattended",
]
.iter()
.any(|p| unit.starts_with(p))
}
/// Applications installed directly on the host: running services that declare a directory.
pub fn from_services(services: &[HostService]) -> Vec<Application> {
let mut apps: Vec<Application> = services
.iter()
.filter(|s| !is_system_unit(&s.unit))
.filter_map(|s| {
let path = s.data_dir()?;
let name = s.unit.trim_end_matches(".service").to_string();
Some(Application {
id: format!("host:{}", s.unit),
name: title_case(&name),
kind: ApplicationKind::Host,
namespace: None,
detail: s.description.clone(),
components: vec![BackupComponent {
id: format!("data:{path}"),
kind: ComponentKind::Data,
label: format!("Directory {path}"),
source: BackupSource::HostPath { path },
recommended: true,
}],
})
})
.collect();
apps.sort_by(|a, b| a.name.cmp(&b.name));
apps
}
#[cfg(test)]
mod tests {
use super::*;
use crate::cluster::{Container, NodeInfo, VolumeClaim, Workload, WorkloadKind};
use std::collections::BTreeMap;
fn labels(pairs: &[(&str, &str)]) -> BTreeMap<String, String> {
pairs
.iter()
.map(|(k, v)| (k.to_string(), v.to_string()))
.collect()
}
fn workload(ns: &str, name: &str, role: &str, instance: &str, claims: &[&str]) -> Workload {
Workload {
namespace: ns.into(),
kind: WorkloadKind::Deployment,
name: name.into(),
ready: 1,
desired: 1,
containers: vec![Container {
name: name.into(),
image: format!("{role}:1"),
}],
labels: labels(&[(INSTANCE_LABEL, instance), (NAME_LABEL, role)]),
claims: claims.iter().map(|c| c.to_string()).collect(),
}
}
fn overview(workloads: Vec<Workload>, claims: Vec<(&str, &str, &str)>) -> ClusterOverview {
ClusterOverview {
nodes: vec![NodeInfo {
name: "n".into(),
version: "v1".into(),
ready: true,
os_image: String::new(),
kernel: String::new(),
container_runtime: String::new(),
}],
namespaces: vec!["gitea".into()],
workloads,
volume_claims: claims
.into_iter()
.map(|(ns, name, cap)| VolumeClaim {
namespace: ns.into(),
name: name.into(),
capacity: cap.into(),
storage_class: "hostpath".into(),
status: "Bound".into(),
})
.collect(),
fetched_at: chrono::Utc::now(),
}
}
#[test]
fn groups_the_workloads_of_a_helm_release_into_one_application() {
let o = overview(
vec![
workload(
"gitea",
"gitea",
"gitea",
"gitea",
&["gitea-shared-storage"],
),
workload(
"gitea",
"gitea-postgresql",
"postgresql",
"gitea",
&["data-gitea-postgresql-0"],
),
workload(
"gitea",
"gitea-valkey-primary",
"valkey",
"gitea",
&["valkey-data-gitea-valkey-primary-0"],
),
],
vec![
("gitea", "gitea-shared-storage", "10Gi"),
("gitea", "data-gitea-postgresql-0", "10Gi"),
],
);
let apps = from_cluster(&o);
assert_eq!(apps.len(), 1);
let app = &apps[0];
assert_eq!(app.id, "k8s:gitea/gitea");
assert_eq!(app.name, "Gitea");
assert_eq!(app.namespace.as_deref(), Some("gitea"));
assert!(app.detail.contains("postgresql"), "{}", app.detail);
let ids: Vec<&str> = app.components.iter().map(|c| c.id.as_str()).collect();
assert_eq!(
ids,
vec![
"data:gitea-shared-storage",
"data:valkey-data-gitea-valkey-primary-0",
"db:gitea-postgresql-0",
"manifests"
],
"the database volume is dropped in favour of the dump"
);
let data = &app.components[0];
assert_eq!(data.kind, ComponentKind::Data);
assert!(data.label.contains("gitea-shared-storage") && data.label.contains("10Gi"));
assert!(data.recommended);
assert_eq!(
data.source,
BackupSource::VolumeClaim {
namespace: "gitea".into(),
pvc: "gitea-shared-storage".into()
}
);
let cache = &app.components[1];
assert!(!cache.recommended, "a cache volume is not preselected");
let db = &app.components[2];
assert_eq!(db.kind, ComponentKind::Database);
assert_eq!(
db.source,
BackupSource::PostgresDump {
namespace: "gitea".into(),
pod: "gitea-postgresql-0".into()
}
);
assert!(db.recommended);
let manifests = app.components.last().unwrap();
assert_eq!(manifests.kind, ComponentKind::Manifests);
assert!(!manifests.recommended);
}
#[test]
fn applications_without_anything_to_back_up_are_skipped() {
let o = overview(
vec![workload(
"ingress",
"controller",
"ingress-nginx",
"ingress",
&[],
)],
vec![],
);
assert!(from_cluster(&o).is_empty());
}
#[test]
fn workloads_without_helm_labels_stand_on_their_own() {
let mut w = workload("apps", "legacy", "legacy", "legacy", &["legacy-data"]);
w.labels.clear();
let apps = from_cluster(&overview(vec![w], vec![("apps", "legacy-data", "5Gi")]));
assert_eq!(apps.len(), 1);
assert_eq!(apps[0].id, "k8s:apps/legacy");
assert_eq!(apps[0].name, "Legacy");
}
#[test]
fn host_services_with_a_directory_become_applications() {
let services = vec![
HostService {
unit: "monitoring.service".into(),
description: "SoftVisor Infrastructure Monitoring".into(),
working_dir: Some("/opt/monitoring".into()),
state_dir: None,
},
HostService {
unit: "tailscaled.service".into(),
description: "Tailscale node agent".into(),
working_dir: None,
state_dir: Some("tailscale".into()),
},
// no directory to back up
HostService {
unit: "ssh.service".into(),
description: "OpenSSH".into(),
working_dir: None,
state_dir: None,
},
// the container runtime is not an application
HostService {
unit: "snap.microk8s.daemon-kubelite.service".into(),
description: "microk8s".into(),
working_dir: Some("/var/snap/microk8s/8702".into()),
state_dir: None,
},
];
let apps = from_services(&services);
assert_eq!(
apps.iter().map(|a| a.name.as_str()).collect::<Vec<_>>(),
vec!["Monitoring", "Tailscaled"]
);
assert_eq!(apps[0].id, "host:monitoring.service");
assert_eq!(apps[0].kind, ApplicationKind::Host);
assert_eq!(apps[0].detail, "SoftVisor Infrastructure Monitoring");
assert_eq!(
apps[0].components[0].source,
BackupSource::HostPath {
path: "/opt/monitoring".into()
}
);
assert_eq!(
apps[1].components[0].source,
BackupSource::HostPath {
path: "/var/lib/tailscale".into()
},
"a state directory is relative to /var/lib"
);
}
}

View File

@ -39,6 +39,12 @@ pub struct Workload {
pub ready: i32,
pub desired: i32,
pub containers: Vec<Container>,
/// Object labels; `app.kubernetes.io/instance` groups the workloads of one application.
#[serde(default)]
pub labels: std::collections::BTreeMap<String, String>,
/// Persistent volume claims this workload uses.
#[serde(default)]
pub claims: Vec<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]

View File

@ -54,6 +54,31 @@ impl Inventory {
}
}
/// A systemd service running on the host.
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct HostService {
pub unit: String,
pub description: String,
/// `WorkingDirectory=` of the unit, if it is an absolute path other than `/`.
pub working_dir: Option<String>,
/// `StateDirectory=` of the unit; systemd places it under `/var/lib`.
pub state_dir: Option<String>,
}
impl HostService {
/// The directory holding this service's data, if it declares one.
pub fn data_dir(&self) -> Option<String> {
if let Some(state) = self.state_dir.as_ref().filter(|s| !s.is_empty()) {
let first = state.split_whitespace().next()?;
return Some(format!("/var/lib/{first}"));
}
self.working_dir
.as_ref()
.filter(|w| w.starts_with('/') && w.as_str() != "/")
.cloned()
}
}
/// Debian package names: lowercase letters, digits, `+`, `-`, `.`; at least two characters.
pub fn validate_package_name(name: &str) -> Result<(), crate::DomainError> {
let ok = name.len() >= 2

View File

@ -1,5 +1,6 @@
//! Domain layer: entities, value objects, errors and the ports (traits) the application
//! layer depends on. No I/O here.
pub mod application;
pub mod auth;
pub mod backup;
pub mod cluster;

View File

@ -5,7 +5,7 @@ use uuid::Uuid;
use crate::auth::{AccessClaims, AuthEvent, RefreshToken};
use crate::backup::{BackupRecord, BackupSource, BackupStrategy, BackupTarget, RemoteFile};
use crate::cluster::{ClusterOverview, WorkloadRef};
use crate::host::{Inventory, OsInfo, Package};
use crate::host::{HostService, Inventory, OsInfo, Package};
use crate::image::ImageUpdate;
use crate::jobs::{JobKind, JobRun, JobStatus};
use crate::settings::SmtpSettings;
@ -92,6 +92,8 @@ pub trait JobRunRepository: Send + Sync {
pub trait HostInspector: Send + Sync {
async fn os_info(&self) -> Result<OsInfo, DomainError>;
async fn packages(&self) -> Result<Vec<Package>, DomainError>;
/// Running systemd services, for discovering applications installed on the host.
async fn services(&self) -> Result<Vec<HostService>, DomainError>;
}
#[async_trait]