Files
truenas-truecloud-patch/tests/test_truecloud_nested.py
T
flan 605231b39f
CI / shell (shellcheck + syntax) (push) Successful in 10s
CI / python 3.11 (push) Successful in 13s
CI / python 3.13 (push) Successful in 17s
CI / python 3.12 (push) Successful in 18s
TrueNAS compatibility / compat (push) Successful in 9s
feat: TrueNAS 26 support; enumerate datasets and snapshots from ZFS, not middleware
TrueNAS 26 deletes plugins/zfs_/ outright, taking the private zfs.dataset.query,
zfs.snapshot.query and zfs.snapshot.delete with it. All three were on the nested
module's critical path, so nested snapshots were BROKEN on 26.

Snapshot deletion now resolves its namespace at runtime: pool.snapshot on 25.10
and 26, zfs.snapshot on 24.10 and 25.04. No single namespace spans every supported
release. tools/compat.py checks the same list the runtime uses, so what CI verifies
and what runs cannot drift apart.

Enumeration does NOT move to pool.dataset.query / pool.snapshot.query, and that is
the point of this commit. Those methods exist, are documented, and are covered by
iX's deprecation policy — and they are not like-for-like replacements. They apply a
visibility policy that hides ix-apps/*, .system/* and .ix-virt/*: 84 of 270 datasets
on a real pool, including live application data. Staging from that view omits them
silently, and plan_staging never sees them, so they do not even reach the skipped
list. The snapshot query hides the same datasets' snapshots, so the sweep orphans one
per hidden dataset on every run.

So: read the truth from ZFS, make changes through middleware. zfs list cannot be
filtered by policy and behaves identically on every release. A failing zfs list raises
rather than returning an empty list — "no datasets" and "the command broke" must never
look the same.

No shipped release is affected: v0.6.1 and earlier use the private zfs.dataset.query,
which returns all 270 datasets. The bug existed only in this port.

Verified on a real TrueNAS 26.0.0-BETA.1 install: 274-snapshot recursive backup of a
292-dataset pool, zero orphaned snapshots, zero leaked mounts, and a byte-identical
restore of a four-level-deep child dataset that pool.dataset.query hides.
2026-07-13 22:43:20 +00:00

1181 lines
47 KiB
Python

"""Tests for nested-dataset snapshot staging.
Two rules are under test above all else:
1. A tree that cannot be staged completely must fail LOUDLY. A silently
incomplete backup is the exact failure that stock TrueNAS's "no further
nesting" guard exists to prevent.
2. Every snapshot we cause to exist must be cleaned up. ``zfs.snapshot.delete``
is non-recursive by default and stock calls it with no options, so a
recursive snapshot would otherwise orphan one snapshot per descendant dataset
on EVERY run.
"""
import os
import sys
import pytest
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "patch"))
from truecloud_nested import ( # noqa: E402
StagingError,
apply_plan,
cleanup_all,
cleanup_task,
current_mounts_under,
delete_snapshot_tree,
plan_staging,
sidecar_for,
snapshot_tree_names,
staging_root_for,
teardown,
verify_staged,
)
SNAP = "cloud_backup-5-20260712030000"
ROOT = "/run/truecloud-nested/cloud_backup-5"
def ds(name, mountpoint, mounted="yes"):
return {
"name": name,
"properties": {
"mountpoint": {"value": mountpoint},
"mounted": {"value": mounted},
},
}
# Mirrors the real layout: apps are datasets, several with their own children.
DATASETS = [
ds("Tap", "/mnt/Tap"),
ds("Tap/apps", "/mnt/Tap/apps"),
ds("Tap/apps/lidarr", "/mnt/Tap/apps/lidarr"),
ds("Tap/apps/lidarr/config", "/mnt/Tap/apps/lidarr/config"),
ds("Tap/apps/immich", "/mnt/Tap/apps/immich"),
ds("Tap/apps/immich/pgdata", "/mnt/Tap/apps/immich/pgdata"),
]
def yes(_path):
return "ok"
def plan(datasets=DATASETS, base_dataset="Tap", base_mp="/mnt/Tap",
path="/mnt/Tap", probe=yes):
return plan_staging(base_dataset, base_mp, path, SNAP, datasets, ROOT, probe=probe)
class TestPlanStaging:
def test_stages_every_descendant_dataset(self):
mounts, skipped = plan()
assert skipped == []
assert len(mounts) == 6 # root + 5 descendants
assert mounts[0] == (f"/mnt/Tap/.zfs/snapshot/{SNAP}", ROOT)
by_target = {t: s for s, t in mounts}
assert by_target[f"{ROOT}/apps"] == f"/mnt/Tap/apps/.zfs/snapshot/{SNAP}"
assert by_target[f"{ROOT}/apps/immich/pgdata"] == (
f"/mnt/Tap/apps/immich/pgdata/.zfs/snapshot/{SNAP}"
)
def test_parents_are_mounted_before_children(self):
# A child's mountpoint dir only exists inside its parent's snapshot, so
# mounting a child first would fail.
mounts, _ = plan()
seen = set()
for _src, target in mounts:
if target != ROOT:
assert os.path.dirname(target) in seen
seen.add(target)
def test_backup_path_below_dataset_root(self):
mounts, _ = plan(base_dataset="Tap/apps", base_mp="/mnt/Tap/apps",
path="/mnt/Tap/apps")
assert mounts[0] == (f"/mnt/Tap/apps/.zfs/snapshot/{SNAP}", ROOT)
targets = [t for _s, t in mounts]
assert f"{ROOT}/lidarr" in targets
assert f"{ROOT}/apps/lidarr" not in targets
def test_base_dataset_is_not_a_descendant_of_itself(self):
mounts, _ = plan(datasets=[ds("Tap", "/mnt/Tap")])
assert len(mounts) == 1
class TestScoping:
def test_unrelated_datasets_are_ignored_silently(self):
# Regression: scoping by mountpoint first dragged in every
# mountpoint-less dataset on the box (all of Tank/.system/*), burying the
# warnings that actually matter.
noisy = DATASETS + [
ds("Tank/.system", "none"),
ds("Tank/.system/cores", "legacy"),
ds("Tank/backups", "/mnt/Tank/backups"),
]
mounts, skipped = plan(datasets=noisy)
assert len(mounts) == 6
assert skipped == [], "datasets outside the base dataset must not be reported"
def test_in_scope_dataset_without_mountpoint_is_reported(self):
datasets = DATASETS + [ds("Tap/apps/weird", "none")]
_mounts, skipped = plan(datasets=datasets)
assert ("Tap/apps/weird", "mountpoint is none") in skipped
def test_unmounted_dataset_is_skipped_but_never_silently(self):
datasets = DATASETS + [ds("Tap/apps/vault", "/mnt/Tap/apps/vault", mounted="no")]
mounts, skipped = plan(datasets=datasets)
assert f"{ROOT}/apps/vault" not in [t for _s, t in mounts]
assert ("Tap/apps/vault", "dataset is not mounted (locked/encrypted?)") in skipped
def test_descendant_mounted_outside_the_path_is_not_an_omission(self):
datasets = DATASETS + [ds("Tap/elsewhere", "/mnt/other")]
mounts, skipped = plan(datasets=datasets)
assert len(mounts) == 6
assert skipped == []
class TestSilentOmissionGuard:
"""The whole point of the feature. These are the tests that matter."""
@staticmethod
def _missing_pgdata(path):
return "missing" if "/mnt/Tap/apps/immich/pgdata/" in path else "ok"
@staticmethod
def _denied_pgdata(path):
if "/mnt/Tap/apps/immich/pgdata/" in path:
return "cannot be read (Permission denied)"
return "ok"
def test_missing_snapshot_on_descendant_raises(self):
with pytest.raises(StagingError, match="incomplete tree"):
plan(probe=self._missing_pgdata)
def test_error_names_the_offending_dataset(self):
with pytest.raises(StagingError, match="Tap/apps/immich/pgdata"):
plan(probe=self._missing_pgdata)
def test_missing_and_unreadable_are_reported_differently(self):
# os.path.isdir() collapses both into False, which would report a
# permission problem as "has no snapshot" and send you hunting for a
# snapshot that is sitting right there. Both abort -- but say which.
with pytest.raises(StagingError, match="has no snapshot"):
plan(probe=self._missing_pgdata)
with pytest.raises(StagingError, match="Permission denied"):
plan(probe=self._denied_pgdata)
class TestSnapshotTreeNames:
"""zfs.snapshot.delete is non-recursive; we must sweep children ourselves."""
ALL = [
"Tap@cloud_backup-5-20260712030000",
"Tap/apps@cloud_backup-5-20260712030000",
"Tap/apps/lidarr/config@cloud_backup-5-20260712030000",
"Tap@auto-2026-07-12_03-00", # unrelated periodic snapshot
"Tap/apps@cloud_backup-9-20260712030000", # another task
"Tank/backups@cloud_backup-5-20260712030000", # different pool
]
def test_returns_parent_and_all_children(self):
got = snapshot_tree_names("Tap@cloud_backup-5-20260712030000", self.ALL)
assert set(got) == {
"Tap@cloud_backup-5-20260712030000",
"Tap/apps@cloud_backup-5-20260712030000",
"Tap/apps/lidarr/config@cloud_backup-5-20260712030000",
}
def test_never_touches_periodic_or_other_tasks_or_other_pools(self):
got = snapshot_tree_names("Tap@cloud_backup-5-20260712030000", self.ALL)
assert "Tap@auto-2026-07-12_03-00" not in got
assert "Tap/apps@cloud_backup-9-20260712030000" not in got
assert "Tank/backups@cloud_backup-5-20260712030000" not in got
def test_malformed_snapshot_name_yields_nothing(self):
assert snapshot_tree_names("Tap", self.ALL) == []
class FakeMiddleware:
"""middlewared as this module actually uses it: `call_sync`, from a thread.
The module is synchronous on purpose -- see the orchestration note in
truecloud_nested.py. TrueNAS <= 25.10 reaches it through
`await middleware.run_in_thread(...)` and TrueNAS 26 calls it directly, but the
logic below the boundary is the same code either way, so it is tested once.
`snapshot_ns` picks which middleware GENERATION this is, because they do not
agree on what the snapshot service is called:
pool.snapshot 25.10 and 26 (26 has ONLY this -- plugins/zfs_/ is gone)
zfs.snapshot 24.10 and 25.04 (pool.snapshot does not exist yet)
A method in the namespace this box does NOT have raises, exactly as middleware
does ("Method does not exist"). That is what makes the runtime picker testable:
a module that guessed wrong would blow up here instead of silently orphaning
snapshots on somebody's NAS.
"""
def __init__(self, snapshots=None, snapshot_ns="pool.snapshot"):
self.snapshots = list(snapshots or [])
self.calls = []
self.logger = None
self.snapshot_ns = snapshot_ns
def get_service(self, name):
if name != self.snapshot_ns:
raise KeyError(name) # middleware raises KeyError for an unknown ns
return object()
def list_snapshots(self, dataset):
"""Stands in for `zfs list -t snapshot -r <dataset>`.
Enumeration comes from ZFS now, NOT from middleware -- middleware's query
hides internal datasets, and a sweep that cannot see a snapshot can never
collect it. Passed in as `list_snapshots=` so the seam is explicit.
"""
return [
n for n in self.snapshots
if n.split("@")[0] == dataset or n.startswith(dataset + "/")
]
def call_sync(self, method, *args):
self.calls.append((method, args))
namespace, _, op = method.rpartition(".")
if namespace != self.snapshot_ns:
raise RuntimeError("Method does not exist")
if op == "query":
return [{"name": n} for n in self.snapshots]
if op == "delete":
name = args[0]
opts = args[1] if len(args) > 1 else {}
if name not in self.snapshots:
raise RuntimeError("does not exist")
if opts.get("recursive"):
# Real `zfs destroy -r` takes the parent and every child snapshot.
for n in snapshot_tree_names(name, list(self.snapshots)):
self.snapshots.remove(n)
else:
self.snapshots.remove(name)
return True
raise AssertionError(f"unexpected call {method}")
def stub_core(monkeypatch, tn, *, plan=None, order=None, plan_raises=None):
"""Replace the blocking core (plan/apply/verify/teardown) with recorders.
stage_nested calls these directly now, so they are patched by NAME rather than
intercepted at a `run_in_thread` boundary that no longer exists.
"""
def record(name, result):
def fn(*args, **kwargs):
if order is not None:
order.append(name)
if name == "plan_staging" and plan_raises is not None:
raise plan_raises
return result() if callable(result) else result
fn.__name__ = name
return fn
real_write = tn._write_sidecar
def write_sidecar(*args, **kwargs):
if order is not None:
order.append("_write_sidecar")
return real_write(*args, **kwargs)
monkeypatch.setattr(tn, "_write_sidecar", write_sidecar)
monkeypatch.setattr(tn, "plan_staging", record("plan_staging", plan or ([], [])))
monkeypatch.setattr(tn, "apply_plan", record("apply_plan", True))
monkeypatch.setattr(tn, "verify_staged", record("verify_staged", True))
monkeypatch.setattr(tn, "teardown", record("teardown", []))
class TestDeleteSnapshotTree:
def test_deletes_parent_and_every_child(self):
mw = FakeMiddleware([
"Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap", "Tap@keepme",
])
delete_snapshot_tree(mw, "Tap@snap", list_snapshots=mw.list_snapshots)
assert mw.snapshots == ["Tap@keepme"]
def test_is_idempotent_when_stock_already_removed_the_parent(self):
# Stock's finally can win the race once our mounts are released.
mw = FakeMiddleware(["Tap/apps@snap", "Tap/apps/lidarr@snap"])
delete_snapshot_tree(mw, "Tap@snap", list_snapshots=mw.list_snapshots)
assert mw.snapshots == []
def test_uses_a_single_recursive_delete_not_252_individual_ones(self):
# 252 sequential deletes are slow AND not atomic: a run killed part-way
# through leaves exactly the orphans this function exists to prevent.
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap", "Tap/apps/lidarr@snap"])
delete_snapshot_tree(mw, "Tap@snap", list_snapshots=mw.list_snapshots)
assert mw.snapshots == []
deletes = [a for m, a in mw.calls if m.endswith(".delete")]
assert len(deletes) == 1, "should be ONE recursive call, not one per snapshot"
assert deletes[0][1] == {"recursive": True}
assert not [m for m, _a in mw.calls if m.endswith(".query")], (
"no enumeration needed on the fast path"
)
def test_survives_recursive_and_query_failure_by_deleting_the_parent(self):
class Broken(FakeMiddleware):
def call_sync(self, method, *args):
if method.endswith(".query"):
raise RuntimeError("boom")
if method.endswith(".delete") and len(args) > 1:
raise RuntimeError("recursive delete unavailable")
return super().call_sync(method, *args)
mw = Broken(["Tap@snap"])
delete_snapshot_tree(mw, "Tap@snap", list_snapshots=mw.list_snapshots)
assert mw.snapshots == []
def test_leaves_unrelated_snapshots_alone_when_the_tree_is_gone(self):
mw = FakeMiddleware(["Tap@unrelated"])
delete_snapshot_tree(mw, "Tap@snap", list_snapshots=mw.list_snapshots)
assert mw.snapshots == ["Tap@unrelated"]
class TestStageNestedOrdering:
def test_sidecar_is_written_before_anything_is_mounted(self, tmp_path, monkeypatch):
# middlewared can die at any moment. If the snapshot were recorded only
# after apply_plan, a crash in that window would orphan a 160-snapshot
# tree -- the precise failure the sidecar exists to prevent.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
order = []
stub_core(monkeypatch, tn, order=order,
plan=([("/src", str(tmp_path / "cloud_backup-5"))], []))
tn.stage_nested(
FakeMiddleware(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
"cloud_backup-5", DATASETS,
)
assert order.index("_write_sidecar") < order.index("apply_plan")
def test_reclaims_the_snapshot_tree_left_by_a_crashed_run(self, tmp_path,
monkeypatch):
# teardown() reclaims the crashed run's MOUNTS, but nothing else would
# ever reclaim its SNAPSHOTS -- and we are about to overwrite the only
# record of them. One crash would orphan 160+ snapshots permanently.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(os.path.dirname(root), exist_ok=True)
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
fh.write("Tap@old-crashed-run")
mw = FakeMiddleware(["Tap@old-crashed-run", "Tap/apps@old-crashed-run"])
stub_core(monkeypatch, tn, plan=([("/src", root)], []))
tn.stage_nested(
mw, "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
"cloud_backup-5", DATASETS,
)
assert mw.snapshots == [], "the crashed run's snapshot tree must be reclaimed"
def test_sidecar_is_KEPT_when_staging_fails(self, tmp_path, monkeypatch):
# The caller sweeps the snapshot tree on the way out, and anything still busy
# SURVIVES that sweep -- with the sidecar as its only record. Removing the
# sidecar here would orphan those snapshots permanently.
#
# The asymmetry is the point: a sidecar left behind when the tree is already
# gone costs one no-op delete on the next run; a sidecar removed while the tree
# still exists is unrecoverable.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
stub_core(monkeypatch, tn, plan_raises=StagingError("boom"))
with pytest.raises(StagingError):
tn.stage_nested(
FakeMiddleware(), "/mnt/Tap", "Tap@snap", "Tap", "/mnt/Tap",
"cloud_backup-5", DATASETS,
)
assert os.path.exists(sidecar_for(root)), (
"sidecar removed on staging failure — any snapshot the caller's sweep "
"cannot delete is now orphaned forever"
)
with open(sidecar_for(root), encoding="utf-8") as fh:
assert fh.read().strip() == "Tap@snap"
class TestCleanupTask:
def test_recovers_snapshot_from_sidecar_after_middlewared_restart(self, tmp_path,
monkeypatch):
# The sidecar is the ONLY record of the pinned snapshot, precisely so a
# middlewared restart cannot orphan the tree.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5", base=str(tmp_path))
os.makedirs(root, exist_ok=True)
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
fh.write("Tap@snap")
mw = FakeMiddleware(["Tap@snap", "Tap/apps@snap"])
monkeypatch.setattr(tn, "teardown", lambda *_a, **_k: [])
cleanup_task(mw, "cloud_backup-5")
assert mw.snapshots == []
assert not os.path.exists(sidecar_for(root))
def test_is_a_noop_when_never_staged(self, tmp_path, monkeypatch):
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path / "nope"))
mw = FakeMiddleware(["Tap@snap"])
cleanup_task(mw, "cloud_backup-5")
assert mw.calls == []
assert mw.snapshots == ["Tap@snap"]
class TestVerifyStaged:
"""Anti-regression guard: proves the staged tree is real before we back it up."""
def test_passes_when_every_target_is_mounted_and_root_non_empty(self):
mounts = [("/src", ROOT), ("/src/a", f"{ROOT}/a")]
assert verify_staged(mounts, ismount=lambda p: True, listdir=lambda p: ["apps"])
def test_raises_when_a_target_is_not_actually_mounted(self):
# This is the case that would produce a silently-empty backup.
mounts = [("/src", ROOT), ("/src/a", f"{ROOT}/a")]
with pytest.raises(StagingError, match="not a mountpoint"):
verify_staged(mounts, ismount=lambda p: p == ROOT, listdir=lambda p: ["apps"])
def test_raises_when_staging_root_is_empty(self):
with pytest.raises(StagingError, match="empty"):
verify_staged([("/src", ROOT)], ismount=lambda p: True, listdir=lambda p: [])
def test_raises_on_empty_plan(self):
with pytest.raises(StagingError):
verify_staged([])
class FakeRunner:
def __init__(self, fail_on=None):
self.fail_on = fail_on
self.calls = []
def __call__(self, cmd):
self.calls.append(cmd)
class R:
returncode = 0
stderr = ""
if self.fail_on and cmd[0] == "mount" and cmd[2] == self.fail_on:
R.returncode = 32
R.stderr = "mount failed"
return R
class TestApplyPlanRollback:
def test_rolls_back_mounts_when_one_fails(self, tmp_path):
# A half-built tree must never reach the backup tool.
root = str(tmp_path / "root")
mounts = [("/src", root), ("/src/a", root + "/a"), ("/src/b", root + "/b")]
runner = FakeRunner(fail_on="/src/b")
with pytest.raises(StagingError, match="bind-mount"):
apply_plan(mounts, runner=runner, isdir=lambda _p: True)
umounts = [c[-1] for c in runner.calls if c[0] == "umount"]
assert umounts == [root + "/a", root]
def test_raises_when_target_missing(self, tmp_path):
root = str(tmp_path / "root")
with pytest.raises(StagingError, match="does not exist"):
apply_plan(
[("/src", root), ("/src/a", root + "/a")],
runner=FakeRunner(),
isdir=lambda p: p == root,
)
class TestTeardown:
def test_unmounts_deepest_first(self, tmp_path):
mounts_file = tmp_path / "mounts"
mounts_file.write_text(
f"tmpfs {ROOT} tmpfs rw 0 0\n"
f"tmpfs {ROOT}/apps tmpfs rw 0 0\n"
f"tmpfs {ROOT}/apps/lidarr/config tmpfs rw 0 0\n"
f"tmpfs {ROOT}/apps/lidarr tmpfs rw 0 0\n"
"tmpfs /somewhere/else tmpfs rw 0 0\n"
)
runner = FakeRunner()
teardown(ROOT, runner=runner, mounts_file=str(mounts_file))
order = [c[-1] for c in runner.calls if c[0] == "umount"]
assert order == [
f"{ROOT}/apps/lidarr/config",
f"{ROOT}/apps/lidarr",
f"{ROOT}/apps",
ROOT,
]
assert "/somewhere/else" not in order
def test_is_idempotent_when_nothing_mounted(self, tmp_path):
mounts_file = tmp_path / "mounts"
mounts_file.write_text("tmpfs /somewhere/else tmpfs rw 0 0\n")
runner = FakeRunner()
assert teardown(ROOT, runner=runner, mounts_file=str(mounts_file)) == []
assert runner.calls == []
def test_falls_back_to_lazy_umount(self, tmp_path):
mounts_file = tmp_path / "mounts"
mounts_file.write_text(f"tmpfs {ROOT} tmpfs rw 0 0\n")
class Busy(FakeRunner):
def __call__(self, cmd):
self.calls.append(cmd)
class R:
returncode = 0 if "-l" in cmd else 32
stderr = "target is busy"
return R
runner = Busy()
assert teardown(ROOT, runner=runner, mounts_file=str(mounts_file)) == []
assert ["umount", "-l", ROOT] in runner.calls
class TestCleanupAll:
"""uninstall.sh and recover.sh call this instead of reimplementing teardown."""
def test_reports_orphan_snapshots_before_deleting_their_sidecars(self, tmp_path):
# The sidecar is the only record that an interrupted run's snapshot tree
# is still on disk. Deleting it without naming the snapshot orphans the
# whole tree silently.
base = tmp_path / "stage"
base.mkdir()
(base / "cloud_backup-5.snapshot").write_text("Tap@interrupted")
mounts_file = tmp_path / "mounts"
mounts_file.write_text("")
lines, errors = cleanup_all(
base=str(base), runner=FakeRunner(), mounts_file=str(mounts_file)
)
assert errors == []
assert any("Tap@interrupted" in ln for ln in lines)
assert any("zfs destroy -r" in ln for ln in lines)
# Sidecar cleared only after being reported.
assert not (base / "cloud_backup-5.snapshot").exists()
def test_unmounts_everything_under_the_base_deepest_first(self, tmp_path):
base = tmp_path / "stage"
base.mkdir()
mounts_file = tmp_path / "mounts"
mounts_file.write_text(
f"tmpfs {base} tmpfs rw 0 0\n"
f"tmpfs {base}/cloud_backup-5 tmpfs rw 0 0\n"
f"tmpfs {base}/cloud_backup-5/apps tmpfs rw 0 0\n"
)
runner = FakeRunner()
_lines, errors = cleanup_all(
base=str(base), runner=runner, mounts_file=str(mounts_file)
)
assert errors == []
order = [c[-1] for c in runner.calls if c[0] == "umount"]
assert order == [
f"{base}/cloud_backup-5/apps",
f"{base}/cloud_backup-5",
str(base),
]
def test_keeps_sidecars_when_an_unmount_failed(self, tmp_path):
# If a mount is stuck, the snapshot is still pinned — so the record of it
# must survive for the next run (or the operator) to act on.
base = tmp_path / "stage"
base.mkdir()
(base / "cloud_backup-5.snapshot").write_text("Tap@stuck")
mounts_file = tmp_path / "mounts"
mounts_file.write_text(f"tmpfs {base}/cloud_backup-5 tmpfs rw 0 0\n")
class Stuck(FakeRunner):
def __call__(self, cmd):
self.calls.append(cmd)
class R:
returncode = 32
stderr = "target is busy"
return R
_lines, errors = cleanup_all(
base=str(base), runner=Stuck(), mounts_file=str(mounts_file)
)
assert errors, "a stuck unmount must be reported"
assert (base / "cloud_backup-5.snapshot").exists()
def test_is_a_noop_on_a_clean_system(self, tmp_path):
mounts_file = tmp_path / "mounts"
mounts_file.write_text("")
lines, errors = cleanup_all(
base=str(tmp_path / "absent"), runner=FakeRunner(),
mounts_file=str(mounts_file),
)
assert errors == []
assert lines == [" None active."]
class TestCurrentMountsUnder:
def test_matches_only_the_staging_subtree(self, tmp_path):
mounts_file = tmp_path / "mounts"
# "cloud_backup-50" must NOT match "cloud_backup-5".
mounts_file.write_text(
f"tmpfs {ROOT} tmpfs rw 0 0\n"
"tmpfs /run/truecloud-nested/cloud_backup-50 tmpfs rw 0 0\n"
)
assert current_mounts_under(ROOT, mounts_file=str(mounts_file)) == [ROOT]
class TestStagingRootFor:
def test_stable_per_task(self):
assert staging_root_for("cloud_backup-5") == "/run/truecloud-nested/cloud_backup-5"
def test_sanitises_path_separators(self):
assert "/" not in staging_root_for("evil/name").rsplit("/", 1)[-1]
@pytest.mark.parametrize("name", ["..", ".", "...", "/", ""])
def test_dot_components_cannot_escape_the_staging_base(self, name):
# os.path.join(BASE, "..") normalises to /run — teardown would rmdir it.
root = staging_root_for(name)
assert os.path.normpath(root).startswith("/run/truecloud-nested/")
class TestZfsAutomountKeepsSnapshotsBusy:
""""dataset is busy" is EXPECTED, TRANSIENT, and used to orphan snapshots forever.
Reading `<dataset>/.zfs/snapshot/<snap>/` makes ZFS **automount** that snapshot,
and it stays mounted for zfs_expire_snapshot seconds (300 by default) after the
last access. teardown() unmounts OUR bind mounts, but not the automount underneath
-- so `zfs destroy` refuses with EBUSY for everything restic read recently.
Observed on a real 256-snapshot tree: 253 swept cleanly, and the 3 datasets restic
had touched last failed with "dataset is busy". cleanup_task then removed the
sidecar anyway, so nothing would ever reclaim them. A few snapshots leaked per run,
forever.
"""
MOUNTS = (
"tmpfs /run tmpfs rw 0 0\n"
"Tap/apps/prometheus /mnt/Tap/apps/prometheus/.zfs/snapshot/snap1 zfs ro 0 0\n"
"Tap/apps/standing/data /mnt/Tap/apps/standing/data/.zfs/snapshot/snap1 zfs ro 0 0\n"
"Tap /mnt/Tap/.zfs/snapshot/snap1 zfs ro 0 0\n"
"Tap/other /mnt/Tap/other/.zfs/snapshot/OTHER zfs ro 0 0\n"
)
def _mounts_file(self, tmp_path):
p = tmp_path / "mounts"
p.write_text(self.MOUNTS)
return str(p)
def test_it_finds_the_automounts_for_this_snapshot_only(self, tmp_path):
import truecloud_nested as tn
found = tn.snapdir_automounts("snap1", mounts_file=self._mounts_file(tmp_path))
assert "/mnt/Tap/other/.zfs/snapshot/OTHER" not in found
assert len(found) == 3
def test_deepest_first(self, tmp_path):
# A child's automount must be released before its parent's.
import truecloud_nested as tn
found = tn.snapdir_automounts("snap1", mounts_file=self._mounts_file(tmp_path))
assert found[-1] == "/mnt/Tap/.zfs/snapshot/snap1"
def test_release_snapdirs_unmounts_them(self, tmp_path):
import truecloud_nested as tn
called = []
class R:
returncode = 0
stderr = ""
def runner(cmd):
called.append(cmd)
return R()
errs = tn.release_snapdirs("snap1", runner=runner,
mounts_file=self._mounts_file(tmp_path))
assert errs == []
assert all(c[0] == "umount" for c in called)
assert len(called) == 3
class BusyMiddleware(FakeMiddleware):
"""Deletes fail with EBUSY until `busy_until_attempt` passes -- like a ZFS
automount expiring."""
def __init__(self, snapshots, busy, busy_for=2):
super().__init__(snapshots)
self.busy = set(busy)
self.busy_for = busy_for
self.attempts = 0
def call_sync(self, method, *args):
if method.endswith(".delete"):
name = args[0]
opts = args[1] if len(args) > 1 else {}
if opts.get("recursive"):
raise RuntimeError("cannot destroy snapshot: dataset is busy")
if name in self.busy:
self.attempts += 1
if self.attempts <= self.busy_for * len(self.busy):
raise RuntimeError(f"cannot destroy '{name}': dataset is busy")
return super().call_sync(method, *args)
class TestDeleteRetriesAndReportsSurvivors:
def test_a_transient_busy_is_retried_and_wins(self, monkeypatch):
import truecloud_nested as tn
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
mw = BusyMiddleware(
["Tap@snap", "Tap/apps@snap", "Tap/apps/prometheus@snap"],
busy=["Tap/apps/prometheus@snap"], busy_for=1,
)
survivors = tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None, list_snapshots=mw.list_snapshots)
assert survivors == []
assert mw.snapshots == []
def test_a_permanently_busy_snapshot_is_REPORTED_not_swallowed(self, monkeypatch):
import truecloud_nested as tn
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
mw = BusyMiddleware(
["Tap@snap", "Tap/apps/prometheus@snap"],
busy=["Tap/apps/prometheus@snap"], busy_for=99,
)
survivors = tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None, list_snapshots=mw.list_snapshots)
assert survivors == ["Tap/apps/prometheus@snap"]
assert mw.snapshots == ["Tap/apps/prometheus@snap"]
def test_the_automounts_are_released_before_deleting(self, monkeypatch):
import truecloud_nested as tn
order = []
monkeypatch.setattr(tn, "release_snapdirs",
lambda name, **k: order.append(("release", name)) or [])
mw = FakeMiddleware(["Tap@snap"])
real = mw.call_sync
def spy(method, *args):
order.append((method, args[0] if args else None))
return real(method, *args)
mw.call_sync = spy
tn.delete_snapshot_tree(mw, "Tap@snap", sleep=lambda _s: None, list_snapshots=mw.list_snapshots)
assert order[0] == ("release", "snap"), order
class TestSidecarSurvivesAnIncompleteSweep:
def test_the_sidecar_is_KEPT_when_snapshots_could_not_be_deleted(
self, tmp_path, monkeypatch
):
# It is the ONLY record those snapshots exist. Removing it orphans them
# permanently -- which is exactly what happened on the real box.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
monkeypatch.setattr(tn, "release_snapdirs", lambda *a, **k: [])
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(root, exist_ok=True)
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
fh.write("Tap@snap")
mw = BusyMiddleware(["Tap@snap", "Tap/apps/prometheus@snap"],
busy=["Tap/apps/prometheus@snap"], busy_for=99)
monkeypatch.setattr(tn, "delete_snapshot_tree",
lambda m, s, logger=None: ["Tap/apps/prometheus@snap"])
tn.cleanup_task(mw, "cloud_backup-5")
assert os.path.exists(sidecar_for(root)), (
"sidecar removed despite survivors — they are now orphaned forever"
)
def test_the_sidecar_is_removed_on_a_clean_sweep(self, tmp_path, monkeypatch):
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(root, exist_ok=True)
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
fh.write("Tap@snap")
monkeypatch.setattr(tn, "delete_snapshot_tree", lambda m, s, logger=None: [])
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
assert not os.path.exists(sidecar_for(root))
class TestTheSidecarCarriesEveryPendingTree:
"""The sidecar holds a LIST, and that is a bug fix, not a generalisation.
It used to hold ONE snapshot. So a run that reclaimed an older tree, FAILED to
finish reclaiming it, and then recorded its own snapshot would **overwrite the only
record of the survivor** — orphaning it permanently, via the exact code written to
prevent orphans.
Observed live: a snapshot survived one run; the next run's reclaim also failed
(ZFS's 300s automount window had not elapsed, because the two runs were minutes
apart); the record was overwritten; the snapshot was orphaned for good.
"""
def test_round_trips_a_list(self, tmp_path):
import truecloud_nested as tn
root = str(tmp_path / "cloud_backup-5")
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
assert tn._read_sidecar(root) == ["Tap@a", "Tap@b"]
def test_reads_the_old_single_line_format(self, tmp_path):
# Boxes upgrading from an older version have a one-line sidecar on disk.
import truecloud_nested as tn
root = str(tmp_path / "cloud_backup-5")
os.makedirs(os.path.dirname(sidecar_for(root)), exist_ok=True)
with open(sidecar_for(root), "w", encoding="utf-8") as fh:
fh.write("Tap@legacy")
assert tn._read_sidecar(root) == ["Tap@legacy"]
def test_a_failed_reclaim_is_carried_forward_not_overwritten(
self, tmp_path, monkeypatch
):
# THE bug. stage_nested reclaims an old tree, cannot finish, then records its
# own snapshot -- the survivor must still be in the sidecar afterwards.
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(os.path.dirname(root), exist_ok=True)
tn._write_sidecar(root, ["Tap@old"])
# The reclaim of Tap@old leaves one snapshot behind (still busy).
monkeypatch.setattr(
tn, "delete_snapshot_tree",
lambda m, s, logger=None: ["Tap/apps/x@old"] if s == "Tap@old" else [],
)
stub_core(monkeypatch, tn, plan=([("/src", root)], []))
tn.stage_nested(FakeMiddleware(), "/mnt/Tap", "Tap@new", "Tap", "/mnt/Tap",
"cloud_backup-5", DATASETS)
recorded = tn._read_sidecar(root)
assert "Tap/apps/x@old" in recorded, (
"the failed reclaim's survivor was dropped — orphaned forever"
)
assert "Tap@new" in recorded, "our own snapshot must also be recorded"
def test_cleanup_sweeps_every_pending_tree_and_records_only_survivors(
self, tmp_path, monkeypatch
):
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(root, exist_ok=True)
tn._write_sidecar(root, ["Tap@old", "Tap@new"])
swept = []
def fake_delete(m, s, logger=None):
swept.append(s)
return ["Tap/apps/x@new"] if s == "Tap@new" else []
monkeypatch.setattr(tn, "delete_snapshot_tree", fake_delete)
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
assert swept == ["Tap@old", "Tap@new"], "both pending trees must be swept"
# Only the SURVIVOR is written back -- re-recording Tap@old would make every
# future run re-sweep a tree that is already gone.
assert tn._read_sidecar(root) == ["Tap/apps/x@new"]
def test_a_fully_clean_sweep_removes_the_sidecar(self, tmp_path, monkeypatch):
import truecloud_nested as tn
monkeypatch.setattr(tn, "STAGING_BASE", str(tmp_path))
root = tn.staging_root_for("cloud_backup-5")
os.makedirs(root, exist_ok=True)
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
monkeypatch.setattr(tn, "delete_snapshot_tree", lambda m, s, logger=None: [])
tn.cleanup_task(FakeMiddleware(), "cloud_backup-5")
assert not os.path.exists(sidecar_for(root))
def test_cleanup_all_reports_each_pending_snapshot_on_its_own_line(self, tmp_path):
# It formats them for a human during uninstall. A list rendered into an
# f-string would print "['Tap@a', 'Tap@b']" at them.
import truecloud_nested as tn
root = str(tmp_path / "cloud_backup-5")
tn._write_sidecar(root, ["Tap@a", "Tap@b"])
lines, _errors = tn.cleanup_all(
base=str(tmp_path),
glob_fn=lambda _p: [sidecar_for(root)],
mounts_file=os.devnull,
)
notes = [ln for ln in lines if "left snapshot" in ln]
assert len(notes) == 2
assert "'Tap@a'" in notes[0] and "'Tap@b'" in notes[1]
assert "[" not in "".join(notes)
class TestGarbageCollectorSelection:
"""`stale_snapshot_names` DELETES DATA on a name match.
A name match is a weaker claim than a recorded fact, so every way it could be wrong
is a test. It exists because the sidecar — which IS a recorded fact — lives in /run,
which is tmpfs: a reboot mid-backup destroys it and orphans a 250-snapshot tree with
nothing left pointing at it. This is the only thing that would ever find those.
"""
import datetime as _dt
NOW = _dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=_dt.UTC)
CURRENT = "Tap@cloud_backup-5-20260714115900" # 1 minute ago
OLD = "Tap/apps/x@cloud_backup-5-20260713030000" # ~33 hours ago
def collect(self, names, **kw):
import truecloud_nested as tn
return tn.stale_snapshot_names(
"cloud_backup-5", self.CURRENT, names, self.NOW, **kw
)
def test_it_collects_our_own_leftovers(self):
assert self.collect([self.OLD]) == [self.OLD]
def test_it_NEVER_touches_the_current_run(self):
# Both the parent and its children share the current snapname.
names = [self.CURRENT, "Tap/apps/x@cloud_backup-5-20260714115900"]
assert self.collect(names) == []
def test_it_NEVER_touches_a_periodic_snapshot(self):
assert self.collect(["Tap/apps/x@auto-2026-07-13_03-00"]) == []
def test_it_NEVER_touches_a_human_made_snapshot(self):
assert self.collect(["Tap@before-i-broke-everything"]) == []
def test_it_NEVER_touches_another_TASK(self):
# cloud_backup-5 must not match cloud_backup-50. This is why the prefix
# carries the trailing dash.
assert self.collect(["Tap/apps/x@cloud_backup-50-20260713030000"]) == []
assert self.collect(["Tap/apps/x@cloud_backup-7-20260713030000"]) == []
def test_it_NEVER_touches_a_one_time_backup(self):
assert self.collect(["Tap@cloud_backup-onetime-20260713030000"]) == []
def test_it_NEVER_touches_a_snapshot_that_is_MOUNTED(self):
# An in-flight run pins its own snapshots. This — not the age heuristic — is
# what actually protects a concurrent backup.
assert self.collect([self.OLD], in_use={self.OLD}) == []
def test_it_NEVER_touches_a_snapshot_younger_than_the_minimum_age(self):
# Covers the seconds-long window between `zfs snapshot -r` and the mounts
# appearing, when a live run's snapshots look exactly like garbage.
young = "Tap/apps/x@cloud_backup-5-20260714113000" # 30 minutes ago
assert self.collect([young]) == []
assert self.collect([young], min_age=60) == [young]
def test_a_name_it_cannot_parse_is_left_alone(self):
assert self.collect(["Tap@cloud_backup-5-not-a-timestamp"]) == []
assert self.collect(["Tap@cloud_backup-5-"]) == []
def test_a_realistic_mixed_pool(self):
names = [
self.CURRENT, # ours, running
"Tap/apps/x@cloud_backup-5-20260714115900", # ours, running (child)
self.OLD, # ours, orphaned <-
"Tap/apps/y@cloud_backup-5-20260712030000", # ours, orphaned <-
"Tap/apps/x@auto-2026-07-13_03-00", # periodic
"Tap/apps/x@cloud_backup-7-20260713030000", # another task
"Tap@manual-keepme", # human
]
assert sorted(self.collect(names)) == sorted(
[self.OLD, "Tap/apps/y@cloud_backup-5-20260712030000"]
)
class TestMountedSnapshots:
def test_it_reads_snapshot_names_out_of_the_mount_table(self, tmp_path):
import truecloud_nested as tn
mounts = tmp_path / "mounts"
mounts.write_text(
"tmpfs /run tmpfs rw 0 0\n"
"Tap/apps/x@snap1 /run/truecloud-nested/t/apps/x zfs ro 0 0\n"
"Tap/apps/y@snap1 /mnt/Tap/apps/y/.zfs/snapshot/snap1 zfs ro 0 0\n"
"Tap/live /mnt/Tap/live zfs rw 0 0\n"
)
live = tn.mounted_snapshots(str(mounts))
assert live == {"Tap/apps/x@snap1", "Tap/apps/y@snap1"}
assert "Tap/live" not in live # a live dataset is not a snapshot
class TestGarbageCollectorExecution:
def test_it_deletes_the_stale_ones_and_nothing_else(self, monkeypatch, tmp_path):
import datetime as dt
import truecloud_nested as tn
mounts = tmp_path / "mounts"
mounts.write_text("")
now = dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC)
mw = FakeMiddleware([
"Tap@cloud_backup-5-20260714115900", # current run
"Tap/apps/x@cloud_backup-5-20260713030000", # orphan <-
"Tap/apps/x@auto-2026-07-13_03-00", # periodic
"Tap/apps/x@cloud_backup-7-20260713030000", # other task
])
remaining = tn.gc_stale_snapshots(
mw, "cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
now=now, mounts_file=str(mounts),
list_snapshots=mw.list_snapshots,
)
assert remaining == []
assert mw.snapshots == [
"Tap@cloud_backup-5-20260714115900",
"Tap/apps/x@auto-2026-07-13_03-00",
"Tap/apps/x@cloud_backup-7-20260713030000",
]
def test_a_busy_orphan_is_reported_not_swallowed(self, monkeypatch, tmp_path):
import datetime as dt
import truecloud_nested as tn
mounts = tmp_path / "mounts"
mounts.write_text("")
now = dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC)
orphan = "Tap/apps/x@cloud_backup-5-20260713030000"
mw = BusyMiddleware(
["Tap@cloud_backup-5-20260714115900", orphan],
busy=[orphan], busy_for=99,
)
remaining = tn.gc_stale_snapshots(
mw, "cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
now=now, mounts_file=str(mounts),
list_snapshots=mw.list_snapshots,
)
assert remaining == [orphan]
def test_it_collects_NOTHING_when_the_query_fails(self, tmp_path):
# Cannot enumerate => cannot know what is ours => delete nothing.
import datetime as dt
import truecloud_nested as tn
mounts = tmp_path / "mounts"
mounts.write_text("")
class Broken(FakeMiddleware):
def call_sync(self, method, *args):
if method.endswith(".query"):
raise RuntimeError("middleware is having a day")
return super().call_sync(method, *args)
assert tn.gc_stale_snapshots(
Broken(["Tap/apps/x@cloud_backup-5-20260713030000"]),
"cloud_backup-5", "Tap@cloud_backup-5-20260714115900",
now=dt.datetime(2026, 7, 14, 12, 0, 0, tzinfo=dt.UTC),
mounts_file=str(mounts),
) == []
class TestEnumerationComesFromZfsNotMiddleware:
"""The bug that a source check can never catch, found only by running it.
Porting the deleted private `zfs.dataset.query` to the public
`pool.dataset.query` looked obviously right: the method exists, it is
documented, iX will not delete it. Every test passed and `compat.py` went green
on TrueNAS 26.
It was wrong. The public query applies a VISIBILITY POLICY -- on a real box it
returns 205 of 274 datasets, hiding `ix-apps/*`, `.system/*` and `.ix-virt/*`.
On the production pool that is 84 of 270, and `ix-apps` holds LIVE APPLICATION
DATA. The staging plan would have omitted every one of them, and `plan_staging`
would never have seen them, so they would not even appear in `skipped`. A green
backup, silently missing data -- the precise failure this module exists to
prevent.
The snapshot query lies the same way, so the sweep would orphan one snapshot per
hidden dataset, forever.
Hence: READ from ZFS, MUTATE through middleware. These tests hold that line.
"""
def test_query_filesystems_shells_out_to_zfs(self):
import truecloud_nested as tn
seen = []
class R:
returncode = 0
stdout = (
"scratch\t/mnt/scratch\tyes\n"
"scratch/ix-apps\t/mnt/scratch/ix-apps\tyes\n" # middleware HIDES this one
"scratch/.system\tlegacy\tno\n"
)
stderr = ""
def runner(cmd):
seen.append(cmd)
return R()
rows = tn.query_filesystems(runner=runner)
assert seen and seen[0][0] == "zfs", "must read ZFS, not call middleware"
names = [r["name"] for r in rows]
assert "scratch/ix-apps" in names, (
"ix-apps is exactly what pool.dataset.query hides, and exactly what "
"holds live app data. If it is not here the backup omits it silently."
)
# ...and it still speaks the shape the planner expects.
row = next(r for r in rows if r["name"] == "scratch/.system")
assert row["properties"]["mountpoint"]["value"] == "legacy"
assert row["properties"]["mounted"]["value"] == "no"
def test_a_failing_zfs_raises_rather_than_returning_an_empty_list(self):
# The whole failure class in one assertion. "No datasets" and "the command
# broke" must never look the same: a caller that cannot tell them apart
# stages nothing, sweeps nothing, and reports success.
import truecloud_nested as tn
class R:
returncode = 1
stdout = ""
stderr = "cannot open 'scratch': no such pool"
with pytest.raises(tn.ZfsError, match="no such pool"):
tn.query_filesystems(runner=lambda cmd: R())
with pytest.raises(tn.ZfsError):
tn.list_snapshot_names("scratch", runner=lambda cmd: R())
def test_list_snapshot_names_reads_the_whole_tree_from_zfs(self):
import truecloud_nested as tn
seen = []
class R:
returncode = 0
stdout = "scratch@s\nscratch/ix-apps@s\n"
stderr = ""
def runner(cmd):
seen.append(cmd)
return R()
names = tn.list_snapshot_names("scratch", runner=runner)
assert names == ["scratch@s", "scratch/ix-apps@s"]
assert "-t" in seen[0] and "snapshot" in seen[0] and "-r" in seen[0]
assert "scratch/ix-apps@s" in names, (
"pool.snapshot.query hides this; a sweep that cannot see it orphans it "
"on every single run, forever"
)